mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-09-30 12:39:42 +02:00
Compare commits
230
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
09bf00c815 | ||
|
|
2e19fc0430 | ||
|
|
30f35490f6 | ||
|
|
322801052b | ||
|
|
736a6b8b7b | ||
|
|
6525ade530 | ||
|
|
947ff6f6fa | ||
|
|
ce405a4cff | ||
|
|
8e5691b05c | ||
|
|
98e37bf895 | ||
|
|
5ded2ed1a3 | ||
|
|
76ea090a67 | ||
|
|
d30cac4440 | ||
|
|
f3cb7696f0 | ||
|
|
5080390e2c | ||
|
|
f7e2975883 | ||
|
|
f60bf93c99 | ||
|
|
f4ba4d2cb1 | ||
|
|
fa8ebe0068 | ||
|
|
b0e493d462 | ||
|
|
631913f04c | ||
|
|
bb959c4aac | ||
|
|
24ed43935c | ||
|
|
cb9149879d | ||
|
|
f44d597450 | ||
|
|
cdbde9f36f | ||
|
|
f07905b193 | ||
|
|
05c94f5ac0 | ||
|
|
aaf22909bc | ||
|
|
94908ffdb5 | ||
|
|
82fe3cf684 | ||
|
|
6946ca0b8a | ||
|
|
ea4b940cef | ||
|
|
c6f428e687 | ||
|
|
c8f3981b0c | ||
|
|
499d35566b | ||
|
|
210da991d5 | ||
|
|
da999b130e | ||
|
|
cbc54fc98d | ||
|
|
4e2c1b9989 | ||
|
|
1c94995290 | ||
|
|
f485085174 | ||
|
|
19aabe34d2 | ||
|
|
98fa8c00d1 | ||
|
|
869a507482 | ||
|
|
854bcb99aa | ||
|
|
9ee6bf113b | ||
|
|
66d4c483c7 | ||
|
|
ff13234b3d | ||
|
|
0af80b417c | ||
|
|
52d113ab12 | ||
|
|
74662dd788 | ||
|
|
0a89505358 | ||
|
|
5387587a64 | ||
|
|
9c0a9bf8e3 | ||
|
|
210154f96f | ||
|
|
bbc960a8ff | ||
|
|
f18097cb23 | ||
|
|
62b0039dc5 | ||
|
|
174976fc40 | ||
|
|
5ae54536cb | ||
|
|
2d2a455dd2 | ||
|
|
943f04ba53 | ||
|
|
69d8a9ea6f | ||
|
|
405eb50ba3 | ||
|
|
b6f15b30c6 | ||
|
|
736f35da7f | ||
|
|
6866a617a8 | ||
|
|
a415948736 | ||
|
|
d68cba9432 | ||
|
|
497cbe55bd | ||
|
|
9a0e665f72 | ||
|
|
4bbe2b7ff6 | ||
|
|
a7928f5c64 | ||
|
|
4fa44f2e55 | ||
|
|
829b202f51 | ||
|
|
86234db1ef | ||
|
|
86c78fece3 | ||
|
|
c790166564 | ||
|
|
f4dcfbe6ca | ||
|
|
d19895651d | ||
|
|
c5b59633d8 | ||
|
|
f39beb3326 | ||
|
|
cf3183abf7 | ||
|
|
15a43894f9 | ||
|
|
e20aa1d4d8 | ||
|
|
aa28ef048c | ||
|
|
67f6ed3168 | ||
|
|
2d4616f059 | ||
|
|
35f8f9d19f | ||
|
|
c992784681 | ||
|
|
3a7be356ae | ||
|
|
a6a572e635 | ||
|
|
26416f98de | ||
|
|
084d7b7328 | ||
|
|
a4cdb352be | ||
|
|
d81454b6f9 | ||
|
|
00f1b9228a | ||
|
|
13d069e1e5 | ||
|
|
fa4c36c2a5 | ||
|
|
fe2c03b2cc | ||
|
|
4e3f7ac36b | ||
|
|
089283e0b3 | ||
|
|
3b85001fed | ||
|
|
1513067a7f | ||
|
|
623fedf5b7 | ||
|
|
1410362e5b | ||
|
|
92ae46246c | ||
|
|
6831d79127 | ||
|
|
b01ed611c4 | ||
|
|
8d094b086c | ||
|
|
ecc6f30e24 | ||
|
|
7da9fb4d53 | ||
|
|
b025047cbf | ||
|
|
f11bee72f5 | ||
|
|
78356d7fd0 | ||
|
|
0da7f652b4 | ||
|
|
6ccab925b1 | ||
|
|
4b51ba306e | ||
|
|
aaad031510 | ||
|
|
29efd0e970 | ||
|
|
a6cf4c2b2a | ||
|
|
831af88579 | ||
|
|
3a106bd048 | ||
|
|
752374abc7 | ||
|
|
adfc4fbb1c | ||
|
|
b0b058891c | ||
|
|
4a1ad8d194 | ||
|
|
193ce6348d | ||
|
|
8668b4b352 | ||
|
|
312ca541e6 | ||
|
|
40ce91f098 | ||
|
|
250a53125a | ||
|
|
14ea9f630f | ||
|
|
c8ac04662d | ||
|
|
7c2a49d432 | ||
|
|
8fcfdb1e6e | ||
|
|
45ad9de89e | ||
|
|
a070fc43ea | ||
|
|
aa35c1a0c4 | ||
|
|
c13b3c55d3 | ||
|
|
053a6d238d | ||
|
|
9b9f2c21e9 | ||
|
|
a80eda8e4c | ||
|
|
5d42f64393 | ||
|
|
c891a8045d | ||
|
|
c942bb5dfb | ||
|
|
f98922063a | ||
|
|
5671c20076 | ||
|
|
d5375d7f0b | ||
|
|
1692238531 | ||
|
|
f9510f8a54 | ||
|
|
62ca7f1381 | ||
|
|
93df8188a5 | ||
|
|
94abcf29dc | ||
|
|
23d91a6ee1 | ||
|
|
8a6570e22d | ||
|
|
0aafabd28d | ||
|
|
4add38c4b1 | ||
|
|
e6df0c4094 | ||
|
|
6bb3d66004 | ||
|
|
161f1da2eb | ||
|
|
87e787e934 | ||
|
|
3533c4332b | ||
|
|
6fc772f697 | ||
|
|
527ce10491 | ||
|
|
26a4dd2879 | ||
|
|
e087198056 | ||
|
|
2e266380f8 | ||
|
|
3363d25876 | ||
|
|
b793ff3294 | ||
|
|
a68b2c5bc5 | ||
|
|
89f9e0becb | ||
|
|
6cc7b4328b | ||
|
|
ce22c2a608 | ||
|
|
338f0e460d | ||
|
|
8595e84c56 | ||
|
|
696339fe12 | ||
|
|
6c744f8677 | ||
|
|
c50bb02e62 | ||
|
|
086ea4dd7c | ||
|
|
b03780dfd2 | ||
|
|
ff10a50bc0 | ||
|
|
3e568511f8 | ||
|
|
64b33eb630 | ||
|
|
1e1db947c5 | ||
|
|
b1614e89fc | ||
|
|
0aa16cd4d3 | ||
|
|
4ed86aa0cd | ||
|
|
a15b81db77 | ||
|
|
341c7ccc59 | ||
|
|
c1719e04e5 | ||
|
|
d33f3803a1 | ||
|
|
477e73039c | ||
|
|
b374032699 | ||
|
|
b6efdfccf4 | ||
|
|
b191f3c2c6 | ||
|
|
9fd856a918 | ||
|
|
be449e6e9e | ||
|
|
04de943b7f | ||
|
|
9e7c537e14 | ||
|
|
55bff4a4bf | ||
|
|
fa02bd4503 | ||
|
|
5bde897752 | ||
|
|
6c55ce3f8d | ||
|
|
a30524060a | ||
|
|
00fb3b0908 | ||
|
|
5aa59c70cc | ||
|
|
ffccde4f7d | ||
|
|
bec3da3d31 | ||
|
|
6e89eb9ec1 | ||
|
|
94aa53c65b | ||
|
|
e88b971bb7 | ||
|
|
8406c497e2 | ||
|
|
40b4aba043 | ||
|
|
4b44988bfc | ||
|
|
316d0a4c82 | ||
|
|
1184720648 | ||
|
|
b067aad9b6 | ||
|
|
19a3d7c773 | ||
|
|
085f4acb60 | ||
|
|
d26f26fe34 | ||
|
|
091df2b6d8 | ||
|
|
5f775b1ab1 | ||
|
|
da51193264 | ||
|
|
fa18eeef35 | ||
|
|
a9f26bd03a | ||
|
|
0afd4e1cdc | ||
|
|
b6293959d2 | ||
|
|
c01edcbbb8 |
@@ -0,0 +1,89 @@
|
||||
# Contributing to Codeman
|
||||
|
||||
Thanks for wanting to help! Codeman is a small project with a fast loop: issues usually get a response within a day, good PRs get reviewed quickly, and every release credits its contributors and bug reporters by name in the release notes. This guide gets you from clone to merged PR without stepping on the traps.
|
||||
|
||||
## The short version
|
||||
|
||||
1. **Bugs**: open an issue with your OS, install method (installer / npm / git clone), browser, and which CLI + version the session was running.
|
||||
2. **Questions and ideas**: use [Discussions](https://github.com/Ark0N/Codeman/discussions), not issues.
|
||||
3. **Small fixes** (docs, typos, a new skin, a translation): just send the PR.
|
||||
4. **Anything bigger**: open an issue or Discussion first and get a nod before building. Codeman has strong architectural invariants, and a design chat up front is what turns a big idea into a merged PR instead of a stalled one. This flow works: features like Clone Repo (#236) went idea, then design discussion, then review, then shipped.
|
||||
5. **Security issues**: never a public issue. See [SECURITY.md](SECURITY.md).
|
||||
|
||||
## Dev setup
|
||||
|
||||
Requirements: Node.js 22+ (see `.nvmrc`), tmux, and at least one supported agent CLI on your PATH (Claude Code is the primary one).
|
||||
|
||||
```bash
|
||||
git clone https://github.com/Ark0N/Codeman.git
|
||||
cd Codeman
|
||||
npm install # postinstall builds the vendored xterm addon bundles
|
||||
npm run dev # dev server on http://localhost:3000
|
||||
```
|
||||
|
||||
The frontend is plain JS served from `src/web/public/` with no bundler in dev: edit a `.js`/`.css` file and reload the page. The one exception is `index.html`, which is read once at server start, so markup changes need a server restart.
|
||||
|
||||
## Before you push
|
||||
|
||||
CI runs all of these, so save yourself a round trip:
|
||||
|
||||
```bash
|
||||
npm run typecheck # tsc --noEmit, strict mode
|
||||
npm run lint
|
||||
npm run format:check
|
||||
npm run check:frontend-syntax # syntax-checks the plain-JS frontend modules
|
||||
```
|
||||
|
||||
### Tests
|
||||
|
||||
```bash
|
||||
npm test # the gate — exactly what CI runs
|
||||
npm test -- test/<file>.test.ts # one file
|
||||
```
|
||||
|
||||
`npm test` is the same suite CI runs, so a green run locally means a green run there. It leaves out three suites that cannot pass on an arbitrary machine, each with its own command:
|
||||
|
||||
```bash
|
||||
npm run test:browser # Playwright + chromium (+ a live server; codex-predictive-echo needs a real codex binary)
|
||||
npm run test:mobile # the above plus environment-specific PNG baselines
|
||||
npm run test:perf # wall-clock benchmarks — run on an otherwise idle machine
|
||||
npm run test:all # literally everything, environmental failures included
|
||||
```
|
||||
|
||||
Expect `test:browser`/`test:mobile`/`test:perf` to fail where the machine cannot provide what they need; read that as "not runnable here", not as a regression. `config/test-suites.ts` holds the globs, and both configs derive from it, so the exclusions and those runners cannot drift apart.
|
||||
|
||||
If you add a test that binds a port, pick a unique one at 3150 or above (search the repo for `const PORT =` first). Never 3000.
|
||||
|
||||
Tests are tmux-safe by design: under vitest, the tmux layer becomes an in-memory mock, so tests cannot touch real sessions.
|
||||
|
||||
## Finding your way around
|
||||
|
||||
- Every source file starts with a `@fileoverview` JSDoc block. Read it before diving into the file, it is the map.
|
||||
- [`CLAUDE.md`](../CLAUDE.md) at the repo root is the densest architecture primer in the repo. It is written for AI coding agents, but the invariants and gotchas in it apply to humans exactly the same, and most review feedback on PRs traces back to something already written there.
|
||||
- Deep mechanisms and the history behind each rule live in [`docs/architecture-invariants.md`](../docs/architecture-invariants.md).
|
||||
- Third-party extension surfaces are documented in [`docs/extending-codeman.md`](../docs/extending-codeman.md).
|
||||
|
||||
## Great first contributions
|
||||
|
||||
These are well-fenced areas where a first PR is genuinely easy to get right:
|
||||
|
||||
- **A new theme skin.** A skin is four things kept in sync: the `html[data-skin="…"]` token block in `styles.css`, the xterm ANSI palette in `terminal-ui.js`, the pre-paint allowlist and the settings picker (both in `index.html`). `test/skin-themes.test.ts` statically checks the sync, so if the test passes, your skin works.
|
||||
- **A new language.** `src/web/public/i18n.js` is dependency-free, English is the canonical source, and `zh-CN` is a complete example to copy. Add your language's entries and register it in `SUPPORTED_LANGUAGES`.
|
||||
- **Docs.** If you got stuck on something and then figured it out, the sentence that would have unstuck you is a PR.
|
||||
- Anything labeled [`good first issue`](https://github.com/Ark0N/Codeman/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22).
|
||||
|
||||
Bigger extension points worth discussing first: new CLI backends (the pluggable resolver pattern has absorbed six CLIs so far; `docs/extending-codeman.md` and `docs/opencode-integration.md` show the shape), and real-device testing reports, especially mobile, which always find things emulation cannot.
|
||||
|
||||
## PR expectations
|
||||
|
||||
- **One change per PR.** Small and focused reviews fast; a grab-bag stalls.
|
||||
- Target the `master` branch.
|
||||
- **Keep your branch mergeable.** A PR with conflicts silently gets no CI runs at all (GitHub quirk), so rebase or merge master when conflicts appear.
|
||||
- Include or update tests when you change behavior. Route handlers have a lightweight pattern in `test/routes/` using `app.inject()` (no live server needed).
|
||||
- Formatting is Prettier with a deliberately narrow scope (`npm run format`), several frontend files are hand-formatted on purpose and excluded via `.prettierignore`. Don't "fix" a file by adding it back into Prettier's scope.
|
||||
- Don't bump versions or touch `CHANGELOG.md`; releases are handled by the maintainer via changesets after merge.
|
||||
- AI-assisted contributions are welcome (much of Codeman is built that way), with one condition: you must understand what you're submitting and have actually run it. "The model said it works" is not a test.
|
||||
|
||||
## Conduct
|
||||
|
||||
Be kind, be direct, assume good faith. Report unacceptable behavior privately via the contact in [SECURITY.md](SECURITY.md).
|
||||
@@ -87,10 +87,25 @@ jobs:
|
||||
fi
|
||||
|
||||
- name: Run unit & integration tests
|
||||
# Excludes the browser-driven mobile suite (test/mobile/**); see config/vitest.ci.config.ts.
|
||||
# Excludes the suites that need chromium, per-machine PNG baselines or a
|
||||
# quiet machine — see config/test-suites.ts for the list and the reason
|
||||
# behind each entry. Identical to what `npm test` runs locally.
|
||||
# Safe in CI: TmuxManager no-ops all shell commands under VITEST (test/setup.ts).
|
||||
run: npm run test:ci
|
||||
|
||||
# Note: The browser-driven mobile suite (test/mobile/**) is excluded from CI —
|
||||
# it needs a live server + chromium + environment-specific PNG baselines.
|
||||
# Run it locally/manually. All other tests run via the `test` job above.
|
||||
- name: Run xterm-zerolag-input package tests
|
||||
# Layers 1-3 of the predictive-echo suites (unit laws, fixture replay,
|
||||
# seeded fuzz): deterministic, no browser, no live server. Depends on
|
||||
# the ROOT `npm ci` above — workspaces hoist the package's vitest into
|
||||
# the root node_modules; do not add a separate install here.
|
||||
run: npx vitest run
|
||||
working-directory: packages/xterm-zerolag-input
|
||||
|
||||
# Note: three suites are excluded from CI, each with its own local runner:
|
||||
# npm run test:browser Playwright + chromium (+ a live server, and a real
|
||||
# codex binary for codex-predictive-echo)
|
||||
# npm run test:mobile the above plus environment-specific PNG baselines
|
||||
# npm run test:perf wall-clock benchmarks; need an otherwise idle machine
|
||||
# config/test-suites.ts holds the globs; the configs derive from it so the
|
||||
# exclusions here and those runners cannot drift apart. Everything else runs in
|
||||
# the `test` job above, which is the same thing `npm test` runs.
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
name: Sync Wiki
|
||||
|
||||
# Publishes docs/wiki/ to the repository's GitHub wiki.
|
||||
#
|
||||
# The wiki is a separate git repo with no CI and no review, so the source of truth
|
||||
# lives in docs/wiki/ and this workflow mirrors it. Browser edits to the wiki are
|
||||
# overwritten by the next sync; fix pages with a PR against docs/wiki/ instead.
|
||||
#
|
||||
# One-time setup: GitHub only creates <repo>.wiki.git once the first page has been
|
||||
# saved in the browser. Save a stub page at /wiki/_new before the first run.
|
||||
#
|
||||
# Token: GITHUB_TOKEN can push to the wiki on most repos but not all. If a run fails
|
||||
# with 403, add a fine-grained PAT with wiki write access as the WIKI_TOKEN secret;
|
||||
# it is preferred automatically when present. Note the 403 usually surfaces on the
|
||||
# PUSH, not the clone: this repo is public, so a read-only token still clones the
|
||||
# wiki fine. Both steps carry the hint.
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [master]
|
||||
paths:
|
||||
- 'docs/wiki/**'
|
||||
- '.github/workflows/wiki-sync.yml'
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency: ${{ github.workflow }}
|
||||
|
||||
jobs:
|
||||
sync:
|
||||
name: Push docs/wiki to the wiki
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
steps:
|
||||
- name: Checkout repo
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Clone wiki
|
||||
env:
|
||||
WIKI_TOKEN: ${{ secrets.WIKI_TOKEN || secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if ! git clone "https://x-access-token:${WIKI_TOKEN}@github.com/${GITHUB_REPOSITORY}.wiki.git" wiki 2>"${RUNNER_TEMP}/clone-err.txt"; then
|
||||
cat "${RUNNER_TEMP}/clone-err.txt"
|
||||
echo "::error::Could not clone ${GITHUB_REPOSITORY}.wiki.git. If this says 'Repository not found', the wiki has never had a page: save one at https://github.com/${GITHUB_REPOSITORY}/wiki/_new and re-run. If it says 403, add a WIKI_TOKEN secret."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Mirror pages
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
# The mirror deletes before it copies, so an empty source would wipe
|
||||
# every published page and the commit step would happily push that. A
|
||||
# MISSING directory already fails safely (cp aborts under set -e); an
|
||||
# empty one does not, so check explicitly. This is the one failure mode
|
||||
# here that destroys something a browser edit cannot get back.
|
||||
if [ ! -d docs/wiki ]; then
|
||||
echo "::error::docs/wiki does not exist. Refusing to mirror, which would delete the entire published wiki."
|
||||
exit 1
|
||||
fi
|
||||
pages=$(find docs/wiki -maxdepth 1 -name '*.md' | wc -l)
|
||||
if [ "$pages" -eq 0 ]; then
|
||||
echo "::error::docs/wiki contains no .md pages. Refusing to mirror, which would delete the entire published wiki."
|
||||
exit 1
|
||||
fi
|
||||
echo "Mirroring ${pages} pages."
|
||||
|
||||
find wiki -mindepth 1 -maxdepth 1 ! -name '.git' -exec rm -rf {} +
|
||||
cp -R docs/wiki/. wiki/
|
||||
|
||||
- name: Stamp the documented version
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# _Footer.md renders on every page and used to carry a hand-written
|
||||
# version, which went stale on every release because nothing refreshed
|
||||
# it. It carries {{VERSION}} instead and the series is stamped here.
|
||||
series="$(node -p "require('./package.json').version.split('.').slice(0,2).join('.') + '.x'")"
|
||||
# grep exits 1 when it matches nothing, which under `set -o pipefail`
|
||||
# would fail the step instead of warning, so test before substituting.
|
||||
if grep -rlq '{{VERSION}}' wiki/; then
|
||||
grep -rlZ '{{VERSION}}' wiki/ | xargs -0 -r sed -i "s/{{VERSION}}/${series}/g"
|
||||
else
|
||||
echo "::warning::No {{VERSION}} placeholder found in docs/wiki. The published version line can no longer be refreshed automatically."
|
||||
fi
|
||||
if grep -rq '{{VERSION}}' wiki/; then
|
||||
echo "::error::A {{VERSION}} placeholder survived substitution and would be published verbatim."
|
||||
exit 1
|
||||
fi
|
||||
echo "Stamped version ${series}."
|
||||
|
||||
- name: Commit and push
|
||||
run: |
|
||||
set -euo pipefail
|
||||
cd wiki
|
||||
git config user.name 'github-actions[bot]'
|
||||
git config user.email '41898282+github-actions[bot]@users.noreply.github.com'
|
||||
git add -A
|
||||
if git diff --quiet --cached; then
|
||||
echo "Wiki already up to date."
|
||||
exit 0
|
||||
fi
|
||||
git commit -m "docs: sync wiki from docs/wiki @ ${GITHUB_SHA:0:7}"
|
||||
if ! git push 2>"${RUNNER_TEMP}/push-err.txt"; then
|
||||
cat "${RUNNER_TEMP}/push-err.txt"
|
||||
echo "::error::Could not push to ${GITHUB_REPOSITORY}.wiki.git. A 403 here means the token can read the wiki but not write it, which is the usual GITHUB_TOKEN case: add a fine-grained PAT with wiki write access as the WIKI_TOKEN secret."
|
||||
exit 1
|
||||
fi
|
||||
@@ -10,7 +10,7 @@ sections here.
|
||||
Quick pointers:
|
||||
|
||||
- Type check: `tsc --noEmit` · Lint: `npm run lint` · Format: `npm run format:check`
|
||||
- Targeted tests only: `npm test -- test/<file>.test.ts` (bare `npm test` is unsafe in managed sessions)
|
||||
- Tests: `npm test` (the CI gate, safe to run bare) or `npm test -- test/<file>.test.ts` for one file
|
||||
- Route tests use `app.inject()`; new tests needing ports must pick a unique `const PORT =`
|
||||
- Branch off `master` for all work; Conventional Commit-style messages (`fix(mobile): ...`)
|
||||
- Never commit secrets or local state from `~/.codeman/`
|
||||
|
||||
+693
@@ -1,5 +1,698 @@
|
||||
# aicodeman
|
||||
|
||||
## 1.19.6
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Wiki user manual, a phone tab tap-zone fix, per-parent lineage colours, and two robustness fixes.
|
||||
- **Wiki**: `docs/wiki/` is now a 30-page user manual (installation, quick start, the dashboard, agent CLIs, remote/Docker cases, hooks, security, HTTP API, troubleshooting and more), published to the GitHub wiki by a sync workflow on every push that touches it.
|
||||
- **Phone tabs**: on a narrow phone the active tab's geometric centre could land on its gear icon, so a thumb aiming at the tab opened Session Options instead of switching. The active tab's name now reserves a minimum width, and a static test recomputes the clearance from the stylesheet so widening the icons fails there rather than on a phone.
|
||||
- **Lineage lines**: the arcs between a tab and the tabs it spawned are now coloured per SPAWNING tab, so every arc leaving one tab shares a colour and the strip reads as "these came from w1, those from w2". A child that spawns in turn gets its own colour, so a chain changes colour at each generation.
|
||||
- **File access**: `validateSessionFilePath()` now canonicalizes the workspace as well as the candidate path before comparing them. Resolving only the candidate made a workspace reached through a symlink (`/tmp` on macOS, symlinked project dirs, bind-mounted case paths) report a spurious escape and refuse every read and write in that session. Escapes are still refused.
|
||||
- **Respawn**: a cycle step that is stopped mid-write no longer revives the state machine. `stop()` could land during the `await` on the kickstart / update / clear / init write, after which the controller set itself back to a waiting state and kept running.
|
||||
|
||||
### Thanks
|
||||
- @aakhter for the symlink-safe workspace confinement fix (#314) and the respawn stop-race fix (#315).
|
||||
|
||||
- 98e37bf: Session List Layout gains a third option, "Left sidebar", whose rows carry the same per-session detail the home screen shows.
|
||||
|
||||
The sidebar previously had one row style: a name and a folder. That is the whole story a tab can tell, but a docked column is not a tab strip — it has width to spare and a row per session either way, and the information that was missing is exactly the information the desktop home rail and the phone overview already put on screen. So the new option lifts it onto the rows: when the session was first created, how long it has been in the state it is in, and a status pill naming that state.
|
||||
- The old "Left sidebar" is now **"Left sidebar simple"** and is unchanged, down to the byte — the stored value stays `sidebar`, so anyone already using it keeps exactly the layout they chose. The new option is `sidebar-rich`.
|
||||
- Both sidebar values are the SAME layout and both set `data-session-list="sidebar"`; row detail rides on a separate `data-sidebar-detail` attribute. That is deliberate: every `isSessionSidebarActive()` call site and every `html[data-session-list="sidebar"]` rule in styles.css and mobile.css keeps matching both, untouched.
|
||||
- Which state a session is in, and which stamp measures it, come from `_mobileOverviewState()` / `_mobileOverviewSince()` rather than being re-derived — the sidebar, the home rail and the phone overview cannot disagree about what "working" means. A working row is measured from the turn's last Enter, not from its last repaint, so a running turn reads `working 12m` instead of `0m`.
|
||||
- The stamps refresh in place on a 20s clock instead of re-rendering: a rebuild would restart every load spinner and alert animation in the list, twice a minute. The clock only runs while rich rows are on screen.
|
||||
- The column widens to 300px for the extra line, and the collapsed 44px rail and the handheld drawer are explicitly held back from that width.
|
||||
|
||||
- 947ff6f: `npm test` is now the CI gate and is safe to run bare; the suites it cannot run each got their own command.
|
||||
|
||||
`npm test` ran the everything-config, which fails ~87 tests on a clean master on any machine without chromium, a free port and per-machine PNG baselines. That made the repo's most obvious command useless as a pass/fail signal, and the docs had accumulated "never run bare `npm test`" warnings in four files to work around it. It now runs `config/vitest.ci.config.ts` — exactly what CI runs — so local green means CI green.
|
||||
- New: `test:browser` (5 Playwright files), `test:perf` (2 wall-clock benchmarks), `test:all` (the old everything-behaviour, kept reachable). `test:ci` and `test:mobile` are unchanged; `test:watch` and `test:coverage` follow `test` onto the gate's config.
|
||||
- The exclusion list moved to `config/test-suites.ts`, with the reason each suite cannot run in CI. Every config derives from it, so the gate's excludes and the runners' includes cannot drift.
|
||||
- That drift was a silent hole, not a tidiness problem: a file excluded from CI and added to no runner is tested by NOTHING, and every command stays green, because vitest counts "no files matched" as success. `test/test-suite-partition.test.ts` now fails if any test file is reachable by no runner or by two.
|
||||
- ⚠️ A file filter must match its runner: `npm test -- test/mobile/keyboard.test.ts` matches nothing and exits green having run zero tests, because the gate excludes that path. Use `npm run test:mobile -- <file>`. Documented in CLAUDE.md, and the one place that recommended the old form was corrected.
|
||||
- Docs synced: CLAUDE.md, AGENTS.md, .github/CONTRIBUTING.md, both READMEs, and two ci.yml comments that claimed only `test/mobile/**` was excluded (it is three suites, and 5 Playwright files rather than 3).
|
||||
|
||||
## 1.19.5
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Closing the session you are looking at now always moves you to the next tab.
|
||||
|
||||
The delete request and its own `session_deleted` broadcast raced each other: the close path selected the next tab, while the broadcast handler cleared the active session and showed the home screen, and whichever ran first decided what you saw. On one build, closing a tab either switched sessions or dumped you on the welcome screen depending on timing. The close now owns that handoff from beginning to end, and the broadcast handler stays out of the way for a close started in that tab. A session deleted from somewhere else still returns you to the home screen, which is the honest answer when what you were looking at was taken away.
|
||||
|
||||
The next tab is also picked from sessions that still exist, so a stale entry in the tab order can no longer name a tab that is already gone.
|
||||
|
||||
## 1.19.4
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Only a human opening a session clears its yellow "waiting for input" tab alert.
|
||||
|
||||
1.19.2 made that clear durable and cross-device, which also meant the app itself could spend it: restoring your last session on page load, a popped-out window opening its target, and the fallback to another tab after you close the active one all counted as "I checked it", so a yellow tab could clear itself before you ever saw it. Those three app-driven selections are now marked and skip the acknowledgement, so the alert survives until you actually open the session.
|
||||
|
||||
Everything a human does still clears it, on every surface: tapping a tab, tapping a row on the phone home screen, the keyboard tab shortcuts, and submitting a prompt into the session. The flag defaults to user-initiated, so a selection path nobody marked keeps acknowledging rather than leaving an alert nothing can clear.
|
||||
|
||||
## 1.19.3
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Red "needs you" tab alerts now follow the dialog instead of the keyboard.
|
||||
|
||||
Typing in the terminal no longer clears a red alert. It used to clear every pending alert on the device you typed on, but a permission or question dialog ignores keystrokes that are not one of its options, so the dialog was still open and still blocking: the other devices stayed red and a reload brought the red back on the first one. Input now spends the yellow idle alert only, and it does that through the server-side acknowledgement added in 1.19.2, so the clear is durable and reaches every device.
|
||||
|
||||
A dialog answered in the terminal now clears by itself. Claude Code fires no "permission answered" hook, so the item stayed pending until the whole turn ended, and any page load in between re-armed a red alert for a dialog that was long gone. Listing approvals now re-captures the pane and resolves items whose dialog is no longer on screen, using the same conservative check the answer path already uses: only an item whose original frame parsed numbered options can be dropped this way, so an unreadable capture keeps the alert rather than losing a live one. Measured against a real AskUserQuestion dialog: the stale item cleared 5 seconds ahead of the stop hook that used to be the only signal, while a dialog still on screen survived 11 consecutive listings over 55 seconds untouched.
|
||||
|
||||
## 1.19.2
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Yellow "waiting for input" tab alerts now stay cleared once you have checked them, on every device.
|
||||
|
||||
Viewing a session used to clear its idle alert in that browser's memory only. The server-side approval store still held the prompt, so the next page load seeded the alert straight back and a tab you had already checked went yellow again, while your other devices never heard about the click at all. Opening a session now acknowledges its pending idle prompt server-side (`POST /api/approvals/session/:sessionId/viewed`, a new `acknowledgedAt` field on approval items, broadcast as `approval:updated`), so the clear survives reloads and reaches every connected client.
|
||||
|
||||
Acknowledgement is deliberately not resolution: the prompt is still unanswered, so the item stays in the Approvals Inbox, stays answerable, and stays available as Read My Mind context, it just stops arming the tab alert. Permission and question dialogs are never acknowledged this way, since looking at a dialog does not answer it, so the red "needs you" alert survives being viewed. Clicking the tab you are already on now clears the alert as well; that path returned early before, so an alert armed on the active tab could not be cleared by clicking at all.
|
||||
|
||||
## 1.19.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Follow-up hardening from the 1.19.0 reviews, across all three of that release's areas (#309, #310, #311).
|
||||
|
||||
Home screens: the activity ordering introduced in 1.19.0 now stays truthful. Hook events push a session state broadcast, so a blocked session ranks by a fresh stamp instead of whatever the page loaded with; a working row with no recorded submit shows the same stamp it sorts by; Alt+1..9 resolves through the live sessions the tabs actually paint, so a stale id in the saved order can no longer shift every number off its target; and the "most recently quiet" ordering survives restarts, since recovery now restores each session's previous activity stamp from state.json instead of restamping everything at boot (previously every deploy flattened the ordering to tab order).
|
||||
|
||||
Files and sidebar: playable media extensions are pinned to the attachment registry by a parity test, so an in-workspace .m4a/.flac/.opus opens the preview player instead of the log viewer; /etc paths no longer render as links that can only 403; the sidebar session count counts the rows actually on screen (web tabs included, filtered rows excluded) and follows the filter box; connectors re-anchor on incremental renders in sidebar layout; and ~/.claude.json plus ~/.claude/settings(.local).json are blocked from file serving, home-anchored only, so case-level .claude files stay viewable.
|
||||
|
||||
Workspace hooks: the install-vs-refresh decision is one shared core that every claude create path routes through, so the workspaceHooksEnabled setting now also applies to cron jobs, legacy scheduled runs, and plan-orchestrator one-shots; a shell session in a docker case no longer authors a hooks block; the boot sweep no longer resurrects a deleted workspace as an empty directory; and the statusLine exporter got the same remote-attach and cwd-fallback guards as the hooks install.
|
||||
|
||||
## 1.19.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- c01edcb: Add an optional collapsible left session sidebar as an alternative to the header tab strip.
|
||||
|
||||
With many concurrent sessions the horizontal strip wraps into several rows and stops being scannable. The new layout puts the session list in a vertical `<aside>` with a filter box and a live session count, collapsible to a 44px rail that keeps the status dots and task badges visible.
|
||||
|
||||
Opt-in via Settings → Layout → Tabs → Session List Layout; the default stays the header strip, so nothing changes unless you switch. Both layouts share one `#sessionTabs` element that is re-parented between mount points, so every existing affordance (status, mode badge, alerts, drag-reorder, keyboard navigation, web tabs, subagent windows) behaves identically in both. Below 1024px the sidebar is an off-canvas drawer that overlays the terminal instead of shrinking it. Collapse state persists per device; `Alt+B` toggles it.
|
||||
|
||||
- Codeman hooks now install into every claude workspace at session create, not just cases Codeman created (#304). Linked cases and cloned repos previously ran hook-blind: tab alerts, the Approvals Inbox, and the agent skill's stop/blocked wait signals were silently dead there. The install is an add-only merge that preserves user-authored hooks and leaves malformed files untouched, and a boot sweep heals sessions recovered from a restart. Opt out with the new synced `workspaceHooksEnabled` setting. Note: a `.claude/settings.local.json` can now appear in repos you link as cases; it contains no secrets. Remote SSH attaches and creates without a `workingDir` never write hooks.
|
||||
|
||||
File paths an agent prints are now clickable in both the terminal and the response viewer, opening the file preview overlay, including paths outside the session workspace (#306). Out-of-workspace paths are served through the attachment routes' extension allowlist, realpath confinement, and sensitive-path blocklist; Codeman's own credential-bearing files (`settings.json`, `push-keys.json`, `intents.json`, `state*.json`) are blocked from serving.
|
||||
|
||||
Both home screens (the desktop home tab rail and the phone overview) sort sessions by activity instead of tab order (#303): blocked sessions first with the longest-blocked on top, then running sessions longest-running first, then quiet sessions most recently active first. A turn starting now pushes a session state broadcast so the ordering stays live after page load.
|
||||
|
||||
The codeman agent skill docs teach hook presence as a setting to check rather than a consequence of who created the workspace, and the §0 preamble stamp is bumped to 1.19.0 (#305).
|
||||
|
||||
### Thanks
|
||||
- @christianhaberl designed and built the collapsible left session sidebar (#307)
|
||||
|
||||
## 1.18.4
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Faster agent-skill workers, retuned multi-color lineage arcs, a per-tab pop-out option, reliable tab alerts, and the community launch.
|
||||
- Agent skill: SKILL.md now forbids the standalone preamble check and the pre-spawn reconnaissance turns that were costing whole model turns; the same two-worker spawn measured at 28.6s end to end now runs 20.2s cold and 12.8s warm, with the spawn machinery itself unchanged.
|
||||
- Session lineage lines: arcs now hang from the tab strip's bottom edge (dip cap 104px to 64px, no stacked row offsets), fixing the deep bow on wrapped tab strips and keeping same-row arcs off the second row's tab labels; each spawned worker's arc gets its own color (skin blue first, then matrix green, pink, violet, red, turquoise, orange), assigned per child and stable across re-renders.
|
||||
- Session Options > Session: new "Pop-out button on this tab" per-tab override on top of the general App Settings toggle (per-device).
|
||||
- Tab alerts: pending permission/question alerts now survive page reloads regardless of the Approvals Inbox setting (the alert state machine seeds from the server-side approval store on every load), stay visible on the selected tab until the prompt is actually resolved (the alert paints on a ::before overlay the active tab's styling cannot bury), and render as a steady red/yellow ring with glow and a colored status dot instead of a blink that spent half of every cycle looking like a normal tab. The README carries a live capture of the new alerts.
|
||||
- Community launch: README Community section, .github/CONTRIBUTING.md (dev setup, test safety, great first contributions, PR expectations), and GitHub Discussions.
|
||||
- docs: worker warm-pool design sketch with the measured baselines.
|
||||
|
||||
## 1.18.3
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Fix skill-spawned workers losing their lineage arcs and spawning slowly: a stale user-level agent skill copy (`~/.claude/skills/codeman`, written once by `codeman skill install`) shadowed the fresh per-case injections, so agents ran old recipes (serial spawns with pid polls, no `X-Codeman-Parent-Session` header). Session create now refreshes a marker-owned user-level copy (refresh-only, never installs, foreign/symlink copies untouched) and pre-seeds the skill's preamble into `${XDG_CACHE_HOME:-~/.cache}/codeman-agent-<id>.sh` (0600, local claude sessions only), single-sourced from the new `skills/codeman/preamble.sh` and pinned byte-identical to the SKILL.md heredoc by test. The skill's bootstrap is now a two-line loader with the full block as fallback, cutting measured prompt-to-workers-spawned time from 35s to 10.6s; `spawn_worker` also sends `parentSessionId` in the request body as defense in depth, and the preamble stamp is bumped to 1.18.3 so pre-fix cached preambles self-heal.
|
||||
|
||||
## 1.18.2
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Draw session lineage lines in blue for contrast. The violet arcs sat close to the
|
||||
terminal's own dim foreground, so they lost contrast exactly where they cross text;
|
||||
the colour now comes from each skin's own `--session-blue` token, and the layer is
|
||||
separated from subagent lines by shape, weight and dash pattern rather than hue.
|
||||
- f18097c: Make the `codeman` agent skill spawn workers fast instead of deliberating first.
|
||||
|
||||
Measured against a live server, the API does the whole job (spawn two claude workers,
|
||||
task them, read both answers) in about 10 seconds, so the delay users saw was
|
||||
agent-side: the skill taught serial spawning, made the happy path something to
|
||||
reassemble from five sections on every run, and cost ~16k tokens of mostly failure
|
||||
modes before the first call.
|
||||
- The §0 preamble now defines the verbs instead of describing them: `spawn_worker`,
|
||||
`spawn_workers` (concurrent), `sendwait` and `last_text`. §1 composes them into the
|
||||
whole job in one Bash call, and says to stop reading there.
|
||||
- Dropped two ceremonies the measurements retired: the pid-poll loop (`wait-output`
|
||||
already blocks on the composer) and the agent-driven hooks check, which is now folded
|
||||
into `spawn_worker` itself as a single local grep of the resolved `casePath`, so a
|
||||
name that resolves to a linked case or a hook-less pre-existing directory is refused
|
||||
instead of silently running the job there. Linked cases and raw paths still require
|
||||
the by-hand check, where its absence silently breaks send-and-wait.
|
||||
- The bootstrap's write condition now greps the version stamp, so a stale or truncated
|
||||
preamble file self-heals instead of failing and asking you to `rm` it by hand.
|
||||
- `sendwait` picks a fresh `seq` per call (a fixed default made every second prompt to
|
||||
the same worker a silently-swallowed duplicate) and self-heals stranded delivery: an
|
||||
Ink repaint occasionally eats the Enter, leaving the prompt typed but unsubmitted
|
||||
(observed live), so a timed-out first wait sends one bare `\r` and re-waits by
|
||||
resending the identical frame as a tagged duplicate.
|
||||
- §5 moved to `reference/verbs.md`, leaving an index. SKILL.md is the only part paid on
|
||||
every load and drops from ~16.4k to roughly 9k tokens (~35KB); section numbers and
|
||||
anchors are unchanged, so existing `§5.x` references still resolve.
|
||||
|
||||
## 1.18.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Terminal history and scroll position fixes, a seekable file-viewer video player, and clearer session lineage lines.
|
||||
|
||||
**Terminal scroll position (#259).** Three paths dragged the terminal to the bottom while the user was reading scrollback. Opening or closing the mobile keyboard forced it unconditionally; scroll intent is now captured before the keyboard reflow and restored afterwards. Live writes preserved the viewport only inside a 1500ms window, so a user who scrolled up and then actually read for longer was dragged along by the next repaint; that is now based on position rather than recency. The backpressure refresh, which is server-triggered and so has no gesture to blame, now holds the reader's place too.
|
||||
|
||||
**Terminal history loss (#259 follow-on).** The backpressure refresh rebuilt the terminal from a 1MB tail, which measured as an 869-row buffer coming back with 158 rows: the routine meant to repair the display was discarding most of the scrollback every time SSE backpressure cleared. It now restores full history, falling back to the tail only when the capture would shrink the buffer, so repaint-mode panes are unaffected. It also bails if the user switches tabs mid-fetch, which would otherwise paint one session's history into another's terminal.
|
||||
|
||||
**History truncation is now visible and recoverable (#258).** Truncation was reported by a grey line written into the terminal, which scrolled away with the output it described and read the same whether the rest was one click away or gone forever. `GET /api/sessions/:id/terminal` now reports `truncationReason` (`tail` for an intentional partial replay whose remainder is still retained, `capped` for the byte ceiling) plus `retainedBytes`, and the browser shows a dismissible banner outside terminal output with three honest states: recoverable, which offers a Load full history button, at-ceiling, and exhausted. The button bypasses the scroll cooldown but not the downgrade guard, so it cannot destroy history on a repaint-mode pane.
|
||||
|
||||
**File viewer video (#284).** Closing the preview left the video playing with audible audio and no visible player, since hiding the overlay does not stop a media element and detaching one does not either. Media is now paused, unsourced and reloaded on close and on re-open, which also aborts the in-flight download. The scrub bar was inert because raw file bodies were served as a single `200` with no `Accept-Ranges`, so Chrome reported `video.seekable` as `[0, 0]` and Safari refused to start the media at all. Raw bodies are now streamed and range-aware (`Accept-Ranges` on every response, `206` with `Content-Range` for a range request, `416` past EOF, malformed specs ignored per RFC 9110), with pure, unit-tested parsing in `src/web/http-range.ts`. The attachments raw route gets the same treatment.
|
||||
|
||||
**Session lineage lines (#285).** The arcs joining a tab to the workers it spawned were tuned for two adjacent tabs and flattened into a straight thread across the terminal at the 800-1500px spans they are actually used at, drew a flat overprinted line inside the row gap on a wrapped strip, and were too faint to see at 1:1. Every pair now uses one U-bridge shape anchored on both tabs' bottom edges, with a deeper span-scaled dip and heavier, higher-contrast strokes.
|
||||
|
||||
**Docs.** The pi run mode is now listed in the mode lists that the sixth-backend sweep missed.
|
||||
|
||||
## 1.18.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- Heal a stalled SSE stream with a heartbeat and a client-side staleness watchdog, and make a tab rename apply immediately.
|
||||
|
||||
An `EventSource` that stops delivering does not always error. A proxy that idle-closed the connection, a laptop resumed from sleep, a tailnet reconnect: `onerror` never fires, the header dot stays green, and every SSE-driven surface (tab status dots, sessions created on another device, renames) freezes until the user reloads. Nothing on the client tracked stream liveness at all.
|
||||
- **`sse:heartbeat` is a new named event** under a new Transport category in the registry (155 constants now, both the backend list and the frontend `SSE_EVENTS` copy updated). The server already wrote a keepalive every 15s, but as an SSE `:keepalive` **comment**, and comments are invisible to `EventSource` by spec, so there was nothing a client could observe. `cleanupDeadClients()` now writes the named frame (`{"t":<epoch ms>}`) instead; interval, tunnel padding and dead-socket eviction are unchanged, and the write stays per-client rather than going through `broadcast()` because the frame carries no session data and so needs no multi-user owner routing.
|
||||
- **Client watchdog.** `computeSseStale()` in `constants.js` is a pure policy beside `computeConnectionLossUi`: stale only when the transport believes it is `connected`, the device is online, and no frame has arrived for 45s (three missed heartbeats). That `connected`-only guard doubles as the loop breaker, since a forced reconnect leaves the state immediately and the watchdog cannot re-fire while one is in flight. The liveness stamp is applied inside `addListener` itself so every registered listener feeds it from one place instead of three that can drift, and the heartbeat's own listener is a deliberate no-op that exists only to be registered (`EventSource` drops named events nobody listens for). A 5s watchdog forces `connectSSE()`, `visibilitychange` to visible checks too (a background tab's timers are throttled, and a wake is exactly when a stream comes back zombie), and the forced reconnect logs one diagnostic line so a middlebox that strips or delays heartbeats does not present as an undebuggable "silently reconnects every 45s".
|
||||
- **Renaming a tab appeared to do nothing** until a full page reload. The `PUT` always succeeded; what was broken is how the tab strip learned the result. `finishRename()` re-renders from the client-side `app.sessions` map and nothing wrote the new name into it, so the rename depended on the `session:updated` SSE frame to carry its own write back, which is precisely what a quiet stream never delivers. `_applyLocalSessionName()` now writes the confirmed name locally and refreshes cached subagent parent names. A rejected rename also used to read as success and silently drop the edit, because `_apiPut` turns a network error into a null Response so the old `try`/`catch` could never fire; a failure now restores the old label and toasts.
|
||||
|
||||
Tests: `test/sse-staleness.test.ts` (node VM over `constants.js`, threshold boundaries and every not-stale guard), `test/sse-heartbeat.test.ts` (drives `cleanupDeadClients()` with fake replies: named frame not a comment, parseable payload, padding only with a tunnel, dead clients still evicted), and `test/inline-rename.test.ts` (the name applies with no SSE frame dispatched, and a 500 leaves the map untouched).
|
||||
|
||||
Event names are part of the stable `/api/v1` contract, so this is a minor bump.
|
||||
|
||||
- c5b5963: Add Pi (pi.dev) as a sixth CLI run mode (#206).
|
||||
|
||||
`SessionMode` gains `'pi'`, a first-class backend alongside Claude Code, OpenCode, Codex, Gemini and Antigravity: its own PTY, tmux session, rose tab identity, welcome button, run-mode entry, cron `agentType`, Docker and remote-SSH command defaults, and clone-repo Brain option.
|
||||
- **New resolver** `src/utils/pi-cli-resolver.ts`. Unlike the sibling resolvers it sanity-probes `pi --version` and requires semver-shaped output, because `pi` is a short generic name that a stray binary on `$PATH` can shadow; the rejected path is logged. `GET /api/pi/status` returns `{ available, path, version }` so a misresolution is diagnosable.
|
||||
- **`PiConfig`** maps to `--model` (accepts `provider/id` and a `:thinking` suffix), `--provider`, `--thinking`, `--session`/`-c`, and the tri-state `--approve` / `--no-approve`. Every value is regex-allowlisted and dropped on failure. `--api-key` is deliberately never wired: it would put a provider secret on the spawn command line.
|
||||
- **No bypass flag.** Pi has no permission prompts and no sandbox, so there is no `--dangerously-skip-permissions` analog. Its privilege-shaped knob is `approveProjectTrust`, which makes pi load and execute repo-local `.pi/extensions` TypeScript and install missing project packages. `clampExternalCliBypassForOwner()` therefore puts pi in the **materialize** branch: a non-granted multi-user owner gets `--no-approve` even when no config was sent, because pi's own default is an interactive prompt the session user could answer themselves. The same materialization applies to cron-fired jobs (`clampCronExternalCliConfigs`), which carry no per-CLI config and would otherwise launch on pi's own default. Both helpers had no test coverage at all; they now do, for every CLI.
|
||||
- **Env allowlist gains only the `PI_*` prefix.** Pi's ~34 provider key vars share no prefix and `ALLOWED_ENV_PREFIXES` is one global list with no mode context, so admitting them would widen the allowlist for every mode at once. Users authenticate via pi's `/login` or the server process's own environment.
|
||||
- **Pi stays out of `isAltScreenStripMode()`.** Its default TUI renders into the main screen with terminal-owned scrollback and is mouse-aware, so it consumes `\x1b[3J` and the mouse DECSETs that the full strip removes, unlike an Ink TUI repainting in place. Note what exclusion does NOT do: pi is tmux-backed, so it still falls through to the narrow `isMuxAltScreenOnlyStripMode()` strip and its alt-screen toggles are dropped either way. Pi's runtime-switchable fullscreen TUI therefore paints into the main buffer, exactly like vim inside a tmux `shell` session.
|
||||
- **Docker**: pi installs in its own `--ignore-scripts` step so that flag cannot affect the other four CLIs, and its credentials are seeded per-file (`auth.json`, `settings.json`, `trust.json`, `models.json`, `models-store.json`) rather than whole-dir, since `~/.pi/agent` also holds sessions, extensions and installed package trees.
|
||||
- **Local echo**: pi lands on the buffer overlay. Verified that codex's per-keystroke starvation does not reproduce: pi's slash picker re-filters on the whole composer content, so a one-shot flush behaves identically to per-keystroke typing.
|
||||
- **Mode-list parity**: pi is excluded from the Ralph tracker auto-enable on `POST /api/sessions/:id/interactive` (like every other external CLI, whose output the tracker never parses), carries a `REMOTE_CLI_BIN` entry so a remote-SSH pi session reports its CLI version, and gets its own badge in the desktop home rail instead of rendering like Claude. The packaged agent skill's mode enumerations list pi too, and it now documents the per-CLI availability probes (`GET /api/<mode>/status`) that agents should check before spawning a worker on a backend the server may not have installed. Both are pinned by a new guard that derives the mode set from the Zod schema instead of restating it.
|
||||
- **`codeman doctor` and the run mode agree about pi.** The registry entry resolved a bare `which pi` while `pi-cli-resolver` demanded semver output, so the Dependencies panel could report an installed Pi CLI that sessions refuse to launch. Both now share one exported regex, and the registry's new `requireVersionMatch` reports a non-semver `pi` as missing rather than installed. Only pi sets it; every other tool keeps its existing behaviour.
|
||||
- Installer detection, docs (`docs/pi-integration.md`), READMEs, and the architecture invariants are updated. Tests: `test/pi-mode.test.ts` and `test/routes/external-cli-bypass-clamp.test.ts`, plus extensions to the run-mode, mobile-overview, render-index-html, system-routes and local-echo suites.
|
||||
|
||||
## 1.17.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- Agent skill rework, session lineage lines, and a sharper endpoint drift guard.
|
||||
|
||||
**The packaged agent skill is rewritten around learning it, not just being correct** (`skills/codeman/`, ~2000 lines changed across four files). It previously opened with about fifty lines of credential archaeology before a single working call, and interleaved every recipe with the rationale for its own warnings.
|
||||
- `SKILL.md` is restructured into: a 12-line "Hello, worker" that runs as written, a verb table an agent can act correctly from without reading anything else, a ten-line rules digest, the safety rules, the recipes, and setup/credentials last.
|
||||
- **The preamble is no longer re-pasted.** A bootstrap writes it once to a `$HOME`-derived 0600 file and later calls source it and check a version stamp. Shell state does not survive between tool calls, but the filesystem does. The stamp is the last line written, so a truncated file leaves it unset and the guard aborts instead of running a half-written preamble.
|
||||
- **New: where to spawn.** The only documented spawn used to create a scratch case, so "spin up workers on this repo" led an agent to do correct-looking work in the wrong directory. The rule is now explicit: hooks (and therefore `stop`/`blocked`) exist only where Codeman created the directory, so a linked case or a raw `workingDir` must synchronize on output markers. `wait:true` is still accepted there and silently degrades to a heuristic `idle`, which is documented as its own trap.
|
||||
- **New verbs**: interrupt a runaway worker with ESC instead of deleting it, `active-tools` and `run-summary` as structured liveness signals, `auto-resume` for usage limits, the workspace as a high-bandwidth channel, and `GET /api/events` as a fleet watcher.
|
||||
- `reference/messaging.md` gains a fleet protocol for Claude Code cross-session messaging: peer refs are injected and never discovered (a worker calling `ListAgents` sees the user's real sessions), every message costs a billed turn in both sessions, plus review pairs, mid-task questions, relay chains, mixed fleets, and their failure modes.
|
||||
- `reference/recipes.md` is renumbered to a flat Flow 1-7 and gains Flow 7, one whole job start to finish: worktree fleet, tasks, gather, a review pass, report, cleanup.
|
||||
- `reference/endpoints.md` gains an auth section, a symptom gallery keyed on what you actually see in the JSON, and a consolidated limits table.
|
||||
- **Corrections found by auditing the old text against source**: the input cap is 65536 characters and not 100000 (65537-100000 passes Zod then 400s at the route); `wait.ended` is returned by a _live_ session whose write did not land, so "the session is gone" was wrong recovery advice and `delivered:false` is the discriminator; `DELETE /api/subagents` clears the map rather than killing anything; the trust-dialog auto-accept reads the rendered pane, not the output stream; `claudeMode` is readable globally though not per session; `run-summary` is envelope-wrapped (`.data.summary`); `active-tools` is not empty for `shell` mode; and a session does inherit the server's `CODEMAN_PASSWORD`.
|
||||
|
||||
**Session lineage lines** (`sessionLineageLines`, per-device, desktop default on). A create request may name the session that spawned it, as a `parentSessionId` body field on `POST /api/sessions` and `POST /api/quick-start`, or as an `X-Codeman-Parent-Session` header, and the web UI draws an arc from the parent's tab to each child's. The skill's preamble sets the header once, so every spawn recipe carries it. The value is **resolved rather than trusted**: exact id or a unique prefix of at least eight characters (ids reach agents truncated), it must be a live session the caller can see with the same owner, and anything unresolvable is dropped rather than returning a 400, so a cosmetic field can never fail a worker spawn. It confers no permission and no lifecycle meaning. Rendering is an additional layer on the existing connection-line pass, sharing one batched reflow; desktop only, because the mobile header would bury the overlay.
|
||||
|
||||
**The endpoint drift guard now covers routes it silently could not see.** `test/agent-skill-endpoints-doc.test.ts` matched only bare `app.<method>('path')` registrations under `src/web/routes/`, so routes registered on the server itself (`/api/events`, `/api/events/subscribe`) and any registered with Fastify generics (the approvals routes) were unverifiable. It now scans `server.ts` too and tolerates generics, taking it from about 200 to 216 recognized routes.
|
||||
|
||||
## 1.16.6
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Phone home screen now shows session ages, plus three mobile input fixes.
|
||||
|
||||
**Phone overview: started / how long stamps.** Every live session row on the "C" home screen carries a third line: when the session first started, and how long it has been in the state it is in ("started 3d ago · idle 12m"). Idle, waiting, error and ended states measure from the pane's last output, which for a Claude pane sitting at its composer is exactly when the turn ended; a WORKING session measures from its last Enter instead, because a running pane repaints about once a second and would otherwise report every turn as 0m. A 20s clock rewrites the values in place rather than re-rendering, so no row's blink or pulse restarts.
|
||||
|
||||
**Fix: a recovered session was restamped as new on every restart.** Boot recovery never passed `createdAt`, so each server start reset it to `Date.now()` and a week-old pane reported "created 2m ago" (and sorted as the newest thing in the unified session list). It now comes from the tmux session's own birth time, which mux-sessions.json already carried. The desktop home rail's "created" stamp is fixed by the same change.
|
||||
|
||||
**Fix: a selection dialog locked the on-screen keyboard out of the terminal (regression in 1.16.5).** The check that decides whether a tap belongs to the TUI scanned the whole viewport for a numbered menu, so while a Claude question or permission dialog was on screen EVERY tap in the terminal counted as actionable and blurred the input. The keyboard could not be opened at all until the dialog was answered, which left tapping an option, the one gesture that commits an answer, as the only interaction a phone had. The menu test is now row-local: the dialog's own rows still report the tap and keep the keyboard down, while the question title, the transcript and blank space summon the keyboard so a digit can be typed at the dialog instead of aimed at it.
|
||||
|
||||
**Fix: the accessory bar's arrow keys bypassed the local-echo overlay.** On a phone the text you type is buffered in the browser and has never reached the PTY, so an arrow tapped on the bar arrived at a composer the CLI still considered empty: Up recalled a history entry into it while the overlay went on painting the draft over the same row and still believed it was pending, and the next Enter submitted the two mixed together. The four arrows now flush the draft first and hand the session to plain PTY echo, the same contract a nav key typed on a hardware keyboard has had since #218. The CLI stashes the flushed draft, so Down brings it back. Tab now shares that one flush helper instead of its own copy.
|
||||
|
||||
## 1.16.5
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Mobile keyboard dismissal, and a tidier Save/Close pair in the phone settings sheet.
|
||||
|
||||
**The on-screen keyboard can finally be closed from inside the app.** The terminal
|
||||
keeps focus on a hidden textarea and nothing ever released it, so once the keyboard
|
||||
was up it covered roughly half the screen with no way out but the OS back gesture.
|
||||
Two gestures now dismiss it:
|
||||
- **A tap outside the terminal** (header, tab strip, empty page chrome). Deliberately
|
||||
narrow: it only fires while the terminal input actually holds focus, never inside
|
||||
the terminal (tap classification owns that decision), and never on a control, since
|
||||
anything focusable is about to take focus itself and the keyboard accessory bar
|
||||
exists to be used _while_ the keyboard is open. A scroll ends in `touchend` too, so
|
||||
finger travel is tracked from `touchstart` and only a near-stationary gesture counts
|
||||
as a tap, sharing the terminal's own 8px threshold so both agree on tap-vs-scroll.
|
||||
Scrolling to read something mid-compose no longer drops the composer.
|
||||
- **A second tap on inert transcript content.** Every terminal tap used to re-focus,
|
||||
which left the accessory bar's chevron as the only way out. Scoped to inert rows on
|
||||
purpose: the prompt row keeps focus-then-position, so a second tap there still
|
||||
places the caret, and actionable rows (readbacks, `esc to interrupt` status rows,
|
||||
menu selections) still blur as before.
|
||||
|
||||
**Settings sheet header on phones.** Below 860px Save moves into the header, which
|
||||
left the two ways out of the sheet as a fat accent pill beside a bare glyph. Save and
|
||||
Close now share a recessed tray with matching 36px pill geometry, reading as one
|
||||
44px cluster the height of the phone header. Tray colors come from skin tokens, so
|
||||
the light skins keep their look, and the tray stays off the sheets that carry a lone
|
||||
close button.
|
||||
|
||||
Also fixes a test that could never have caught a regression: the case asserting that
|
||||
tapping a control does _not_ dismiss the keyboard was picking a button from the
|
||||
hidden welcome overlay, whose rect still measures while the hit-test lands on the
|
||||
terminal underneath, so it passed for the wrong reason and stayed green even with the
|
||||
exemption deleted. All four guards in the dismiss handler are now individually
|
||||
pinned.
|
||||
|
||||
## 1.16.4
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- **Voice dictation through your Claude Code login (no API key).** The mic button can now transcribe using this machine's existing Claude Code subscription, via the same speech-to-text service the CLI's own `/voice` mode uses. Off by default (`claudeVoiceEnabled`, synced): turning it on spends the server owner's Claude subscription on transcription for anyone who can reach the UI. The OAuth token never leaves the server process, credentials are read-only (Codeman never refreshes them, which would rotate the refresh token out from under the CLI), streams are capped at 5 minutes and 4 concurrent, and the WebSocket carries the same allowed-Host + same-site Origin guard as the terminal socket. A new Speech engine picker (Auto / Claude / Deepgram / Browser) sits alongside the existing Deepgram and Web Speech paths, which are untouched.
|
||||
|
||||
**One settings surface.** Session Options and Add Case now use the same `set-*` chrome as App Settings instead of the old modal-tab chrome, with a left rail, grouped rows, per-group device/synced scope badges and a search box. App Settings leads with version + update; the Session Options rail stays a real switcher (one section at a time) because Summary and Respawn are each long enough to bury the other. Collapsed Add Case blocks gained a disclosure chevron.
|
||||
|
||||
**Read My Mind: rethink steer note (phase 3 part 2).** Rethink now carries an optional free-text note ("no, I meant the mobile bug") sent as `steer`, the highest-authority signal the predictor gets. It stays in the field across re-runs, clears on each open, and the empty-result copy points at it. The modal footer moved to the styled `btn-toolbar` convention; the bare `btn btn-*` classes it shipped with match no CSS in this codebase and rendered as unstyled browser buttons.
|
||||
|
||||
**Mobile terminal taps no longer fight the keyboard.** Taps on TUI-owned rows (expandable readbacks, tool results, decision menus, the working/status row) now act on the CLI without popping the keyboard, while a tap on inert transcript text keeps the keyboard reachable. Rows are told apart by the affordance the CLI prints (`ctrl+r to expand`, `tap to collapse`, `esc to interrupt`) rather than by row titles, which vary per CLI and per version. A tap with the viewport scrolled up sends no mouse report at all but still restores focus, so the keyboard is reachable after every tab switch. Thanks to @Lint111.
|
||||
|
||||
**Path labels abbreviate `$HOME` on both platforms.** The "show `~/project`" rule had three implementations and two were platform-specific in opposite directions: the Run menu's matched `/home/<user>/` only, so on macOS every Recent Sessions row spent its first ~19 characters on an identical `/Users/<user>/` prefix and ellipsized away the tail that identifies it (#273); the case-manage list's matched `/Users/<user>` only, so no Linux case path was ever abbreviated. Both now route through one helper, with a static guard against a fourth copy appearing.
|
||||
|
||||
**Run menu Recent Sessions rows are legible.** Rows now read as folder, worktree pill, dimmed parent path, timestamp, with only the parent path allowed to shrink, so truncation can never hide which project (or which worktree) a row refers to. `<repo>/.claude/worktrees` is dropped from the parent path as noise. Thanks to @jordan8037310. Follow-up fix: the widened menu was not actually usable by its rows, since `.run-mode-history` is a block scroller and its `<button>` rows stayed shrink-to-fit at ~250px inside a full-window-width menu; rows now fill the menu and it is capped at the 760px one full row costs.
|
||||
|
||||
**Desktop home screen** no longer clips, and shows full tab names.
|
||||
|
||||
## 1.16.3
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Session rows that name their worktree, a shell keyboard bar for phones, App Settings as one scrolling document, and the Read My Mind modal on phones.
|
||||
- **#265 / #266**: a past session whose directory no longer exists used to report
|
||||
`$HOME` as its working directory, because history rows reconstructed a path by
|
||||
stat-walking the filesystem and fell back to `$HOME` when nothing resolved.
|
||||
Deleting a worktree is the normal end of its life, so every past worktree
|
||||
session collapsed onto the same indistinguishable row. History rows now read
|
||||
the literal `cwd` Claude Code stamps on its own records, out of buffers the
|
||||
scanner had already loaded, so it costs no extra file reads and survives the
|
||||
directory being removed. Sessions that ran in a worktree also carry a
|
||||
`⑂ name · branch` pill in the Resume list and the Cmd+K session manager, and
|
||||
both are searchable by worktree name and branch. Measured on a real install:
|
||||
the cwd was recoverable for 215 of 216 transcripts, 212 of them from the first
|
||||
16KB, and 28 rows that previously read `$HOME` now report their real path.
|
||||
Reported and implemented by @jordan8037310.
|
||||
- **#262**: a shell session now gets its own mobile accessory bar
|
||||
(`Ctrl · Esc · Tab · ↑ · ↓ · ← · → · Paste · ⌄`), with Ctrl as a one-shot
|
||||
modifier: tap it, and the next character goes out as its control byte. That
|
||||
puts Ctrl+C/D/Z/R/L/A/E/W/U/K on a nine-button bar without a button per chord.
|
||||
The modifier is applied on the CJK input path too, where the textarea owns the
|
||||
keyboard and an armed modifier could previously neither fire nor be spent, so
|
||||
it survived until a later keystroke and turned that one into a control byte.
|
||||
Agent sessions keep the existing bar unchanged. Proposed by @DodgyBadger.
|
||||
- **#257**: with several tabs open on a phone, the rightmost ones could not be
|
||||
reached. Selecting a tab never scrolled the strip, and every ambient rebuild
|
||||
reset `scrollLeft` to 0, so a strip the user had just swiped snapped back a
|
||||
moment later. Reported by @DodgyBadger.
|
||||
- **App Settings** is now a left rail acting as a table of contents over one
|
||||
scrolling document instead of 8 tabs that wrapped onto two rows. Nine sections,
|
||||
all mounted at once, so find-in-page works across the whole thing. The model
|
||||
controls stop contradicting each other: the base model lives on cards and "1M
|
||||
context window" is a switch that composes onto it, retiring the old pair of
|
||||
settings that each claimed precedence over the other.
|
||||
- **Read My Mind** suggestions beyond the first are no longer discarded. The
|
||||
alternates render as tappable rows with their kind badge, tapping one swaps it
|
||||
into the editable field without losing an in-progress edit, and Rethink now
|
||||
records the whole shown set as rejected. The modal is sized for phones and
|
||||
reachable from the phone keyboard bar.
|
||||
- The desktop welcome screen carries the open tabs as a rail docked to the left
|
||||
edge, with created and last-active stamps refreshed in place.
|
||||
- The README now documents cloning a GitHub repository straight into a case
|
||||
(**Add Case → Clone Repo**), which shipped in 1.16.2 but was only described in
|
||||
the architecture docs.
|
||||
|
||||
- 5d42f64: Home screen: make the past-conversation list usable, and let search find past sessions.
|
||||
- **#260**: "Resume Conversation" showed 4 rows and then dumped every remaining
|
||||
one into a fixed 240px box, with no ordering or filtering. The list now opens
|
||||
with 10 rows, "Show more"/"Show less" grows and shrinks the box itself (the
|
||||
height cap is class-driven instead of fixed), and the header carries a filter
|
||||
box (matches name, folder, `#case` label and the conversation's prompts), a
|
||||
sort control (recent / name A–Z / folder A–Z, pinned rows still first) and a
|
||||
shown-of-total count. Filtering implies expansion, so every match is visible.
|
||||
- **#261**: the search box could not match a past project by folder name: its
|
||||
session corpus was the live in-memory map, while past sessions come from
|
||||
`/api/sessions/unified`. Search now also harvests a bounded snapshot of that
|
||||
unified list, refreshed OUTSIDE the request path (published by
|
||||
`/api/sessions/unified`, plus a fire-and-forget rebuild when stale), so the
|
||||
search path keeps its no-filesystem-reads property. Results for a closed
|
||||
session resume the conversation instead of trying to select a tab that no
|
||||
longer exists, and are badged `RESUME`. In multi-user mode the snapshot is
|
||||
re-scoped per row on read, matching what `/api/sessions/unified` exposes.
|
||||
|
||||
Reported by @jordan8037310.
|
||||
|
||||
## 1.16.2
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Clone a Git repository straight into a case, predict the prompt you were about to type, and point a session at a separate Claude account.
|
||||
|
||||
**Clone Repo (#251, proposed by @DodgyBadger in #236)**: Add Case gains a **Clone Repo** tab that clones a repository into `codeman-cases/<name>` and registers it as a normal local case. A live verdict under the URL field answers, while you type, whether the URL is cloneable without credentials, what its default branch is, and which branches and tags exist (`POST /api/cases/clone-preflight` behind `git ls-remote --symref`). The case name fills in from the parsed repo, refs come from the remote as a datalist, shallow clone is optional, and a Brain picker (installed CLIs only) points the Run button at the agent you chose. Starting a session stays opt-in, and the tab hides itself when the server has no `git`.
|
||||
|
||||
**Every settings writer now refuses to write through a symlink (from the #251 review, affects existing cases too)**: case contents can be foreign, and a repository can ship `.claude` or `.claude/settings.local.json` as a symlink pointing anywhere on this machine. Since `writeFile` follows links, a scaffold write could land outside the case, up to and including replacing your own `~/.claude/settings.json`. All seven writers that touch a case's `settings.local.json` (`writeHooksConfig`, `ensureCodemanHooks`, `refreshStaleCodemanHooks`, `updateCaseModel`, `updateCaseEnvVars`, `stripCaseEnvKeys`, `applyStatusLineConfig`) now go through one `withSafeSettingsWrite()` gate that runs the symlink check inside the per-path settings lock. A refusal is a warning rather than a throw, so hooks degrade to output-based idle detection instead of failing the operation. If you have deliberately symlinked a case's `.claude` or its `settings.local.json`, Codeman will now decline to write there and say so; replace the link with a real file or directory to get hooks, model and statusLine writes back.
|
||||
|
||||
The clone endpoint (`POST /api/cases/clone`) is synchronous by design: no job store, no polling, bounded by `GIT_CLONE_TIMEOUT_MS` (default 5 minutes). Security decisions live in a pure half of `src/git-clone.ts` so each is unit-testable without spawning anything: `<name>::<payload>` transports are refused as a family (any of them dispatches to a `git-remote-<name>` helper, which turns a clone into arbitrary command execution), a leading `-` is refused and `--` precedes every operand, argv arrays are used rather than a shell, URLs carrying credentials are refused, and non-interactive means more than `GIT_TERMINAL_PROMPT=0` (empty `GIT_ASKPASS`/`SSH_ASKPASS`, `SSH_ASKPASS_REQUIRE=never`, empty `DISPLAY`, `GCM_INTERACTIVE=never`, `ssh -oBatchMode=yes`), since with the request held open any one of those left open is a hang instead of an error. Timeouts signal the process group, because `git clone` fans out into `git-remote-https`/`index-pack` and SIGTERM to the parent alone can leave the fetch running. Repository contents beat scaffolding: an existing `CLAUDE.md` is kept, hooks merge into whatever `.claude/settings.local.json` the repo shipped, and a repo shipping its own `.claude/settings*` is reported back as a warning, because those hooks run locally as soon as a session starts.
|
||||
|
||||
**Read My Mind phase 2 (#256)**: phase 1 (1.16.1) gave each case an intent profile; this turns it into the feature as pitched. Press 🧠 on a Claude session and Codeman predicts the prompt you were about to type, from your stated goals, your recent prompts in your own voice, the last assistant reply, tool activity, git state, away context, sibling sessions, and any dialog the session is waiting on. The context assembler is pure and budgeted with trust tiers, so user-stated intent outranks observed content and terminal output alone can never justify a suggestion. One shot at opus (`readMyMindModel` overrides), a strict JSON contract, and 1 to 3 suggestions typed continue / verify / redirect. The modal keeps the suggestion editable: Send, Insert (drops it on the composer without Enter), Rethink (rejections feed back into the next attempt), Dismiss. Nothing is ever auto-sent, the click is the boundary. Opt-in via App Settings, Panels (synced, default OFF), desktop header only. Agents get the same verb through the Codeman skill (`POST /api/sessions/:id/readmymind`).
|
||||
|
||||
**Per-session `CLAUDE_CONFIG_DIR` (#255, designed and specified by @jordan8037310)**: `schemas.ts` gains an exact-key tier (`ALLOWED_ENV_KEYS`) beside `ALLOWED_ENV_PREFIXES`, admitting `CLAUDE_CONFIG_DIR` so a case can run on a separate Claude subscription (client-billed accounts). Exact match only: other `CLAUDE_*` keys and near misses like `CLAUDE_CONFIG_DIR_EXTRA` stay rejected, blocked keys stay blocked. The key survives `getEnvOverridesForPersist()` because it is a path rather than a secret, and dropping it would silently switch a rebuilt session back to the default account after a reboot. Caveat worth knowing: a relocated config dir writes transcripts outside `~/.claude/projects`, so the response viewer, subagent windows, ultracode panel and Read My Mind go blind for that session unless `projects` is symlinked back into the shared tree.
|
||||
|
||||
## 1.16.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- 161f1da: Read My Mind phase 1: per-case intent profiles (docs/readmymind-plan.md). Codeman can now capture the prompts a user actually submits (from the Claude session transcript, opt-in via the new synced readMyMindEnabled setting, default OFF) into a per-case intent profile alongside user-stated goals, stored in ~/.codeman/intents.json (mode 0600, never searched). New endpoints GET/PUT/DELETE /api/sessions/:id/intent (ownership-scoped, strict schemas), a transcript:user_prompt event on TranscriptWatcher, and agent-skill coverage (SKILL.md recipe + endpoints.md rows) so agents can read and record the user's intent. Groundwork for the phase-2 predictor button: nothing is ever auto-sent.
|
||||
- Home screen and phone touch targets.
|
||||
|
||||
The desktop welcome screen now lists your open tabs as a vertical column down its left gutter, which was previously dead space: one row per live session plus any saved web tabs, in tab order so the row badges match Alt+1..9, with case, backend and state on each row. Clicking a row enters that session. The column is width-gated (1180px and up) and never moves the centered welcome content.
|
||||
|
||||
Working state now reads the same everywhere it appears. A busy session shows a pulsing green dot ringed by the same spinner a tab draws while it loads, with a green halo, on the desktop home column, the phone home screen and the tab strip alike. Phone tabs got the bigger 9px glowing dot for the same reason.
|
||||
|
||||
Phone touch targets: the brand "C" that returns you to the home screen was roughly a 12x13px hit area, well under the 44px minimum. It is now a real 44x44 button, and the phone header grew from 36px to 44px to make that possible, which gives every other header control the same 8px. The simple keyboard accessory bar also swaps /clear for Tab (/clear and /compact stay in the extended bar), flushing locally buffered text to the terminal first so completion applies to what you just typed.
|
||||
|
||||
## 1.16.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- Approvals Inbox, truthful idle detection, a revived trust-dialog auto-accept, and an unmistakable offline state.
|
||||
|
||||
**Approvals Inbox (#245, opt-in, default OFF)**: one cross-session inbox for every prompt that is waiting on a human (permission dialogs, AskUserQuestion questions, idle prompts). Enable "Approvals Inbox" in App Settings -> Panels (synced setting `approvalsInboxEnabled`); until then no new UI renders anywhere. Desktop gets a header bell (visible only while something is pending, with a count badge) opening a drawer of cards answerable in place: session, tool/message summary, the captured dialog frame, and one button per parsed dialog option (fallback: Approve / Deny-Esc). The phone overview's NEEDS YOU rows gain compact answer strips, and push notification action buttons were fixed along the way.
|
||||
|
||||
**Sessions no longer report idle while working (#246)**: every working Claude session flipped to `status: "idle"` about two seconds into its turn, and tabs, notifications, respawn and the phone overview all read that bad value. The `❯` prompt redraws throughout a turn, so readiness now requires a sustained repaint streak plus a capture-pane probe that recognizes the live working line (`✻ ... (Xs)`), and the UI shows a working state you can actually see.
|
||||
|
||||
**Workspace trust dialog auto-accept has been dead and now works (#249)**: a session started in a directory Claude had not seen before sat on the workspace-trust dialog until a human pressed Enter, because tmux delivers cursor-forward sequences rather than spaces. Detection now goes through the capture-pane text added in #246 and the dialog is answered reliably.
|
||||
|
||||
**A dead connection is unmistakable instead of a red dot (#248)**: the service worker serves the cached app shell, so opening Codeman with nothing reachable rendered a normal-looking empty dashboard with only an 8px red header dot as a clue. Now a connection-loss overlay (retry button, server host, actionable hints) plus a persistent banner make the state obvious on desktop and phone, and clear the moment the server answers again.
|
||||
|
||||
- 1e1db94: Cross-session messaging integration, two halves. **Workers now carry their Codeman session names as messaging peer names**: local claude spawns pass `--name <session name>` when the installed CLI is 2.1.224+ (the cross-session-messaging release). The gate is fail-closed, since an older claude aborts startup on an unknown option: an unknown or older version yields a spawn command byte-identical to before, the value is allowlist-sanitized before shell interpolation, and docker/remote spawns never carry the flag (their CLI is not the probed binary). Verified end to end on an isolated instance: the worker lists as its session name in `ListAgents`, and its replies arrive tagged `from-name="<session name>"`.
|
||||
|
||||
**The Codeman agent skill teaches cross-session messaging**: drive claude workers over `ListAgents`/`SendMessage` where available, map rows to Codeman sessions via the `tmux codeman-<id8>` column, deliver multi-line exactly-once task messages (including mid-turn steering), collect results as latched replies instead of polling, and fall back to the HTTP recipes whenever the feature is absent (version, feature flag, telemetry-disabling env vars, Docker/remote cases, non-claude modes). Adds `reference/messaging.md` (ships automatically, the installer enumerates `reference/*.md`), fan-out Flow 5 in `reference/recipes.md`, troubleshooting rows in `reference/endpoints.md`, and safety rules for the shared peer namespace (message only workers you created, no permission laundering in either direction). All mechanics verified live against claude-cli 2.1.226.
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- c50bb02: The File Viewer can show hidden files and folders.
|
||||
|
||||
`GET /api/sessions/:id/files` has always accepted `showHidden=true`, but the panel
|
||||
hardcoded `showHidden=false`, so dot-prefixed entries were unreachable from the
|
||||
tree: no `.gitignore`, no `.github/`, no `.env.example`, and nothing under them.
|
||||
Opening one meant guessing its path.
|
||||
|
||||
The panel header gains a `.*` toggle. It re-fetches rather than re-rendering the
|
||||
cached tree, because the filtering happens server-side, and it keeps the expanded
|
||||
directories so toggling does not collapse the tree you just navigated. The state
|
||||
is per-device (its own `codeman:fileBrowserShowHidden` key rather than the
|
||||
app-settings object, which is rebuilt from the settings-modal DOM on save and
|
||||
would drop a key toggled from outside it), defaults to OFF, and survives a reload.
|
||||
|
||||
Generated and version-control directories (`.git`, `node_modules`, `.next`,
|
||||
`.venv`, ...) stay excluded either way: that list is about tree size, not about
|
||||
hiding dotfiles.
|
||||
|
||||
Closes #221.
|
||||
|
||||
- ce22c2a: The filesystem path picker can show hidden files and folders, and the shared secret blocklist grew to make that safe.
|
||||
|
||||
The picker behind Link Existing's "Browse" and the mobile keyboard's `Path` key
|
||||
refused every path with a dot-prefixed segment, so `.github/workflows/ci.yml`
|
||||
could not be selected and a hidden folder could not even be opened. It now has
|
||||
the same `.*` toggle as the File Viewer, default OFF, per-device, and it applies
|
||||
to both the listing and the preview endpoint (which re-resolves the path
|
||||
independently).
|
||||
|
||||
That filter was quietly doing security work. With every hidden path unreachable,
|
||||
`isSensitivePath` never had to name the credentials that live in dot-directories,
|
||||
because the picker's roots include Home. Lifting the filter removes that
|
||||
accident, so the blocklist now covers them explicitly: SSH keys at any depth (not
|
||||
only under `$HOME`), GPG keyrings, AWS/GCloud/Azure/Docker/Kubernetes
|
||||
credentials, npm, Yarn, git, `gh`, netrc, PyPI, RubyGems, Cargo and Terraform
|
||||
tokens, `.pgpass` and `.my.cnf`, and the Claude and Codeman agent credentials.
|
||||
`~/.codeman/` and `~/.claude/` stay attachable as trees, since the publish skill
|
||||
and the review-card loop read from them; only their secret-bearing members are
|
||||
named.
|
||||
|
||||
Blocked trees, sensitive files, root confinement and symlink-escape checks are
|
||||
all unchanged and still apply with the toggle on: a hidden entry that resolves
|
||||
to a secret is dropped from the listing, and opening it is refused.
|
||||
|
||||
Follows #221.
|
||||
|
||||
## 1.15.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- 55bff4a: Zero-lag predictive echo for Codex sessions (mosh-style write-through prediction).
|
||||
|
||||
Codex's per-keystroke composer forced 1.12.2 to disable the local-echo overlay (issues #218/#219/#220/#222), leaving Codex typing at full round-trip latency on remote links. This release adds a second echo mode instead of re-enabling the first: every keystroke still goes to the PTY exactly as before (byte-identical wire behavior, pinned by vm-level and end-to-end trace-equality tests), while the new `PredictiveEchoAddon` in `xterm-zerolag-input` 0.2.0 paints the predicted glyph at the predicted cell. When the real echo lands, the prediction is confirmed and its span removed (an invisible swap); mispredictions self-heal via a two-pass mismatch cascade and a TTL.
|
||||
- Reconciliation reads the parsed terminal buffer, never the raw stream: full-line redraws, ECH gap painting and tmux's in-place deltas all converge to the same cells. Confirmation requires the cell match PLUS a cursor advance, so placeholder glyphs and identical repaints never false-confirm; blank cells are neutral (codex clears its placeholder on the first echo).
|
||||
- Predictions paint only while the cursor sits on the measured Codex composer row (`/^› /`, codex-cli 0.147): trust/approval modals and wrapped continuation rows get no ghosts, deliberately falling back to real echo.
|
||||
- Ships as a SEPARATE `vendor/xterm-predictive-echo.js` bundle: the existing zerolag bundle is byte-identical (sha256-verified), and a missing or broken bundle degrades Codex to exact 1.12.2 behavior. The per-device `localEchoEnabled` toggle is the kill switch.
|
||||
- Claude/Gemini/OpenCode/Antigravity keep buffer mode untouched; shell stays off.
|
||||
- A post-build adversarial review added the anchor-hold rule: after an unpredicted wire edit (backspace into echoed text, cleared input, IME text commits) new predictions hold until the next parsed write, so a stale displayed cursor can never mis-anchor a run.
|
||||
- Tests: 55 new package tests including replay suites driven by fixtures recorded from a real codex TUI through the production tmux+strip pipeline (`scripts/dev/record-codex-frames.mjs`) and a 500-iteration seeded fuzz; new vm policy/wire-neutrality suites; a 10-scenario Playwright E2E against real codex covering the #218/#219/#220/#222 retests, byte-identity, and a simulated 300ms-RTT run. The package test suite now runs in CI.
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Agent-skill hardening, plus a fix for the mobile browser suite.
|
||||
|
||||
## The Codeman agent skill
|
||||
|
||||
Twelve issues found by auditing the skill against a live instance, and fixing them meant measuring things rather than reasoning about them.
|
||||
|
||||
**Readiness now works in every permission mode.** The ladder matched `bypass`, which is the status bar of only ONE mode. Measured one pane per mode against claude-cli 2.1.226:
|
||||
|
||||
| how Codeman spawned it | statusline | `shift+tab` | `bypass` |
|
||||
| ------------------------------------------ | ----------------------- | ----------- | -------- |
|
||||
| `--dangerously-skip-permissions` (default) | `bypass permissions on` | yes | yes |
|
||||
| `--permission-mode auto` | `auto mode on` | yes | no |
|
||||
| `--allowedTools …` | `don't ask on` | yes | no |
|
||||
| neither (`normal`) | `don't ask on` | yes | no |
|
||||
| `--permission-mode plan` | `plan mode on` | yes | no |
|
||||
|
||||
Every mode ends `(shift+tab to cycle)`, and the `claudeMode` setting is not exposed on `GET /api/v1/sessions/:id`, so there was nothing to branch on. The ladder matches `shift+tab` now: universal, and space-free, which is what makes it survive the TUI stream. A non-default worker used to be reported broken after burning the full budget. ⚠️ The `+` means it only works through `--data-urlencode`; a hand-built query silently searches for `shift tab`.
|
||||
|
||||
**`.status` is documented as unreliable in both directions.** Measured on a live worker reading `idle` while mid-turn and actively producing output, with `lastActivityAt` equal to the moment of the call. A worker that dies inside its pane also reads `idle`. Synchronize on `stop` or an output marker; to judge from outside, sample `terminal?tail=` twice and compare.
|
||||
|
||||
**The self-delete guard is fail-closed.** Documented in 1.14.2; the reference files and every recipe now route through it consistently.
|
||||
|
||||
**Reads work on macOS.** The ANSI-strip pipelines used `sed 's/\x1b…'`, and BSD sed has no `\xHH` escape, so on macOS they silently stripped nothing and handed the agent raw ANSI.
|
||||
|
||||
**Injection is atomic and no longer silent.** `installAgentSkillInto()` wrote each file with a bare `writeFile`, so two sessions created concurrently in one repo could leave a reader observing a truncated SKILL.md; writes now go through temp+rename under the same lock every sibling mutator uses. And both server call sites discarded the outcome, so a `foreign` refusal (a user-authored skill is present) or a `symlink` refusal was invisible: turning the setting on, seeing nothing, and having no way to find out why. Refusals are logged now; injection stays best-effort and still cannot fail session creation.
|
||||
|
||||
**Reference corrections**: the `FORBIDDEN` 403 row and which auth responses are plain text rather than the JSON envelope, the input size cap, the undocumented `killMux` parameter on DELETE, and the fact that zero, negative and non-integer timeouts are rejected with a 400 rather than clamped.
|
||||
|
||||
**README.zh-CN.md taught a recipe that could not work**: its input example had no trailing `\r`, so Enter was never sent and the prompt sat unsubmitted, and its read step used `/output`, whose `textOutput` is always empty for interactive sessions. Its agent section is now in line with the English one. CLAUDE.md's single-line gotcha also gained the `\r` rule.
|
||||
|
||||
**Tests**: the `codeman skill install`/`uninstall` CLI had none, including the linked-case resolution shipped in 1.14.2; the `POST /api/sessions` injection call site was never exercised because the shared route mock hardcoded the gate off; and nothing guarded `reference/endpoints.md` against drifting from the routes it documents. All three covered now.
|
||||
|
||||
## Mobile browser suite
|
||||
|
||||
The suite drives a real browser against a server started from TypeScript source, so it serves `src/web/public`, while `npm run build` puts the xterm vendor bundles in `dist/web/public`. Without them every `/vendor/xterm*` request 404s, `Terminal` is never defined, and every test touching `app.terminal` dies on a null. A `pretest:mobile` step now prepares them.
|
||||
|
||||
Hardened after two review rounds, each defect reproduced: the freshness cache trusted mtime alone, so a bundle left without its alias tail (or truncated by an interrupted `npm install`) was reported "up to date" forever while the suite died on `LocalEchoOverlay is not defined`; it now verifies content and size, and repairs what an earlier run poisoned. Builds go to a temp file private to the run and rename into place, so a partial write can never be published and two concurrent runs cannot corrupt each other. Temps whose owning process is gone are reclaimed, and only those. Freshness tracks every input the bundle derives from, not just the entry, so editing a sibling of the addon no longer leaves the suite testing a stale overlay. `npx` runs with the repo as cwd, so it uses the pinned esbuild instead of fetching an unpinned one.
|
||||
|
||||
## 1.14.2
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Four reported bugs fixed, and the Codeman agent skill from 1.14.1 gets its first published build with the fixes below alongside it.
|
||||
|
||||
## The Codeman agent skill
|
||||
|
||||
Introduced in 1.14.1 and the headline of this line. `skills/codeman` is a Claude Code skill that lets an agent running **inside** a Codeman session drive the HTTP API: start worker sessions, send them prompts, block until they finish, read their answers and clean up. It ships in the npm package and self-gates, so outside a Codeman session (`CODEMAN_MUX` unset) it refuses to act and costs unrelated sessions nothing.
|
||||
|
||||
### Installing it
|
||||
|
||||
```bash
|
||||
codeman skill install # ~/.claude/skills/codeman, every new Claude Code session sees it
|
||||
codeman skill install --case myproject # just that case; linked cases resolve by name too
|
||||
codeman skill uninstall # reverses either one
|
||||
```
|
||||
|
||||
Or turn on **App Settings > Agent Skill** (`agentSkillEnabled`, synced, default off) and Codeman injects the skill into each case when a Claude session is created there.
|
||||
|
||||
Installs are marker-owned: a `skills/codeman` that Codeman did not write is never touched, a stale managed copy is refreshed in place, and a symlinked skill directory is refused rather than written through. Re-run `codeman skill install` after upgrading to refresh the copy. Turning `agentSkillEnabled` back off does **not** remove already-injected copies, because a create-time sweep would yank the skill out from under other live sessions sharing that `.claude/` directory; remove them per case with `codeman skill uninstall --case <name>`.
|
||||
|
||||
### Using it
|
||||
|
||||
Ask for orchestration in plain language ("spin up three workers, have them lint, typecheck and test in parallel, then report back") and the skill supplies the guard, the safety rules and the recipes. The flow it runs:
|
||||
1. **Guard.** Re-runs a preamble on every shell call that refuses outside `CODEMAN_MUX=1`, reads `CODEMAN_API_URL` and `CODEMAN_SESSION_ID`, recovers a password from the data dir `.env` or the install's service definition if one is set, and defines a fail-closed `delete_session`. It re-runs it every call because shell state does not survive between an agent's tool calls.
|
||||
2. **Start a worker** with `POST /api/v1/quick-start` (`mode` is any of `claude`, `shell`, `opencode`, `codex`, `gemini`, `antigravity`), checking `.success` before reading `.data.sessionId`.
|
||||
3. **Wait until it is really ready.** A new session reports `idle` before its CLI has spawned, and a brand-new case shows a trust dialog first, so the skill waits for the composer's own status bar and treats the dialog as a bounded fallback.
|
||||
4. **Send and wait in one call**: `wait`/`waitTimeout` on `POST /api/v1/sessions/:id/input`. It registers the waiter before typing, closing the race where a separate wait reports the previous turn's idle state as this turn's answer. For `claude` workers it resolves on the `stop` hook, usually within seconds.
|
||||
5. **Read the answer** from `GET /api/v1/sessions/:id/last-response`, which returns clean transcript text rather than a screen scrape.
|
||||
6. **Clean up** with `delete_session`, for ids it created and nothing else.
|
||||
|
||||
Hook-less modes (`shell` and the external CLIs) have no `stop` signal and coarse lifecycle transitions, so the skill synchronizes those with a unique split marker and `wait-output ... from=buffer`. Worked fan-out flows, the per-mode signal table, error codes and the Docker/remote caveats live in the skill's `reference/` files, loaded on demand.
|
||||
|
||||
### The rules it encodes
|
||||
|
||||
Each of these silently wastes a run, which is why they are written down: every input must end with `\r` or Enter is never sent; input is single-line; a wait timeout is HTTP 200 with `wait.timedOut`, not an error; `stop` and `blocked` are `claude`-only; signals are edge-triggered with no history, so never fire-and-forget N prompts and then gather signal-waits one by one; a typed command echoes into the output stream, so markers must be split; a full-screen TUI stream is space-less, so match single tokens; and `pid != null` proves startup, not life, so `wait?until=exit` is the death check.
|
||||
|
||||
## Bug fixes
|
||||
- **Web tabs: long-running proxied requests were aborted after 30 seconds with no server log (#237).** The proxy wrapped each upstream fetch in a 30s `AbortSignal.timeout`, which bounds the entire exchange rather than the wait for response headers, so a dashboard endpoint doing model inference and any actively streaming response both died at 30s as a generic unlogged 502 that read as an intermittent network error. The timeout now bounds time-to-headers only and is cleared the moment headers arrive, with the default raised to 300s (`CODEMAN_WEBVIEW_TIMEOUT_MS`). Header timeouts are logged with a sanitized identity (method plus origin plus path, never the query string, which can carry the dashboard's tokens). A browser that navigates away mid-request now aborts the upstream fetch, guarded by `writableFinished` so a completed response never triggers it. The WebSocket handshake keeps its own 30s budget via the new `CODEMAN_WEBVIEW_WS_HANDSHAKE_TIMEOUT_MS`, since a handshake is connection establishment and waiting minutes on one only delays the browser's reconnect logic.
|
||||
- **Web tabs: sandbox incompatibility with cookie-authenticated reverse proxies documented (#238).** `docs/web-tabs.md` now covers cookie auth in front of Codeman itself (Cloudflare Access and similar), where a sandboxed frame's asset and API requests carry no auth cookie, bounce to the login provider, and leave the embedded app apparently unstyled while trusted mode works. The Test button's result now states its own scope: it verifies server-to-upstream reachability, not how the page behaves in a sandboxed frame.
|
||||
- **A described session tab now shows just the description (#232).** A session named `w2-foo-bar: some description` rendered both halves, so the generated id ate the width the chosen part needed. The tab shows the description alone, the `w<n>-<case>` id moves to the tooltip and stays in the session settings modal, and `aria-label` deliberately keeps the full name so screen readers still get the id. Undescribed tabs are unchanged. Right-click a tab to rename it inline. This also fixed a re-render loop: the incremental update compared against the full name, which a described tab never matched, so those tabs re-rendered on every pass.
|
||||
- **`codeman status` now probes the running server (#230).** The command runs in its own fresh process and reported that process's always-stopped Ralph loop under a bare "Status:", which reads as "the server is down" while the service is running fine and agents are reachable. It now probes the real server (`CODEMAN_API_URL`, else https then http on the local port, overridable with `--url`) and reports reachability, version and live session state; any HTTP answer proves the server is up, including a 401 from a password-protected install. The Ralph loop keeps its own `codeman ralph status`. This complements `codeman web --status` from the daemon work: that answers "did I start a daemon", this answers "is a server running at all".
|
||||
|
||||
## 1.14.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- The Codeman agent skill is now installable, so an agent running inside a Codeman session can drive the API without you pasting docs into its prompt. Plus six fixes to the packaged skill, each found by running it live against a real instance.
|
||||
|
||||
## What the skill is
|
||||
|
||||
`skills/codeman` is a Claude Code skill that teaches an agent inside a Codeman session how to start worker sessions, send them prompts, block until they finish, read their answers and clean up. It ships in the npm package. It self-gates: outside a Codeman session (`CODEMAN_MUX` unset) it refuses to act, so installing it globally costs unrelated sessions nothing.
|
||||
|
||||
## Installing it
|
||||
|
||||
Three ways, pick one:
|
||||
|
||||
```bash
|
||||
codeman skill install # ~/.claude/skills/codeman, every new Claude Code session sees it
|
||||
codeman skill install --case myproject # just that case; linked cases resolve by name too
|
||||
codeman skill uninstall # reverses either one
|
||||
```
|
||||
|
||||
Or turn on **App Settings > Agent Skill** (`agentSkillEnabled`, synced, default off) and Codeman injects the skill into each case when a Claude session is created there.
|
||||
|
||||
Installs are marker-owned: a `skills/codeman` that Codeman did not write is never touched, a stale managed copy is refreshed in place, and a symlinked skill directory is refused rather than written through. Re-run `codeman skill install` after upgrading Codeman to refresh the copy.
|
||||
|
||||
Note that turning `agentSkillEnabled` back off does **not** remove already-injected copies, because a create-time sweep would yank the skill out from under other live sessions sharing that `.claude/` directory. Remove them per case with `codeman skill uninstall --case <name>`.
|
||||
|
||||
## Using it
|
||||
|
||||
Once installed, just ask: "spin up three workers and have them lint, typecheck and test in parallel, then report back". The skill supplies the guard, the safety rules and the recipes. What it does under the hood:
|
||||
|
||||
**1. Guard.** Every Bash call re-runs a preamble that refuses outside `CODEMAN_MUX=1`, reads `CODEMAN_API_URL` and `CODEMAN_SESSION_ID`, recovers a password from the data dir `.env` or the install's service definition if one is set, and defines a fail-closed `delete_session`. It re-runs it every call because shell state does not survive between an agent's tool calls.
|
||||
|
||||
**2. Start a worker.**
|
||||
|
||||
```bash
|
||||
Q=$("${CURL[@]}" -X POST "$API/api/v1/quick-start" -H 'Content-Type: application/json' \
|
||||
-d '{"caseName":"worker-1","mode":"claude"}')
|
||||
SID=$(jq -r 'if .success then .data.sessionId else empty end' <<<"$Q")
|
||||
```
|
||||
|
||||
`mode` is any of `claude`, `shell`, `opencode`, `codex`, `gemini`, `antigravity`.
|
||||
|
||||
**3. Wait until it is actually ready.** A new session reports `idle` before its CLI has spawned, and a brand-new case shows a trust dialog first, so the skill waits for the composer's own status bar and treats the dialog as a bounded fallback.
|
||||
|
||||
**4. Send a prompt and wait for the turn to end.**
|
||||
|
||||
```bash
|
||||
BODY=$(jq -n --arg p "$PROMPT" '{input:($p+"\r"),useMux:true,clientId:"codeman-agent-1",seq:1,wait:true,waitTimeout:60000}')
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" -H 'Content-Type: application/json' --data-binary "$BODY"
|
||||
```
|
||||
|
||||
Send-and-wait registers the waiter before typing, which closes the race where a separate wait reports the previous turn's idle state as this turn's answer. For `claude` workers it resolves on the `stop` hook, typically within seconds.
|
||||
|
||||
**5. Read the answer.**
|
||||
|
||||
```bash
|
||||
"${CURL[@]}" "$API/api/v1/sessions/$SID/last-response" | jq -r '.data.text'
|
||||
```
|
||||
|
||||
**6. Clean up.** `delete_session "$SID"`, for ids you created and nothing else.
|
||||
|
||||
Hook-less modes (`shell` and the external CLIs) have no `stop` signal and coarse lifecycle transitions, so the skill synchronizes those with a unique split marker and `wait-output ... from=buffer` instead. Worked fan-out flows, the per-mode signal table, error codes and the Docker/remote caveats live in the skill's `reference/` files, loaded on demand.
|
||||
|
||||
## The rules that bite
|
||||
|
||||
The skill documents these because each one silently wastes a run:
|
||||
- **Every input must end with `\r`** or Enter is never sent and the text sits unsubmitted on the worker's prompt. `delivered:true` means "written to the pane", not "submitted".
|
||||
- **Input is single-line.** Newlines are stripped.
|
||||
- **A wait timeout is HTTP 200** with `wait.timedOut:true`, not an error. Loop over short waits; timeouts clamp to [1s, 600s] and the applied value comes back as `wait.timeoutMs`.
|
||||
- **`stop` and `blocked` are `claude`-only.** Requesting them elsewhere is a 400.
|
||||
- **Signals are edge-triggered with no history.** One that fires while no waiter is registered is unobservable afterwards, so never fire-and-forget N prompts and then gather signal-waits worker by worker.
|
||||
- **Your typed command echoes into the output stream**, so a marker that appears verbatim in the input line matches before the command runs. Split it.
|
||||
- **A full-screen TUI stream is space-less**, so match a single space-free token, never a phrase.
|
||||
- **`pid != null` proves startup, not life.** A worker that dies inside its pane keeps `status:"idle"` and a pid. `wait?until=exit` is the death check.
|
||||
|
||||
## Fixes to the packaged skill
|
||||
- **The self-delete guard failed open.** The old `is_self "$SID" || curl -X DELETE ...` shape meant an undefined `is_self` exited 127, the `||` branch fired, and the agent deleted its own session with the one guard bypassed. That is reachable because shell state does not survive between tool calls, so a partially re-pasted preamble was enough. The DELETE now lives inside a fail-closed `delete_session`, which also refuses an empty id and refuses when `$SELF` is unset or too short to prove the target is not the caller.
|
||||
- **`clientId` was built from `$$`.** The pid changes between tool calls, so the documented "resend the identical request" loop stopped being recognized as a duplicate and retyped the prompt, submitting the turn twice. It is a fixed literal now.
|
||||
- **`GET /api/v1/sessions/:id/last-response` was undocumented.** It returns the agent's final message as clean transcript text; the terminal scrape the skill previously recommended returns a wall of TUI repaint noise with the answer buried in it. It is now the documented read path for `claude` and `codex`, with the terminal buffer demoted to diagnosis and hook-less modes. Because the transcript flush lags the `stop` signal, the recipes poll it instead of reading once.
|
||||
- **`quick-start` responses were never checked for `.success`.** On failure `.data.sessionId` is absent, `jq -r` prints the string `null`, and the flow burned its full readiness budget against `/api/v1/sessions/null` before reporting jq noise instead of the cause.
|
||||
- **`codeman skill install --case <name>` could not resolve a linked case.** It hardcoded `~/codeman-cases/<name>` while the server resolves through `linked-cases.json` first, so it failed with "Case not found" for a case the web UI handled fine.
|
||||
- **Documentation corrections**: `SESSION_BUSY` on `quick-start` is the 50-session cap rather than the waiter cap; `caseName` resolves linked cases, so a generic name can land a worker in a real repo; and the claim that a toggle-off sweep exists was wrong, so the per-case `skill uninstall` cleanup is now stated in both the README and the code.
|
||||
|
||||
## Also in this release
|
||||
- **Terminal**: the wheel is no longer forwarded to codex, which ignores SGR mouse reports.
|
||||
|
||||
## 1.14.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- Daemon mode and service install, plus subagent hook hardening and terminal/idle-checker fixes.
|
||||
|
||||
**New: run Codeman in the background without a terminal (#239, closes #231)**
|
||||
- `codeman web -d` starts the server detached: it survives closing the shell, logs to `~/.codeman/web.log`, records a pidfile, and only reports success after the server actually answers `/api/status` (a port clash or missing dependency can never read as a clean start). `codeman web --status` and `codeman web --stop` manage it; `--stop` verifies the pid still looks like a Codeman server before signalling, so a recycled pid is never SIGTERMed.
|
||||
- `codeman service install` / `status` / `uninstall`: installs a systemd user unit (Linux) or LaunchAgent (macOS) so the server comes back after reboots. The unit carries the installing shell's PATH (launchd's default PATH finds neither an nvm/Homebrew `node` nor `tmux`/`claude`), never contains `CODEMAN_PASSWORD`, and uses the same instance-scoped unit names as `install.sh` and the self-updater so no second copy can end up supervised.
|
||||
- Both refuse to start a second server on one data dir (pidfile check plus a live probe): two servers on the shared tmux socket would attach to each other's sessions.
|
||||
- Why `-d` exists at all: `nohup` does not protect a Node process, Node re-arms SIGHUP even when it inherits "ignore", so `nohup codeman web &` still dies on HUP. The detached relaunch (setsid) removes the controlling terminal instead.
|
||||
|
||||
**Subagent background-work hooks (#233, thanks @Lint111)**
|
||||
- The background Bash rewake helper now also watches the top-level parent transcript when the hook fires inside a subagent: Claude records a subagent's Bash result in its own `subagents/agent-*.jsonl` but queues the completion in the lead session transcript, so subagents previously never woke. It can also inline a `CODEMAN_RESULT_BEGIN/END` marked report (up to 64 KiB) from the task output file into the wake feedback.
|
||||
- New SubagentStop guard: a subagent that still owns live Monitor or background Bash processes is kept working instead of publishing an intermediate progress line as its final report. Ownership is verified against live process descriptors on `tasks/<id>.output`, so stale transcript text alone never blocks, and the guard fails open on systems without `/proc`.
|
||||
- Existing cases self-heal to the new hooks on next launch.
|
||||
|
||||
**AI idle checker: stderr kept out of the verdict (#234, thanks @Lint111)**
|
||||
|
||||
The `claude -p` verdict command no longer merges stderr into the verdict file, where CLI warnings could turn a valid verdict into a parse error. On failures, the first 200 chars of stderr are attached to the diagnostic instead.
|
||||
|
||||
**Terminal: large final batches drain fully (#235, thanks @Lint111)**
|
||||
|
||||
A render-scheduling flag was cleared after the flush instead of before it, so when a large batch left a remainder behind, the remainder stayed unrendered until unrelated output arrived. This looked like truncated responses or shell commands that never finish. The flush now reschedules itself until the queue is empty.
|
||||
|
||||
**Docs and tests**
|
||||
- README documents daemon mode and service install.
|
||||
- Unique test port for the daemon-control suite.
|
||||
|
||||
## 1.13.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- Agent wait primitives, the Codeman agent skill, a fix for hooks dying silently on HTTPS installs, and the tab-strip UX improvements from the previous batch.
|
||||
|
||||
**Agent wait primitives (new API surface, the reason this is a minor).** Three bounded long-polls let an agent driving Codeman from a shell block instead of poll:
|
||||
- `GET /api/v1/sessions/:id/wait` blocks until a lifecycle signal fires (`until=stop,idle,working,blocked,exit`, `fresh=1` to require a new transition).
|
||||
- `GET /api/v1/sessions/:id/wait-output` blocks until a literal substring appears in the session's output (`match=`, `nocase=`, `from=now|buffer`; never regex, by design).
|
||||
- `wait`/`waitTimeout` on `POST /api/v1/sessions/:id/input` (send-and-wait) registers the waiter before typing, closing the race where a separate wait reports the previous turn's idle state as this turn's answer.
|
||||
|
||||
Shared semantics: a timeout is HTTP 200 with `wait.timedOut: true` (callers loop over short waits; tunnels cut idle connections), timeouts are clamped to [1s, 600s] and echoed back as `wait.timeoutMs`, all three nest the result under `data.wait`, and `status`/`limitPaused` ride along. `stop`/`blocked` exist for `claude` mode only: requesting them explicitly elsewhere is a 400, the default set silently narrows and echoes what it waited on. Capacity caps (16 waiters per session, 128 process-wide) answer 409/429, waiter slots release on client hang-up, and shutdown resolves parked waiters instead of stranding them. Bounds are operator-tunable via `CODEMAN_WAIT_*` env vars.
|
||||
|
||||
Reliability details that came out of three verification rounds: a worker that dies inside its tmux pane is now detected at the mux layer (pane-death probe, ~750ms cache, a 3s watcher for waits already parked), so a corpse answers `exit` instead of `idle` and send-and-wait rolls back its dedup seq when the write went nowhere; output matching normalizes charset-designation escapes (a stock bash prompt's `ESC ( B` no longer breaks `match=tnode:`) and holds back partial escapes at chunk boundaries, so matches straddling PTY chunks are found.
|
||||
|
||||
**Codeman agent skill (`skills/codeman`).** A packaged skill that teaches an agent running inside a Codeman session to drive the API safely: guard preamble (refuses outside `CODEMAN_MUX=1`, resolves credentials from the data dir `.env` or the install's service definition), self-protection (`is_self` prefix check in both directions), readiness for claude workers (composer-first, trust dialog as bounded fallback), send-and-wait loops that cannot report a never-submitted prompt as success, marker-synchronized shell flows, fan-out patterns, and cleanup discipline. Ships in the npm package via the `files` entry.
|
||||
|
||||
**Hooks were dying silently on every HTTPS install (bug fix).** The generated hook curls lacked `-k`, so on `--https` installs (self-signed cert) every hook event (`stop`, `permission_prompt`, `elicitation_dialog`, `idle_prompt`, `teammate_idle`, `task_completed`) failed TLS verification and the failure was swallowed, taking respawn's definitive idle signals with it. Hooks are now generated with `curl -sk`, and a staleness detector regenerates the on-disk hook config of already-created cases the next time a session starts in them. Relatedly, `CODEMAN_API_URL` is no longer exported with a guessed `http://localhost:3000` fallback (wrong scheme on HTTPS installs); it is omitted unless the server has stamped the real URL, so in-session guards fail closed.
|
||||
|
||||
**Tab strip (from the previous batch, reported by christianhaberl):** action icons (kill/pop-out) now appear on the active tab only, middle-click closes a tab, tab hover uses a fixed width with a sliding title instead of resizing the strip, and the pop-out button is opt-in (default off).
|
||||
|
||||
**Docs.** `docs/api-reference.md` gained the full long-polling contract (signals by mode, readiness, what the matcher sees, response discriminators); `docs/extending-codeman.md` and the README carry verified copy-paste orchestration recipes; `docs/architecture-invariants.md` records the load-bearing ordering, liveness, and edge-triggered-signal invariants. Net +163 tests (4300 passing in the CI sweep).
|
||||
|
||||
## 1.12.2
|
||||
|
||||
### Patch Changes
|
||||
|
||||
@@ -13,10 +13,10 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
||||
| Task | Command |
|
||||
| ----------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Dev server | `npm run dev` (or `npx tsx src/index.ts web`) |
|
||||
| Type check | `tsc --noEmit` |
|
||||
| Type check | `npm run typecheck` (= `tsc --noEmit`) |
|
||||
| Lint | `npm run lint` (fix: `npm run lint:fix`) |
|
||||
| Format | `npm run format` (check: `npm run format:check`) |
|
||||
| Single test | `npm test -- test/<file>.test.ts` (or `npx vitest run --config config/vitest.config.ts test/<file>.test.ts`) — ⚠ **never** run bare `npm test`, see Testing section |
|
||||
| Tests | `npm test` (the CI gate — safe to run bare) · one file: `npm test -- test/<file>.test.ts` · see Testing for the excluded suites |
|
||||
| Build | `npm run build` (esbuild via `scripts/build.mjs`, NOT tsc — `tsc --noEmit` is type-check only) |
|
||||
| Production | `npm run build && systemctl --user restart codeman-web` |
|
||||
|
||||
@@ -43,7 +43,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
||||
2. **Frontend changes**: Use Playwright to load the page and assert the UI renders correctly. Use `waitUntil: 'domcontentloaded'` (not `networkidle` — SSE keeps the connection open). Wait 3-4s for polling/async data to populate, then check element visibility, text content, and CSS values
|
||||
3. **Only after verification passes**, proceed with COM
|
||||
|
||||
The production server caches static files for 1 year, `immutable` (`maxAge: '1y'` in `server.ts`). To avoid stale frontend after a deploy, `renderIndexHtml` runs `cacheBustAssets(html)` — it appends `?v=<mtime>` to **every same-origin `.js`/`.css`** reference (mtime memoized ~1s so a burst of renders is cheap; external/already-versioned/missing refs untouched). Because `index.html` is served `no-cache`, a **normal reload now picks up edited modules/styles — no hard refresh needed** (the gesture bundle is injected separately with its own `?v=`). If you add an asset referenced by an _absolute_ URL or from JS rather than a `<script>/<link>` tag, it won't be auto-busted.
|
||||
The production server caches static files for 1 year, `immutable` (`maxAge: '1y'` in `server.ts`). To avoid stale frontend after a deploy, `renderIndexHtml` runs `cacheBustAssets(html)` — it appends `?v=<mtime>` to **every same-origin `.js`/`.css`** reference (mtime memoized ~1s so a burst of renders is cheap; external/already-versioned/missing refs untouched). Because `index.html` is served `no-cache`, a **normal reload now picks up edited modules/styles — no hard refresh needed** (the gesture bundle is injected separately with its own `?v=`). If you add an asset referenced by an _absolute_ URL or from JS rather than a `<script>/<link>` tag, it won't be auto-busted. ⚠️ **`index.html` itself is the exception: it is read ONCE into `indexHtmlTemplate` in the `WebServer` constructor**, so editing markup in dev needs a server restart (edited `.js`/`.css` do not) — otherwise you debug a "CSS class that doesn't apply" that is really an element still missing from the served HTML.
|
||||
|
||||
## COM Shorthand (Deployment)
|
||||
|
||||
@@ -70,17 +70,18 @@ When user says "COM":
|
||||
4. **Sync CLAUDE.md version**: Update the `**Version**` line below to match the new version from `package.json`
|
||||
5. **Commit and deploy**: verify the branch first (`git branch --show-current`), then stage EXPLICIT paths — never `git add -A`, which has swept another session's WIP into a release. `git status --short` and account for every line before committing:
|
||||
`git add <paths> && git commit -m "chore: version packages" && git push && npm run build && systemctl --user restart codeman-web`
|
||||
6. **Wait for CI**: after `git push`, TWO workflows fire per master push — `CI` and `Release` (the npm publish + GitHub release). List both runs for the pushed commit with `gh run list --commit $(git rev-parse HEAD) --json databaseId,workflowName` and watch EACH with `gh run watch <id> --exit-status`. Confirm both pass before considering the release done (`gh run list -L 1` returns only one of the two).
|
||||
6. **Refresh the getcodeman.com version badge**: the landing page's status bar carries the release version (`v<x.y.z> · getcodeman.com · MIT`), so it goes stale on every release if nobody bumps it. The site source and its deploy script are maintained outside this repository, on the maintainer's machine only; follow the local site handbook there, which also covers the numbers strip and `sitemap.xml` refresh that belong in the same pass. Poll production (`curl -s https://getcodeman.com/ | grep v<x.y.z>`) before calling it done, since the edge lags a deploy by up to a minute. Not applicable to contributor clones — skip it and say so.
|
||||
7. **Wait for CI**: after `git push`, TWO workflows fire per master push — `CI` and `Release` (the npm publish + GitHub release). List both runs for the pushed commit with `gh run list --commit $(git rev-parse HEAD) --json databaseId,workflowName` and watch EACH with `gh run watch <id> --exit-status`. Confirm both pass before considering the release done (`gh run list -L 1` returns only one of the two).
|
||||
|
||||
CI runs `npm run check:lockfile` on every push/PR, so lockfile drift fails the build even if the `version-packages` script is bypassed.
|
||||
|
||||
**Version**: 1.12.2 (must match `package.json`)
|
||||
**Version**: 1.19.6 (must match `package.json`)
|
||||
|
||||
## Project Overview
|
||||
|
||||
Codeman is a Claude Code session manager with web interface and autonomous Ralph Loop. Spawns Claude CLI via PTY, streams via SSE, supports respawn cycling for 24+ hour autonomous runs.
|
||||
|
||||
**Tech Stack**: TypeScript (ES2022/NodeNext, strict mode), Node.js, Fastify, node-pty, xterm.js. Supports Claude Code, OpenCode, Codex (OpenAI), Gemini (Google, enterprise-only since Google's June 2026 consumer cutover), and Antigravity (`agy`, Google) CLIs via pluggable CLI resolvers (`SessionMode = 'claude' | 'shell' | 'opencode' | 'codex' | 'gemini' | 'antigravity'`).
|
||||
**Tech Stack**: TypeScript (ES2022/NodeNext, strict mode), Node.js, Fastify, node-pty, xterm.js. Supports Claude Code, OpenCode, Codex (OpenAI), Gemini (Google, enterprise-only since Google's June 2026 consumer cutover), Antigravity (`agy`, Google) and Pi (pi.dev) CLIs via pluggable CLI resolvers (`SessionMode = 'claude' | 'shell' | 'opencode' | 'codex' | 'gemini' | 'antigravity' | 'pi'`).
|
||||
|
||||
**TypeScript Strictness** (see `tsconfig.json`): `noUnusedLocals`, `noUnusedParameters`, `noImplicitReturns`, `noImplicitOverride`, `noFallthroughCasesInSwitch`, `allowUnreachableCode: false`, `allowUnusedLabels: false`.
|
||||
|
||||
@@ -98,7 +99,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
| Override window title hostname | `npx tsx src/index.ts web --title-hostname <name>` (default: `os.hostname()` — `codeman:<name>` is used for tab title, title-flash, and OS desktop notification prefix) |
|
||||
| Bind a non-loopback host | `npx tsx src/index.ts web --host 0.0.0.0` (or `-H`; env `CODEMAN_HOST`; default `127.0.0.1`). Without `CODEMAN_PASSWORD` it **starts but warns loudly** — see Common Gotchas + `docs/security-architecture.md` |
|
||||
| Continuous typecheck | `tsc --noEmit --watch` |
|
||||
| Watch-mode test | `npm run test:watch -- test/<file>.test.ts` (always pass a file — bare watch includes the browser suites) |
|
||||
| Watch-mode test | `npm run test:watch -- test/<file>.test.ts` (runs the CI gate's config; pass a file to narrow it) |
|
||||
| Test coverage | `npm run test:coverage` |
|
||||
| Dead-code sweep | `npm run knip` (config in `config/knip.json`, passed via `--config`) |
|
||||
| Rebuild gesture overlay | `npm run build:gesture` (esbuild `packages/gesture-control/src/codeman/entry.ts` → `src/web/public/gesture/gesture-codeman.js`; commit the result) |
|
||||
@@ -106,28 +107,30 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
| Gesture playground | `npm run dev` **in** `packages/gesture-control/` (standalone vite demo, fake tabs) |
|
||||
| Check public-asset formatting | `npm run check:public-assets` (prettier-checks `src/web/public/**` text assets; `scripts/check-public-assets.mjs`) |
|
||||
| Frontend JS syntax check | `npm run check:frontend-syntax` (`scripts/check-frontend-syntax.mjs`; runs in CI) |
|
||||
| CI-equivalent test sweep | `npm run test:ci` (full suite minus browser/perf — see Testing) |
|
||||
| Excluded-suite runners | `npm run test:browser` · `npm run test:mobile` · `npm run test:perf` · `npm run test:all` (everything, environmental failures included) — see Testing |
|
||||
| Production start | `npm run start` |
|
||||
| Production logs | `journalctl --user -u codeman-web -f` |
|
||||
| Detached server | `codeman web -d` (`--status`, `--stop`; pidfile+log at `dataPath('web.pid'/'web.log')`). ⚠ Refuses to start a 2nd server on one data dir — see Instance isolation |
|
||||
| Install/remove the service | `codeman service install` / `status` / `uninstall` (systemd user unit on Linux, LaunchAgent on macOS; names from `config/service-names.ts`) |
|
||||
|
||||
**CI**: `.github/workflows/ci.yml` (push to master/main + PRs, Node 22) runs two jobs: **(1)** `check:lockfile`, `typecheck`, `lint`, `check:frontend-syntax`, `format:check`, then a **server boot smoke test** (`tsx src/index.ts web --port 3151` must answer `/api/status` within 30s); **(2)** the **unit/integration test suite** via `npm run test:ci` (`config/vitest.ci.config.ts` — excludes the browser-driven `test/mobile/**` suite, `perf-*` benchmarks, and 3 Playwright tests). Tests are tmux-safe in CI: `TmuxManager` no-ops all shell commands under `VITEST` (see Testing).
|
||||
**CI**: `.github/workflows/ci.yml` (push to master/main + PRs, Node 22) runs two jobs: **(1)** `check:lockfile`, `typecheck`, `lint`, `check:frontend-syntax`, `format:check`, then a **server boot smoke test** (`tsx src/index.ts web --port 3151` must answer `/api/status` within 30s); **(2)** the **unit/integration test suite** via `npm run test:ci` (`config/vitest.ci.config.ts` — excludes the browser-driven `test/mobile/**` suite, `perf-*` benchmarks, and 5 Playwright tests; globs live in `config/test-suites.ts`). `npm test` runs this same config, so local green == CI green. Tests are tmux-safe in CI: `TmuxManager` no-ops all shell commands under `VITEST` (see Testing).
|
||||
|
||||
**Code style**: Prettier (`singleQuote: true`, `printWidth: 120`, `trailingComma: "es5"`) — config lives in the **`"prettier"` key of `package.json`**, not a `.prettierrc` (keeps the repo root short; editors read it natively). `.prettierignore` stays at the root because Prettier resolves it relative to cwd. ESLint flat config (`config/eslint.config.js`) allows `no-console`, warns on `@typescript-eslint/no-explicit-any`. Ignores: `app.js`, `scripts/**/*.mjs`, `src/web/public/vendor/**`, `scripts/remotion/**`.
|
||||
|
||||
**Prettier scope is deliberately narrow.** `npm run format` globs only `src/**/*.ts` and `src/web/public/**`, and `.prettierignore` then exempts most of `src/web/public/*.js` (app.js, styles.css, index.html, and 14 hand-formatted modules) plus `CLAUDE.md`. Those files are hand-formatted by design; `npm run check:public-assets` and `check:frontend-syntax` are what guard them (NUL bytes + JS syntax), not Prettier. Do not "fix" a file by adding it back to Prettier's scope.
|
||||
**Prettier scope is deliberately narrow.** `npm run format` globs only `src/**/*.ts` and `src/web/public/**`, and `.prettierignore` then exempts most of `src/web/public/*.js` (app.js, styles.css, **mobile.css**, index.html, upload.html, and 15 hand-formatted modules) plus `CLAUDE.md`. Those files are hand-formatted by design; `npm run check:public-assets` and `check:frontend-syntax` are what guard them (NUL bytes + JS syntax), not Prettier. Do not "fix" a file by adding it back to Prettier's scope.
|
||||
|
||||
## Common Gotchas
|
||||
|
||||
- **Single-line prompts only** — `writeViaMux()` sends text+Enter separately; multi-line breaks Ink
|
||||
- **Single-line prompts only** — `writeViaMux()` sends text+Enter separately; multi-line breaks Ink. ⚠️ **Input must END with `\r` or Enter is never sent**: `sendInput()` only issues `send-keys Enter` when the payload contains a carriage return, a `\r`-less `POST /api/sessions/:id/input` still succeeds (send-and-wait even reports `delivered:true`) while the text sits unsubmitted on the composer, and any `wait` burns its whole timeout on a turn that never started. Embedded newlines are stripped, not rejected, so `"echo A\necho B\r"` runs the joined `echo Aecho B`
|
||||
- **ESM only** — Never `require()`, use `await import()`. `tsx` masks CJS/ESM issues in dev but production breaks
|
||||
- **Package ≠ product name** — npm: `aicodeman`, product: **Codeman**. Release renames tags accordingly. Both `aicodeman` and `codeman` bin aliases are installed (`package.json` `bin`)
|
||||
- **Global regex `lastIndex`** — Shared `g`-flag patterns in loops must reset `lastIndex = 0` first, or use the `execPattern()` helper in `utils/regex-patterns.ts` (resets automatically)
|
||||
- **`envOverrides` flow `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `GEMINI_*` / `GOOGLE_*` / `ANTIGRAVITY_*` env vars** — Set via `POST /api/sessions { envOverrides }`, stored on `Session._envOverrides`, exported by `tmux-manager.buildEnvExports()` at spawn time, persisted in `SessionState.envOverrides`. **Do NOT** write these to `<case>/.claude/settings.local.json` — that's the old path and creates UI/disk drift. (`GOOGLE_*` is the deliberately-broad Vertex-AI namespace for Gemini — see Multi-CLI prefix discipline.)
|
||||
- **`envOverrides` flow `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `GEMINI_*` / `GOOGLE_*` / `ANTIGRAVITY_*` / `PI_*` env vars, plus exact-key `CLAUDE_CONFIG_DIR`** — Set via `POST /api/sessions { envOverrides }`, stored on `Session._envOverrides`, exported by `tmux-manager.buildEnvExports()` at spawn time, persisted in `SessionState.envOverrides`. **Do NOT** write these to `<case>/.claude/settings.local.json` — that's the old path and creates UI/disk drift. (`GOOGLE_*` is the deliberately-broad Vertex-AI namespace for Gemini — see Multi-CLI prefix discipline.) `CLAUDE_CONFIG_DIR` (#255, exact match via `ALLOWED_ENV_KEYS` in `schemas.ts`) points a session at a separate Claude account/config dir for per-client subscriptions; it persists to state.json (a path, not a secret; losing it on restart would silently switch accounts). ⚠️ A relocated config dir writes transcripts outside `~/.claude/projects`, so the response viewer, subagent windows, ultracode panel and Read My Mind capture go blind for that session unless the user symlinks `projects` back into the shared tree (`ln -s ~/.claude/projects <configDir>/projects`). → [architecture-invariants#per-session-env-overrides-exact-key-allowlist-and-claude_config_dir](docs/architecture-invariants.md#per-session-env-overrides-exact-key-allowlist-and-claude_config_dir)
|
||||
- **Effort is NOT an env var** — never carry effort as `CLAUDE_CODE_EFFORT_LEVEL`: the env var hard-locks effort and blocks in-session `/effort` switching (incl. ultracode). It flows as the dedicated `effort` payload field → `Session._effort` → `claude --effort <level>` for regular levels incl. `max` (the settings `effortLevel` key is `enum(["low","medium","high","xhigh"]).catch(undefined)` — `max` gets SILENTLY dropped there), or `claude --settings '{"ultracode":true}'` for ultracode (rejected by `--effort`). Both are soft defaults the user can override anytime. Legacy env-var entries are auto-migrated by the Session constructor and unset from tmux sessions in `applyEnvOverrides()`. See `buildEffortCliArgs()` in `session-cli-builder.ts`, tests in `test/effort-injection.test.ts`
|
||||
- **Model choice flows via `settings.local.json`, NOT `--model` or env** — the App Settings **Claude Model** picker (`claudeModel` in `settings.json`) is read by `session-ui.js` at session create (wins over the legacy 1M-Opus toggles `opusContext1m`/`opusContext1mEnabled`), sent as the `modelOverride` payload field, and `updateCaseModel()` (`hooks-config.ts`) writes/deletes the `model` key in `<case>/.claude/settings.local.json`. This is the intended exception to the envOverrides rule above: model legitimately lives in `settings.local.json` (a soft default — in-session `/model` still works); env vars do not
|
||||
- **Multi-CLI prefix discipline** — env-var prefix is CLI-specific (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `GEMINI_*` vs `ANTIGRAVITY_*`) and the `ALLOWED_ENV_PREFIXES` allowlist in `schemas.ts` enforces this. Gemini additionally allowlists the **broad `GOOGLE_*`** namespace (intentional: Vertex AI auth needs `GOOGLE_CLOUD_PROJECT`/`GOOGLE_APPLICATION_CREDENTIALS`/`GOOGLE_GENAI_USE_VERTEXAI`; it is the loosest allowlist entry, affecting only the user's own spawned CLI). When adding a setting, decide which CLI(s) it applies to and gate the env export accordingly. Never blanket-forward all prefixes. Resolver design pattern: `docs/opencode-integration.md`
|
||||
- **Multi-CLI prefix discipline** — env-var prefix is CLI-specific (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `GEMINI_*` vs `ANTIGRAVITY_*` vs `PI_*`) and the `ALLOWED_ENV_PREFIXES` allowlist in `schemas.ts` enforces this; non-prefix exceptions are exact keys in `ALLOWED_ENV_KEYS` (currently only `CLAUDE_CONFIG_DIR`), never a widened prefix. Gemini additionally allowlists the **broad `GOOGLE_*`** namespace (intentional: Vertex AI auth needs `GOOGLE_CLOUD_PROJECT`/`GOOGLE_APPLICATION_CREDENTIALS`/`GOOGLE_GENAI_USE_VERTEXAI`; it is the loosest allowlist entry, affecting only the user's own spawned CLI). When adding a setting, decide which CLI(s) it applies to and gate the env export accordingly. Never blanket-forward all prefixes. ⚠️ Pi is the case that proves the rule: its ~34 provider keys (`ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `HF_TOKEN`, …) share NO prefix, and the allowlist is one GLOBAL list applied by a refine with no mode context, so admitting them for pi would widen it for every mode at once — they stay out, and pi users authenticate via `/login` or the server process's own env. Resolver design pattern: `docs/opencode-integration.md`, `docs/pi-integration.md`
|
||||
- **Zod `.optional()` rejects `null`** — accepts `undefined` only. When the frontend builds a request body with `JSON.stringify`, an explicit `null` field is preserved on the wire and fails validation with `INVALID_INPUT`. Convert `null` → `undefined` before stringifying (e.g. `field: value ?? undefined`), or declare the schema `.nullish()`. This has caused real shipped bugs twice
|
||||
- **`xterm-zerolag-input` is single-source** — the local-echo overlay source lives ONLY in `packages/xterm-zerolag-input/src/`, and is bundled into the **gitignored** `src/web/public/vendor/xterm-zerolag-input.js` (dev, by `scripts/postinstall.js`) and `dist/.../vendor/` (prod, by `scripts/build.mjs`). `app.js` only **consumes** it via `new LocalEchoOverlay(terminal)`; there is no inline copy. So: change the package source, then rerun the bundle step (`npm install` for dev, `npm run build` for prod). **Never hand-edit `app.js` for overlay behavior, and never commit the gitignored vendor bundle.** Always test on mobile after touching it. → [architecture-invariants#xterm-zerolag-input-is-single-source](docs/architecture-invariants.md#xterm-zerolag-input-is-single-source), `docs/local-echo-overlay-plan.md`
|
||||
- **`xterm-zerolag-input` is single-source** — BOTH echo addons live ONLY in `packages/xterm-zerolag-input/src/`, bundled into TWO **gitignored** vendor files: `vendor/xterm-zerolag-input.js` (buffer overlay, entry `zerolag-input-addon.ts`) and `vendor/xterm-predictive-echo.js` (codex write-through, entry `predictive-echo-addon.ts`) — dev by `scripts/postinstall.js`, prod by `scripts/build.mjs`. `app.js`/terminal-ui.js only **consume** them via `new LocalEchoOverlay(terminal)` / `new PredictiveEchoOverlay(terminal)`; there is no inline copy. So: change the package source, then rerun the bundle step (`npm install` for dev, `npm run build` for prod). **Never hand-edit `app.js` for overlay behavior, and never commit the gitignored vendor bundles.** Always test on mobile after touching it. → [architecture-invariants#xterm-zerolag-input-is-single-source](docs/architecture-invariants.md#xterm-zerolag-input-is-single-source), `docs/local-echo-overlay-plan.md`
|
||||
- **Default bind is loopback-only; non-loopback without a password starts but warns** — the server defaults to `--host 127.0.0.1`. Binding non-loopback (`--host`/`-H`/`CODEMAN_HOST`) without `CODEMAN_PASSWORD` starts anyway but prints a loud warning; `--allow-unauthenticated-network` / `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK=1` acknowledges it. ⚠️ The production systemd unit passes no `--host`, so prod binds **localhost only**: reach it via `tailscale serve`/tunnel to `127.0.0.1`. A loopback bind is reachable through a same-host tunnel but NOT by a browser hitting the box's LAN IP. `install.sh` is separate and prompts for the binding (defaulting to LAN + a password), and preserves the existing binding on re-runs. → [architecture-invariants#default-bind-and-the-non-loopback-warning-path](docs/architecture-invariants.md#default-bind-and-the-non-loopback-warning-path), `docs/security-architecture.md`
|
||||
- **Instance isolation / multi-instance attach danger** — the data dir (`~/.codeman`) and tmux socket (`tmux -L codeman`) are PROCESS-WIDE and shared by every Codeman on the machine, derived from `CODEMAN_INSTANCE` via `src/config/instance.ts`. ⚠️ A 2nd instance on the SAME socket **discovers and attaches PTYs to the first instance's live sessions**, resizing and mutating them. `$HOME` isolation is NOT enough because tmux is system-global. To run two instances, give each a distinct `CODEMAN_INSTANCE` (scopes dir + socket together), or set `CODEMAN_TMUX_SOCKET` + `CODEMAN_DATA_DIR` individually; `scripts/run-beta.sh` does this for a beta alongside prod. **Any new `~/.codeman/...` path MUST go through `dataPath()`**, never `join(homedir(), '.codeman', …)`. → [architecture-invariants#instance-isolation-and-the-multi-instance-attach-danger](docs/architecture-invariants.md#instance-isolation-and-the-multi-instance-attach-danger)
|
||||
- **node-pty's macOS `spawn-helper` ships without `+x`** (issues #6, #204): `node-pty@1.1.0` publishes `prebuilds/darwin-<arch>/spawn-helper` as mode 0644, and macOS launches every PTY through it, so a stock macOS install fails every session start with `Error: posix_spawnp failed.` **Linux can never reproduce it**: `spawn-helper` is an `OS=="mac"` gyp target and node-pty ships no Linux prebuild, so node-gyp always emits an executable helper there. ⚠️ Look in **`prebuilds/<platform>-<arch>/`**, not just `build/Release/`, which does not exist on macOS. Repair is a chmod, never a mandatory rebuild (that would require Xcode CLI tools and deletes `prebuilds/` before compiling): `npm run fix:node-pty` chmods every helper then proves it by really opening a PTY. `spawnPtyWithHelperRepair()` (`utils/node-pty-repair.ts`) wraps every `pty.spawn()` in `session.ts` and self-heals a broken install on the first failure. → [architecture-invariants#node-ptys-macos-spawn-helper-must-be-executable](docs/architecture-invariants.md#node-ptys-macos-spawn-helper-must-be-executable)
|
||||
@@ -141,7 +144,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
| Domain | Key files | Notes |
|
||||
| ---------------- | -------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------- |
|
||||
| **Entry** | `src/index.ts`, `src/cli.ts` | |
|
||||
| **Entry** | `src/index.ts`, `src/cli.ts`, `daemon-control`, `service-installer`, `config/service-names` | The last three back `web -d` / `service install` |
|
||||
| **Session** | `src/session.ts` ★, `session-manager`, `session-auto-ops`, `session-cli-builder`, `session-task-cache`, `session-order` (pure), `session-pty-exit-breaker`, `usage-limit-patterns`, `usage-telemetry`; `src/services/unified-session-service.ts` | Pure/unit-tested helpers are split out of `session.ts` on purpose |
|
||||
| **Mux** | `src/mux-interface.ts`, `src/mux-factory.ts`, `src/tmux-manager.ts` ★ | |
|
||||
| **Respawn** | `src/respawn-controller.ts` ★ + 4 helpers (`-adaptive-timing`, `-health`, `-metrics`, `-patterns`) | Read `docs/respawn-state-machine.md` first |
|
||||
@@ -151,23 +154,23 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
| **Agents** | `src/subagent-watcher.ts` ★, `team-watcher`, `bash-tool-parser`, `transcript-watcher`, `workflow-run-watcher` | `workflow-run-watcher` is STANDALONE and never touches `subagent-watcher` |
|
||||
| **AI** | `src/ai-checker-base.ts`, `ai-idle-checker.ts`, `ai-plan-checker.ts` | |
|
||||
| **Tasks** | `src/task.ts`, `task-queue.ts`, `task-tracker.ts` | |
|
||||
| **State** | `src/state-store.ts`, `run-summary.ts`, `session-lifecycle-log.ts` | |
|
||||
| **State** | `src/state-store.ts`, `run-summary.ts`, `session-lifecycle-log.ts`, `intent-store.ts` | |
|
||||
| **Infra** | `src/hooks-config.ts`, `push-store`, `tunnel-manager`, `image-watcher`, `file-stream-manager`, `remote-hosts` + `remote-reconnect` (pure), `docker-hosts` + `docker-export` | Remote/docker case overlays; see Key Patterns |
|
||||
| **Web tabs** | `src/webview-store.ts`, `webview-capabilities.ts`, `src/web/webview-proxy.ts` (pure), `src/web/routes/webview-routes.ts` | Dashboard URLs as tabs; NOT a SessionMode |
|
||||
| **Search** | `src/search-service.ts` | Pure in-memory core for `GET /api/search` |
|
||||
| **Attachments** | `src/attachment-registry.ts`, `attachment-magic`, `generated-artifact-attachments`, `session-attachment-history`, `document-preview-cache`, `document-thumbnailer`, `document-conversion-limiter`, `config/attachment-guard` | See Key Patterns |
|
||||
| **Plan** | `src/plan-orchestrator.ts`, `src/prompts/*.ts`, `src/templates/` (`claude-md.ts` + `case-template.md`) | `templates/` holds the CLAUDE.md scaffold generated into new cases |
|
||||
| **Web** | `src/web/server.ts` ★, `sse-events.ts`, `routes/*.ts` (20 modules + barrel; `session-routes.ts` ★), `route-helpers.ts`, `ports/*.ts`, `middleware/auth.ts`, `schemas.ts`, `self-update.ts`, `plan-usage-latest.ts`, `ws-connection-registry.ts`, `heic-jpeg-converter.ts` + `heic-jpeg-worker.ts` | |
|
||||
| **Frontend** | `src/web/public/app.js` (~5K lines, core) + 25 modules + `sw.js` | See Frontend section for the load order, which is authoritative |
|
||||
| **Types** | `src/types/index.ts` (barrel) → 20 domain files; also `src/types.ts` root re-export | See `@fileoverview` in index.ts |
|
||||
| **Web** | `src/web/server.ts` ★, `sse-events.ts`, `routes/*.ts` (24 modules + barrel; `session-routes.ts` ★), `route-helpers.ts`, `ports/*.ts`, `middleware/auth.ts`, `schemas.ts`, `self-update.ts`, `plan-usage-latest.ts`, `ws-connection-registry.ts`, `heic-jpeg-converter.ts` + `heic-jpeg-worker.ts` | |
|
||||
| **Frontend** | `src/web/public/app.js` (~5K lines, core) + 30 modules + `sw.js` | See Frontend section for the load order, which is authoritative |
|
||||
| **Types** | `src/types/index.ts` (barrel) → 22 domain files; also `src/types.ts` root re-export | See `@fileoverview` in index.ts |
|
||||
|
||||
★ = Large, central file (>50KB) — read its `@fileoverview` first. All files have `@fileoverview` JSDoc — read that before diving in. Discovery aid: `grep -l '@fileoverview' src/web/routes/*.ts` lists all route modules; same grep works for `src/types/`, `src/web/public/*.js`.
|
||||
|
||||
**Local packages**: `packages/xterm-zerolag-input/` (local echo overlay, single-source, see Gotchas). `packages/gesture-control/` (`codeman-gesture-control`, hand-tracking overlay source, built via `npm run build:gesture`).
|
||||
|
||||
**Config**: `src/config/` — 17 files, no barrel (`index.ts`) exists; import from the specific file.
|
||||
**Config**: `src/config/` — 21 files, no barrel (`index.ts`) exists; import from the specific file.
|
||||
|
||||
**Utilities**: `src/utils/` — re-exported via index. Key: `CleanupManager`, `LRUMap` (⚠ NOT in the barrel — import from `./utils/lru-map.js` directly), `StaleExpirationMap`, `BufferAccumulator`, `stripAnsi`, `Debouncer`, `KeyedDebouncer`. Also: `claude-cli-resolver`/`opencode-cli-resolver`/`codex-cli-resolver`/`gemini-cli-resolver` (CLI path resolution), `string-similarity` (fuzzy matching), `regex-patterns` (ANSI/token/spinner patterns), `assertNever` (exhaustive checks), `token-validation` (auth tokens), `nice-wrapper` (process priority).
|
||||
**Utilities**: `src/utils/` — re-exported via index. Key: `CleanupManager`, `LRUMap` (⚠ NOT in the barrel — import from `./utils/lru-map.js` directly), `StaleExpirationMap`, `BufferAccumulator`, `stripAnsi`, `Debouncer`, `KeyedDebouncer`. Also: `claude-cli-resolver`/`opencode-cli-resolver`/`codex-cli-resolver`/`gemini-cli-resolver`/`antigravity-cli-resolver`/`pi-cli-resolver` (CLI path resolution; ⚠ `pi-cli-resolver` additionally version-probes the binary, since `pi` is a generic name), `string-similarity` (fuzzy matching), `regex-patterns` (ANSI/token/spinner patterns), `assertNever` (exhaustive checks), `token-validation` (auth tokens), `nice-wrapper` (process priority).
|
||||
|
||||
### Data Flow
|
||||
|
||||
@@ -180,8 +183,12 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
**Input**: `session.writeViaMux()` for programmatic/curl input via tmux `send-keys -l` + `send-keys Enter`, single-line only. Interactive **browser** input goes through a durable **exactly-once** layer: a stable `clientId` + monotonic per-session `seq` persisted to localStorage until the server ACKs, so a dropped link cannot lose or double-deliver a prompt. `ws-connection-registry.ts` supersedes only same-TAB reconnects, so two tabs on one session coexist. → [architecture-invariants#input-delivery-and-ws-resilience](docs/architecture-invariants.md#input-delivery-and-ws-resilience)
|
||||
|
||||
**Agent wait primitives**: bounded long-polls so an agent driving Codeman from a shell can block instead of poll: `GET /api/sessions/:id/wait` (lifecycle signal), `GET /api/sessions/:id/wait-output` (literal substring, **never** regex) and `wait`/`waitTimeout` on `POST /api/sessions/:id/input`. Registry in `session-wait-registry.ts` (pure, no `Session` reference), bounds in `config/agent-wait.ts`. ⚠️ **A timeout is a 200** (`wait.timedOut`), never an error, so callers loop over short waits. ⚠️ `stop`/`blocked` come from Claude Code hooks and therefore fire for **`claude` mode ONLY** (`shell` installs none either); asking for one explicitly on another mode is a 400, the default set silently drops them. ⚠️ Send-and-wait registers the waiter BEFORE the write (a separate POST-then-wait races and reports the PREVIOUS turn), and both teardown paths must `notifySignal('exit')` BEFORE `cancelAll()`. ⚠️ Client-hangup abort listens on **`reply.raw`** guarded by `writableFinished`: on `req.raw`, `close` fires when the request BODY ends, which on a POST killed every send-and-wait instantly and no `app.inject()` test could see it. ⚠️ Worker liveness cannot come from `session.pid` — for a tmux session that is the local attach client, which outlives a worker dying inside its pane — so it is probed at the mux layer (`isPaneDead`, ~750 ms cache) on blocking waits only, never on the input hot path. ⚠️ Signals are edge-triggered with no history: one that fires with no waiter registered is unobservable afterwards, so gather fan-outs with send-and-wait or latched `wait-output` markers, never fire-and-forget-then-sequential-signal-waits. The primitives are packaged as the **`skills/codeman` agent skill**: installable via `codeman skill install [--case <name>]` / `skill uninstall`, or auto-injected into a case's `.claude/skills/` on Claude session create behind `agentSkillEnabled` (SYNCED, default OFF). Injection is ADD-ONLY at create, marker-owned (`applyAgentSkill` in `hooks-config.ts` never touches an unmarked user copy) and refuses symlinks (this repo's own `.claude/skills/codeman` is a symlink to the source, which the injector must never write through). ⚠️ Claude Code loads a same-named USER-LEVEL skill (`~/.claude/skills/codeman`, written once by `codeman skill install` with no `--case`) over the per-case copy, and nothing used to refresh it: a stale Aug-9 user copy shadowed every fresh injection (2026-08-14: agents ran the old recipes, spawned workers serially and lost their lineage arcs), so session create now also refreshes a marker-owned user copy (`refreshUserAgentSkill`; refresh-only, never installs, foreign/symlink refused). Session create additionally pre-seeds the skill's §0 preamble cache (`seedAgentSessionPreamble` → `${XDG_CACHE_HOME:-~/.cache}/codeman-agent-<id>.sh`, local claude sessions only), single-sourced from `skills/codeman/preamble.sh` and pinned byte-identical to SKILL.md's §0 heredoc by `test/agent-skill.test.ts`, so the skill's bootstrap is a two-line loader instead of a ~150-line paste the model types out (~47 s of generation, measured live). → [architecture-invariants#agent-wait-primitives](docs/architecture-invariants.md#agent-wait-primitives), `docs/api-reference.md`
|
||||
|
||||
**Idle detection**: Multi-layer (completion message → AI check → output silence → token stability). See `docs/respawn-state-machine.md`.
|
||||
|
||||
⚠️ **A `❯` sighting is NOT the end of a turn, and neither is silence.** Claude redraws the composer (`❯`) about once a second all through a turn, so the old "saw a ❯, wait 2s → idle" rule flipped every working session to idle two seconds in (measured: a session mid-tool-call at 17 minutes reporting `status:"idle"`). Its working indicator is `✻ Actualizing… (13m 23s · ↓ 47.5k tokens)`: the glyph animates through `· ✢ ✳ ∗ ✻ ✽`, the gerund is randomized, and the finished line (`✻ Cooked for 2m 49s`) carries the same glyph, so neither `SPINNER_PATTERN` (braille, not what current versions draw) nor a keyword list can see it. Matching the new line in the STREAM does not work either: tmux ships partial repaints, so the whole line reaches the PTY only every few tens of seconds. So: `_confirmIdle()` (session.ts) requires the pane to go quiet, and then asks the SCREEN via `capturePaneText()` + `CLAUDE_WORKING_LINE_PATTERN` before believing it; a sustained run of repaints (`session-activity.ts`, pure + unit tested) is what marks a turn as started, with the same screen probe vetoing keystroke echo. Idle now lands ~3-5s after a turn ends instead of 2s into one. Claude-mode only, since an external CLI has no `❯`, so nothing would ever arm the confirmation and the session would latch busy.
|
||||
|
||||
**Auto-resume on usage limit** (opt-in per session, top of the Respawn tab): when Claude halts on a subscription limit, `usage-limit-patterns.ts` (pure, unit-tested) parses the reset time and `SessionAutoOps` arms a timer for reset+2min, then sends Esc + `continue`. ⚠️ Respawn cycles are blocked while paused (`isLimitPaused` guard in `onIdleDetected`), which is what prevents `/clear` from wiping the paused conversation. Claude-mode only. → [architecture-invariants#auto-resume-on-usage-limit](docs/architecture-invariants.md#auto-resume-on-usage-limit)
|
||||
|
||||
**Plan-usage chip** (statusLine telemetry, `showPlanUsageLimits`, per-device: desktop default **ON**, handhelds OFF via the mobile block in `getDefaultSettings()`): resolve it ONLY through `planUsageChipEnabled()` in settings-ui.js, which backs all three call sites (the App Settings checkbox, the chip's visibility, and the `statusLineTelemetry` flag on session create). A chip shown without telemetry renders `—` forever. Codeman injects its own `statusLine.command` exporter which POSTs Claude's `rate_limits` blob to `POST /api/status-telemetry`. The exporter is identified by a marker, so it only ever adds/updates/removes a statusLine that is **ours**, never a user's hand-authored one, and it prints the footer through so the in-terminal statusline is not blanked. Claude-mode only; distinct from auto-resume, which reacts to the limit *message* rather than showing live %. → [architecture-invariants#plan-usage-chip-statusline-telemetry](docs/architecture-invariants.md#plan-usage-chip-statusline-telemetry), `docs/usage-limits-display-plan.md`
|
||||
@@ -194,13 +201,21 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
**Docker cases**: a case can point at a **container**, with any of the five CLI backends running inside it. Like remote-SSH this is a **LOCATION OVERLAY on cases, never a sixth `SessionMode`**. Exactly one long-lived container **per case**, shared by all its sessions, so killing a session kills only that session's in-container tmux and **never** `docker stop` while siblings remain. The workspace is a real host dir bind-mounted at the **same absolute path**, which is what keeps file-routes/watchers on real host bytes and makes the in-container transcript projHash match the host. Credentials are **seeded** (RO mount, copied into the container once) rather than shared RW, so in-container CLIs never write refreshed tokens back to the host, and bind mounts are excluded from `docker commit` so exports stay secret-free. **NEVER a create-time `-e` for secrets, NEVER `--privileged`, NEVER the docker socket.** Config drift is detected via a label hash and a drifted launch is REFUSED rather than silently launched with stale config. ⚠️ On the loopback-only prod bind a container cannot reach 127.0.0.1, so in-container hooks need `CODEMAN_DOCKER_BRIDGE_HOOKS=1`; otherwise idle detection falls back to output-based. → [architecture-invariants#docker-cases](docs/architecture-invariants.md#docker-cases), `docs/docker-cases.md` (user guide), `docs/docker-cases-plan.md` (design)
|
||||
|
||||
**External CLI modes (OpenCode, Codex, Gemini, Antigravity)**: `isExternalCliMode()` in `session.ts` gates Claude-specific behavior off (Ralph tracker, BashToolParser, token/CLI-info parsing, ❯-prompt readiness); these CLIs render their own TUIs, so readiness is output stabilization instead. All four **require tmux with no direct PTY fallback**, because secrets are injected via socket-scoped `tmux setenv` and never on the spawn command line. ⚠️ `run*()` in `session-ui.js` MUST unwrap the `{success,data}` envelope; reading the raw shape silently breaks the run. ⚠️ **The local-echo overlay is DISABLED for codex sessions** (`_updateLocalEchoState` in terminal-ui.js, same branch as shell): codex's composer reacts per keystroke ("/" pops a live-filtering picker, arrows edit server-side state, the composer grows as it wraps), so buffer-until-Enter starved it into issues #218/#219/#220/#222. Codex also **drops keystrokes that share a PTY read with a bracketed paste**, so flushed text and the paste sequence must go out as separate delayed writes (mirroring the Enter branch's delayed `\r`). Tests: `test/local-echo-codex-gating.test.ts`. → [architecture-invariants#external-cli-modes-opencode-codex-gemini](docs/architecture-invariants.md#external-cli-modes-opencode-codex-gemini)
|
||||
**External CLI modes (OpenCode, Codex, Gemini, Antigravity, Pi)**: `isExternalCliMode()` in `session.ts` gates Claude-specific behavior off (Ralph tracker, BashToolParser, token/CLI-info parsing, ❯-prompt readiness); these CLIs render their own TUIs, so readiness is output stabilization instead. All five **require tmux with no direct PTY fallback**, because secrets are injected via socket-scoped `tmux setenv` and never on the spawn command line. ⚠️ `run*()` in `session-ui.js` MUST unwrap the `{success,data}` envelope; reading the raw shape silently breaks the run. ⚠️ **Codex sessions use PREDICTIVE WRITE-THROUGH echo, never the buffer overlay** (`_localEchoPolicy` in `_updateLocalEchoState`, terminal-ui.js): codex's composer reacts per keystroke ("/" pops a live-filtering picker, arrows edit server-side state, the composer grows as it wraps), so buffer-until-Enter starved it into issues #218/#219/#220/#222 and stays disabled (`_localEchoEnabled` remains false for codex). Instead, `PredictiveEchoAddon` (separate `vendor/xterm-predictive-echo.js` bundle) paints each keystroke at the predicted cell while the wire path stays BYTE-IDENTICAL: the onData hook (`_predictHookOnData`) is a plain statement with no `return`, so control always falls through into the untouched send path — pinned by vm and E2E byte-identity tests. Predictions reconcile against the parsed buffer and only while the cursor sits on the measured composer row (`isCodexComposerRow`, `/^› /`). Codex also **drops keystrokes that share a PTY read with a bracketed paste**, so flushed text and the paste sequence must go out as separate delayed writes (mirroring the Enter branch's delayed `\r`). Tests: `test/local-echo-codex-gating.test.ts`, `test/codex-predictive-echo.test.ts` (E2E vs real codex), `packages/xterm-zerolag-input/test/codex-replay.test.ts`. ⚠️ **Pi is the opposite kind of CLI and needs the opposite instincts**: it has NO permission prompts and no sandbox, so there is no bypass flag to send and Codeman must not invent one; its privileged knob is the tri-state `approveProjectTrust` (`--approve`/`--no-approve`), which makes pi EXECUTE repo-local `.pi/extensions` TypeScript, so the multi-user clamp puts pi in the **materialize** branch (an absent config still yields `--no-approve` for a non-granted owner) and `--api-key` is never wired. Pi stays OUT of `isAltScreenStripMode()` (main-screen TUI, and its 0.84.0 fullscreen mode is runtime-switchable via `/settings`, where the alt screen is load-bearing), and lands on the `'buffer'` echo policy via the `_updateLocalEchoState` fallthrough. Pi's own tests: `test/pi-mode.test.ts`, `test/routes/external-cli-bypass-clamp.test.ts`; user guide `docs/pi-integration.md`. → [architecture-invariants#external-cli-modes-opencode-codex-gemini-antigravity-pi](docs/architecture-invariants.md#external-cli-modes-opencode-codex-gemini-antigravity-pi)
|
||||
|
||||
**Run launch synchronization**: the Run entrypoint holds an in-flight lock and disables `#runBtn` for the whole launch (≥500ms), so a double click cannot create duplicate sessions with the same `w<n>-<case>` name. `_ensureCreatedSessionVisible()` runs before `selectSession()`, and `_onSessionCreated()` stays an idempotent upsert, so POST-first and SSE-first ordering both produce exactly one rendered tab. → [architecture-invariants#run-launch-synchronization](docs/architecture-invariants.md#run-launch-synchronization)
|
||||
**Run launch synchronization**: the Run entrypoint holds an in-flight lock and disables `#runBtn` for the whole launch (≥500ms), so a double click cannot create duplicate sessions with the same `w<n>-<case>` name. `_ensureCreatedSessionVisible()` runs before `selectSession()`, and `_onSessionCreated()` stays an idempotent upsert, so POST-first and SSE-first ordering both produce exactly one rendered tab. ⚠️ **Closing has the mirror-image race and one owner**: `closeSession()` reads `wasActive` BEFORE its `await` and announces the delete via `_closingSessions`, while `_onSessionDeleted` skips the active-session handoff for an id in that set. Both used to read `activeSessionId` after the fact, so the `session_deleted` broadcast for your own delete could null it first and closing the tab you were on landed on the welcome screen instead of the next session, on the same build, depending on timing. The fallback also picks the first order entry that is still in `sessions` (a dead id can linger in `sessionOrder`, same reason Alt+N indexes a live-filtered list). A delete from ANOTHER client still shows the welcome screen, which is the honest answer when what you were looking at was taken away. Tests: `test/session-close-fallback.test.ts`. → [architecture-invariants#run-launch-synchronization](docs/architecture-invariants.md#run-launch-synchronization)
|
||||
|
||||
**Session lineage lines** (tab → tab it spawned, `sessionLineageLines`, per-device, desktop default ON): a create request may name the session that spawned it, as a `parentSessionId` body field on `POST /api/sessions` / `POST /api/quick-start` or the `X-Codeman-Parent-Session` header (the agent skill sets that once on its shared curl invocation, so every spawn recipe carries it). `resolveParentSessionId()` (route-helpers.ts) **resolves rather than trusts** it: exact id, else a UNIQUE ≥8-char prefix (ids reach agents truncated), it must be a live session the caller can see AND carry the same owner, and **anything unresolvable is DROPPED, never a 400** — a cosmetic field must not be able to fail a worker spawn. It rides `toState()` into `session_created`, so there is no new SSE event. ⚠️ Rendering is an ADDITIONAL LAYER on the existing SVG pass (`_appendLineageConnectionLines` called at the tail of `_updateConnectionLinesImmediate()`, exactly like ultracode), sharing one batched read→write reflow and the `tab:<id>` rect cache; geometry is pure in `computeLineagePath()` (constants.js). ⚠️ **ONE shape, and the second one was the bug**: every pair (flat strip or wrapped) gets a U-bridge hanging below the strip, anchored on both tabs' BOTTOM edges. A wrapped strip used to get a parent-bottom → child-TOP bezier with a ~14px row gap to bend in, which drew a flat line hidden in the gap with siblings overprinting. ⚠️ The dip is a **mis-tuned-in-both-directions corridor** (44px cap = straight thread at strip-wide spans, #285; 104px cap + full row offset = ~106px over-bow into the terminal, 2026-08-15): it now hangs from the **STRIP's bottom edge** (fallback: lower tab bottom), capped at 64px, with NO per-row offsets stacked on top — the strip-bottom baseline is also what keeps a row-1 pair's arc from drawing through row 2's tab labels. ⚠️ **Colors are keyed on the SPAWNING tab, not per child**: every arc leaving one tab is the same color however many workers it spawns, so the strip reads as "these five came from w1, those two came from w2" — per-child coloring gave one tab's own children a different color each, which is the distinction the colors exist to make. A child that spawns in turn is a parent in its own right and gets its own color for the arcs below it, so a chain changes color at each generation while each generation's fan-out stays uniform. Assignment cycles `CodemanLineage.COLORS` in first-seen order per parent id (first entry empty = the skin-tuned `--session-blue`, so the first spawning tab keeps it; the rest vivid fixed hexes), memoized rather than derived from draw index (the SVG is wiped and rebuilt constantly, so an index-based color would flicker), and set inline as `--lineage-color` so styles.css keeps owning opacity/glow/dash. `test/session-lineage-lines.test.ts` drives the real `_appendLineageConnectionLines()` and asserts the painted property, since testing the color function alone would pass just as happily with the child id passed back in. ⚠️ **Desktop only**: the overlay is `z-index: 999` and the desktop header is 100 (arcs paint over it, which is what lets them touch tab bottoms), but under 1024px mobile.css makes the header `fixed; z-index: 1200` and would bury them. ⚠️ Paths carry `data-agent-id="lineage:<childId>"` because that is what `_applyLineEntrances()` queries — that one attribute is what gives them the entrance animation and its negative-`animation-delay` resume across `svg.innerHTML=''`. ⚠️ `.session-tabs` is `overflow-x: auto`, so a scrolled-out tab still HAS a rect (over the logo); edges with an endpoint outside the strip are skipped, and a passive `scroll` listener re-anchors the rest.
|
||||
|
||||
**Unified session list**: `GET /api/sessions/unified` merges live sessions, persisted state, lifecycle-log history, and Claude transcript files into one deduped list (pure core in `src/services/unified-session-service.ts`). Transcript rows fold into their owning session via a `claudeSessionId → Codeman id` alias map, so resumed and `/clear`-respawned sessions do not appear twice. No terminal buffers in the response, unlike `/api/sessions`. Backs the Cmd+K Session Manager, plus pinning and cross-device tab order (`PUT /api/session-order`; pure merge helpers in `src/session-order.ts`, pushing device wins and server-only ids are never dropped). → [architecture-invariants#unified-session-list-and-session-manager](docs/architecture-invariants.md#unified-session-list-and-session-manager)
|
||||
|
||||
**Hook events**: Claude Code hooks trigger via `/api/hook-event`. Key events: `permission_prompt`, `elicitation_dialog`, `idle_prompt`, `stop`, `teammate_idle`, `task_completed`. See `src/hooks-config.ts`; upstream hook semantics mirrored in `docs/claude-code-hooks-reference.md`.
|
||||
**Hook events**: Claude Code hooks trigger via `/api/hook-event`. Key events: `permission_prompt`, `elicitation_dialog`, `elicitation_complete`, `elicitation_response`, `idle_prompt`, `stop`, `teammate_idle`, `task_completed`. See `src/hooks-config.ts`; upstream hook semantics mirrored in `docs/claude-code-hooks-reference.md`. ⚠️ **Every claude session INSTALLS the hooks block into its workspace** (`applyWorkspaceHooks` in hooks-config.ts → `ensureCodemanHooks`, an add-only merge that keeps a user's own handlers), from EVERY claude create path — both interactive routes, cron fires, legacy scheduled runs, the plan-orchestrator one-shots — and from `restoreMuxSessions()` for sessions recovered on server start (that boot sweep skips a workspace that no longer exists, so a deleted repo with a surviving tmux session is never resurrected as an empty dir). Before 2026-08-15 hooks were written ONLY when Codeman created the case DIRECTORY, so a linked case / cloned repo — where most sessions actually run — had no hooks at all and every hook-driven surface was silently dead there: an AskUserQuestion dialog blocked the pane while the tab and the phone overview both read a calm `idle`, with no Approvals Inbox item, no push, no definitive `stop`/`idle_prompt` for respawn and no `stop`/`blocked` for the wait endpoints. The escape hatch is the synced `workspaceHooksEnabled` setting (App Settings → Agents & CLIs → Claude, **default ON**); OFF restores the old behavior, where a Codeman block that is already there is still refreshed when stale (COD-91) but one is never added. ⚠️ Route the decision through `applyWorkspaceHooks` rather than calling `ensureCodemanHooks` at a new site, or the setting silently stops applying to that path. ⚠️ Claude Code RE-READS `settings.local.json`, so an already-running session starts firing hooks without a restart (measured 2026-08-15) — and the notification for a blocking dialog is delayed by Claude Code (~30s), so the alert trails the dialog. ⚠️ An AskUserQuestion / plan-selection dialog arrives as **`permission_prompt`**, not `elicitation_dialog` (that one is MCP elicitation), so it renders as the RED "needs you" alert, not the yellow idle one.
|
||||
|
||||
**Approvals Inbox** (cross-session queue of prompts waiting on a human; `approvalsInboxEnabled`, SYNCED, default OFF: every surface is opt-in; only the store and answer endpoints run regardless, so flipping it ON shows anything already pending): `web/approval-inbox.ts` is a `sessionWaits`-style singleton fed by `/api/hook-event`, holding at most ONE item per session (a new prompt supersedes), claude-mode only, in-memory. Cards are answered via `POST /api/approvals/:id/answer`, which sends a digit / Esc / idle-prompt text through `writeViaMux` (menu answers never carry `\r`). ⚠️ `option` digits are accepted ONLY when they match options parsed from the captured pane frame, and the answer path RE-CAPTURES the pane first (a dialog that no longer parses on screen means the keystroke would land in the composer, so refuse with 409). ⚠️ Resolution on the heuristic `working` signal is restricted to `idle` items; permission/question items clear only on definitive signals (`stop`, `elicitation_complete`/`elicitation_response`, exit/delete, answer, supersede, 12h TTL). ⚠️ **Viewing a session ACKNOWLEDGES its idle item, it does not resolve it** (`POST /api/approvals/session/:sessionId/viewed` → `acknowledgedAt` → `approval:updated`): the item stays pending (still answerable, still Read My Mind context) and only stops arming the yellow tab alert. That flag is what makes the clear durable, since the view-clears-idle rule used to live in one browser's memory and `seedApprovals()` re-armed the alert on the next reload while other devices never heard about it at all; the local half is `markIdleAlertSeen()` (app.js), called from BOTH `selectSession` paths, including the already-active early return, where a click could otherwise never clear the alert. ⚠️ **Only a HUMAN opening a session acknowledges**: `selectSession(id, { auto: true })` marks the three selections the APP makes (boot restore, a solo window opening its target, the fallback after the active session is closed) and skips the acknowledgement, so a page load cannot silently spend an alert the user never saw. The flag defaults to user-initiated, so an untagged call site fails toward acknowledging rather than toward an alert nothing can clear; `test/session-select-ack-gate.test.ts` pins both the gate and the tagged call sites. Idle-only by construction (`acknowledge()` defaults to `['idle']`): looking at a permission/question dialog does not answer it. ⚠️ Same rule on the input path: `_ackDelivery` (app.js) spends the IDLE alert only, via that same `markIdleAlertSeen()`. It used to `clearPendingHooks(sessionId)` with no kind, so one keystroke wiped a RED alert on that device while the dialog was still up, the other devices stayed red, and a reload re-seeded it. ⚠️ Claude Code fires no "permission answered" hook (only `elicitation_complete`/`elicitation_response`, i.e. the question flavor), so an answered-in-the-terminal dialog would otherwise sit pending until `stop`: `GET /api/approvals` therefore runs a **staleness sweep** over the caller's own items via `verifyStillAnswerable()`, which is deliberately the conservative check the answer path uses (only an item whose ORIGINAL frame parsed options can be dropped, so an unreadable capture keeps the alert rather than losing a live one). The frontend seeds from `GET /api/approvals` in `handleInit` **regardless of the setting**: the seed re-arms the tab-alert state machine (`setPendingHook`) unconditionally, and only populating `this.approvals` (the inbox surfaces) is gated — seeding used to be gated wholesale, which left a reloaded page with NO red tab while a permission dialog sat blocking a session (2026-08-15); `_onApprovalResolved` clears the pending-hook alert unconditionally for the same reason. ⚠️ The red/yellow tab alert itself is a STEADY border/background/dot with a pulse on top: the original keyframes swung to transparent at 0%/100%, so half of every cycle looked like a normal tab. Push Approve/Deny buttons stay gated on the setting (`sendPushNotifications` strips `actions`/`approvalId` when OFF) and are answered from `sw.js` directly so they work with no tab open. Surfaces (all gated on the setting): header bell (marker-hidden until count > 0, phones never show it) + drawer (`approvals-ui.js`), phone overview NEEDS YOU answer strips (`mobile-overview.js`). Design: `docs/approvals-inbox-plan.md`.
|
||||
|
||||
**Read My Mind intent profiles** (phase 1 of `docs/readmymind-plan.md`; `readMyMindEnabled`, SYNCED, default OFF): per-CASE profiles (user-stated `goals` + the user's recent real prompts), keyed by owner + realpath(workingDir) so they survive `/clear`/respawns and multi-user scoping is structural. Capture rides the transcript (`transcript:user_prompt` from `transcript-watcher.ts`), NOT the input paths: `POST /input` sees only programmatic prompts and the WS channel is raw keystrokes. The listener lives inside `startTranscriptWatcher()`'s `if (!watcher)` block (outside it would duplicate per hook event) and is claude-only + gated on the setting per event. Store: `src/intent-store.ts` singleton, `intents.json` written 0600 tmp+rename (prompts can contain secrets; never fed to `/api/search`). Endpoints: GET/PUT/DELETE `/api/sessions/:id/intent` + POST `/api/sessions/:id/readmymind` (`readmymind-routes.ts`, ownership via `findSessionOrFail` WITH `req`; registrations stay the bare `app.<method>('path')` shape, the endpoints.md drift scanner cannot see generics). **Phase 2 (predictor + 🧠 button)**: `readmymind-context.ts` is the PURE budgeted assembler (9 ranked sources, drop order siblings→away→workspace→tools, sections 1-4 truncate only); IO lives in `readmymind-collectors.ts` (transcript TAIL read — the live watcher keeps only a 500-char snippet — + git signals, skipped for remote-SSH cases) and the route; `readmymind-predictor.ts` reuses the AiCheckerBase spawn mechanics standalone (verdict-shaped base vs freeform JSON) as a mutable singleton routes call and tests stub. Claude-mode only (400), one in flight per session (409 CONFLICT), model = `readMyMindModel` setting defaulting to `AI_CHECK_MODEL` (opus, decided). Frontend `readmymind-ui.js`: header 🧠 marker-hidden (`btn-readmymind--hidden`) until the setting is ON; phones hide it in mobile.css and get a keyboard-accessory 🧠 key instead (ships in BOTH bar templates, revealed by the `rmm-enabled` class on the BAR element — setMode() rebuilds button innerHTML, so per-key state would be wiped; synced at init + every `applyHeaderVisibilitySettings()`). Alternate suggestions render as tappable rows that swap into the editable field without losing edits; Rethink rejects the whole shown set and carries the optional steer note (`#readMyMindSteer`, sent as `steer`, shown in ready + empty-result phases, cleared on each open). Suggestions render via value/`textContent` ONLY and Send/Insert go through `POST /input` (server-side, so the sendEnterKey/local-echo trap does not apply) — nothing auto-sends, ever. User guide: `docs/readmymind.md`.
|
||||
|
||||
**Voice dictation via Claude** (`claudeVoiceEnabled`, SYNCED, default OFF): the mic button can transcribe through this machine's Claude Code login instead of a Deepgram key, using the same speech-to-text service the CLI's own `/voice` mode uses. ⚠️ **Claude Code's voice mode itself is unusable here**: it opens the HOST's microphone (`sox`/`arecord`), and the CLI runs in a headless tmux pane while the human is in a browser elsewhere. So Codeman captures in the browser and borrows only the backend. Audio goes browser → Codeman → Anthropic (`src/web/voice-stream.ts`): the OAuth token never reaches the page, and the browser only sends PCM and receives text. ⚠️ Credentials are **read-only** (`src/claude-credentials.ts`) and Codeman never refreshes them — a refresh rotates the refresh token and could sign the user out of their own CLI; an elapsed token reports `expired` instead. ⚠️ Capture MUST be linear16/16 kHz/mono, so it uses an **AudioWorklet**, not MediaRecorder (which cannot emit raw PCM); `voice-pcm-worklet.js` is fetched from JS, so it is invisible to `cacheBustAssets` and borrows voice-input.js's `?v=` token — **edit the two together**. ⚠️ Transcript frames carry the WHOLE running transcript, not deltas: the Claude path replaces where the Deepgram path appends. Provider choice is `voiceSettings.provider` (`auto` prefers Claude → Deepgram → Web Speech). → `docs/claude-voice-plan.md`
|
||||
|
||||
**Agent Teams**: `TeamWatcher` polls `~/.claude/teams/`, matches to sessions via `leadSessionId`. Teammates are in-process threads appearing as subagents. Enable: `CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`. See `docs/agent-teams/`.
|
||||
|
||||
@@ -208,19 +223,27 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
**Full-scrollback replay**: `GET /api/sessions/:id/terminal?full=1` returns the entire tmux scrollback, bounded by the configured history limit. On success the capture is returned ALONE (`source='mux-full-history'`), superseding the byte buffer so nothing duplicates. The first load of EACH session per page load requests `full=1` (`_fullHistoryLoaded` Set); tab switches keep the cheap `?tail=` path, and scrolling up at the TOP of the buffer re-pulls `full=1` on demand (cooldown-guarded — tmux repaints bursty output in place, so browser scrollback shrinks while tmux's history stays complete). ⚠️ That re-pull must never DOWNGRADE the buffer: a repaint-mode CLI pane keeps no tmux history, so its capture is one frame and the reset+rewrite would delete history mid-scroll — `_replayWouldShrinkBuffer()` refuses it and slows that session's cooldown to 60s. → [architecture-invariants#full-scrollback-replay](docs/architecture-invariants.md#full-scrollback-replay)
|
||||
|
||||
**Terminal scrollback strip + wheel/touch forwarding** (#205): codex/claude/gemini get the FULL strip (alt-screen, `3J`, mouse DECSETs); tmux-backed shell/opencode/antigravity get a NARROW strip (alt-screen toggles only — it removes tmux's own attach-time `smcup`, which otherwise parks xterm in the scrollback-less alt buffer and turns the wheel into arrow keys). ⚠️ Gated on `useMux`: direct-PTY fallback sessions must keep the alt screen for vim/less/htop. Wheel AND touch forward to the CLI transcript for codex/claude ≥ 2.1.187 at ANY scroll position (snap-to-bottom first); Shift+wheel and the `terminalWheelLocalScrollback` setting stay local. `_wheelScrollLines()` reads `ev.deltaMode` (Firefox = LINE units). ⚠️ When that gate is FALSE on a claude session whose local buffer is hollow (`baseY === 0`), the gesture becomes coalesced PageUp/PageDown key sends (`_maybePageCliTranscript`) instead of a no-op; ⚠️ and `getClaudeCliVersion()` must never cache a FAILED probe (one timeout used to disable forwarding process-wide until restart). `_logScrollRouting()` prints the routing decision and its inputs once per session — read it before diagnosing a scroll report. → [architecture-invariants#terminal-scrollback-strip-flavors-and-wheeltouch-forwarding](docs/architecture-invariants.md#terminal-scrollback-strip-flavors-and-wheeltouch-forwarding)
|
||||
**Terminal scrollback strip + wheel/touch forwarding** (#205): codex/claude/gemini get the FULL strip (alt-screen, `3J`, mouse DECSETs); tmux-backed shell/opencode/antigravity get a NARROW strip (alt-screen toggles only — it removes tmux's own attach-time `smcup`, which otherwise parks xterm in the scrollback-less alt buffer and turns the wheel into arrow keys). ⚠️ Gated on `useMux`: direct-PTY fallback sessions must keep the alt screen for vim/less/htop. Wheel AND touch forward to the CLI transcript for **claude ≥ 2.1.187 ONLY** at ANY scroll position (snap-to-bottom first); Shift+wheel and the `terminalWheelLocalScrollback` setting stay local. ⚠️ Codex was in that list and must never go back without a fresh measurement: codex-cli 0.147.0 ignores SGR wheel reports entirely (`mouse_any_flag=0`, inline viewport, transcript pushed into terminal scrollback), so forwarding produced a dead wheel (#227 follow-up). `_wheelScrollLines()` reads `ev.deltaMode` (Firefox = LINE units). ⚠️ When that gate is FALSE on a claude session whose local buffer is hollow (`baseY === 0`), the gesture becomes coalesced PageUp/PageDown key sends (`_maybePageCliTranscript`) instead of a no-op; ⚠️ and `getClaudeCliVersion()` must never cache a FAILED probe (one timeout used to disable forwarding process-wide until restart). `_logScrollRouting()` prints the routing decision and its inputs once per session — read it before diagnosing a scroll report. → [architecture-invariants#terminal-scrollback-strip-flavors-and-wheeltouch-forwarding](docs/architecture-invariants.md#terminal-scrollback-strip-flavors-and-wheeltouch-forwarding)
|
||||
|
||||
**Self-update** (App Settings → Updates): in-app updater for git-clone installs supervised by systemd/launchd (`systemd`, `launchd`, `launchd-daemon`, else `none` → "restart manually"). The update restarts the very process running it, so the real work runs in a DETACHED `scripts/self-update.sh` that outlives the restart and writes progress to `update-status.json`, which the browser polls across the connection drop. `src/web/self-update.ts` splits pure helpers (unit-tested) from IO wrappers. npm installs report as non-updatable. → [architecture-invariants#self-update](docs/architecture-invariants.md#self-update)
|
||||
**Detached start + service install** (issue #231): `codeman web -d` relaunches the SAME entry script with `detached:true` (setsid), so there is no controlling terminal and no shell job entry. ⚠️ `nohup` is NOT what makes this work: Node re-arms SIGHUP to its default disposition even when it inherits "ignore", and `cli.ts` handles SIGHUP with a graceful shutdown, so a delivered HUP still stops the server. ⚠️ Both `-d` and `service install` must REFUSE when a server is already up on this data dir (pidfile check + `/api/status` probe): a second instance on the shared tmux socket attaches PTYs to the first one's live sessions. ⚠️ Neither may report success it has not observed — the parent polls `/api/status` until the child answers or dies, since `launchctl load` and a clean spawn are both silent about a server that starts and immediately exits. `--stop` verifies the pid still LOOKS like a Codeman server (`ps -o command=`) before signalling, because pids get recycled. Unit/label names live in `config/service-names.ts` so install.sh, `detectSupervisor()` and `service install` cannot drift into supervising two copies; they are instance-scoped, and identical to the historical names for the default instance. `service install` bakes the installing shell's PATH into the unit (launchd gives a job `/usr/bin:/bin:/usr/sbin:/sbin`, which finds neither a Homebrew/nvm `node` nor `tmux`/`claude`) and never writes `CODEMAN_PASSWORD` into it. → [architecture-invariants#detached-start-and-service-install](docs/architecture-invariants.md#detached-start-and-service-install)
|
||||
|
||||
**Self-update** (App Settings → System → Updates): in-app updater for git-clone installs supervised by systemd/launchd (`systemd`, `launchd`, `launchd-daemon`, else `none` → "restart manually"). The update restarts the very process running it, so the real work runs in a DETACHED `scripts/self-update.sh` that outlives the restart and writes progress to `update-status.json`, which the browser polls across the connection drop. `src/web/self-update.ts` splits pure helpers (unit-tested) from IO wrappers. npm installs report as non-updatable. → [architecture-invariants#self-update](docs/architecture-invariants.md#self-update)
|
||||
|
||||
**Attachments** (live external document references; all wiring in `file-routes.ts`): a **registry** maps a stable `attachmentId` to a realpath-resolved, extension-allowlisted absolute path, so browser requests never carry arbitrary absolute paths. ⚠️ The **magic-link scanner** (`codeman://attach?...` in terminal output) is **prompt-injectable**, so its scan path is force-confined to the session workspace; a hostile prompt could otherwise exfiltrate arbitrary host files over SSE. The security gate is an extension **allowlist**, not a blocklist. `document-conversion-limiter.ts` caps converter spawns globally: without it, N large docs detected at once fork N multi-minute processes, which is a resource-exhaustion vector. → [architecture-invariants#attachments](docs/architecture-invariants.md#attachments)
|
||||
|
||||
**File-path links (terminal + chat)**: a path an agent prints is clickable on BOTH surfaces and opens the file-preview overlay. ⚠️ ONE pattern (`FILE_PATH_LINK_PATTERN` / `absoluteFilePathPattern()` in constants.js) feeds the xterm link provider AND the response viewer's `_linkifyFilePaths()`; a fresh instance per call, since `lastIndex` is per-object state. The chat linkifier walks TEXT NODES with DOM APIs (the source is model output; never rebuild sanitized markup as a string) and skips subtrees already inside an `<a>`. ⚠️ **An out-of-workspace path is served through the ATTACHMENT routes, not the file routes** — `file-content`/`file-raw` are workspace-confined and 404 exactly the paths agents print most (a `/tmp` capture, Claude's scratchpad), so `openFilePreview()` registers such a path via `POST /api/sessions/:id/attachments` with **`notify: false`** (suppresses only the `attachment:detected` broadcast — same guard, same routes; without it every click also popped a card announcing the file already on screen) and renders by id. The click is an explicit action on the explicit, Origin-guarded route, which is what distinguishes it from the force-confined magic-link scanner. ⚠️ **Media extensions are single-sourced** (`VIDEO_ATTACHMENT_EXTENSIONS`/`AUDIO_ATTACHMENT_EXTENSIONS` in `attachment-registry.ts`, imported by `file-content`'s classification) so a clip plays the same in or out of the workspace; a player needs all THREE of allowlist + a real `MIME_TYPES` entry (octet-stream renders a dead player) + the range-aware body. ⚠️ **`TEXT_ATTACHMENT_EXTENSIONS` IS `EDITABLE_EXTENSIONS`** (never a second list): if the viewer would edit it inside the workspace, it can be read outside. Widening READ must never widen RUN, so `html`/`htm` joined `svg` in `serveRawFile`'s download-only branch, other text goes out as inert `text/plain`+`nosniff`, and `~/.codeman*/state.json` joined `isSensitivePath` (it persists `envOverrides`, which can hold `GEMINI_API_KEY`). ⚠️ The terminal sends an **out-of-workspace** path to the preview instead of the log viewer (that one spawns `tail -f` and reaches only workspace + `/var/log` + `~/logs`); in-workspace text keeps the tail viewer and `file-stream-manager`'s allowlist is untouched. The image-watcher keeps its own narrow detection list, so none of this cards every file an agent writes. → [architecture-invariants#file-path-links-terminal--response-viewer](docs/architecture-invariants.md#file-path-links-terminal--response-viewer)
|
||||
|
||||
**Filesystem path picker** (Link Existing "Browse" + the mobile keyboard's `📁 Path` key): lazy one-directory browsing via `GET /api/filesystem/browse`, with `GET /api/filesystem/preview` for the tapped file. Inserts the path **without** Enter, so the prompt is never submitted; the sibling `⌫ All` key clears only the unsent prompt and must never send the agent's `/clear`. ⚠️ This is a **second file-serving surface and inherits neither the attachment confinement nor its ownership scoping** — it allowlists Home, `CASES_DIR`, `/mnt/d` and `CODEMAN_FILE_PICKER_ROOTS`, blocks sensitive trees, and rejects symlink escapes **after** `realpath`. ⚠️ The optional `sessionId` is an ownership boundary that must be `canAccessOwned`-checked by hand (it does not go through `findSessionOrFail`), and in multi-user mode a non-admin gets only their own `userSpacePath` as a root: per-user spaces live INSIDE `homedir()`, so a `Home` root exposes every other user's workspace. Previews go through the same global conversion limiter, and Markdown/TXT/JSON are served as inert `text/plain`. → [architecture-invariants#filesystem-path-picker](docs/architecture-invariants.md#filesystem-path-picker)
|
||||
|
||||
**File Viewer edit mode** (issue #212): the file-preview overlay edits workspace text files in place — `GET .../file-content?edit=1` + `PUT /api/sessions/:id/file-content`, policy in `src/config/file-editing.ts`. This is a **third file surface and the only one that WRITES**: read-path confinement (realpath + workspace + ownership) plus sensitive/blocked/`.git` denies and an extension **allowlist**; writes are `wx`-temp + rename (no `O_CREAT` anywhere = edit-in-place is structural); optimistic concurrency via sha256 `baseHash` → 409. ⚠️ `edit=1` never truncates and the client must never save a plain-preview buffer (the 500-line truncation would silently delete the rest). ⚠️ CRLF/UTF-8 guards: EOL re-applied server-side, non-UTF-8 refused via round-trip compare. → [architecture-invariants#file-viewer-edit-mode](docs/architecture-invariants.md#file-viewer-edit-mode), `docs/file-viewer-edit-plan.md`
|
||||
|
||||
**Raw file bodies are streamed and range-aware**: `file-raw` and the attachments `/raw` route always advertise `Accept-Ranges: bytes` and answer a `Range` header with `206` + `Content-Range` (single-range only; parser is pure + unit-tested in `src/web/http-range.ts`, a malformed spec is ignored → 200 while an out-of-bounds one is a 416). ⚠️ A 200-only response is what made the File Viewer's `<video>` unseekable: Chrome then reports `video.seekable` as `[0, 0]`, the scrub bar is inert and `currentTime = x` silently reverts (measured on an 18MB mp4), and Safari refuses to start the media at all. ⚠️ These bodies go out through `reply.hijack()`, which bypasses Fastify's status handling — `sendRawStream` must copy the status onto `reply.raw` by hand or a partial body ships labelled `200` and the browser treats a slice as the whole file. ⚠️ Closing the preview must **pause and unload** the media (`_stopFilePreviewMedia` in panels-ui.js): dropping the overlay's `visible` class is `display:none` and nothing else, and a DETACHED `HTMLMediaElement` keeps playing, which is how the X button used to leave a video audible with no player to pause.
|
||||
|
||||
**Ultracode / workflow-run visualization** (opt-in, default OFF): the Workflow tool writes a completion artifact only at run *end*, so live in-flight runs exist solely as transcript dirs. `workflow-run-watcher.ts` therefore synthesizes ACTIVE runs from transcripts until the completion artifact appears and supersedes them. It is **STANDALONE** and deliberately never imports or touches `subagent-watcher.ts`, despite reading the same tree. Two independent toggles: `showUltracodeAgents` (docked panel) and `ultracodeFloatingWindows` (floating windows); the watcher starts if **either** is on. → [architecture-invariants#ultracode--workflow-run-visualization](docs/architecture-invariants.md#ultracode-and-workflow-run-visualization)
|
||||
|
||||
**Cross-session search**: `GET /api/search` federates an in-memory search over session metadata, run-summary events, and attachment-history entries. The pure core `searchSources()` does substring matching with hard per-type caps: **no regex (so no ReDoS) and no filesystem reads (so no traversal)**. The server-private `externalPath` is never read. → [architecture-invariants#cross-session-search](docs/architecture-invariants.md#cross-session-search)
|
||||
**Clone a repository as a case** (issue #236, Add Case → **Clone Repo**): `POST /api/cases/clone` clones a public repo into the caller's case space synchronously (request held open, bounded by `GIT_CLONE_TIMEOUT_MS`, no job store); `POST /api/cases/clone-preflight` reports whether the URL can be cloned anonymously plus its real branches/tags. Core in `src/git-clone.ts`. ⚠️ **The URL is a code-execution surface**: `ext::sh -c <cmd>` (and ANY `<name>::<payload>` helper) makes git run a command, so every `::` form is refused, a leading `-` is refused, and every spawn is an argv array with `--` before the operands. ⚠️ **Non-interactive or the open request hangs** — `gitNonInteractiveEnv()` closes the terminal/askpass/ssh/GCM prompt paths; `HOME`/`PATH` stay inherited, so a user's OWN credential helper may authenticate (Codeman still never collects or stores credentials, and refuses a `user:password@` URL). ⚠️ Timeout kills the process GROUP (clone fans out into child processes), the destination is removed only if this attempt created it, and repository contents win over scaffolding (existing `CLAUDE.md` kept, hooks merged, repo-shipped `.claude/settings*` reported as a warning since its hooks run locally). The **Brain** picker sets the toolbar run mode on success. → [architecture-invariants#clone-a-repository-as-a-case](docs/architecture-invariants.md#clone-a-repository-as-a-case)
|
||||
|
||||
**Cross-session search**: `GET /api/search` federates an in-memory search over session metadata, run-summary events, and attachment-history entries. The pure core `searchSources()` does substring matching with hard per-type caps: **no regex (so no ReDoS) and no filesystem reads (so no traversal)**. The server-private `externalPath` is never read. PAST sessions (#261) come from `session-history-index.ts`, a capped snapshot of the unified list filled **outside** the request path (`/api/sessions/unified` publishes it; a stale one is rebuilt fire-and-forget), that indirection is what keeps the no-fs property. ⚠️ The snapshot is stored UNSCOPED with a per-row owner and MUST be re-filtered through `canAccessOwned()` on read; history rows carry `jumpTo.kind:'resume-session'`, since a closed session has no tab to select. → [architecture-invariants#cross-session-search](docs/architecture-invariants.md#cross-session-search)
|
||||
|
||||
**Web tabs** (dashboard URLs as tabs): a saved URL renders as a tab beside agent sessions. **NOT a sixth `SessionMode`** (no PTY, no tmux, no respawn), same reasoning that keeps Docker/remote-SSH as case overlays. Dashboards are **proxied through Codeman's own origin** by default, because a direct iframe fails three ways at once: prod is HTTPS so `http://` targets are blocked as mixed content, many dashboards send `X-Frame-Options: DENY`, and our own `default-src 'self'` CSP blocks cross-origin frames. Proxying leaves the prod CSP unchanged (`/webview/...` is `'self'`). ⚠️ The proxy is **NOT an API surface**: it authenticates on an in-memory capability in the path and is correspondingly exempt from the cookie + Origin checks; that exemption is fenced to safe methods and non-`/api` paths and is pinned by `test/webview-auth-exemption.test.ts`. ⚠️ Iframes omit `allow-same-origin` unless a dashboard is explicitly marked `trusted`, and `Authorization`/`codeman_session` are stripped upstream in **both** modes so `CODEMAN_PASSWORD` cannot leak. ⚠️ A sandboxed frame is **opaque-origin**, which breaks two things `curl` can never reproduce: its runtime-built root-absolute URLs escape `<base>` (fixed by an injected `runtimeUrlShim()`), and its same-host `fetch`/XHR are CORS-checked with `Origin: null` (fixed by `buildProxyCorsHeaders()` plus exempting the proxy from the global `OPTIONS`-204 short-circuit in `registerSecurityHeaders`). Both present as the dashboard's own "Failed to fetch" while the page renders fine. → [architecture-invariants#web-tabs](docs/architecture-invariants.md#web-tabs), `docs/web-tabs.md`
|
||||
|
||||
@@ -234,16 +257,28 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
### Frontend
|
||||
|
||||
Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `sanitize-html.js`(5.6) → `app.js`(6) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `ultracode-panel.js`(11.5) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData).
|
||||
Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `sanitize-html.js`(5.6) → `app.js`(6) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `readmymind-ui.js`(11.3) → `ultracode-panel.js`(11.5) → `approvals-ui.js`(11.6) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `home-sessions.js`(12.56) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `session-lineage.js`(15.6) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData).
|
||||
|
||||
**Entrance animations** (`entrance-animations.js`, all OFF by default): opt-in animations for the four things that appear when work starts, chosen per surface via `data-tab-anim` / `data-term-anim` / `data-win-anim` / `data-line-anim` on `<html>`. Defaults are the `legacy` theme, so an untouched install behaves exactly as before and every hook short-circuits on its first line. ⚠️ Tabs and connection lines are **destroyed mid-animation** on every re-render (`_fullRenderSessionTabs()` replaces the strip's innerHTML; `_updateConnectionLinesImmediate()` does `svg.innerHTML = ''`), so both are tracked by id and re-applied to the fresh element with a **negative `animation-delay`** to resume rather than restart. ⚠️ The terminal-pane styles may animate **transform / opacity / clip-path only**, xterm's FitAddon derives rows+cols from `getComputedStyle(parent).width/height`, so animating width/height/padding there would resize the PTY. ⚠️ Window styles other than `beam` transform the window, which moves the rect its connection line is aimed at; `beam` deliberately animates opacity/filter only so its line can draw toward a stable target. Persisted to its own `codeman:*Anim` localStorage keys (per-device, deliberately NOT in the `.strict()` `SettingsUpdateSchema`); picker in App Settings → Appearance, full per-surface lab at `?animlab=1`.
|
||||
|
||||
**Mobile tab strip scrolling** (issue #257): under 768px the tab strip is a horizontal scroller (desktop wraps to a second row instead), so the active tab can sit off-screen. Three rules keep it reachable and they only work together: `_updateActiveTabImmediate()` scrolls the selected tab into view via `computeTabScrollLeft()` (pure, in constants.js) using **rect math on the strip's own `scrollLeft`**, never `scrollIntoView()`, which would also scroll the document under a fixed header; `_fullRenderSessionTabs()` **restores `scrollLeft`** across the `innerHTML` rebuild, since ambient rebuilds (a task badge appearing, a session created elsewhere) otherwise snap a mid-swipe strip back to 0; and it re-reveals the active tab **only when it changed** (`_lastRenderedActiveTabId`), so browsing the far end of the strip is not undone by background renders. ⚠️ **The ACTIVE tab is the only one with action icons, and on a phone they can eat it**: `.session-tab.active .tab-name` reserves `min-width: 44px` in the ≤430px block, because a short session name rendered a 13px label against a 50px gear+close cluster, putting the tab's geometric CENTRE on the gear, so a thumb aiming at the tab opened Session Options instead of switching (measured at 360/393/430px; only long names cleared it). ⚠️ **The floor is set by the 10th tab onward, not by the tabs you can see**: `.tab-number` renders only for `_tabIdx < 9`, so tab 10 loses 16px + a gap off its left and its centre sits 10px further right. The centre clears the icons when `reserved > icons + rightEdge - leftRunUp - gap` (= 50 + 9 - 17 - 4 = **38px**), hit-testing snaps to whole pixels so 39px still lands on the gear, and the practical floor is 40px — a NUMBERED tab clears it at 20px, which is exactly why reasoning from the tabs on screen would put the centre back on the gear. `test/mobile-tab-tap-zones.test.ts` recomputes that inequality from the stylesheet, so widening the gear or the padding fails there rather than on a phone. The guarantee is centre-off-the-ICONS, not centre-inside-the-label (on a numberless tab it lands in the gap between them, which still switches). Non-active tabs keep their icons hidden and stay tappable end to end. ⚠️ Mobile no longer hoists the active session to the front of the strip: that reordering ran on full renders only, so tab order flipped depending on which render path fired, and it renumbered the Alt+N badges. Scroll-into-view replaces it; do not reintroduce it.
|
||||
|
||||
**Session list layout: header strip or left sidebar** (`sessionListLayout`, App Settings → Appearance → Tabs, default `header`; per-device policy — it IS in `SettingsUpdateSchema` and persists server-side, but `displayKeys` makes a device keep its own value): with many sessions the horizontal strip stops being scannable, so the list can move into a vertical `<aside>` with a filter box and a live count, collapsible to a 44px rail (`--sidebar-width` 260 / `--sidebar-width-collapsed` 44) via **Alt+B** (`toggleSessionSidebar`; Alt, not Ctrl+B, which must reach tmux/readline in the terminal). ⚠️ **There is ONE `#sessionTabs` element and it is MOVED between two hosts** (`#sessionTabsHost` in the header, `#sessionSidebarList` in the aside), never a second list — so every render path, drag-reorder handler and Alt+N index keeps working unchanged, and `applySessionListLayout()` is the only thing that reparents it. ⚠️ It sets `data-session-list` / `data-sidebar` on `<html>` and must run BEFORE `applyTabWrapSettings()`, which is the one owner of `tabs-two-rows`/`tabs-show-folder` and reads those attributes. ⚠️ Leaving sidebar mode **clears `_sidebarFilter`**: the filter box only exists in the aside, so a stale filter would hide sessions from the header strip with no reachable control to clear it. ⚠️ On handhelds the aside is an off-canvas overlay rather than a docked rail, and a closed drawer keeps `display: flex`, so it is marked `inert` + `aria-hidden` (`_isSessionSidebarOverlay()`) or its filter box and ~4 tab stops per session stay in the tab order; the DOCKED desktop rail must never be inerted, its rows are still clickable. The desktop home rail (`home-sessions.js`) defers to it, since both dock the session list flush left.
|
||||
|
||||
**Phone overview home screen** (`mobile-overview.js`, phones only, per-device `mobileOverviewEnabled`, default ON): under 430px the "C" logo shows a session overview (NEEDS YOU / CURRENT SESSIONS / PAST SESSIONS) instead of the welcome overlay; tablet and desktop are unchanged. The branch lives in `showWelcome()`/`hideWelcome()` (terminal-ui.js) behind `shouldUseMobileOverview()`, which is **width-driven** (`getDeviceType() === 'mobile'`) because this is a layout decision, unlike the settings namespace which stays handheld-based. ⚠️ The container ships with the `hidden` attribute and only this module removes it: never give `.mobile-overview` a bare `display` rule, since desktop does not load `mobile.css` (`media="(max-width: 1023px)"`) and would then render it unstyled. Live re-renders ride on the tail of `_renderSessionTabsImmediate()` (every state change it needs already funnels there); PAST rows come from one `_fetchUnifiedSessions(60)` per home-screen visit and resume through the shared `resumeHistorySession()`, so they behave exactly like the welcome screen's Resume list. ⚠️ Two things must stay in lockstep with surfaces outside this module, because divergence reads as a bug rather than a style: the split Run button carries the **toolbar's own classes** (`btn-toolbar btn-run mode-<backend>` / `btn-run-gear`) so the per-backend gradient and the light-skin overrides apply unchanged (mobile.css must therefore set no `background`/`color` on it), and row status uses the **session-tab language** (green dot when fine, `pulse` while working, yellow blinking row when waiting for input, red blinking row when a question is pending, mirroring `tab-alert-idle`/`tab-alert-action`). The picker mirrors the toolbar run-mode menu (`setRunMode()` + `run()`, `openWebviewFromMenu()` for saved dashboards) and deliberately omits its Recent-Sessions block, since PAST SESSIONS is that. Status pills carry `data-i18n-skip` (generic words like "idle" collide with state strings elsewhere).
|
||||
|
||||
**Desktop home tab rail** (`home-sessions.js`, desktop only): the welcome overlay centers ~560px of content in a ~1400px window, so its left gutter is dead space; it carries the open tabs as a rail **docked flush to the left edge, full height** (a vertically centered card floating mid-gutter read as debris). Rows are in **overview order** (see below), and each carries a **created** stamp plus the **state duration** the order is computed from (`created 3d ago · working 12m`, word and anchor from `_mobileOverviewSince()` so both home screens say the same thing). A rail sorted by a number it does not show reads as arbitrarily shuffled, and a working row's plain last-active stamp always says "just now". ⚠️ The number badge is the **Alt+1..9 index**, i.e. the position in the TAB STRIP, so on a sorted rail it deliberately does NOT run 1,2,3 downward: it names a shortcut, not a row position, and renumbering it to look tidy would make every badge lie. State classification is REUSED from mobile-overview.js (`_mobileOverviewState`/`_mobileOverviewCaseFor`), which is why the module loads after it. ⚠️ The rail is `position: absolute` so the centered content never moves, which is exactly why it needs a **width gate in two places** — `HOME_SESSIONS_MIN_WIDTH` (1180) in the JS plus a `max-width: 1179px` media query as the backstop for a resize that outruns the matchMedia listener; drift between them means a rail overlapping the search panel, and `test/home-sessions.test.ts` pins them equal. ⚠️ `.home-sessions` is `display: flex`, so `[hidden]` must be re-asserted as `display: none` or the module's only visibility lever does nothing. ⚠️ Size scales with the viewport off **one knob**: `width: clamp(250px, 19vw, 430px)` plus a fluid `font-size` on `.home-sessions`, with every child sized in `em` — reintroducing `rem`/px type inside the block silently breaks the scaling, and widening the clamp past the gutter reintroduces the overlap the gate exists to prevent. The age stamps are refreshed **in place** by a 20s clock (`_tickHomeSessionsTimes()`, disarmed in `hideHomeSessions()`), never by re-rendering, which would restart every row's blink and working ring. Working state is deliberately byte-identical to the phone's: pulsing green dot + the `tab-load-spin` ring reused from the tab strip + the same green halo (added to `.mobile-overview-dot--working` at the same time), so "working" reads the same on every surface; **idle** is deliberately NOT that green — dot and pill mix toward `--text-muted` so a glance separates running from sitting. Live re-renders ride the tail of `_renderSessionTabsImmediate()` alongside the phone overview.
|
||||
|
||||
**Home-screen session order** (`CodemanSessionOrder` in constants.js, pure + unit-tested in `test/session-overview-order.test.ts`): BOTH home screens (phone overview and desktop rail) order rows through this ONE comparator, because they list the same sessions and must answer "which of these wants me next?" the same way. Rank is `needs` → `error` → `waiting` → `working` → `idle` → `done`, and ⚠️ **the tiebreak flips direction halfway down**: states a session is still IN sort **oldest-first** (blocked longest / running longest = most urgent), states it has STOPPED in sort **newest-first** (the session that just went quiet is the one you came back for). ⚠️ The running group keys off **`lastSubmitAt`** (the pane's last Enter), never `lastActivityAt`: a working Claude pane repaints about once a second, so its last-activity stamp is always "now" and would rank every running turn as freshly started. A working pane with no submit stamp falls back to last activity, which lands it at the SHORT end of the group rather than falsely leading it. ⚠️ A **0 stamp means "unknown", not "the epoch"**, and it sorts last within its state either way, or a brand-new session would head every oldest-first group. Final tiebreak is the user's tab order (`orderIndex`), so the list is deterministic and cannot shuffle between renders. The tab strip itself is NOT sorted by this; it stays user-ordered and drag-reorderable.
|
||||
|
||||
**Welcome "Resume Conversation" list** (terminal-ui.js): `loadHistorySessions()` fetches once and caches the corpus on `_historyAll`/`_historyCases`; every subsequent view (filter box, sort select, expand, the periodic refresh in panels-ui.js) goes through `_renderHistoryList()`, so never append rows to `#historyList` directly or re-fetch to re-sort. ⚠️ The box height is **class-driven**: expanding the list without `.history-list.expanded` leaves the collapsed `max-height` in place and just deepens a scroll well, which is the bug #260 reported (35 sessions in a ~4-row box). ⚠️ The A–Z sort keys off `_historyRowLabel()`, the SAME string the row renders (`name || firstPrompt || path`), most rows are transcript-backed and have no session name, so sorting on `name` alone silently does nothing. ⚠️ A filter implies expansion, and `_renderSearch()` hides `#historyHeader` (title + controls) as one unit while a search is active. Tests: `test/history-list-controls.test.ts`.
|
||||
|
||||
**Command palette + shortcut registry**: `Ctrl/Cmd/Alt+K` opens the session palette; shortcuts live in a rebindable registry (`DEFAULT_SHORTCUTS`/`getShortcutRegistry()`/`matchesShortcutEvent()` in app.js, overrides in `settings.shortcutOverrides`). ⚠️ Palette-chord keys must ALSO be swallowed in `attachCustomKeyEventHandler` (terminal-ui.js) or xterm writes the control byte (0x0B) into the PTY. ⚠️ `saveAppSettings()` rebuilds settings from the DOM, so keys edited elsewhere (`shortcutOverrides`, `showTokenCount`, `showCost`) need explicit `_prev` carry-over. ⚠️ **Smart copy (`Ctrl+C`)** lives in that same handler: with a selection it copies, with none it must `return true` **without** `preventDefault()` or the interrupt is lost. `copyTerminalSelection` is deliberately absent from `SHORTCUT_ACTIONS` because the generic capture loop preventDefaults every match it dispatches. → [architecture-invariants#command-palette-and-shortcut-registry](docs/architecture-invariants.md#command-palette-and-shortcut-registry)
|
||||
|
||||
**Per-device vs synced settings**: the `displayKeys` set in settings-ui.js is a **client-side merge policy**, not a wire filter. A display key seeds from the server only when localStorage has no value for it, which is what prevents one device overwriting another; `showPlanUsageLimits` is additionally `delete`d from the incoming payload outright. Separately, `SettingsUpdateSchema` is `.strict()` and simply **does not declare** `skin`, `showFileViewerButton`, `showCronButton`, `webglRendererEnabled`, `localEchoEnabled`, `cjkInputEnabled`, or `extendedKeyboardBar`, so sending one of those is a validation error. The rest (`showResponseViewer`, `showPlanUsageLimits`, `language`, and most `show*` keys) ARE in the schema and do persist server-side; they are per-device by client policy only. ⚠️ Adding a new per-device setting means deciding **both** questions: membership in `displayKeys`, and presence in the schema.
|
||||
|
||||
**Settings surface** (`#appSettingsModal` + `#sessionOptionsModal` + `#createCaseModal`): the `set-*` language (left rail, groups of rows, control pinned right) is shared by all three modals through ONE `:is(#appSettingsModal, #sessionOptionsModal, #createCaseModal)` scope in styles.css: an `:is()` list takes its most specific argument's specificity, so every rule keeps the id weight it had and nothing downstream shifts. **App Settings** is a rail that is a **table of contents over ONE scrolling document**, not a tab switcher: every section stays mounted (`.set-section`, ids `settings-updates|terminal|layout|appearance|models|clis|notifications|voice|shortcuts|system`, in that order, the version and the updater leading and the rest of the system settings tailing), and `switchSettingsTab(id)` keeps its historical name but SCROLLS instead of hiding. **Session Options** and **Add Case** use the same surface with a rail that really SWITCHES (`switchOptionsTab` / `switchCaseModalTab` show one `.set-section` and `.hidden` the rest, since Summary owns its own scroller, Respawn is long, and Add Case is six independent forms). ⚠️ They also take a deliberate **size-up** that App Settings does not (900px shell, 236px rail, `height:auto` between `min(560px,80vh)` and 88vh, vs App Settings' tight 760×620): they are short task panels, not a document you scan, and at scanning density they read as a few fields marooned in an empty frame. Those per-modal blocks are the design, not drift. Phones (≤860px) give App Settings the sticky `#appSettingsJump` pill and give the other two a horizontal rail strip, which neither has a pill for. ⚠️ The Session Options rail entry labelled **Session** still keys off `context` (`data-tab="context"`, `#context-tab`, `switchOptionsTab('context')`), the rename is label-only. Add Case keeps its legacy `.form-row` markup (six panels of it, every id read back by session-ui.js) and is mapped onto the look by an adapter block scoped to `#createCaseModal .set-doc`. Do not restructure those forms just to reach the row classes. ⚠️ That adapter's `summary { display:flex }` **kills the native disclosure triangle**, so every `<details>` there needs the explicit `.set-adv-chev` and both marker suppressions (`list-style` + `::-webkit-details-marker`); without it five collapsed blocks render as plain headings nobody clicks. ⚠️ **The load/save contract is `getElementById` by id**: `openAppSettings()`/`saveAppSettings()`/`openSessionOptions()` read every control by a fixed id, so moving a control between sections is free but renaming or dropping one silently stops it loading or saving. Static guards: `test/app-settings-structure.test.ts` + `test/session-options-structure.test.ts` (rail↔section pairing, one-visible-section, the `data-claude-only` entries external CLIs drop). ⚠️ Model cards (`#appSettingsModelCards`) and the effort segment are **views over hidden `<select>`s** that remain the source of truth; the cards hold the BASE model and the "1M context window" switch composes `base + [1m]` back into `claudeModel`, which is what retires the old "takes precedence over the toggle below" trap. ⚠️ `.modal-tabs`/`.modal-tab-btn`/`.modal-tab-content` are RETIRED: no modal uses them and their CSS is deleted, and a reappearance means a modal drifted off the shared surface. ⚠️ The **Header & Panels live preview** is a scale model rebuilt from the chips (`_syncLayoutPreview`); it owns NO icons, it CLONES `.set-chip-ico` out of the chip, so each icon has exactly one copy in index.html. A chip joins it via `data-preview` (slot) + `data-preview-order`, or `data-preview-text` for readouts that are not buttons. Its frame is painted from skin tokens only (hardcoded black alphas turned it into a grey slab on the light skins) and is `data-i18n-skip`. ⚠️ In Session Options → Respawn, auto-resume is a `.set-callout` whose `<label>` **wraps its own switch with no `for=`** (nesting associates them; the label+`for` pair has historically double-fired), and the cycle steps are real checkboxes (`.set-checks`), not chips. ⚠️ `admin-ui.js` injects the multi-user Users entry into `.set-rail-items` + `.set-doc`, so those hooks must survive any restructure. → [architecture-invariants#settings-surface-app-settings-session-options-add-case](docs/architecture-invariants.md#settings-surface-app-settings-session-options-add-case)
|
||||
|
||||
**Header button visibility**: most header controls are opt-in and hidden by a marker class (`btn-multimonitor--hidden`, `btn-response-viewer-header--hidden`, `btn-file-viewer--hidden`, `btn-cron--hidden`) that `applyHeaderVisibilitySettings()` (settings-ui.js) toggles after settings load; the multi-monitor button is instead stripped at render by `renderIndexHtml`. ⚠️ Hiding must go through the marker class: the base rules are `display:inline-flex !important`, so an inline style cannot override them. Current desktop default is WS/CPU/MEM + File Viewer + gear, with the token chip and lifecycle-log button OFF. ⚠️ New header controls must not leak onto phones; `test/mobile-header-buttons-policy.test.ts` is the static guard. → [architecture-invariants#header-button-visibility-multi-monitor-response-viewer-file-viewer-cron](docs/architecture-invariants.md#header-button-visibility-multi-monitor-response-viewer-file-viewer-cron)
|
||||
|
||||
**Gesture control** (camera hand-tracking overlay, opt-in, default OFF): `CODEMAN_GESTURE=1` makes the feature *available*; `gestureControlEnabled` turns it on. The bundle is injected by `renderIndexHtml` only when enabled, which is why that method is `async` and reads settings with `readSettings(true)` (a fresh read: a post-save reload lands inside the 2s cache TTL and would otherwise render the pre-toggle state). **Source lives in `packages/gesture-control/`; edit there, run `npm run build:gesture`, and commit the regenerated bundle** because dev serves the committed bundle with no runtime bundler. The MediaPipe wasm + model are fetched separately and gitignored. ⚠️ Keep `MP_VERSION` in `fetch-gesture-assets.mjs` in sync with `@mediapipe/tasks-vision`. → [architecture-invariants#gesture-control-the-source-package](docs/architecture-invariants.md#gesture-control-the-source-package)
|
||||
@@ -254,13 +289,21 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L
|
||||
|
||||
**WebGL renderer toggle** (`webglRendererEnabled`, per-device): the GPU-stall watchdog's sticky `codeman-webgl-disabled` marker survives page loads and is cleared only by an explicit OFF→ON save or `?webgl=force`. `?nowebgl` forces the DOM renderer per-load. → [architecture-invariants#webgl-renderer-toggle](docs/architecture-invariants.md#webgl-renderer-toggle)
|
||||
|
||||
**Shell keyboard accessory bar + one-shot Ctrl** (issue #262, `keyboard-accessory.js`): a **shell**-mode session automatically swaps the mobile accessory bar for terminal controls (Ctrl, Esc, Tab, four arrows, paste, dismiss); every other mode keeps the agent bar. `setMode()` now records the user's `extendedKeyboardBar` preference as the **base** layout and `refreshForActiveSession()` (called from `selectSession`) resolves base-vs-shell, so a settings save during a shell session cannot yank the bar away and switching back restores the user's choice. ⚠️ **Ctrl is a ONE-SHOT modifier applied in `terminal.onData`, not in a keydown handler**: a virtual keyboard emits no usable key events, so the character only exists as onData text. The hook sits AFTER `shouldSuppressTerminalQueryResponse` (xterm answers DA/CPR through onData too, and one of those would silently spend the modifier) and BEFORE every send path, so the control byte follows the normal control-char route. ⚠️ **Not every onData chunk is a keystroke**, and the query filter is not enough on its own: xterm ALSO emits mouse and focus reports on its own initiative, so the hook skips them via `isTerminalFocusOrMouseReport()` (they still reach the PTY, they just don't count as the next key). The mouse half is live — a shell session keeps the NARROW strip, so mouse DECSETs reach the browser and one tap while vim/htop runs spent the armed modifier silently (measured). The focus half is defense in depth: `FOCUS_ESCAPE_FILTER` in `session.ts` strips `\x1b[?1004h` from every PTY read, so `sendFocusMode` never turns on today; if it ever did, the bar's own post-key refocus would emit `\x1b[I` and eat the modifier before the user typed. ⚠️ It must disarm on ALL of: use, second tap, any other accessory key, session switch, keyboard dismissal, and a layout swap; a modifier left armed turns the next innocent keystroke into a control byte. ⚠️ **onData is not the only input path** — with `cjkInputEnabled` on, the CJK textarea owns the keyboard (onData returns early for everything it swallows, and the focus router sends `terminal.focus()` there, which is where the bar refocuses after every key), so `_handleCjkInput()` applies the modifier too. It is that module's single choke point to the PTY, so one call covers typed characters, IME flushes, Enter, backspace and arrows. Without it an armed modifier could neither fire NOR be spent, and survived to a later keystroke. Mapping is `ctrlByteFor()` (`code & 0x1f` over @A-Z[\]^_ and a-z, plus Ctrl+Space=NUL / Ctrl+?=DEL); characters with no control equivalent pass through unchanged, like a hardware keyboard. ⚠️ The armed style is `.accessory-btn.accessory-btn-ctrl.armed` (0,3,0) in BOTH stylesheets, and it cannot outrank mobile.css's light-skin repaint at **(0,3,1)** (`:is()` inherits its most specific argument, and that list holds `.btn-toolbar.btn-shell`) — so that rule excludes the state by hand as `.accessory-btn:not(.armed)`. Without the exclusion the armed button renders identically to a resting one on all four light skins, which is worse than no armed style at all.
|
||||
|
||||
**Dismissing the on-screen keyboard** (PRs #279/#280, `terminal-ui.js`): the terminal parks focus on a hidden textarea that nothing used to release, so TWO gestures now blur it, and they own different regions. **(1)** `_installMobileKeyboardDismiss()` — a document-level `touchend` that fires only while the terminal input actually holds focus, **never inside `#terminalContainer`** (tap classification owns that) and **never on a control** (`MOBILE_KEYBOARD_DISMISS_EXEMPT_SELECTOR`, matched with `closest()` so an icon inside a button counts). Session tabs are covered by the selector's `[tabindex]:not([tabindex="-1"])` arm, which is what stops a tab tap from blurring and then being re-focused by `selectSession()`. **(2)** In `_handleMobileTerminalTap`, a second tap on **inert `content`** (`startedWithTerminalFocus`) blurs instead of re-focusing. ⚠️ Scoped to `content` on purpose: the prompt row (`input`) keeps focus-then-position so a second tap still places the caret, and actionable rows blur earlier via `_isActionableMobileTerminalTap`. ⚠️ **A scroll ends in `touchend` too** — dismissing there closes the keyboard and drops the composer mid-read, so travel is tracked from `touchstart` and multi-touch is never a tap. Both classifiers MUST share one threshold: `initTerminal`'s `TAP_THRESHOLD` reads `MOBILE_KEYBOARD_DISMISS_TAP_SLOP`, since a gesture the terminal calls a scroll and the dismiss handler calls a tap is exactly that bug. ⚠️ **The gate excludes `test/mobile/**`, so CI cannot see the only test covering (1)** — run `npm run test:mobile -- test/mobile/keyboard.test.ts` by hand and diff the FAIL list against master. (Not `npm test --`: the gate's config excludes that path, so a file filter pointing into it matches nothing and exits green having run zero tests.) That blind spot is why merging the two PRs, which conflicted semantically but not textually, produced a red suite with two green CI checks.
|
||||
|
||||
**Phone toolbar: Enter replaces Shell** (post-1.8.0): inside `@media (max-width: 430px)` `btn-shell` is `display:none` and `btn-enter` takes its slot (`order: 4`); starting a shell moved into the Run dropdown (`Terminal / Shell` → `setRunMode('shell')` → `run()` → `runShell()`, button label "Run SH"). `runMode` is `z.string().max(20)` server-side, so new modes need no schema change. Desktop and tablet keep the green Run Shell button unchanged.
|
||||
|
||||
⚠️ **`sendEnterKey()` MUST go through `terminal._core.coreService.triggerDataEvent('\r', true)`** — not `sendInput()`, and never a raw POST to `/api/sessions/:id/input`. `localEchoEnabled` defaults to `MobileDetection.isTouchDevice()`, so on every phone the characters you type are buffered in the `LocalEchoOverlay` and have **never reached the PTY**; the `onData` Enter branch in terminal-ui.js is what flushes `pendingText` first and only then sends `\r` (after an 80ms delay so text lands first). Sending a bare `\r` submits an empty line and strands the typed text on screen, so the button looks dead. Replaying the keypress reuses the overlay flush, the flushed-offset cleanup and the ordering instead of reimplementing them. `KeyboardAccessory.sendKey()` is for escape sequences (arrows/Esc) and is the WRONG template to copy for input.
|
||||
|
||||
⚠️ **Skin overrides outrank plain class rules.** `styles.css` nests its skin block inside `html:not([data-skin="og"]) { … }`, so a bare `.btn-toolbar` rule in there resolves to specificity **(0,2,1)** and beats a `.btn-toolbar.btn-x` rule **(0,2,0)** in `mobile.css` regardless of load order. Toolbar-button colors set from mobile.css therefore need `!important` — that is why mobile.css leans on it so heavily. Symptom: only your `!important` properties land and everything else silently renders in generic toolbar grey.
|
||||
|
||||
**Z-index layers**: subagent windows (1000), plan agents (1100), mobile/tablet fixed header (1200, `mobile.css`), modals on ≤768px (1300 — must beat the fixed header or the modal close button is buried), log viewers (2000), image popups (3000), local echo overlay (7).
|
||||
**Connection-loss UI** (`computeConnectionLossUi()` in constants.js, writer `_updateConnectionLossUi()` in app.js): the service worker serves the cached app shell, so an unreachable server (phone off the tailnet, VPN down, server stopped) used to render a normal-looking empty dashboard whose only tell was the 8px header dot, which reads as "no sessions", not "no connection". Two surfaces now: a full-screen **overlay** while no server state has loaded this page load (nothing behind it is worth preserving), and a non-blocking **banner** once it has (the terminal scrollback stays readable). ⚠️ A **2.5s grace** is load-bearing: a COM deploy restarts the server and SSE is back in ~200ms, and a banner on every deploy trains the user to ignore it. `navigator.onLine === false` skips the grace, since that is never a blip. Retry re-arms SSE **and** the terminal WS (`planWsReconnect` can 'give-up', and the SSE backoff caps at 30s).
|
||||
|
||||
**SSE staleness watchdog** (`computeSseStale()` in constants.js, `_checkSseStale()` + a 5s interval in app.js): an `EventSource` that stops delivering does not always error, so `onerror` never fires, the header dot stays green, and every SSE-driven surface (tab status dots, sessions created on another device, renames) freezes until the user reloads. ⚠️ The 15s server keepalive was an SSE **comment** (`:keepalive`), and comments are **invisible to `EventSource` by spec**, so there was nothing a client could observe: it is now the named `sse:heartbeat` event (`cleanupDeadClients()`, sse-stream-manager.ts), which is exactly why the frame had to change type. ⚠️ Staleness is judged **only while the status is `connected`** and the device is online; that guard is the loop breaker, since a forced `connectSSE()` leaves `connected` immediately and cannot re-fire while a reconnect is in flight. ⚠️ The liveness stamp is applied inside `addListener` itself, so every registered handler (the `_SSE_HANDLER_MAP` wrappers AND the directly-registered ones) feeds it from one place; the heartbeat's own listener is a no-op that exists **only** to be registered, since `EventSource` drops named events nobody listens for. ⚠️ The watchdog interval is cleared at the top of `connectSSE()` and nowhere else (its only teardown path); clearing it elsewhere stacks intervals. Recovery needs no new sync path: the reconnect re-runs `handleInit` → `_resetAllAppState()`. The forced reconnect logs one diagnostic line, because a middlebox that strips heartbeats presents as "silently reconnects every 45s".
|
||||
|
||||
**Z-index layers**: subagent windows (1000), plan agents (1100), mobile/tablet fixed header (1200, `mobile.css`), modals on ≤768px (1300 — must beat the fixed header or the modal close button is buried), log viewers (2000), connection-loss overlay (2500, above the fixed header and modals), image popups (3000), response viewer (5000, backdrop 4999), file-preview overlay (5100 — must outrank the response viewer, which can launch it; at its old 2000 a path clicked in the chat opened BEHIND the chat), toasts/path picker (10000+, deliberately above the preview), local echo overlay (7).
|
||||
|
||||
**Respawn presets**: `solo-work` (3s/60min), `subagent-workflow` (45s/240min), `team-lead` (90s/480min), `ralph-todo` (8s/480min), `overnight-autonomous` (10s/480min).
|
||||
|
||||
@@ -288,11 +331,11 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L
|
||||
|
||||
### SSE Event Registry
|
||||
|
||||
149 event constants in `src/web/sse-events.ts` (backend) and `SSE_EVENTS` in `constants.js` (frontend). **Both must be kept in sync** — they are currently exactly in sync, and the backend file's `@fileoverview` carries the per-category breakdown.
|
||||
155 event constants in `src/web/sse-events.ts` (backend) and `SSE_EVENTS` in `constants.js` (frontend). **Both must be kept in sync**, and `test/sse-registry-parity.test.ts` is the guard that pins it (currently exactly in sync, 155 = 155, no drift either direction). The backend file's `@fileoverview` carries the per-category breakdown.
|
||||
|
||||
### API Routes
|
||||
|
||||
~199 handlers across 21 route files in `src/web/routes/`: system (45), sessions (32), cases (27), files (16), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), me (2), teams (2), search (1), hooks (1), clipboard (1), status-telemetry (1), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details.
|
||||
~217 handlers across 24 route files in `src/web/routes/`: system (48), sessions (34), cases (29), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), me (2), teams (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details.
|
||||
|
||||
**HTTP contract** (stable since 0.9.x, see `docs/versioning-policy.md`; full envelope/status/error-code/SSE spec in `docs/api-reference.md`): responses use the `ApiResponse<T>` envelope — `{ success: true, data? }` or `{ success: false, error, errorCode }` (`src/types/api.ts`). `/api/v1/*` is a versioned alias of `/api/*` (URL rewrite in `server.ts`).
|
||||
|
||||
@@ -310,24 +353,36 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L
|
||||
|
||||
## State Files
|
||||
|
||||
All in `~/.codeman/`: `state.json` (sessions, settings, respawn, orchestrator, cron jobs/runs), `mux-sessions.json` (tmux recovery), `settings.json` (user prefs), `push-keys.json` + `push-subscriptions.json`, `session-lifecycle.jsonl` (audit log), `update-status.json` (self-updater progress, polled across the service restart), `linked-cases.json`, `webviews.json` (saved web-tab dashboard URLs), `remote-hosts.json` + `remote-cases.json`, `docker-hosts.json` + `docker-cases.json` + `docker-exports/`, `subagent-window-states.json` + `subagent-parents.json` (subagent window layout, GET/PUT `/api/subagent-window-states`/`-parents`), `hook-secret` (per-instance), `users.json` (multi-user, mode 0600) + `admin-audit.jsonl`, `certs/` (self-signed TLS for `--https`), `.env` (CODEMAN_USERNAME/PASSWORD fallback for the `codeman attach` CLI). Transient: `self-update-runner.sh`. Multi-user case spaces live OUTSIDE the data dir at `~/codeman-users/<username>/cases` (shared across instances like `~/codeman-cases`, override `CODEMAN_USER_SPACES_DIR`).
|
||||
All in `~/.codeman/`: `state.json` (sessions, settings, respawn, orchestrator, cron jobs/runs), `mux-sessions.json` (tmux recovery), `settings.json` (user prefs), `push-keys.json` + `push-subscriptions.json`, `session-lifecycle.jsonl` (audit log), `update-status.json` (self-updater progress, polled across the service restart), `linked-cases.json`, `webviews.json` (saved web-tab dashboard URLs), `remote-hosts.json` + `remote-cases.json`, `docker-hosts.json` + `docker-cases.json` + `docker-exports/`, `subagent-window-states.json` + `subagent-parents.json` (subagent window layout, GET/PUT `/api/subagent-window-states`/`-parents`), `hook-secret` (per-instance), `users.json` (multi-user, mode 0600) + `admin-audit.jsonl`, `intents.json` (Read My Mind intent profiles, mode 0600), `certs/` (self-signed TLS for `--https`), `.env` (CODEMAN_USERNAME/PASSWORD fallback for the `codeman attach` CLI). Transient: `self-update-runner.sh`. Multi-user case spaces live OUTSIDE the data dir at `~/codeman-users/<username>/cases` (shared across instances like `~/codeman-cases`, override `CODEMAN_USER_SPACES_DIR`).
|
||||
|
||||
**Generated top-level dirs** (all gitignored — don't edit or commit): `dist/` (esbuild output), `out/`, `coverage/`, `test-results/`, `tmp/`, `screenshots-echo-diag/`. The committed gesture bundle (`src/web/public/gesture/gesture-codeman.js`) IS tracked, but its runtime wasm/model assets (`src/web/public/gesture/wasm/`, `*.task`) are fetched and gitignored.
|
||||
|
||||
## Testing
|
||||
|
||||
**Never run the bare full suite** (`npm test` with no file argument): the default config includes the browser-driven suites (`test/mobile/**` and 3 other Playwright tests), which need a live server + chromium + environment-specific PNG baselines and will fail/hang locally. Run individual files, or `test:ci` for a broad sweep:
|
||||
**`npm test` is the gate and is safe to run bare** — it runs `config/vitest.ci.config.ts`, exactly what CI runs, so local green means CI green.
|
||||
|
||||
```bash
|
||||
npm test -- test/<specific-file>.test.ts # Single file (SAFE, uses config/vitest.config.ts)
|
||||
npm test -- -t "pattern" # By name (SAFE)
|
||||
npm run test:ci # Everything except browser/perf suites — what CI runs
|
||||
# npm test # DON'T — includes browser/visual suites
|
||||
npm test # The gate — what CI runs
|
||||
npm test -- test/<specific-file>.test.ts # Single file
|
||||
npm test -- -t "pattern" # By name
|
||||
```
|
||||
|
||||
Raw `npx vitest` skips `config/vitest.config.ts`; always use `npm test --` or pass `--config config/vitest.config.ts`.
|
||||
Three suites are deliberately left out, because they cannot pass on an arbitrary machine. Each has its own runner, and a failure there means "not runnable here", not a regression:
|
||||
|
||||
**Config**: Vitest with `globals: true`, `fileParallelism: false`. Timeout 30s, teardown 60s. `config/vitest.ci.config.ts` = same minus the browser/perf excludes — keep the two configs in sync when changing shared options.
|
||||
```bash
|
||||
npm run test:browser # Playwright + chromium, live server; codex-predictive-echo also needs a real codex binary
|
||||
npm run test:mobile # the above plus environment-specific PNG baselines (own config, own pretest vendor step)
|
||||
npm run test:perf # wall-clock benchmarks — need an otherwise idle machine
|
||||
npm run test:all # literally everything; fails ~87 tests on a clean master here, which is why it is not the default
|
||||
```
|
||||
|
||||
⚠️ **`npm test` cannot see those suites**, so a change touching mobile/gesture/terminal-render behaviour needs the matching runner by hand — diff its FAIL list against master rather than reading it as pass/fail. That blind spot is what let two semantically-conflicting PRs merge green (see the on-screen-keyboard note above).
|
||||
|
||||
⚠️ **A file filter must match the runner.** `npm test -- test/mobile/keyboard.test.ts` matches nothing and exits GREEN having run zero tests, because the gate's config excludes that path — an excluded file needs its own runner (`npm run test:mobile -- <file>`, `npm run test:browser -- <file>`, `npm run test:perf -- <file>`). Vitest treats "no files matched a filter" as success, so read the file count, not just the colour.
|
||||
|
||||
Raw `npx vitest` skips the config (and with it `setup.ts`); always use `npm test --` or pass `--config`.
|
||||
|
||||
**Config**: Vitest with `globals: true`, `fileParallelism: false`. Timeout 30s, teardown 60s. `config/vitest.config.ts` is the everything-config behind `test:all`; `config/vitest.ci.config.ts` is the gate and derives its excludes from `config/test-suites.ts`, which is also what `vitest.browser.config.ts` and `vitest.perf.config.ts` derive their includes from — so the exclusions and the runners cannot drift apart. Keep shared options in sync across them.
|
||||
|
||||
**Tmux safety**: under vitest (`VITEST` env var, set automatically), `TmuxManager` no-ops ALL shell commands and becomes a pure in-memory mock — tests physically cannot create/kill/attach real tmux sessions (`IS_TEST_MODE` in `src/tmux-manager.ts`). Every docker IO path is no-op'd the same way. `Session` is test-gated too: instead of attaching a real tmux client, it spawns a raw-mode echo PTY (`TEST_PTY_SCRIPT` in `src/session.ts`), so integration tests get a live input/output loop that echoes each byte exactly once. `test/setup.ts` gives every test file a temporary `HOME`/`USERPROFILE` (all `homedir()`-derived state, `~/.codeman` and `~/codeman-cases` included, resolves into a per-file fixture; the Playwright browser cache path is preserved), and additionally strips `CODEMAN_PASSWORD`/`CODEMAN_USERNAME` (so auth state from the running instance can't leak into tests) and `CODEMAN_GESTURE` (a shell-exported gesture flag would flip render-injection assertions). ⚠️ Raw `npx vitest` without `--config` skips `setup.ts` and with it the temp-HOME isolation.
|
||||
|
||||
@@ -361,6 +416,6 @@ Two constraints worth knowing before you touch them: the env-derived PTY buffer
|
||||
|
||||
## Scripts & Tunnel
|
||||
|
||||
**`install.sh`** (repo root, 69KB) is the public entry point: `curl -fsSL <raw url> | bash` installs Node/tmux if missing, clones to `~/.codeman/app`, builds, and offers a systemd/launchd service. The network-access prompt is 3-way: **Tailscale** (loopback bind + guided `tailscale serve --bg <port>` HTTPS setup: install/login/operator/tailnet-HTTPS-toggle, then curl-verified end-to-end), **LAN** (0.0.0.0 + password prompt), or **local-only**; it preserves the existing binding on re-runs via `read_existing_binding()`. Tailscale state is detected dynamically from `tailscale serve status --json` (no marker files); the installer must NEVER `tailscale serve reset` or touch serve mappings other than 443→Codeman's port (users have unrelated serve config). `install.sh update`, `install.sh uninstall`, and `install.sh tailscale` (retrofit Tailscale access onto an existing install) also exist; `CODEMAN_NONINTERACTIVE=1` approves system changes for automation, `CODEMAN_TAILSCALE=1` presets the Tailscale choice (never installs Tailscale non-interactively).
|
||||
**`install.sh`** (repo root, 92KB) is the public entry point: `curl -fsSL <raw url> | bash` installs Node/tmux if missing, clones to `~/.codeman/app`, builds, and offers a systemd/launchd service. The network-access prompt is 3-way: **Tailscale** (loopback bind + guided `tailscale serve --bg <port>` HTTPS setup: install/login/operator/tailnet-HTTPS-toggle, then curl-verified end-to-end), **LAN** (0.0.0.0 + password prompt), or **local-only**; it preserves the existing binding on re-runs via `read_existing_binding()`. Tailscale state is detected dynamically from `tailscale serve status --json` (no marker files); the installer must NEVER `tailscale serve reset` or touch serve mappings other than 443→Codeman's port (users have unrelated serve config). `install.sh update`, `install.sh uninstall`, and `install.sh tailscale` (retrofit Tailscale access onto an existing install) also exist; `CODEMAN_NONINTERACTIVE=1` approves system changes for automation, `CODEMAN_TAILSCALE=1` presets the Tailscale choice (never installs Tailscale non-interactively).
|
||||
|
||||
Other key scripts: `scripts/tmux-manager.sh` (safe tmux mgmt), `scripts/tunnel.sh [quick|named] start|stop|status|url` (quick = random trycloudflare URL, default; `named setup|enable` = fixed-hostname tunnel via `scripts/codeman-tunnel-named.service`; bare `start|stop|url` still means quick), `scripts/run-beta.sh` (isolated beta instance), `scripts/build-agent-image.mjs` (docker base image), `scripts/self-update.sh` (detached updater). Production services: `scripts/codeman-web.service`, `scripts/codeman-tunnel.service`. **Always set `CODEMAN_PASSWORD`** before exposing via tunnel.
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
<h2 align="center">Mission control for AI coding agents</h2>
|
||||
|
||||
<p align="center">
|
||||
<em>Claude Code • OpenCode • Codex • Antigravity • Gemini • Terminal - One Dashboard • Any Device</em>
|
||||
<em>Claude Code • OpenCode • Codex • Antigravity • Gemini • Pi • Terminal - One Dashboard • Any Device</em>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
@@ -27,7 +27,7 @@
|
||||
<img src="docs/images/subagent-demo-20260724.gif" alt="Codeman — parallel subagent visualization" width="900">
|
||||
</p>
|
||||
|
||||
**Codeman** is a self-hosted mission control for AI coding agents. It spawns Claude Code, OpenCode, Codex, Antigravity, or Gemini CLI inside persistent tmux sessions, streams the real terminal to any browser, and keeps agents productive after you walk away: it re-prompts on idle, resumes when a usage limit resets, runs scheduled jobs, and shows every background agent working in real time.
|
||||
**Codeman** is a self-hosted mission control for AI coding agents. It spawns Claude Code, OpenCode, Codex, Antigravity, Gemini, or Pi inside persistent tmux sessions, streams the real terminal to any browser, and keeps agents productive after you walk away: it re-prompts on idle, resumes when a usage limit resets, runs scheduled jobs, and shows every background agent working in real time.
|
||||
|
||||
Get started in one line (macOS & Linux, Windows via WSL):
|
||||
|
||||
@@ -42,7 +42,7 @@ codeman web
|
||||
|
||||
The installer asks before every system change, and re-running the same line updates in place. Full details: [Quick Start - Installation](#quick-start---installation).
|
||||
|
||||
- **One dashboard, five CLIs** - run [Claude Code, OpenCode, Codex, Antigravity, or Gemini](#more-features) per session (plus plain shell), locally, [in Docker](#isolated-docker-sessions), or [over SSH](#remote-ssh-sessions)
|
||||
- **One dashboard, six CLIs** - run [Claude Code, OpenCode, Codex, Antigravity, Gemini, or Pi](#more-features) per session (plus plain shell), locally, [in Docker](#isolated-docker-sessions), or [over SSH](#remote-ssh-sessions)
|
||||
- **Truly phone-friendly** - a [touch-optimized terminal](#mobile-optimized-web-ui) with instant local echo, QR login, swipe navigation, and push notifications
|
||||
- **Runs while you sleep** - [idle detection + respawn cycling](#respawn-controller) and auto-resume when a subscription limit resets, for 24+ hour unattended runs
|
||||
- **See your agents think** - [live floating windows](#live-agent-visualization) for every subagent and teammate, with real-time transcripts
|
||||
@@ -68,7 +68,7 @@ This installs Node.js and tmux if missing, clones Codeman to `~/.codeman/app`, a
|
||||
- **Re-run to update.** The same one-liner updates a finished install in place: local changes in `~/.codeman/app` are stashed (never discarded), and a running service is restarted and verified. If a first install was interrupted, re-running resumes the full setup instead. `install.sh update` and `install.sh uninstall` also exist.
|
||||
- **CI / headless:** without a terminal attached, steps that would change your system abort with instructions instead of running silently. Set `CODEMAN_NONINTERACTIVE=1` to approve them for automation.
|
||||
|
||||
You'll need at least one AI coding CLI installed — [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), or [Gemini CLI](https://github.com/google-gemini/gemini-cli) (any combination works; Gemini CLI is enterprise-only since Google's consumer cutover, and Antigravity is its successor). The installer detects whichever of the five is present; if none is found, it offers to install Claude Code or OpenCode, or you can skip and install one yourself later. After install:
|
||||
You'll need at least one AI coding CLI installed — [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), or [Pi](https://pi.dev) (any combination works; Gemini CLI is enterprise-only since Google's consumer cutover, and Antigravity is its successor). The installer detects whichever of the six is present; if none is found, it offers to install Claude Code or OpenCode, or you can skip and install one yourself later. After install:
|
||||
|
||||
```bash
|
||||
codeman web
|
||||
@@ -85,9 +85,29 @@ codeman web --multiuser # named logins + per-user case spaces
|
||||
Details in [Multi-User Mode](#multi-user-mode-opt-in) below.
|
||||
|
||||
<details>
|
||||
<summary><strong>Run as a background service</strong></summary>
|
||||
<summary><strong>Keep it running in the background</strong></summary>
|
||||
|
||||
The installer's final menu sets this up for you (option 2) and verifies the service actually comes up before claiming success. To configure it manually instead:
|
||||
To outlive the shell you started it in, without setting anything up:
|
||||
|
||||
```bash
|
||||
codeman web -d # detach; logs to ~/.codeman/web.log
|
||||
codeman web --status # is it up, and on which pid
|
||||
codeman web --stop # graceful SIGTERM; agents keep running in tmux
|
||||
```
|
||||
|
||||
`-d` waits until the server actually answers before reporting success, and refuses to start a second one on the same data dir (two servers sharing a tmux socket attach to each other's sessions).
|
||||
|
||||
To have it come back after a reboot, install it as a service instead. The installer's final menu does this for you (option 2); `codeman service` is the equivalent for an `npm i -g aicodeman` install:
|
||||
|
||||
```bash
|
||||
codeman service install # systemd user unit (Linux) or LaunchAgent (macOS)
|
||||
codeman service status
|
||||
codeman service uninstall
|
||||
```
|
||||
|
||||
`service install` writes the unit with your current PATH baked in, which matters more than it sounds: launchd hands a job `/usr/bin:/bin:/usr/sbin:/sbin`, so a Homebrew or nvm `node`, `tmux` or `claude` is invisible to a hand-written plist. It never copies `CODEMAN_PASSWORD` into the unit file; add that yourself if the service needs auth.
|
||||
|
||||
To write the unit by hand instead:
|
||||
|
||||
**Linux (systemd):**
|
||||
|
||||
@@ -151,7 +171,7 @@ launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/com.codeman.web.plist
|
||||
wsl bash -c "curl -fsSL https://getcodeman.com/install | bash"
|
||||
```
|
||||
|
||||
Codeman requires tmux, so Windows users need [WSL](https://learn.microsoft.com/en-us/windows/wsl/install). If you don't have WSL yet: run `wsl --install` in an admin PowerShell, reboot, open Ubuntu, then install your preferred AI coding CLI inside WSL ([Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), or [Gemini CLI](https://github.com/google-gemini/gemini-cli)). After installing, `http://localhost:3000` is accessible from your Windows browser.
|
||||
Codeman requires tmux, so Windows users need [WSL](https://learn.microsoft.com/en-us/windows/wsl/install). If you don't have WSL yet: run `wsl --install` in an admin PowerShell, reboot, open Ubuntu, then install your preferred AI coding CLI inside WSL ([Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), or [Pi](https://pi.dev)). After installing, `http://localhost:3000` is accessible from your Windows browser.
|
||||
|
||||
</details>
|
||||
|
||||
@@ -220,6 +240,8 @@ codeman web # localhost:3000 (loopback only — safe defau
|
||||
codeman web --port 8080 # custom port (or set CODEMAN_PORT)
|
||||
codeman web --https # self-signed TLS (only needed for remote access)
|
||||
codeman web -H 0.0.0.0 # bind LAN — REQUIRES CODEMAN_PASSWORD (see Security)
|
||||
codeman web -d # detach: survives closing the shell (--status, --stop)
|
||||
codeman service install # systemd/launchd service: comes back after reboots
|
||||
```
|
||||
|
||||
Open the printed URL. The page is a single dashboard; everything below happens there.
|
||||
@@ -230,9 +252,9 @@ Click **+ New Session** (or **Quick Start**). A session is one AI CLI running in
|
||||
|
||||
| Field | What it does |
|
||||
| ---------------------------- | ------------------------------------------------------------------------------------------------------------------- |
|
||||
| **Working directory / case** | The folder the agent operates in. A "case" is just a named working dir Codeman remembers. |
|
||||
| **CLI / run mode** | `Claude` (default), `OpenCode`, `Codex`, `Antigravity`, `Gemini`, or `Terminal` (plain shell). |
|
||||
| **Model** | Per-session model (App Settings → Claude Model). A soft default — `/model` still works in-session. |
|
||||
| **Working directory / case** | The folder the agent operates in. A "case" is just a named working dir Codeman remembers. **Add Case** creates one from scratch, links an existing folder, or clones a GitHub repo straight into one (**Clone Repo**). |
|
||||
| **CLI / run mode** | `Claude` (default), `OpenCode`, `Codex`, `Antigravity`, `Gemini`, `Pi`, or `Terminal` (plain shell). |
|
||||
| **Model** | Per-session model (App Settings → Models → New Claude sessions). A soft default — `/model` still works in-session. |
|
||||
| **Effort / Ultracode** | Reasoning effort (`low`–`max`) or `ultracode` for dynamic multi-agent workflows. Switchable anytime with `/effort`. |
|
||||
|
||||
Hit start — Codeman spawns the CLI via a real PTY and streams it to your browser over SSE.
|
||||
@@ -256,7 +278,7 @@ Hit start — Codeman spawns the CLI via a real PTY and streams it to your brows
|
||||
| ---------------- | --------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------- |
|
||||
| **Respawn** | Long unattended runs — auto-restarts the CLI on idle/limit, with adaptive timing. Presets: `solo-work`, `overnight-autonomous`, … | Respawn tab |
|
||||
| **Orchestrator** | Turn one goal into a phased plan and drive it to completion across agents. | Orchestrator panel |
|
||||
| **Cron** | Saved, named jobs on a schedule (`once`/`interval`/`daily`/`weekly`) that spawn a session and send a prompt when due. | ⏰ Cron button _(opt-in: App Settings → Display → Header Displays)_ |
|
||||
| **Cron** | Saved, named jobs on a schedule (`once`/`interval`/`daily`/`weekly`) that spawn a session and send a prompt when due. | ⏰ Cron button _(opt-in: App Settings → Header & Panels → Scheduling)_ |
|
||||
| **Auto-resume** | Automatically continue after a subscription rate-limit resets. | Respawn tab (top) |
|
||||
|
||||
### 6. Reach it from anywhere
|
||||
@@ -268,7 +290,8 @@ Hit start — Codeman spawns the CLI via a real PTY and streams it to your brows
|
||||
### 7. Operate & maintain
|
||||
|
||||
- **App Settings** — model, effort, permission startup mode, theme/skin, notifications, display toggles, per-CLI options, a synced custom display name, and per-device English/Simplified Chinese UI language.
|
||||
- **Self-update** — git-clone installs update in place from **Settings → Updates**.
|
||||
- **Run it in the background** — `codeman web -d` detaches from your shell (`--status`, `--stop`); `codeman service install` makes it a systemd user unit / macOS LaunchAgent that survives reboots. Both verify the server actually answers before reporting success, and both refuse to start a second server on one data dir. See [Keep it running in the background](#quick-start---installation).
|
||||
- **Self-update** — git-clone installs update in place from **App Settings → System → Updates**.
|
||||
- **Deploy your own changes** — see [Development](#development).
|
||||
|
||||
> ⚠️ **Safety:** if you're working _inside_ a Codeman-managed session (`echo $CODEMAN_MUX` → `1`), never run `tmux kill-session` / `pkill claude` directly — use the web UI or `./scripts/tmux-manager.sh`.
|
||||
@@ -383,6 +406,14 @@ The title is templated into the served HTML on first byte, so it's correct from
|
||||
| **110k tokens** | Auto `/compact` | Context summarized, work continues |
|
||||
| **140k tokens** | Auto `/clear` | Fresh start with `/init` |
|
||||
|
||||
### Tab Alerts
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/tab-alerts-glow-20260815.gif" alt="Session tabs: a regular active tab beside a yellow waiting-for-input tab and a red needs-decision tab, both with a breathing glow" width="900">
|
||||
</p>
|
||||
|
||||
Every tab tells you its state at a glance. A running session keeps its green status dot. When a session stops and waits for input, its tab turns **yellow**: steady ring, tinted background, yellow dot, with a slow breathing glow on top. When a permission prompt or question is **blocking** the agent, the tab turns **red** with a faster pulse. The base tint never blinks off, so even a split-second glance (or a screenshot) reads the true state; the ring stays visible while the tab is selected, and a page reload re-arms pending alerts from the server, so a blocked session can never hide behind a fresh-looking tab.
|
||||
|
||||
### Notifications
|
||||
|
||||
Real-time desktop alerts when sessions need attention — `permission_prompt` and `elicitation_dialog` trigger critical red tab blinks, `idle_prompt` triggers yellow blinks. Click any notification to jump directly to the affected session. Hooks auto-configured per case directory.
|
||||
@@ -403,16 +434,18 @@ PTY Output → 16ms Server Batch → DEC 2026 Wrap → SSE → Client rAF → xt
|
||||
|
||||
## More Features
|
||||
|
||||
- **Self-update** — git-clone installs under systemd/launchd update in place from **App Settings → Updates**: it detects the latest release, auto-stashes a dirty tree, and streams build progress across the service restart (npm installs report as non-updatable)
|
||||
- **Multi-CLI** — run **Claude Code**, **OpenCode**, **Codex**, **Antigravity**, or **Gemini** per session; env-var prefixes auto-gate (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `ANTIGRAVITY_*` vs `GEMINI_*`/`GOOGLE_*`). See [`docs/opencode-integration.md`](docs/opencode-integration.md)
|
||||
- **Background daemon & service install** — `codeman web -d` runs the server detached with a pidfile, `~/.codeman/web.log`, and verified startup (it polls the server until it answers, so a port clash never reads as success); `codeman service install` writes a systemd user unit (Linux) or LaunchAgent (macOS) with your shell's PATH baked in, so an nvm or Homebrew `node`, `tmux` and `claude` are actually found. Secrets are never written into unit files
|
||||
- **Self-update** — git-clone installs under systemd/launchd update in place from **App Settings → System → Updates**: it detects the latest release, auto-stashes a dirty tree, and streams build progress across the service restart (npm installs report as non-updatable)
|
||||
- **Clone a GitHub repo as a case** — paste a repository URL into **Add Case → Clone Repo** and Codeman clones it into `~/codeman-cases/<name>` and registers it as a normal case, ready to run an agent in. It preflights the URL while you type (tells you whether it can be cloned anonymously and offers the repo's real branches and tags for the optional branch/tag field), fills the case name in from the URL, and lets you pick which CLI the Run button should use. Public repositories over `https://`; Codeman never collects or stores credentials
|
||||
- **Multi-CLI** — run **Claude Code**, **OpenCode**, **Codex**, **Antigravity**, **Gemini**, or **Pi** per session; env-var prefixes auto-gate (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `ANTIGRAVITY_*` vs `GEMINI_*`/`GOOGLE_*` vs `PI_*`). See [`docs/opencode-integration.md`](docs/opencode-integration.md) and [`docs/pi-integration.md`](docs/pi-integration.md)
|
||||
- **Docker sessions** — run a case inside an isolated, hardened container. One checkbox on **Create New** spins up a container with sensible defaults and starts the agent inside it; multiple sessions share one per-case container; export a container + its workspace to a portable `.tar.gz` to move it to another machine. See [`docs/docker-cases.md`](docs/docker-cases.md)
|
||||
- **Remote SSH sessions** — point a case at another machine and run the agent there inside a durable remote tmux: survives SSH drops, auto-reconnects, and can discover + attach sessions already running on the host. See [`docs/remote-sessions.md`](docs/remote-sessions.md)
|
||||
- **Effort & Ultracode** — set a per-session default effort (`low`–`max`) or enable **ultracode** (dynamic multi-agent workflows). Soft defaults only — switchable anytime with `/effort` in-session. Extended-thinking budget is configurable too
|
||||
- **Voice input** — dictate prompts with Deepgram Nova-3 (Web Speech API fallback): toggle recording, auto-silence stop, live level meter (`Ctrl+Shift+V`)
|
||||
- **Image input** — paste or drag-and-drop images straight into a session
|
||||
- **Gesture control** _(opt-in)_ — a MediaPipe hand-tracking overlay to grab/drag session windows and pinch buttons, hands-free. Enable with `CODEMAN_GESTURE=1` + App Settings → Display
|
||||
- **Gesture control** _(opt-in)_ — a MediaPipe hand-tracking overlay to grab/drag session windows and pinch buttons, hands-free. Enable with `CODEMAN_GESTURE=1` + App Settings → Terminal & Input
|
||||
- **Multi-monitor span** _(macOS)_ — one click opens a browser window maximized across all displays, so floating agent/gesture panels can cross the physical seam
|
||||
- **File Viewer button** _(opt-in)_ — a header button that toggles the built-in file browser panel with one tap; enable under App Settings → Display → Header Displays
|
||||
- **File Viewer button** _(opt-in)_ — a header button that toggles the built-in file browser panel with one tap; enable under App Settings → Header & Panels → Header buttons
|
||||
- **CJK / IME input** — full composition support for Chinese / Japanese / Korean
|
||||
- **OS notifications & hostname-aware titles** — desktop alerts and tab titles are prefixed `codeman:<host>` so multi-host setups stay unambiguous
|
||||
|
||||
@@ -426,7 +459,7 @@ Run a case inside its own hardened Docker container instead of directly on your
|
||||
- **Resource templates** — expand the checkbox for a **Small / Medium / Large / GPU** preset (memory, CPUs, GPU), or set your own. **Disk is elastic** — storage grows as data flows in, no fixed cap.
|
||||
- **Shared per-case container** — many sessions can `docker exec` into the same container; killing one session never tears the container out from under the others.
|
||||
- **Hardened by default** — non-root, `--cap-drop ALL`, `no-new-privileges`, PID/memory caps, never `--privileged` or the docker socket; a **sealed** profile (no host credentials, network off) is one toggle away.
|
||||
- **Seamless auth, isolated credentials** — your host Claude / Codex / Antigravity / Gemini / OpenCode logins work inside the container out of the box: credentials are seeded (copied) in at launch and onboarding/trust prompts are pre-answered, so no login wizard appears. The container keeps its own copies and never writes back to your host credential stores; only conversation transcripts are shared, and exports never capture secrets.
|
||||
- **Seamless auth, isolated credentials** — your host Claude / Codex / Antigravity / Gemini / OpenCode / Pi logins work inside the container out of the box: credentials are seeded (copied) in at launch and onboarding/trust prompts are pre-answered, so no login wizard appears. The container keeps its own copies and never writes back to your host credential stores; only conversation transcripts are shared, and exports never capture secrets.
|
||||
- **Move it to another machine** — export a container's whole environment (toolchain + workspace) to a portable `.tar.gz`, `docker load` it on the other side, and import it into a fresh case.
|
||||
- **Durable** — reconnect after a restart lands back in the same live agent; a container stop/reboot resumes the conversation from the bind-mounted transcript.
|
||||
|
||||
@@ -496,7 +529,7 @@ The script auto-installs a systemd user service on first run. The tunnel URL is
|
||||
systemctl --user enable codeman-tunnel
|
||||
loginctl enable-linger $USER
|
||||
|
||||
# Or via the Codeman web UI: Settings → Tunnel → Toggle On
|
||||
# Or via the Codeman web UI: App Settings → System → Remote access → Cloudflare Tunnel
|
||||
```
|
||||
|
||||
</details>
|
||||
@@ -598,7 +631,7 @@ By default Codeman launches sessions with `--dangerously-skip-permissions`, so t
|
||||
- **Loopback by default** — the server binary binds `127.0.0.1`, reachable only from the same machine, so the no-password default is safe out of the box (the guided installer asks about network access and configures the binding + password for you). Binding a non-loopback host without `CODEMAN_PASSWORD` _starts but prints a loud warning_ with three concrete fixes (set a password, loopback + an authenticated tunnel, or explicitly acknowledge with `--allow-unauthenticated-network`)
|
||||
- **Optional auth, real sessions** — HTTP Basic via `CODEMAN_USERNAME` (default `admin`) / `CODEMAN_PASSWORD`. Success issues an opaque 256-bit `codeman_session` cookie (`randomBytes(32)`) — validated server-side, not client-signed, so it can't be forged offline (24h TTL, auto-extend, device-context audit log)
|
||||
- **Per-IP rate limiting** — 10 failed attempts → `429` with `Retry-After` (15-min decay). A valid cookie or correct password recovers _immediately_ even while an attacker hammers the same IP — important because all tunnel traffic shares one loopback IP. QR auth has its own separate limiter
|
||||
- **Configurable permission mode** - `--dangerously-skip-permissions` is only the default. **App Settings → Claude CLI → Startup Mode** can switch new sessions to Anthropic's classifier-guarded `auto` mode (low-prompt, needs Claude Code 2.1.207+), `normal` prompting, or an explicit allowed-tools list. In multi-user mode, non-granted users are forced to `auto`, and shell sessions / skip-permissions require an explicit per-user grant
|
||||
- **Configurable permission mode** - `--dangerously-skip-permissions` is only the default. **App Settings → Agents & CLIs → Claude → Startup Mode** can switch new sessions to Anthropic's classifier-guarded `auto` mode (low-prompt, needs Claude Code 2.1.207+), `normal` prompting, or an explicit allowed-tools list. In multi-user mode, non-granted users are forced to `auto`, and shell sessions / skip-permissions require an explicit per-user grant
|
||||
|
||||
### Always-on browser hardening (v0.9.5)
|
||||
|
||||
@@ -612,7 +645,7 @@ These run for **every** request — before auth, even on the default no-password
|
||||
|
||||
### Input, files & headers
|
||||
|
||||
- **Schema-validated inputs** — every API body is checked with Zod v4 schemas; a `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` env-prefix allowlist gates which settings each CLI can receive
|
||||
- **Schema-validated inputs** — every API body is checked with Zod v4 schemas; a `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` / `PI_*` env-prefix allowlist gates which settings each CLI can receive
|
||||
- **Path containment** — file routes `realpath` before boundary checks (no TOCTOU); `..`, absolute paths, and symlinks resolving outside the working dir are rejected. Caps: 10 MB text preview / 50 MB raw & download; `/api/download` blocklists sensitive paths (`.env`, `*credentials*`, `~/.ssh/`, `.aws/credentials`). SVG/HTML is served `octet-stream` + `nosniff` + attachment so it downloads rather than executes
|
||||
- **Security headers** — `Content-Security-Policy` (`default-src 'self'`, every exception enumerated), `X-Content-Type-Options: nosniff`, `X-Frame-Options: SAMEORIGIN`, HSTS over HTTPS, and CORS reflected **only** for `localhost` / `127.0.0.1` / `::1`
|
||||
|
||||
@@ -650,6 +683,7 @@ Single-digit selection (1-9), color-coded status, token counts, auto-refresh. De
|
||||
| `Ctrl/Cmd+Tab` | Next session |
|
||||
| `Alt/Option+[` / `Alt/Option+]` | Previous / next session |
|
||||
| `Alt/Option+1`-`Alt/Option+9` | Switch to tab N (physical keys, so macOS Option layouts work) |
|
||||
| `Alt/Option+B` | Collapse / expand the session sidebar (sidebar layout only) |
|
||||
| `Ctrl+Shift+{` / `Ctrl+Shift+}` | Move active tab left / right |
|
||||
| `Ctrl/Cmd+C` | Copy selection, or interrupt when nothing is selected |
|
||||
| `Ctrl+Shift+C` | Copy selection (never interrupts) |
|
||||
@@ -667,6 +701,78 @@ Single-digit selection (1-9), color-coded status, token counts, auto-refresh. De
|
||||
|
||||
For AI agents and automation that control Codeman without a browser: an agent that spins up worker sessions, a CI bot, or **Claude Code running _inside_ a Codeman session orchestrating other sessions**. Everything the UI does is HTTP + a CLI, so an agent can do it too.
|
||||
|
||||
### The agent skill (start here)
|
||||
|
||||
Everything in this section also ships as a **Claude Code skill** in [`skills/codeman`](skills/codeman/SKILL.md). Install it once and you never paste API docs into a prompt again. You ask for what you want in plain English, and the agent already sitting inside a Codeman session loads the recipes and drives the API itself.
|
||||
|
||||
#### Step 1: install it
|
||||
|
||||
| How | Command | Scope |
|
||||
| -------------- | ---------------------------------------------------------- | ------------------------------------------------------------------------------------------ |
|
||||
| Skills CLI | `npx skills add Ark0N/Codeman --skill codeman -g` | Global, works for any skills-aware agent |
|
||||
| Bundled CLI | `codeman skill install` | Global (`~/.claude/skills/codeman`), for npm installs that never cloned the repo |
|
||||
| Bundled CLI | `codeman skill install --case <name>` | One case only |
|
||||
| Web UI | App Settings → Agents & CLIs → Claude → **Agent Skill** | Auto-injects into each case on Claude session create (`agentSkillEnabled`, SYNCED, default off) |
|
||||
|
||||
`codeman skill uninstall [--case <name>]` reverses the CLI installs, and never touches a `skills/codeman` you wrote yourself.
|
||||
|
||||
#### Step 2: ask for things
|
||||
|
||||
That is the entire interface. No curl, no endpoint names, no session ids. These prompts work as written:
|
||||
|
||||
| You say | The skill does |
|
||||
| ------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------------------------------- |
|
||||
| _"What sessions are running right now?"_ | Lists them with name, mode and status. Read-only, safe to ask anytime. |
|
||||
| _"Start a shell worker on the `myapp` case, run the test suite, tell me if it passes."_ | Spawns, waits on a split completion marker, reads back the exit code, cleans up. |
|
||||
| _"Spin up 3 workers for lint, typecheck and tests. Run them in parallel, report failures."_ | The fan-out flow: one session per task, all started first, then gathered as each finishes. |
|
||||
| _"Have a claude worker on `refactor-auth` summarize `src/session.ts`, then close it."_ | Spawns, runs the readiness ladder (first-run trust dialog included), send-and-wait, reads the clean transcript answer, deletes. |
|
||||
| _"Watch session w4 and tell me if it gets stuck on a permission prompt."_ | Blocks on the `blocked` signal and surfaces the question to **you**. It never answers another session's prompt itself. |
|
||||
|
||||
#### Step 3: nothing
|
||||
|
||||
The agent deletes every session it started. Watch the tabs appear and disappear in the dashboard while it works.
|
||||
|
||||
#### A real run, start to finish
|
||||
|
||||
> **You:** spin up 3 shell workers, run lint / typecheck / the frontend syntax check in parallel, and tell me which failed.
|
||||
|
||||
```text
|
||||
lint -> 9f2d8e5f dispatched
|
||||
typecheck -> aff9c691 dispatched 3 tabs appear in the dashboard
|
||||
syntax -> be9f1f15 dispatched
|
||||
|
||||
lint DONE_lint_17909 rc=0
|
||||
typecheck DONE_typecheck_3409 rc=0 gathered as each one finishes
|
||||
syntax DONE_syntax_18501 rc=0
|
||||
|
||||
deleted 9f2d8e5f, aff9c691, be9f1f15 tabs disappear
|
||||
```
|
||||
|
||||
Those `DONE_<task>_<random>` strings are the skill's **split marker** trick, and they are why the fan-out is reliable on hook-less `shell` sessions: the typed line contains `${M}_17909`, so only the command's real *output* ever contains `DONE_17909`. An unsplit marker would match the echo of your own keystrokes before the command had even run.
|
||||
|
||||
#### What's in the box
|
||||
|
||||
| File | Contents |
|
||||
| --------------------------------------------------------------------- | --------------------------------------------------------------------------------------------- |
|
||||
| [`SKILL.md`](skills/codeman/SKILL.md) | Safety rules, the ready-made fast path (spawn N workers, task them, collect), and the verb index. Always loaded. |
|
||||
| [`reference/verbs.md`](skills/codeman/reference/verbs.md) | The 14 verbs in detail: readiness, send-and-wait, markers, interrupts, cleanup. On demand. |
|
||||
| [`reference/recipes.md`](skills/codeman/reference/recipes.md) | 6 worked multi-worker flows (fan-out, blocked-worker watch, messaging fan-out). On demand. |
|
||||
| [`reference/endpoints.md`](skills/codeman/reference/endpoints.md) | Full endpoint tables, error codes, per-mode signal table, capacity limits. On demand. |
|
||||
| [`reference/messaging.md`](skills/codeman/reference/messaging.md) | Talking to claude workers directly via Claude Code cross-session messaging. On demand. |
|
||||
|
||||
Every recipe in there was verified against a live server, and the comments record the failure modes that were measured rather than guessed.
|
||||
|
||||
#### Two things worth knowing
|
||||
|
||||
- **It self-gates.** Outside a Codeman session (`CODEMAN_MUX` unset) the skill refuses to act and does not guess an API URL, so a global install costs an unrelated Claude Code session nothing.
|
||||
- **It is deliberately conservative.** Unprompted, it may only spawn sessions, prompt them, and delete ones **it created in that same conversation, by exact id**, through a fail-closed guard that refuses to delete the agent's own session. Deleting a case (which erases a real directory of your code), bulk kills, respawn/ralph/cron/orchestrator changes and settings writes all require you to ask, naming the target.
|
||||
|
||||
⚠️ Turning `agentSkillEnabled` back off **does not remove already-injected copies** (a create-time sweep would yank the skill out from under other live sessions sharing that `.claude/` dir). Remove them per case with `codeman skill uninstall --case <name>`.
|
||||
|
||||
---
|
||||
|
||||
**The rest of this section is the manual path**: the same operations as raw HTTP, for a CI bot, a shell script, or any agent without skill support.
|
||||
|
||||
### Detect that you're inside Codeman
|
||||
|
||||
When a CLI runs in a Codeman-managed session, these environment variables are set — read them instead of hardcoding anything:
|
||||
@@ -680,15 +786,21 @@ When a CLI runs in a Codeman-managed session, these environment variables are se
|
||||
|
||||
### Rules of the road (read before you POST)
|
||||
|
||||
1. **Single-line input only.** Programmatic input is sent as literal text **+ Enter** in one shot. Multi-line strings break the agent TUI (Ink) — send one line, or split into multiple calls.
|
||||
1. **Single-line input, ending in `\r`.** Programmatic input is sent as literal text, and Enter fires **only when the input contains a carriage return**: `{"input":"run tests\r"}`. Without the `\r` the text sits on the session's prompt unsubmitted (and a combined `wait` runs its full timeout on a turn that never started). Embedded newlines are stripped rather than rejected, so `"echo A\necho B\r"` runs the joined command `echo Aecho B`: send one line per call.
|
||||
2. **Make input idempotent.** Include a stable `clientId` and a monotonic per-session `seq` on `POST …/input`. The server de-duplicates, so a retry after a dropped connection can't double-deliver a prompt.
|
||||
3. **Auth.** If `CODEMAN_PASSWORD` is set, send HTTP Basic auth (user `admin` or `CODEMAN_USERNAME`) or a `codeman_session` cookie. The default loopback install is passwordless. A missing `Origin` header is allowed, so plain `curl` works; cross-site browser origins are rejected (CSRF guard).
|
||||
3. **Auth.** If `CODEMAN_PASSWORD` is set, send HTTP Basic auth (user `admin` or `CODEMAN_USERNAME`) or a `codeman_session` cookie. The default loopback install is passwordless. A missing `Origin` header is allowed, so plain `curl` works; cross-site browser origins are rejected (CSRF guard). ⚠️ A `401` replies with the bare string `Unauthorized`, **not** the JSON envelope, so piping it into `jq` throws a parse error instead of showing the failure: check the status before parsing.
|
||||
4. **Response envelope.** Most endpoints return `{ "success": true, "data": … }` (errors: `{ "success": false, "error", "errorCode" }`). A few legacy GETs return bare bodies — **handle both** (`body.data ?? body`).
|
||||
5. **`/api/v1/*`** is a stable alias of `/api/*`.
|
||||
6. **Wait instead of polling, and don't treat a timeout as an error.** The wait endpoints answer with HTTP `200` and `wait.timedOut: true` when nothing happened in time, so loop over short waits (60s is the default) rather than issuing one long call, because tunnels cut idle connections. `wait.timeoutMs` tells you the timeout the server actually applied after clamping (600s ceiling).
|
||||
7. **Only `claude` sessions emit `stop` and `blocked`.** Those two come from Claude Code hooks; `shell` and the external CLIs (opencode/codex/gemini/antigravity/pi) accept only `idle`, `working` and `exit`. Asking for `stop` explicitly on those is a `400`; omitting `until` is always safe. ⚠️ On a `shell` session `idle` fires **once**, at startup, and never again, so send-and-wait there can only time out; synchronize hook-less sessions with a `wait-output` marker.
|
||||
8. **Nothing reports "ready", so wait for it explicitly.** A new session answers `{"signal":"exit","immediate":true}` (that means *not started*, not *crashed*) until its PID exists, and a `claude` worker in a fresh case then sits on the CLI's trust dialog. Prompt it there and the wait resolves on `idle` in ~2s looking exactly like a finished turn, while the text sits stuck in the dialog. Recipe 2b below is the sequence that avoids it.
|
||||
|
||||
### Recipes
|
||||
|
||||
```bash
|
||||
# CODEMAN_API_URL is auto-set inside every Codeman session, correct scheme included.
|
||||
# The fallback below fits a stock install; on a --https install set the https:// URL
|
||||
# yourself and add -k to each curl (self-signed cert).
|
||||
API="${CODEMAN_API_URL:-http://127.0.0.1:3000}"
|
||||
# (add -u admin:"$CODEMAN_PASSWORD" to each call if a password is set)
|
||||
|
||||
@@ -700,18 +812,63 @@ curl -s -X POST "$API/api/quick-start" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"caseName":"refactor-auth","mode":"claude","effort":"high"}' | jq
|
||||
|
||||
# 2b. Wait until that worker is actually READY (see rule 8): composer marker first,
|
||||
# first-run trust dialog only as the fallback. (Probing trust first and sending
|
||||
# a blind Enter misfires on re-runs: the dialog text stays in the buffer forever,
|
||||
# so the probe matches stale text and the Enter lands in a ready composer.)
|
||||
# Match single tokens: TUI text can reach the matcher without its spaces.
|
||||
until [ "$(curl -s "$API/api/sessions/$SID" | jq '.data.pid')" != null ]; do sleep 1; done
|
||||
R=$(curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=bypass' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=5000')
|
||||
if ! jq -e '.data.wait.matched' <<<"$R" >/dev/null; then
|
||||
T=$(curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=trust' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=2000')
|
||||
jq -e '.data.wait.matched' <<<"$T" >/dev/null && \
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" -H 'Content-Type: application/json' \
|
||||
-d '{"input":"\r","useMux":true}' # accept the first-run trust dialog
|
||||
curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=bypass' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=45000' >/dev/null
|
||||
fi
|
||||
|
||||
# 3. Send a prompt into a session (exactly-once: clientId + seq)
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"Run the test suite and summarize failures","useMux":true,"clientId":"agent-1","seq":1}'
|
||||
-d '{"input":"Run the test suite and summarize failures\r","useMux":true,"clientId":"agent-1","seq":1}'
|
||||
|
||||
# 4. Read the terminal back
|
||||
curl -s "$API/api/sessions/$SID/output" | jq -r '.data // .'
|
||||
# 4. Send a prompt and BLOCK until that turn is done (registers the wait before
|
||||
# writing, so it can't answer with the previous turn's idle state)
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"Run the test suite and summarize failures\r","useMux":true,
|
||||
"clientId":"agent-1","seq":2,"wait":"stop,exit","waitTimeout":60000}' \
|
||||
| jq '.data.wait' # -> {"signal":"stop","timedOut":false,"waitedMs":41230,...}
|
||||
# (`stop` is the definitive end-of-turn hook. Adding `idle` makes it resolve on a
|
||||
# spinner pause too, and on anything that redraws a ❯ prompt — like a dialog.)
|
||||
|
||||
# 5. Stream live events (session output, agent activity, status)
|
||||
# 4b. Timed out? That's a 200, not a failure. Loop over short waits.
|
||||
curl -s "$API/api/sessions/$SID/wait?until=stop,exit&timeout=60000" | jq '.data.wait'
|
||||
|
||||
# 4c. Or wait for a marker in the output (works for shell sessions too).
|
||||
# ⚠️ Unique per call (tmux repaints replay old screen text), and SPLIT so the
|
||||
# typed line never contains it: your own keystrokes echo into the output
|
||||
# stream, so an unsplit marker matches before the command has run. from=buffer
|
||||
# catches a marker that printed before the wait landed.
|
||||
N=$RANDOM
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" -H 'Content-Type: application/json' \
|
||||
-d "{\"input\":\"M=DONE; npm test; echo \${M}_$N rc=\$?\r\",\"useMux\":true}"
|
||||
curl -sG "$API/api/sessions/$SID/wait-output" \
|
||||
--data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=60000' | jq '.data.wait'
|
||||
|
||||
# 5. Read the terminal back. ⚠️ Use terminal?tail=, NOT /output: the latter's
|
||||
# textOutput is empty for every tmux-backed (i.e. every interactive) session.
|
||||
# tail counts BYTES, and what comes back is terminal data, ANSI included.
|
||||
curl -s "$API/api/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer'
|
||||
|
||||
# 6. Stream live events (session output, agent activity, status)
|
||||
curl -sN "$API/api/events" # Server-Sent Events
|
||||
|
||||
# 6. Schedule recurring work (cron-style job)
|
||||
# 7. Schedule recurring work (cron-style job)
|
||||
curl -s -X POST "$API/api/cron/jobs" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"name":"nightly-deps","agentType":"claude","workingDir":"/home/me/proj",
|
||||
@@ -719,11 +876,11 @@ curl -s -X POST "$API/api/cron/jobs" \
|
||||
"inputMode":"typed","scheduleType":"daily","dailyTime":"03:00",
|
||||
"enabled":true,"concurrencyPolicy":"warn_only"}' | jq
|
||||
|
||||
# 7. Inspect background sub-agents and their transcripts
|
||||
# 8. Inspect background sub-agents and their transcripts
|
||||
curl -s "$API/api/subagents" | jq '.data // .'
|
||||
curl -s "$API/api/subagents/$AID/transcript" | jq -r '.data // .'
|
||||
|
||||
# 8. Whole-system snapshot (sessions, settings, respawn, stats)
|
||||
# 9. Whole-system snapshot (sessions, settings, respawn, stats)
|
||||
curl -s "$API/api/status" | jq
|
||||
```
|
||||
|
||||
@@ -749,7 +906,7 @@ Codeman registers Claude Code hooks that `POST /api/hook-event` (`permission_pro
|
||||
|
||||
## API
|
||||
|
||||
REST over Fastify — **~190 handlers across 20 route modules**, plus an SSE stream and a WebSocket terminal channel. All responses use the `ApiResponse<T>` envelope (`{success, data}` / `{success, error, errorCode}`); `/api/v1/*` is a stable alias. A representative subset:
|
||||
REST over Fastify — **~200 handlers across 21 route modules**, plus an SSE stream and a WebSocket terminal channel. All responses use the `ApiResponse<T>` envelope (`{success, data}` / `{success, error, errorCode}`); `/api/v1/*` is a stable alias. A representative subset:
|
||||
|
||||
### Sessions
|
||||
|
||||
@@ -757,8 +914,11 @@ REST over Fastify — **~190 handlers across 20 route modules**, plus an SSE str
|
||||
| -------- | -------------------------- | ---------------------------------------------------------------------------------- |
|
||||
| `GET` | `/api/sessions` | List all |
|
||||
| `POST` | `/api/quick-start` | Create case + start session (`{caseName?, mode?, effort?, envOverrides?}`) |
|
||||
| `POST` | `/api/sessions/:id/input` | Send input (`{input, useMux?, clientId?, seq?}` — `clientId`+`seq` = exactly-once) |
|
||||
| `GET` | `/api/sessions/:id/output` | Read terminal output |
|
||||
| `POST` | `/api/sessions/:id/input` | Send input (`{input, useMux?, clientId?, seq?, wait?, waitTimeout?}`: `clientId`+`seq` = exactly-once; `wait` blocks until the turn ends) |
|
||||
| `GET` | `/api/sessions/:id/terminal` | Read terminal output (`?tail=<bytes>`, `?full=1`); the read path for interactive sessions |
|
||||
| `GET` | `/api/sessions/:id/output` | Parsed one-shot output (`textOutput` is empty for tmux-backed sessions) |
|
||||
| `GET` | `/api/sessions/:id/wait` | Block until a signal fires (`?until=stop,idle,exit&timeout=&fresh=`); a timeout is a `200` |
|
||||
| `GET` | `/api/sessions/:id/wait-output` | Block until a literal string appears (`?match=&nocase=&from=now\|buffer&timeout=`) |
|
||||
| `GET` | `/api/sessions/unified` | Unified live + history list (Session Manager) — `?q=&limit=` |
|
||||
| `POST` | `/api/sessions/:id/pin` | Pin/unpin in the Session Manager (`{pinned}`) |
|
||||
| `PUT` | `/api/session-order` | Sync tab order across devices (`{order: [ids]}`) |
|
||||
@@ -846,7 +1006,7 @@ flowchart TB
|
||||
end
|
||||
|
||||
subgraph External["External"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini</small>"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini / Pi</small>"]
|
||||
BG["Background Agents<br/><small>(Task tool)</small>"]
|
||||
end
|
||||
end
|
||||
@@ -877,13 +1037,19 @@ flowchart TB
|
||||
npm install
|
||||
npx tsx src/index.ts web # Dev mode
|
||||
npm run build # Production build
|
||||
npm run test:ci # Run tests (the CI suite; browser suites need extra setup)
|
||||
npm test # Run tests (same suite CI runs; browser/mobile/perf suites have their own commands)
|
||||
```
|
||||
|
||||
See [CLAUDE.md](./CLAUDE.md) for full documentation.
|
||||
|
||||
---
|
||||
|
||||
## Community
|
||||
|
||||
Questions, setup help, and ideas live in [GitHub Discussions](https://github.com/Ark0N/Codeman/discussions): the [Q&A section](https://github.com/Ark0N/Codeman/discussions/categories/q-a) answers the most common ones (phone access, overnight runs, updating), and the roadmap gets decided in [Ideas](https://github.com/Ark0N/Codeman/discussions/categories/ideas). Bugs go to [issues](https://github.com/Ark0N/Codeman/issues); reports usually get a response within a day, and every release credits its reporters and contributors by name. Want to contribute? [CONTRIBUTING.md](.github/CONTRIBUTING.md) has the map: skins, translations, and docs make great first PRs, and bigger features start life as a Discussion. And if you're proud of your rig, post it in [Show and tell](https://github.com/Ark0N/Codeman/discussions/300).
|
||||
|
||||
---
|
||||
|
||||
## Codebase Quality
|
||||
|
||||
The codebase went through a comprehensive 7-phase refactoring that eliminated god objects, centralized configuration, and established modular architecture:
|
||||
|
||||
+99
-29
@@ -5,7 +5,7 @@
|
||||
<h2 align="center">AI 编程智能体的任务控制中心</h2>
|
||||
|
||||
<p align="center">
|
||||
<em>Claude Code • OpenCode • Codex • Antigravity • Gemini • 终端 —— 统一仪表盘 • 任意设备</em>
|
||||
<em>Claude Code • OpenCode • Codex • Antigravity • Gemini • Pi • 终端 —— 统一仪表盘 • 任意设备</em>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
@@ -58,7 +58,7 @@ curl -fsSL https://getcodeman.com/install | bash
|
||||
- **重跑即更新。** 再次运行同一条命令即可原地更新已完成的安装:`~/.codeman/app` 中的本地改动会被 stash(绝不丢弃),运行中的服务会自动重启并校验。若首次安装中途失败,重跑会继续完成完整的安装流程。也可以使用 `install.sh update` 与 `install.sh uninstall`。
|
||||
- **CI / 无终端环境:** 没有终端时,涉及系统改动的步骤会带着说明中止,而不是静默执行;在自动化场景设置 `CODEMAN_NONINTERACTIVE=1` 即可批准这些步骤。
|
||||
|
||||
你至少需要安装一个 AI 编程 CLI —— [Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai)、[Codex](https://developers.openai.com/codex/cli)、[Antigravity](https://antigravity.google) 或 [Gemini CLI](https://github.com/google-gemini/gemini-cli)(任意组合均可;自 Google 面向消费者停售后,Gemini CLI 仅限企业版,Antigravity 是其继任者)。安装器会自动检测这五个中已安装的任意一个;若一个都没有,会提供安装 Claude Code 或 OpenCode 的选项,也可以选择跳过、稍后自行安装。安装完成后:
|
||||
你至少需要安装一个 AI 编程 CLI —— [Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai)、[Codex](https://developers.openai.com/codex/cli)、[Antigravity](https://antigravity.google)、[Gemini CLI](https://github.com/google-gemini/gemini-cli) 或 [Pi](https://pi.dev)(任意组合均可;自 Google 面向消费者停售后,Gemini CLI 仅限企业版,Antigravity 是其继任者)。安装器会自动检测这六个中已安装的任意一个;若一个都没有,会提供安装 Claude Code 或 OpenCode 的选项,也可以选择跳过、稍后自行安装。安装完成后:
|
||||
|
||||
```bash
|
||||
codeman web
|
||||
@@ -141,7 +141,7 @@ launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/com.codeman.web.plist
|
||||
wsl bash -c "curl -fsSL https://getcodeman.com/install | bash"
|
||||
```
|
||||
|
||||
Codeman 依赖 tmux,因此 Windows 用户需要 [WSL](https://learn.microsoft.com/en-us/windows/wsl/install)。如果还没装 WSL:在管理员 PowerShell 中运行 `wsl --install`,重启,打开 Ubuntu,然后在 WSL 内安装你偏好的 AI 编程 CLI([Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai)、[Codex](https://developers.openai.com/codex/cli)、[Antigravity](https://antigravity.google) 或 [Gemini CLI](https://github.com/google-gemini/gemini-cli))。安装完成后,即可从 Windows 浏览器访问 `http://localhost:3000`。
|
||||
Codeman 依赖 tmux,因此 Windows 用户需要 [WSL](https://learn.microsoft.com/en-us/windows/wsl/install)。如果还没装 WSL:在管理员 PowerShell 中运行 `wsl --install`,重启,打开 Ubuntu,然后在 WSL 内安装你偏好的 AI 编程 CLI([Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai)、[Codex](https://developers.openai.com/codex/cli)、[Antigravity](https://antigravity.google)、[Gemini CLI](https://github.com/google-gemini/gemini-cli) 或 [Pi](https://pi.dev))。安装完成后,即可从 Windows 浏览器访问 `http://localhost:3000`。
|
||||
|
||||
</details>
|
||||
|
||||
@@ -221,7 +221,7 @@ codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_
|
||||
| 字段 | 作用 |
|
||||
| ---------------------- | ------------------------------------------------------------------------------------------- |
|
||||
| **工作目录 / case** | 智能体操作的文件夹。「case」就是一个 Codeman 记住的命名工作目录。 |
|
||||
| **CLI / 运行模式** | `Claude`(默认)、`OpenCode`、`Codex`、`Antigravity`、`Gemini` 或 `Terminal`(普通 shell)。 |
|
||||
| **CLI / 运行模式** | `Claude`(默认)、`OpenCode`、`Codex`、`Antigravity`、`Gemini`、`Pi` 或 `Terminal`(普通 shell)。 |
|
||||
| **模型** | 每会话模型(App Settings → Claude Model)。软默认值 —— 会话内 `/model` 依然有效。 |
|
||||
| **Effort / Ultracode** | 推理力度(`low`–`max`),或用 `ultracode` 开启动态多智能体工作流。随时可用 `/effort` 切换。 |
|
||||
|
||||
@@ -394,7 +394,7 @@ PTY 输出 → 16ms 服务端批处理 → DEC 2026 包裹 → SSE → 客户端
|
||||
## 更多特性
|
||||
|
||||
- **自更新** —— systemd/launchd 管理下的 git-clone 安装可在 **App Settings → Updates** 中原地更新:它会检测最新发行版,自动暂存(stash)脏工作树,并在服务重启期间流式展示构建进度(npm 安装会被报告为不可更新)
|
||||
- **多 CLI** —— 每个会话可选 **Claude Code**、**OpenCode**、**Codex**、**Antigravity** 或 **Gemini**;环境变量前缀自动隔离(`CLAUDE_CODE_*`、`OPENCODE_*`、`CODEX_*`、`ANTIGRAVITY_*` 与 `GEMINI_*`/`GOOGLE_*`)。详见 [`docs/opencode-integration.md`](docs/opencode-integration.md)
|
||||
- **多 CLI** —— 每个会话可选 **Claude Code**、**OpenCode**、**Codex**、**Antigravity**、**Gemini** 或 **Pi**;环境变量前缀自动隔离(`CLAUDE_CODE_*`、`OPENCODE_*`、`CODEX_*`、`ANTIGRAVITY_*`、`PI_*` 与 `GEMINI_*`/`GOOGLE_*`)。详见 [`docs/opencode-integration.md`](docs/opencode-integration.md) 与 [`docs/pi-integration.md`](docs/pi-integration.md)
|
||||
- **Docker 会话** —— 在隔离且加固的容器中运行案例。**Create New** 上勾选一个复选框即可用合理的默认值启动容器并在其中启动智能体;同一案例的多个会话共享一个容器;可将容器连同工作区导出为可移植的 `.tar.gz`,迁移到另一台机器。详见 [`docs/docker-cases.md`](docs/docker-cases.md)
|
||||
- **远程 SSH 会话**:把案例指向另一台机器,让智能体在那里一个持久的远程 tmux 中运行:SSH 断连不中断任务、自动重连,还能发现并附着主机上已在运行的会话。详见 [`docs/remote-sessions.md`](docs/remote-sessions.md)
|
||||
- **Effort 与 Ultracode** —— 设置每会话的默认 effort(`low`–`max`),或启用 **ultracode**(动态多智能体工作流)。这些都只是软默认值 —— 会话中可随时用 `/effort` 切换。扩展思考预算也可配置
|
||||
@@ -416,7 +416,7 @@ PTY 输出 → 16ms 服务端批处理 → DEC 2026 包裹 → SSE → 客户端
|
||||
- **资源模板** —— 展开复选框可选 **Small / Medium / Large / GPU** 预设(内存、CPU、GPU),也可以完全自定义。**磁盘是弹性的** —— 存储随数据增长,没有固定上限。
|
||||
- **按案例共享容器** —— 多个会话可以 `docker exec` 进同一个容器;结束某个会话绝不会影响其他会话所在的容器。
|
||||
- **默认加固** —— 非 root、`--cap-drop ALL`、`no-new-privileges`、PID/内存上限,绝不使用 `--privileged` 或 docker socket;**密封(sealed)** 配置(不注入主机凭据、关闭网络)只需一个开关。
|
||||
- **无感认证、凭据隔离** —— 主机上的 Claude / Codex / Antigravity / Gemini / OpenCode 登录在容器内开箱即用:凭据在启动时以只读种子方式复制注入,onboarding/信任提示已预先答复,不会弹出登录向导。容器保留自己的副本,绝不回写主机的凭据存储;跨边界共享的只有对话转录,导出文件也绝不包含机密。
|
||||
- **无感认证、凭据隔离** —— 主机上的 Claude / Codex / Antigravity / Gemini / OpenCode / Pi 登录在容器内开箱即用:凭据在启动时以只读种子方式复制注入,onboarding/信任提示已预先答复,不会弹出登录向导。容器保留自己的副本,绝不回写主机的凭据存储;跨边界共享的只有对话转录,导出文件也绝不包含机密。
|
||||
- **迁移到另一台机器** —— 把容器的完整环境(工具链 + 工作区)导出为可移植的 `.tar.gz`,在另一台机器上导入到新案例即可继续。
|
||||
- **持久耐用** —— Codeman 重启后重连会回到同一个存活的智能体;容器停止/重启后则从绑定挂载的转录恢复对话。
|
||||
|
||||
@@ -602,7 +602,7 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI
|
||||
|
||||
### 输入、文件与响应头
|
||||
|
||||
- **模式校验的输入** —— 每个 API 请求体都用 Zod v4 模式检查;一个 `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` 环境变量前缀允许列表把控每个 CLI 能接收哪些设置
|
||||
- **模式校验的输入** —— 每个 API 请求体都用 Zod v4 模式检查;一个 `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` / `PI_*` 环境变量前缀允许列表把控每个 CLI 能接收哪些设置
|
||||
- **路径限定** —— 文件路由在边界检查前先 `realpath`(无 TOCTOU);`..`、绝对路径、以及解析到工作目录之外的符号链接都会被拒绝。上限:10 MB 文本预览 / 50 MB 原始与下载;`/api/download` 对敏感路径(`.env`、`*credentials*`、`~/.ssh/`、`.aws/credentials`)做黑名单。SVG/HTML 以 `octet-stream` + `nosniff` + attachment 提供,因此会被下载而非执行
|
||||
- **安全响应头** —— `Content-Security-Policy`(`default-src 'self'`,每个例外都逐条列举)、`X-Content-Type-Options: nosniff`、`X-Frame-Options: SAMEORIGIN`、HTTPS 下的 HSTS,以及**仅**对 `localhost` / `127.0.0.1` / `::1` 反射的 CORS
|
||||
|
||||
@@ -657,6 +657,16 @@ sc -l # 列出会话
|
||||
|
||||
面向不经浏览器控制 Codeman 的 AI 智能体与自动化:一个拉起工作会话的智能体、一个 CI 机器人,或是**运行在 Codeman 会话*内部*、编排其他会话的 Claude Code**。UI 能做的一切都是 HTTP + CLI,因此智能体也能做。
|
||||
|
||||
> **捷径:装上打包好的智能体技能。** 下面这一整套(外加多工作会话的实战配方)已经作为 Claude Code 技能随仓库发布在 [`skills/codeman`](skills/codeman/SKILL.md),会话内部的智能体不必等你把文档粘进提示词就能驱动 Codeman。三种获取方式:
|
||||
>
|
||||
> - `npx skills add Ark0N/Codeman --skill codeman -g`:全局安装,任何支持技能的智能体都能用
|
||||
> - `codeman skill install`(全局)或 `codeman skill install --case <name>`:给那些从 npm 安装、从未克隆过仓库的用户;`codeman skill uninstall` 可撤销
|
||||
> - **App Settings → Agent Skill**(`agentSkillEnabled`,默认关闭):开启后,Codeman 会在每次于某个 case 中创建 Claude 会话时把技能注入该 case;case 里用户自己写的 `skills/codeman` 永远不会被覆盖
|
||||
>
|
||||
> 全局安装(`codeman skill install` 或 `npx skills add`)会被**本机每一个新建的 Claude Code 会话**读到,无论它在不在 Codeman 里。技能自带门禁:不在 Codeman 会话中(`CODEMAN_MUX` 未设置)时它拒绝动作,所以全局装上它对无关会话没有代价。
|
||||
>
|
||||
> ⚠️ 把 `agentSkillEnabled` 关回去**不会删掉已经注入的副本**(在创建时做清扫,会把技能从共用同一个 `.claude/` 目录的其他活动会话脚下抽走)。要删就按 case 删:`codeman skill uninstall --case <name>`。
|
||||
|
||||
### 检测自己身处 Codeman 内部
|
||||
|
||||
当 CLI 运行在 Codeman 受管会话中时,以下环境变量会被设置 —— 读取它们,别硬编码任何东西:
|
||||
@@ -670,15 +680,21 @@ sc -l # 列出会话
|
||||
|
||||
### 行路规则(POST 之前先读)
|
||||
|
||||
1. **只发单行输入。** 编程输入会作为字面文本 **+ Enter** 一次性发送。多行字符串会破坏智能体 TUI(Ink)—— 发送一行,或拆成多次调用。
|
||||
1. **只发单行输入,而且必须以 `\r` 结尾。** 编程输入按字面文本发送,**只有当输入里含回车符时才会触发 Enter**:`{"input":"run tests\r"}`。少了 `\r`,文本就停在会话的输入框里不被提交(同一次调用里的 `wait` 还会在一个压根没开始的回合上耗满整个超时)。内嵌的换行会被剥掉而不是报错,因此 `"echo A\necho B\r"` 执行的是拼起来的 `echo Aecho B`:一次调用只发一行。
|
||||
2. **让输入幂等。** 在 `POST …/input` 上带上稳定的 `clientId` 和按会话单调递增的 `seq`。服务端会去重,因此连接中断后的重试不会重复投递提示。
|
||||
3. **认证。** 若设置了 `CODEMAN_PASSWORD`,发送 HTTP Basic 认证(用户 `admin` 或 `CODEMAN_USERNAME`)或 `codeman_session` cookie。默认的环回安装无密码。缺失的 `Origin` 头被允许,因此普通 `curl` 可用;跨站的浏览器 origin 会被拒绝(CSRF 防护)。
|
||||
3. **认证。** 若设置了 `CODEMAN_PASSWORD`,发送 HTTP Basic 认证(用户 `admin` 或 `CODEMAN_USERNAME`)或 `codeman_session` cookie。默认的环回安装无密码。缺失的 `Origin` 头被允许,因此普通 `curl` 可用;跨站的浏览器 origin 会被拒绝(CSRF 防护)。⚠️ `401` 回的是裸字符串 `Unauthorized`,**不是** JSON 信封,直接喂给 `jq` 只会抛解析错误而看不到真正的失败原因:先看状态码,再解析。
|
||||
4. **响应信封。** 多数端点返回 `{ "success": true, "data": … }`(错误:`{ "success": false, "error", "errorCode" }`)。少数遗留 GET 返回裸响应体 —— **两种都要处理**(`body.data ?? body`)。
|
||||
5. **`/api/v1/*`** 是 `/api/*` 的稳定别名。
|
||||
6. **用等待代替轮询,别把超时当成错误。** 等待类端点在没等到事情发生时也以 HTTP `200` 加 `wait.timedOut: true` 应答,所以要循环调用短等待(默认 60 秒),而不是发一个超长的调用:隧道会掐断空闲连接。`wait.timeoutMs` 告诉你服务端钳制之后真正采用的超时(上限 600 秒)。
|
||||
7. **只有 `claude` 会话会发出 `stop` 与 `blocked`。** 这两个来自 Claude Code hook;`shell` 与外部 CLI(opencode/codex/gemini/antigravity/pi)只接受 `idle`、`working` 与 `exit`。在这些模式上显式索要 `stop` 会得到 `400`;不传 `until` 则永远安全。⚠️ `shell` 会话的 `idle` 只在启动时触发**一次**,此后再也不会,所以在那里用「发送并等待」只能等到超时:没有 hook 的会话请用 `wait-output` 标记来同步。
|
||||
8. **没有任何东西会报告「就绪」,得自己显式等。** 新会话在 PID 出现之前一律回答 `{"signal":"exit","immediate":true}`(意思是*还没启动*,不是*崩了*),而全新 case 里的 `claude` 工作会话接着会停在 CLI 的信任对话框上。此时给它发提示,等待会在约 2 秒后因 `idle` 解除,看上去和一个跑完的回合一模一样,而文本其实卡在对话框里。下面的配方 2b 就是避开它的顺序。
|
||||
|
||||
### 常用配方
|
||||
|
||||
```bash
|
||||
# 每个 Codeman 会话里都自动设好了 CODEMAN_API_URL,协议也是对的。
|
||||
# 下面的兜底值适用于标准安装;在 --https 安装上请自己写 https:// 的地址,
|
||||
# 并给每个 curl 加上 -k(自签名证书)。
|
||||
API="${CODEMAN_API_URL:-http://127.0.0.1:3000}"
|
||||
# (若设置了密码,给每个调用加上 -u admin:"$CODEMAN_PASSWORD")
|
||||
|
||||
@@ -690,18 +706,69 @@ curl -s -X POST "$API/api/quick-start" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"caseName":"refactor-auth","mode":"claude","effort":"high"}' | jq
|
||||
|
||||
# 2b. 等这个工作会话真正就绪(见规则 8):先探输入框的标记,信任对话框只作兜底。
|
||||
# (反过来先探信任对话框、再盲发一个 Enter,在重复运行时会误伤:对话框的文字
|
||||
# 会一直留在缓冲区里,探测因此匹配到旧文本,而那个 Enter 落进了已经就绪的输入框。)
|
||||
# 匹配单个词:TUI 的文字到达匹配器时可能已经丢掉了词间空格。
|
||||
until [ "$(curl -s "$API/api/sessions/$SID" | jq '.data.pid')" != null ]; do sleep 1; done
|
||||
R=$(curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=bypass' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=5000')
|
||||
if ! jq -e '.data.wait.matched' <<<"$R" >/dev/null; then
|
||||
T=$(curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=trust' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=2000')
|
||||
jq -e '.data.wait.matched' <<<"$T" >/dev/null && \
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" -H 'Content-Type: application/json' \
|
||||
-d '{"input":"\r","useMux":true}' # 接受首次运行的信任对话框
|
||||
curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=bypass' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=45000' >/dev/null
|
||||
fi
|
||||
|
||||
# 3. 向会话发送提示(精确一次:clientId + seq)
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"Run the test suite and summarize failures","useMux":true,"clientId":"agent-1","seq":1}'
|
||||
-d '{"input":"Run the test suite and summarize failures\r","useMux":true,"clientId":"agent-1","seq":1}'
|
||||
|
||||
# 4. 读回终端内容
|
||||
curl -s "$API/api/sessions/$SID/output" | jq -r '.data // .'
|
||||
# 4. 发送提示并阻塞到这一回合结束(先注册等待再写入,因此不会拿上一回合的状态来应答)
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"Run the test suite and summarize failures\r","useMux":true,
|
||||
"clientId":"agent-1","seq":2,"wait":"stop,exit","waitTimeout":60000}' \
|
||||
| jq '.data.wait' # -> {"signal":"stop","timedOut":false,"waitedMs":41230,...}
|
||||
# (`stop` 是回合结束的权威 hook。加上 `idle` 会让它在转圈停顿时也解除,
|
||||
# 任何重画出 ❯ 提示符的东西同理,比如一个对话框。)
|
||||
|
||||
# 5. 流式接收实时事件(会话输出、智能体活动、状态)
|
||||
# 4b. 超时了?那是 200,不是失败。循环调用短等待即可。
|
||||
curl -s "$API/api/sessions/$SID/wait?until=stop,exit&timeout=60000" | jq '.data.wait'
|
||||
|
||||
# 4c. 或者等输出里出现某个标记(shell 会话也适用)。
|
||||
# ⚠️ 每次调用都要用不同的标记(tmux 重画会重放旧屏幕文字),并且把标记拆开写,
|
||||
# 让敲进去的那一行本身不包含它:你自己的按键会回显进输出流,不拆开的标记会在
|
||||
# 命令还没跑之前就匹配上。from=buffer 用来接住在等待落地之前就已打印的标记。
|
||||
N=$RANDOM
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" -H 'Content-Type: application/json' \
|
||||
-d "{\"input\":\"M=DONE; npm test; echo \${M}_$N rc=\$?\r\",\"useMux\":true}"
|
||||
curl -sG "$API/api/sessions/$SID/wait-output" \
|
||||
--data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=60000' | jq '.data.wait'
|
||||
|
||||
# 5. 读回答案。claude / codex 会话用 last-response:它取自 transcript 而不是屏幕,
|
||||
# 因此不带 TUI 的画框与重画噪声。⚠️ 要轮询,别只读一次:transcript 落盘比 stop
|
||||
# 信号稍晚,紧跟着「发送并等待」返回后立刻读,常常拿到空串。
|
||||
for _ in $(seq 1 10); do
|
||||
TXT=$(curl -s "$API/api/sessions/$SID/last-response" | jq -r '.data.text')
|
||||
[ -n "$TXT" ] && break; sleep 1
|
||||
done
|
||||
printf '%s\n' "$TXT"
|
||||
|
||||
# 5b. 其他模式(shell/opencode/gemini/antigravity/pi)没有 transcript,读终端。
|
||||
# ⚠️ 用 terminal?tail=,不要用 /output:后者的 textOutput 对每个由 tmux 承载的
|
||||
# (也就是每个交互式)会话都是空的。tail 按字节计,返回的是含 ANSI 的终端数据。
|
||||
curl -s "$API/api/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer'
|
||||
|
||||
# 6. 流式接收实时事件(会话输出、智能体活动、状态)
|
||||
curl -sN "$API/api/events" # Server-Sent Events
|
||||
|
||||
# 6. 调度周期性工作(cron 风格任务)
|
||||
# 7. 调度周期性工作(cron 风格任务)
|
||||
curl -s -X POST "$API/api/cron/jobs" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"name":"nightly-deps","agentType":"claude","workingDir":"/home/me/proj",
|
||||
@@ -709,11 +776,11 @@ curl -s -X POST "$API/api/cron/jobs" \
|
||||
"inputMode":"typed","scheduleType":"daily","dailyTime":"03:00",
|
||||
"enabled":true,"concurrencyPolicy":"warn_only"}' | jq
|
||||
|
||||
# 7. 查看后台子智能体及其活动记录
|
||||
# 8. 查看后台子智能体及其活动记录
|
||||
curl -s "$API/api/subagents" | jq '.data // .'
|
||||
curl -s "$API/api/subagents/$AID/transcript" | jq -r '.data // .'
|
||||
|
||||
# 8. 全系统快照(会话、设置、重生、统计)
|
||||
# 9. 全系统快照(会话、设置、重生、统计)
|
||||
curl -s "$API/api/status" | jq
|
||||
```
|
||||
|
||||
@@ -739,20 +806,23 @@ Codeman 会注册 Claude Code hook,它们 `POST /api/hook-event`(`permission
|
||||
|
||||
## API
|
||||
|
||||
基于 Fastify 的 REST —— **20 个路由模块中约 190 个处理器**,外加一条 SSE 流和一条 WebSocket 终端通道。所有响应都使用 `ApiResponse<T>` 信封(`{success, data}` / `{success, error, errorCode}`);`/api/v1/*` 是稳定别名。以下是一个有代表性的子集:
|
||||
基于 Fastify 的 REST —— **21 个路由模块中约 200 个处理器**,外加一条 SSE 流和一条 WebSocket 终端通道。所有响应都使用 `ApiResponse<T>` 信封(`{success, data}` / `{success, error, errorCode}`);`/api/v1/*` 是稳定别名。以下是一个有代表性的子集:
|
||||
|
||||
### 会话(Sessions)
|
||||
|
||||
| 方法 | 端点 | 说明 |
|
||||
| -------- | -------------------------- | ------------------------------------------------------------------------------ |
|
||||
| `GET` | `/api/sessions` | 列出全部 |
|
||||
| `POST` | `/api/quick-start` | 创建 case + 启动会话(`{caseName?, mode?, effort?, envOverrides?}`) |
|
||||
| `POST` | `/api/sessions/:id/input` | 发送输入(`{input, useMux?, clientId?, seq?}` —— `clientId`+`seq` = 精确一次) |
|
||||
| `GET` | `/api/sessions/:id/output` | 读取终端输出 |
|
||||
| `GET` | `/api/sessions/unified` | 统一的活动 + 历史清单(会话管理器):`?q=&limit=` |
|
||||
| `POST` | `/api/sessions/:id/pin` | 在会话管理器中置顶 / 取消置顶(`{pinned}`) |
|
||||
| `PUT` | `/api/session-order` | 跨设备同步标签顺序(`{order: [ids]}`) |
|
||||
| `DELETE` | `/api/sessions/:id` | 删除会话 |
|
||||
| 方法 | 端点 | 说明 |
|
||||
| -------- | ------------------------------- | ---------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `GET` | `/api/sessions` | 列出全部 |
|
||||
| `POST` | `/api/quick-start` | 创建 case + 启动会话(`{caseName?, mode?, effort?, envOverrides?}`) |
|
||||
| `POST` | `/api/sessions/:id/input` | 发送输入(`{input, useMux?, clientId?, seq?, wait?, waitTimeout?}`:`clientId`+`seq` = 精确一次;`wait` 阻塞到这一回合结束) |
|
||||
| `GET` | `/api/sessions/:id/terminal` | 读取终端输出(`?tail=<bytes>`、`?full=1`):交互式会话的读取路径 |
|
||||
| `GET` | `/api/sessions/:id/output` | 一次性的解析输出(tmux 承载的会话里 `textOutput` 为空) |
|
||||
| `GET` | `/api/sessions/:id/wait` | 阻塞到某个信号触发(`?until=stop,idle,exit&timeout=&fresh=`);超时是 `200` |
|
||||
| `GET` | `/api/sessions/:id/wait-output` | 阻塞到某个字面串出现(`?match=&nocase=&from=now\|buffer&timeout=`) |
|
||||
| `GET` | `/api/sessions/unified` | 统一的活动 + 历史清单(会话管理器):`?q=&limit=` |
|
||||
| `POST` | `/api/sessions/:id/pin` | 在会话管理器中置顶 / 取消置顶(`{pinned}`) |
|
||||
| `PUT` | `/api/session-order` | 跨设备同步标签顺序(`{order: [ids]}`) |
|
||||
| `DELETE` | `/api/sessions/:id` | 删除会话 |
|
||||
|
||||
### 重生(Respawn)
|
||||
|
||||
@@ -836,7 +906,7 @@ flowchart TB
|
||||
end
|
||||
|
||||
subgraph External["外部"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini</small>"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini / Pi</small>"]
|
||||
BG["后台智能体<br/><small>(Task 工具)</small>"]
|
||||
end
|
||||
end
|
||||
@@ -867,7 +937,7 @@ flowchart TB
|
||||
npm install
|
||||
npx tsx src/index.ts web # 开发模式
|
||||
npm run build # 生产构建
|
||||
npm run test:ci # 运行测试(CI 套件;浏览器套件需要额外环境)
|
||||
npm test # 运行测试(与 CI 相同;浏览器/移动端/性能套件另有独立命令)
|
||||
```
|
||||
|
||||
完整文档见 [CLAUDE.md](./CLAUDE.md)。
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
/**
|
||||
* The test suites that `npm test` deliberately does NOT run, in one place.
|
||||
*
|
||||
* Why this file exists: the exclusion list used to live only in
|
||||
* config/vitest.ci.config.ts, as literals. Anything excluded there was
|
||||
* therefore reachable only by running the everything-config by hand and reading
|
||||
* past its failures — and a newly excluded file was reachable by nothing at
|
||||
* all, silently, because nothing pointed at it. Both configs now derive their
|
||||
* globs from the arrays below, so adding a suite here puts it in exactly one
|
||||
* runner and takes it out of exactly one gate.
|
||||
*
|
||||
* Adding a new test that cannot run in CI: put its glob in the array that
|
||||
* describes WHY it cannot, not in whichever one is shortest.
|
||||
*/
|
||||
|
||||
/**
|
||||
* Playwright-driven: needs chromium and, in most cases, a live Codeman server
|
||||
* on a real port. Deterministic where the environment provides both, which is
|
||||
* why these are a runnable suite (`npm run test:browser`) rather than skipped.
|
||||
*/
|
||||
export const BROWSER_TEST_GLOBS = [
|
||||
'test/inline-rename.test.ts',
|
||||
'test/opencode-resize.test.ts',
|
||||
'test/webgl-fallback.test.ts',
|
||||
'test/terminal-copy-shortcut.test.ts',
|
||||
'test/codex-predictive-echo.test.ts', // also needs a real codex binary
|
||||
];
|
||||
|
||||
/**
|
||||
* Wall-clock benchmarks. They assert on durations, so a loaded shared runner
|
||||
* fails them for reasons that have nothing to do with the diff under test.
|
||||
*/
|
||||
export const PERF_TEST_GLOBS = ['test/perf-*.test.ts'];
|
||||
|
||||
/**
|
||||
* Browser + visual regression: chromium AND environment-specific PNG baselines
|
||||
* that are generated per machine. Has its own config
|
||||
* (test/mobile/vitest.config.ts) because it needs serial execution, a longer
|
||||
* timeout and the `pretest:mobile` vendor step — run it with
|
||||
* `npm run test:mobile`, not through the configs here.
|
||||
*/
|
||||
export const MOBILE_TEST_GLOBS = ['test/mobile/**'];
|
||||
|
||||
/** Everything `npm test` skips. */
|
||||
export const NON_CI_TEST_GLOBS = [...MOBILE_TEST_GLOBS, ...PERF_TEST_GLOBS, ...BROWSER_TEST_GLOBS];
|
||||
@@ -0,0 +1,34 @@
|
||||
import { resolve } from 'node:path';
|
||||
import { defineConfig } from 'vitest/config';
|
||||
import { BROWSER_TEST_GLOBS } from './test-suites';
|
||||
|
||||
const root = resolve(import.meta.dirname, '..');
|
||||
|
||||
/**
|
||||
* The Playwright-driven suite `npm test` skips — `npm run test:browser`.
|
||||
*
|
||||
* Needs chromium and, for most of these, a live Codeman server on a real port;
|
||||
* codex-predictive-echo also needs a real codex binary. Expect failures where
|
||||
* the machine cannot provide those, and read them as "not runnable here", not
|
||||
* as a regression.
|
||||
*
|
||||
* The mobile suite is NOT here: it needs per-machine PNG baselines, serial
|
||||
* execution and the `pretest:mobile` vendor step, so it keeps its own config
|
||||
* (test/mobile/vitest.config.ts) behind `npm run test:mobile`.
|
||||
*
|
||||
* fileParallelism stays off for the same reason as every other config in this
|
||||
* directory: these bind real ports and drive real tmux sessions, and two files
|
||||
* doing that at once fail each other rather than the code.
|
||||
*/
|
||||
export default defineConfig({
|
||||
test: {
|
||||
root,
|
||||
globals: true,
|
||||
environment: 'node',
|
||||
include: BROWSER_TEST_GLOBS,
|
||||
setupFiles: ['./test/setup.ts'],
|
||||
fileParallelism: false,
|
||||
testTimeout: 60000,
|
||||
teardownTimeout: 60000,
|
||||
},
|
||||
});
|
||||
@@ -1,13 +1,17 @@
|
||||
import { resolve } from 'node:path';
|
||||
import { defineConfig, configDefaults } from 'vitest/config';
|
||||
import { NON_CI_TEST_GLOBS } from './test-suites';
|
||||
|
||||
const root = resolve(import.meta.dirname, '..');
|
||||
|
||||
/**
|
||||
* CI test config — same as vitest.config.ts but EXCLUDES the browser-driven
|
||||
* mobile suite (test/mobile/**). Those are Playwright visual-regression tests
|
||||
* that need a live server + chromium + environment-specific PNG baselines, so
|
||||
* they are run/maintained separately and are not part of the CI gate.
|
||||
* The default gate — what `npm test` and CI both run.
|
||||
*
|
||||
* Same as vitest.config.ts but EXCLUDES the suites that cannot pass on an
|
||||
* arbitrary machine: browser-driven (Playwright + chromium), visual-regression
|
||||
* (per-machine PNG baselines) and wall-clock perf. Those are not unmaintained;
|
||||
* they have their own runners (`test:browser`, `test:mobile`, `test:perf`).
|
||||
* See config/test-suites.ts for the list and the reason behind each entry.
|
||||
*
|
||||
* Keep the rest in sync with config/vitest.config.ts.
|
||||
*/
|
||||
@@ -17,15 +21,7 @@ export default defineConfig({
|
||||
globals: true,
|
||||
environment: 'node',
|
||||
include: ['test/**/*.test.ts'],
|
||||
exclude: [
|
||||
...configDefaults.exclude,
|
||||
'test/mobile/**', // browser/visual (Playwright + chromium)
|
||||
'test/perf-*.test.ts', // timing-sensitive perf benchmarks (flaky in CI)
|
||||
'test/inline-rename.test.ts', // browser (Playwright)
|
||||
'test/opencode-resize.test.ts', // browser (Playwright)
|
||||
'test/webgl-fallback.test.ts', // browser (Playwright)
|
||||
'test/terminal-copy-shortcut.test.ts', // browser (Playwright)
|
||||
],
|
||||
exclude: [...configDefaults.exclude, ...NON_CI_TEST_GLOBS],
|
||||
setupFiles: ['./test/setup.ts'],
|
||||
fileParallelism: false,
|
||||
testTimeout: 30000,
|
||||
|
||||
@@ -3,6 +3,17 @@ import { defineConfig } from 'vitest/config';
|
||||
|
||||
const root = resolve(import.meta.dirname, '..');
|
||||
|
||||
/**
|
||||
* EVERY test in the repo, including the ones that cannot pass on an arbitrary
|
||||
* machine — `npm run test:all`. Reach for it when you want the complete picture
|
||||
* and are prepared to read past environmental failures.
|
||||
*
|
||||
* This is NOT what `npm test` runs. On a machine without chromium, a free port
|
||||
* or per-machine PNG baselines this config fails ~87 tests on a clean master,
|
||||
* which makes it useless as a pass/fail signal: the default gate is
|
||||
* config/vitest.ci.config.ts, and the suites it leaves out each have their own
|
||||
* runner (`test:browser`, `test:perf`, `test:mobile`). See config/test-suites.ts.
|
||||
*/
|
||||
export default defineConfig({
|
||||
test: {
|
||||
root,
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
import { resolve } from 'node:path';
|
||||
import { defineConfig } from 'vitest/config';
|
||||
import { PERF_TEST_GLOBS } from './test-suites';
|
||||
|
||||
const root = resolve(import.meta.dirname, '..');
|
||||
|
||||
/**
|
||||
* The wall-clock benchmarks `npm test` skips — `npm run test:perf`.
|
||||
*
|
||||
* These assert on durations, so run them on an otherwise idle machine: a loaded
|
||||
* runner fails them for reasons that have nothing to do with the diff under
|
||||
* test, which is exactly why they are not part of the default gate.
|
||||
*/
|
||||
export default defineConfig({
|
||||
test: {
|
||||
root,
|
||||
globals: true,
|
||||
environment: 'node',
|
||||
include: PERF_TEST_GLOBS,
|
||||
setupFiles: ['./test/setup.ts'],
|
||||
fileParallelism: false,
|
||||
testTimeout: 60000,
|
||||
teardownTimeout: 60000,
|
||||
},
|
||||
});
|
||||
+10
-1
@@ -44,6 +44,13 @@ RUN curl -fsSL https://antigravity.google/cli/install.sh | bash -s -- --dir /usr
|
||||
&& chmod 755 /usr/local/bin/agy \
|
||||
&& agy --version
|
||||
|
||||
# Pi (pi.dev). Upstream documents --ignore-scripts (pi needs no lifecycle scripts);
|
||||
# kept out of the shared npm block above so the flag cannot silently change how the
|
||||
# other four CLIs install.
|
||||
RUN npm install -g --ignore-scripts @earendil-works/pi-coding-agent \
|
||||
&& npm cache clean --force \
|
||||
&& pi --version
|
||||
|
||||
# `agent` user (gid 0) with an arbitrary-uid-writable HOME. The uid is
|
||||
# auto-assigned (node:22-slim already occupies uid 1000 with its `node` user); at
|
||||
# runtime Codeman overrides with `--user <hostUid>:0` on Linux, so the baked uid
|
||||
@@ -61,9 +68,11 @@ ENV HOME=/home/agent
|
||||
# transcript/rollout dirs (`.claude/projects`, `.codex/sessions`) are bind-mounted from
|
||||
# the host. (gemini/gcloud/opencode are whole seed-copies and need no pre-created dir;
|
||||
# Antigravity nests its state inside `.gemini/antigravity-cli`, so it rides that seed.)
|
||||
# `.pi/agent` IS pre-created: pi is seeded per-FILE (auth/settings/trust/models), and a
|
||||
# per-file seed copy, unlike a whole-dir one, does not create its parent directory.
|
||||
RUN useradd -g 0 -m -d /home/agent -s /bin/bash agent \
|
||||
&& mkdir -p /home/agent/.npm /home/agent/.cache /home/agent/.config /home/agent/.codeman \
|
||||
/home/agent/.claude/projects /home/agent/.codex/sessions \
|
||||
/home/agent/.claude/projects /home/agent/.codex/sessions /home/agent/.pi/agent \
|
||||
&& chgrp -R 0 /home/agent \
|
||||
&& chmod -R g=u /home/agent
|
||||
|
||||
|
||||
@@ -0,0 +1,759 @@
|
||||
# Agent Control Plan: skill packaging + wait primitives
|
||||
|
||||
**Status**: steps 1 to 8 DONE and RELEASED. The wait primitives and the skill itself
|
||||
(steps 1 to 5) shipped in **1.13.0**; the `codeman skill install` CLI, per-case injection
|
||||
and `agentSkillEnabled` (step 6) shipped in **1.14.1** and were republished with fixes in
|
||||
**1.14.2**. Steps 1 to 5 were multi-round verified on 2026-08-08, step 6 on 2026-08-09;
|
||||
see [§7 Build log](#7-build-log-what-actually-happened) for what shipped, what each
|
||||
verification round found, and the two items that genuinely remain open (§2.4's footgun
|
||||
guard and the Part 3 deferrals).
|
||||
|
||||
**Date**: 2026-08-08
|
||||
**Scope**: Part 1 (agent skill) and Part 2 (wait primitives) were specified and built.
|
||||
Parts 3 to 5 are captured so they are not lost, but remain deliberately deferred.
|
||||
|
||||
---
|
||||
|
||||
## 0. Where this came from: what herdr does
|
||||
|
||||
[herdr](https://github.com/herdrdev/herdr) (Rust, Apache-2.0, ~25.8k stars) is a terminal
|
||||
multiplexer built around AI coding agents. Relevant findings from the research pass:
|
||||
|
||||
| Capability | How herdr does it |
|
||||
| --------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Agent state | Four states (`idle`, `working`, `blocked`, `done`) that roll up pane to tab to workspace in a sidebar |
|
||||
| Detection | Lifecycle hooks where the agent supports them (it names Pi and MastraCode), otherwise TOML manifests matched against a live bottom-buffer snapshot. Bundled manifests plus remote updates from herdr.dev, local overrides win |
|
||||
| Control API | Newline-delimited JSON over a Unix socket (`~/.config/herdr/sessions/<name>/herdr.sock`), `{"id":"req_1","method":"pane.split","params":{}}`, dot-notation methods, plus long-lived event subscriptions |
|
||||
| Discoverability | `herdr api schema` prints a machine-readable schema |
|
||||
| Agent skill | `npx skills add herdrdev/herdr --skill herdr -g`, a SKILL.md wrapping the CLI, guarded by `test "${HERDR_ENV:-}" = 1` so an agent outside a herdr pane refuses to act |
|
||||
| Persistence | Background server, detach with `ctrl+b q`, snapshot restore of workspaces/tabs/panes/cwd/layout, experimental screen-history replay, agent resume via native session ids, live PTY handoff across server replacement |
|
||||
| Plugins | `herdr-plugin.toml` manifest, actions, event hooks, plugin panes, link handlers, GitHub-topic marketplace index |
|
||||
|
||||
The commands the skill teaches the agent:
|
||||
|
||||
| Group | Commands |
|
||||
| --------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| workspace | `workspace list`, `workspace create` |
|
||||
| tab | `tab list --workspace <id>`, `tab create` |
|
||||
| pane | `pane current`, `pane list`, `pane layout`, `pane split --current --direction right --cwd <path> --no-focus`, `pane run <id> "<cmd>"`, `pane wait-output <id> --match/--regex <p> --timeout <ms>`, `pane read <id> --source visible\|recent\|detection` |
|
||||
| agent | `agent list`, `agent start <name> --kind <type> --pane <id>`, `agent prompt <name> "<text>" --wait --timeout <ms>`, `agent wait <name> --until <state> --timeout <ms>`, `agent send-keys`, `agent get`, `agent read` |
|
||||
|
||||
### The honest comparison
|
||||
|
||||
herdr and Codeman are not the same product. herdr is a local, keyboard-first multiplexer with
|
||||
no server, no web UI, and no autonomy layer. Codeman is a server with a browser and mobile UI,
|
||||
remote and Docker cases, respawn, Ralph, cron, and the orchestrator, none of which herdr has.
|
||||
|
||||
What herdr genuinely does better is being **callable by the agent running inside it**. For
|
||||
Codeman that is a packaging problem plus one missing primitive, not an architecture problem.
|
||||
|
||||
---
|
||||
|
||||
## 1. Gap analysis
|
||||
|
||||
| herdr capability | Codeman equivalent today | Gap |
|
||||
| ---------------------------- | ------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------- |
|
||||
| `pane split` + `agent start` | `POST /api/quick-start`, `POST /api/sessions` | none, already there |
|
||||
| `agent prompt` | `POST /api/sessions/:id/input` with `clientId`+`seq` exactly-once | no `--wait` |
|
||||
| `pane read` | `GET /api/sessions/:id/output`, `GET /api/sessions/:id/terminal?full=1` | none |
|
||||
| `agent list` / `agent get` | `GET /api/sessions`, `GET /api/sessions/unified`, `GET /api/status` | none |
|
||||
| `agent wait --until <state>` | SSE only (`/api/events`) | **missing**, and SSE is impractical from a shell tool |
|
||||
| `pane wait-output --match` | nothing | **missing** |
|
||||
| Skill file | README section "Driving Codeman from an Agent" | **not packaged**, an agent will never find it |
|
||||
| Env guard `HERDR_ENV=1` | `CODEMAN_MUX=1`, `CODEMAN_API_URL`, `CODEMAN_SESSION_ID` already exported at spawn | none, the guard variables exist |
|
||||
| `blocked` state | hook events (`permission_prompt`, `elicitation_dialog`) plus CSS classes plus the phone overview NEEDS YOU section | not in the wire contract (`SessionStatus = 'idle' \| 'busy' \| 'stopped' \| 'error'`) |
|
||||
| `api schema` | hand-written `docs/api-reference.md` | no machine-readable schema |
|
||||
| Detection manifests | hardcoded in `usage-limit-patterns.ts`, `respawn-*-patterns`, `regex-patterns.ts` | patterns are code, not data |
|
||||
| Plugin runtime | deliberately refused, see `docs/extending-codeman.md` | not a gap, a decision |
|
||||
| Session handoff on restart | tmux owns the PTYs, so they already survive a Codeman restart | not a gap, solved by architecture |
|
||||
|
||||
**Conclusion**: roughly 90% of the capability surface already exists. Parts 1 and 2 below close
|
||||
the two real gaps.
|
||||
|
||||
The table is the 2026-08-08 snapshot that motivated the work, kept as written. The three rows
|
||||
marked missing are closed since: `GET .../wait` and `GET .../wait-output` shipped in 1.13.0, and
|
||||
the skill is packaged at `skills/codeman` (npm tarball included). `blocked` as a wire-contract
|
||||
state, and the machine-readable schema, are still open (Parts 3 and 4).
|
||||
|
||||
---
|
||||
|
||||
## 2. Part 1: the Codeman agent skill
|
||||
|
||||
### 2.1 Goal
|
||||
|
||||
An agent running inside a Codeman session can discover and correctly drive Codeman without the
|
||||
user pasting API docs into the prompt, and without inventing dangerous calls.
|
||||
|
||||
### 2.2 Layout and distribution
|
||||
|
||||
The `npx skills` CLI (vercel-labs/skills) clones a GitHub repo and looks for
|
||||
`skills/<name>/SKILL.md`. Claude Code natively discovers `.claude/skills/<name>/SKILL.md` in a
|
||||
project and `~/.claude/skills/` globally. Both are satisfied with one source of truth plus a
|
||||
symlink, which is the pattern this repo already uses for `remotion-best-practices`.
|
||||
|
||||
```
|
||||
skills/
|
||||
codeman/
|
||||
SKILL.md <- single source of truth
|
||||
reference/
|
||||
endpoints.md <- full endpoint tables, loaded on demand
|
||||
recipes.md <- worked multi-session orchestration examples
|
||||
.claude/skills/codeman -> ../../skills/codeman (symlink, dogfooding in this repo)
|
||||
```
|
||||
|
||||
Adding a `skills/` directory to the repo root costs one entry in the GitHub listing. CLAUDE.md
|
||||
keeps the root short on purpose, so this needs a conscious sign-off; the alternative is
|
||||
`docs/skills/codeman/` with a `--skill` path argument, which breaks the one-liner install.
|
||||
**Recommendation**: accept `skills/` at the root, because the install one-liner is the whole
|
||||
point of shipping a skill.
|
||||
|
||||
Install paths, in order of how a user gets it:
|
||||
|
||||
1. `npx skills add Ark0N/Codeman --skill codeman -g` (global, any agent, matches the herdr flow).
|
||||
2. `codeman skill install [--global | --case <name>]`, a new CLI subcommand writing the same
|
||||
file. This is the path for users who installed via npm and never cloned the repo.
|
||||
3. **Automatic per-case injection**, modeled exactly on `applyStatusLineConfig(casePath, enabled)`
|
||||
in `hooks-config.ts`: write `<case>/.claude/skills/codeman/SKILL.md` at case creation,
|
||||
gated on a new setting. Codeman already writes `<case>/.claude/settings.local.json` hooks
|
||||
through `writeHooksConfig()`, so this is the same mechanism with the same lifecycle.
|
||||
|
||||
Setting name: `agentSkillEnabled`. Synced (not per-device), since it changes on-disk case
|
||||
content rather than display. Default: **ON after the dogfooding phase, OFF in the first
|
||||
release**. Rationale for starting OFF: Claude Code loads every skill's name and description
|
||||
into context on every turn, so an always-on skill has a small permanent token cost, and we
|
||||
should measure that we are buying something with it first.
|
||||
|
||||
### 2.3 SKILL.md content
|
||||
|
||||
Frontmatter, per the skills convention (`name` + `description` required):
|
||||
|
||||
```yaml
|
||||
---
|
||||
name: codeman
|
||||
description: >-
|
||||
Control Codeman, the session manager this agent is running inside: list sessions,
|
||||
start worker sessions, send prompts, read terminal output, and wait for other agents
|
||||
to finish. Only usable when CODEMAN_MUX=1.
|
||||
---
|
||||
```
|
||||
|
||||
Body sections, in order:
|
||||
|
||||
**1. Guard (first thing, non-negotiable).**
|
||||
|
||||
```bash
|
||||
test "${CODEMAN_MUX:-}" = 1 || { echo "not inside a Codeman session"; exit 1; }
|
||||
API="${CODEMAN_API_URL:?CODEMAN_API_URL not set, refusing to guess}"
|
||||
SELF="${CODEMAN_SESSION_ID:-}"
|
||||
```
|
||||
|
||||
If `CODEMAN_MUX` is not `1`, the agent must stop and say it is not running inside a
|
||||
Codeman-managed session. Same shape as herdr's `HERDR_ENV` guard, and the variables are
|
||||
already exported by `tmux-manager.buildEnvExports()`. No fallback URL when
|
||||
`CODEMAN_API_URL` is unset: any guess is the wrong scheme on an HTTPS install (prod is
|
||||
HTTPS with a self-signed cert, hence `curl -sk` throughout), and a server the agent
|
||||
cannot identify is not one it should be driving.
|
||||
|
||||
**2. Rules of the road.** Lifted and tightened from README lines 666 to 745:
|
||||
|
||||
- Single-line input only. Multi-line breaks the agent TUI (Ink).
|
||||
- Always send `clientId` + a monotonic `seq` on `POST .../input` so a retry cannot double-deliver.
|
||||
- Envelope is `{success, data}`; a few legacy GETs are bare, so read `body.data ?? body`.
|
||||
- Add `-u admin:"$CODEMAN_PASSWORD"` when a password is set. Prod is HTTPS, so `curl -sk`.
|
||||
- Prefer `/api/v1/*`, the stable alias.
|
||||
|
||||
**3. Safety rules (the section that does not exist anywhere today).**
|
||||
|
||||
- Never act on `$CODEMAN_SESSION_ID`. That is you.
|
||||
- Only `DELETE` sessions **you created in this conversation**, by exact id. Keep the list.
|
||||
- Never bulk-delete, never loop a `DELETE` over `/api/sessions`. There is no undo.
|
||||
- Never `tmux kill-session`, `pkill tmux`, `pkill claude`. Use the API.
|
||||
- Creating a session consumes a slot against the 50-session cap. Clean up what you start.
|
||||
|
||||
**4. Recipes**, each one a single copy-pasteable curl:
|
||||
|
||||
| Task | Call |
|
||||
| -------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| list sessions | `GET /api/v1/sessions` |
|
||||
| find yourself | match ids by PREFIX of `$CODEMAN_SESSION_ID` (Docker cases truncate it to 8 chars, so an equality check never fires there) |
|
||||
| start a worker | `POST /api/v1/quick-start {caseName, mode, effort}` |
|
||||
| send a prompt | `POST /api/v1/sessions/:id/input {input:"…\r", useMux:true, clientId, seq}` (the trailing `\r` is what sends Enter; without it the text sits on the prompt unsubmitted) |
|
||||
| send prompt and wait | `POST /api/v1/sessions/:id/input {input:"…\r", wait:"stop", waitTimeout:600000}` (Part 2) |
|
||||
| wait for a worker | `GET /api/v1/sessions/:id/wait?until=stop,blocked&timeout=300000` (Part 2) |
|
||||
| wait for a marker | `GET /api/v1/sessions/:id/wait-output?match=DONE_<random>&timeout=120000` (Part 2; unique per call, per §3.3's repaint rule) |
|
||||
| read output | `GET /api/v1/sessions/:id/output` |
|
||||
| read full scrollback | `GET /api/v1/sessions/:id/terminal?full=1` |
|
||||
| watch sub-agents | `GET /api/v1/subagents` |
|
||||
| schedule work | `POST /api/v1/cron/jobs` |
|
||||
| clean up | `DELETE /api/v1/sessions/:id` |
|
||||
|
||||
**5. Pointer to `reference/endpoints.md`** for anything not in the table, so the always-loaded
|
||||
part of the skill stays small.
|
||||
|
||||
### 2.4 An ergonomics guard worth adding server-side
|
||||
|
||||
The skill will tell the agent not to act on itself, but a confused agent can still try. Propose:
|
||||
the skill sends `X-Codeman-Caller-Session: $CODEMAN_SESSION_ID` on every request, and the server
|
||||
refuses destructive operations (`DELETE /api/sessions/:id`, kill, respawn stop) when that header
|
||||
equals the target id, with a clear error.
|
||||
|
||||
This is a **footgun guard, not a security control**: any caller can omit the header. Document it
|
||||
as such so nobody mistakes it for a boundary. It costs about 10 lines in `route-helpers.ts`.
|
||||
|
||||
### 2.5 Verification
|
||||
|
||||
Per the always-end-to-end-test rule, "the skill exists" is not done. Done is:
|
||||
|
||||
1. Symlink it into `.claude/skills/`, start a real throwaway Codeman session, and ask that agent
|
||||
to "start a worker session that runs the test suite and tell me when it finishes".
|
||||
2. Confirm from the outside that exactly one new session appeared, got the prompt, and that the
|
||||
lead agent waited rather than polling in a busy loop.
|
||||
3. Confirm the guard: run the same prompt in a shell with `CODEMAN_MUX` unset and confirm refusal.
|
||||
4. Confirm cleanup: the worker session is deleted by exact id and no other session was touched.
|
||||
|
||||
Never run this against `w1`/`w2`/`w3`.
|
||||
|
||||
### 2.6 Files touched
|
||||
|
||||
- `skills/codeman/SKILL.md` (new), `skills/codeman/reference/*.md` (new)
|
||||
- `.claude/skills/codeman` symlink (new)
|
||||
- `src/cli.ts` (new `skill install` subcommand)
|
||||
- `src/hooks-config.ts` (new `applyAgentSkill(casePath, enabled)`, mirroring `applyStatusLineConfig`)
|
||||
- `src/web/schemas.ts` (`agentSkillEnabled` in `SettingsUpdateSchema`, which is `.strict()`)
|
||||
- `src/web/routes/system-routes.ts` (settings PUT must resolve the flag from `merged`, never
|
||||
from the raw body, per the partial-PUT invariant)
|
||||
- `src/web/public/settings-ui.js` + `index.html` (checkbox)
|
||||
- `package.json` `files` array, so `skills/` ships to npm
|
||||
- README pointer, `docs/extending-codeman.md` seam 3 pointer
|
||||
|
||||
---
|
||||
|
||||
## 3. Part 2: wait primitives
|
||||
|
||||
### 3.1 Goal
|
||||
|
||||
Make Codeman orchestratable from a shell tool. Today the only "tell me when" channel is SSE,
|
||||
which a curl-driven agent cannot practically consume: it would have to hold a streaming
|
||||
connection and parse events inline. herdr solves this with blocking CLI calls. Codeman should
|
||||
solve it with bounded long-poll endpoints.
|
||||
|
||||
All three additions are **additive**, so the versioning policy stays intact (new endpoints and
|
||||
new optional fields are non-breaking).
|
||||
|
||||
### 3.2 The signal model
|
||||
|
||||
A waiter resolves on the first of a set of signals. Sources that already exist:
|
||||
|
||||
| Signal | Source today |
|
||||
| --------- | --------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `idle` | `Session` emits `idle` (session.ts ~1775 for Claude, ~2101 for shell), wired at `session-listener-wiring.ts:402` |
|
||||
| `working` | `Session` emits `working` (session.ts ~1788), wired at `session-listener-wiring.ts:401` |
|
||||
| `stop` | `POST /api/hook-event` with `event: 'stop'`, the definitive "Claude finished responding" signal already used by `controller.signalStopHook()` |
|
||||
| `blocked` | `POST /api/hook-event` with `permission_prompt` or `elicitation_dialog` |
|
||||
| `exit` | `Session` emits `exit` |
|
||||
|
||||
`stop` is the highest-quality signal for "the turn is over" and should be the documented default
|
||||
for orchestration. `idle` is heuristic: output stabilization plus prompt detection, and it can
|
||||
flap mid-turn when a spinner pauses. External CLI modes (`isExternalCliMode()`) have no stop
|
||||
hook at all, so for opencode/codex/gemini/antigravity only `idle`, `working` and `exit` are
|
||||
available. **The skill and the docs must say which signals exist per mode**, otherwise an agent
|
||||
waits forever on `stop` in a codex session.
|
||||
|
||||
### 3.3 Endpoint specs
|
||||
|
||||
#### A. `GET /api/sessions/:id/wait`
|
||||
|
||||
| Param | Type | Default | Notes |
|
||||
| --------- | ---------------------------------------------- | ---------------- | ------------------------------------------------------------ |
|
||||
| `until` | comma list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on first match |
|
||||
| `timeout` | ms | 60000 | clamped to `MAX_WAIT_MS` (600000) |
|
||||
| `fresh` | `0`/`1` | `0` | `1` requires a _transition_, ignoring the state at call time |
|
||||
|
||||
Response (always 200 unless the session is missing or a cap is hit):
|
||||
|
||||
```json
|
||||
{
|
||||
"success": true,
|
||||
"data": {
|
||||
"signal": "stop",
|
||||
"timedOut": false,
|
||||
"immediate": false,
|
||||
"ended": false,
|
||||
"waitedMs": 8421,
|
||||
"status": "idle",
|
||||
"sessionId": "...",
|
||||
"until": ["stop", "idle", "exit"],
|
||||
"limitPaused": false
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`until` is echoed back because the server may narrow it: `stop`/`blocked` are dropped
|
||||
from the DEFAULT set for external CLI modes (asking for them EXPLICITLY is a 400
|
||||
instead, since omitting `until` must never 400). `limitPaused` tells a caller that a
|
||||
timeout was expected rather than a stall worth retrying hard.
|
||||
|
||||
**A timeout is not an error.** `{"timedOut": true, "signal": null}` with HTTP 200, so a caller
|
||||
can loop without treating every poll boundary as a failure. Errors are reserved for
|
||||
`NOT_FOUND` (unknown or not-owned session) and `SESSION_BUSY` (waiter cap exceeded).
|
||||
|
||||
`immediate: true` means the session was already in the requested state and `fresh` was not set.
|
||||
|
||||
#### B. `GET /api/sessions/:id/wait-output`
|
||||
|
||||
| Param | Type | Default | Notes |
|
||||
| --------- | ------------------------------ | -------- | --------------------------------------------------------- |
|
||||
| `match` | literal string, 1 to 200 chars | required | substring match against ANSI-stripped output |
|
||||
| `nocase` | `0`/`1` | `0` | case-insensitive compare |
|
||||
| `from` | `now` \| `buffer` | `now` | `buffer` scans the existing text buffer first, then waits |
|
||||
| `timeout` | ms | 60000 | clamped to `MAX_WAIT_MS` |
|
||||
|
||||
Response: `{ matched: true, timedOut: false, snippet: "...", waitedMs }`.
|
||||
|
||||
**No regex in v1, deliberately.** `search-service.ts` already avoids regex specifically so there
|
||||
is no ReDoS surface, and this endpoint would be even more exposed since the pattern is attacker
|
||||
supplied and the input is a live stream. herdr can offer `--regex` because Rust's regex crate is
|
||||
linear-time with no backtracking; JS `RegExp` is not. If regex is wanted later, the honest
|
||||
options are a length-capped subset compiled once with a match budget, or `re2`. Note it and move on.
|
||||
|
||||
Implementation detail that will bite if missed: a match can straddle two PTY chunks. Keep a
|
||||
carry buffer of `match.length - 1` bytes from the previous chunk and test `carry + chunk`.
|
||||
|
||||
⚠️ **`from=now` does not mean "printed after you asked".** tmux repaints the visible
|
||||
screen on attach, resize, or any TUI redraw, and a repaint arrives as ordinary `terminal`
|
||||
data. Observed live: a marker echoed a minute earlier matched instantly on a fresh
|
||||
`from=now` wait. This is inherent to a terminal multiplexer, not fixable in the registry,
|
||||
so the contract is: **use a marker unique per call** (`echo DONE_$RANDOM`), never a
|
||||
generic one like `BUILD OK`. The skill's recipes must show that.
|
||||
|
||||
The returned snippet is whitespace-collapsed (blank runs to a single newline) for
|
||||
readability only; matching runs on the raw stripped text. Without it, a real pane's
|
||||
`\r\n` padding between the prompt and the match fills the whole context window with
|
||||
nothing, which was the first thing the live test showed.
|
||||
|
||||
#### C. `wait` on the existing input endpoint
|
||||
|
||||
`POST /api/sessions/:id/input` gains two optional fields:
|
||||
|
||||
```json
|
||||
{ "input": "run the tests\r", "useMux": true, "clientId": "agent-1", "seq": 7, "wait": "stop", "waitTimeout": 600000 }
|
||||
```
|
||||
|
||||
(The trailing `\r` is required on every input body: `sendInput` sends Enter only
|
||||
when the input contains a carriage return.)
|
||||
|
||||
Response gains `"wait": { "signal": "stop", "timedOut": false, "waitedMs": 41230 }`.
|
||||
|
||||
This is the important one, because it closes a race the standalone `GET .../wait` cannot: between
|
||||
"input delivered" and "session flips to working" there is a window where a naive
|
||||
send-then-wait sees the _pre-existing_ idle state and returns instantly. The combined endpoint
|
||||
**registers the waiter before writing**, so that window does not exist. This is exactly why herdr
|
||||
ships `agent prompt --wait` as its own thing.
|
||||
|
||||
`wait` accepts `true` (the default signal set) or the same comma grammar as `until`.
|
||||
Both new fields are `.nullish()`, not `.optional()`: a third-party caller building the
|
||||
body with `JSON.stringify` keeps an explicit `null` on the wire, and `.optional()`
|
||||
rejects that with `INVALID_INPUT`. That gotcha has shipped as a real bug twice.
|
||||
|
||||
Two behaviors to preserve carefully:
|
||||
|
||||
- **`useMux` is fire-and-forget today.** The handler responds without awaiting `writeViaMux`, on
|
||||
purpose (a tmux child process must not block the HTTP response). With `wait` present the
|
||||
handler already has to stay open, so it can await delivery, and a `writeViaMux` failure becomes
|
||||
observable for the first time. The non-wait path must keep its current fire-and-forget shape
|
||||
byte for byte.
|
||||
- **Duplicate suppression.** A tagged redelivery (`clientId`+`seq` already applied) returns 200
|
||||
without writing. With `wait` set it still waits, since the caller's intent is "tell me when
|
||||
this settles". But it waits with `requireTransition: false`, unlike a fresh delivery: the
|
||||
original turn may be long over, and requiring a new transition would block a redelivery until
|
||||
timeout for no reason. Fresh delivery requires a transition, a duplicate answers from the
|
||||
current state.
|
||||
- **Capacity rollback.** `shouldApplyInput()` MUTATES (it records the seq), and it runs before
|
||||
the waiter is registered. If registration then fails on a full pool, the handler must call
|
||||
`forgetInputSeq` before returning `SESSION_BUSY`, or the caller's retry is rejected as a
|
||||
duplicate and the input is lost by the very mechanism reliable delivery exists for.
|
||||
|
||||
### 3.4 Module design
|
||||
|
||||
New file `src/web/session-wait-registry.ts`, with the IO-free core unit-testable in isolation
|
||||
(same split as `self-update.ts`):
|
||||
|
||||
```ts
|
||||
type WaitSignal = 'idle' | 'working' | 'stop' | 'blocked' | 'exit';
|
||||
|
||||
waitForSignal(sessionId, { until: Set<WaitSignal>, timeoutMs, requireTransition }): Promise<WaitResult>
|
||||
notifySignal(sessionId, signal: WaitSignal): void
|
||||
waitForOutput(sessionId, { match, nocase, timeoutMs }): Promise<OutputWaitResult>
|
||||
notifyOutput(sessionId, chunk: string): void
|
||||
cancelAll(sessionId, reason): void
|
||||
```
|
||||
|
||||
Wiring points, all existing:
|
||||
|
||||
- `src/web/session-listener-wiring.ts` around lines 190 and 200 already handles `working` and
|
||||
`idle` and broadcasts them. Add a `notifySignal()` call next to each broadcast, plus `exit`.
|
||||
- `src/web/routes/hook-event-routes.ts` already switches on `event` for the respawn controller.
|
||||
Add `notifySignal(sessionId, 'stop' | 'blocked')` in the same switch.
|
||||
- Output: `notifyOutput()` rides the ALREADY-attached `terminal` listener in
|
||||
session-listener-wiring.ts. An earlier draft had the registry hand out attach/detach
|
||||
callbacks so a listener could be added lazily; that was deleted once it was clear no
|
||||
second listener is needed at all. The cost is one Map lookup per PTY chunk, which is why
|
||||
the no-waiter check comes before the ANSI strip.
|
||||
- Session deletion calls `notifySignal('exit')` then `cancelAll()`, so no promise is left
|
||||
hanging. Both are required: `_doCleanupSession` detaches the session's listeners BEFORE
|
||||
`session.stop()`, so on a delete the PTY exit event never reaches the registry, and an
|
||||
`until=exit` caller would otherwise get a bare `ended` instead of its signal. Found by
|
||||
live-testing the delete path, not by the unit tests.
|
||||
|
||||
Memory-leak discipline, per the 24-hour-session rules: every waiter owns a timer that is cleared
|
||||
on resolve, the per-session waiter set is deleted when it empties, and the output listener is
|
||||
removed with it. `test/memory-leak-prevention.test.ts` should grow a case for this.
|
||||
|
||||
Caps in a new `src/config/agent-wait.ts` (limits live in `src/config/`, env-overridable):
|
||||
|
||||
| Constant | Default | Why |
|
||||
| ------------------------- | ------- | --------------------------------------- |
|
||||
| `MAX_WAIT_MS` | 600000 | an unbounded long-poll is a socket leak |
|
||||
| `DEFAULT_WAIT_MS` | 60000 | short enough to survive most proxies |
|
||||
| `MAX_WAITERS_PER_SESSION` | 16 | |
|
||||
| `MAX_WAITERS_TOTAL` | 128 | same reasoning as `MAX_SSE_CLIENTS` |
|
||||
|
||||
Exceeding a cap returns `SESSION_BUSY`, not a silent queue.
|
||||
|
||||
### 3.5 Transport concerns
|
||||
|
||||
Fastify is constructed with defaults in `server.ts:329-331`. `requestTimeout` defaults to 0
|
||||
(disabled) and `keepAliveTimeout` (72s) applies between requests, not to an in-flight one, so a
|
||||
10-minute in-process hold is fine. **Verify this on the real instance before relying on it.**
|
||||
|
||||
Intermediaries are the actual risk. Prod is reached through `tailscale serve`, and users also run
|
||||
cloudflared tunnels; both can cut an idle connection. That is why `DEFAULT_WAIT_MS` is 60s and
|
||||
why the documented pattern is a client-side loop over short waits rather than one 10-minute call.
|
||||
The skill's recipes must show the loop.
|
||||
|
||||
### 3.6 Edge cases to get right
|
||||
|
||||
| Case | Behavior |
|
||||
| ------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Session already idle, `fresh=0` | return immediately, `immediate: true` |
|
||||
| Session already idle, `fresh=1` | wait for the next transition into a requested state |
|
||||
| Session dies mid-wait | resolve with `signal: "exit"` if `exit` was requested, otherwise resolve `timedOut:false, signal:null, ended:true`. Never hang |
|
||||
| Session deleted mid-wait | same, resolve, do not throw. Verified live: `until=exit` gets `signal:"exit"`, a concurrent `until=blocked` gets `ended:true`, both in ~0ms |
|
||||
| Shutdown with a wait pending | `cancelEverything()` in `stop()`. Verified live: SIGTERM with a 300s wait in flight exits in 1s |
|
||||
| External CLI mode | `stop` and `blocked` never fire. Reject `until=stop` for those modes with a clear `INVALID_INPUT` rather than hanging until timeout |
|
||||
| Multi-user | goes through `findSessionOrFail(ctx, id, req)`, which already enforces ownership |
|
||||
| Remote / Docker cases | signals originate from the same `Session` object, so no special casing. Docker hooks need `CODEMAN_DOCKER_BRIDGE_HOOKS=1` for `stop`/`blocked` to arrive at all; without it, only `idle` works. Document it |
|
||||
| Respawn `/clear` mid-wait | a respawn cycle emits `idle`. Callers waiting on `stop` are unaffected; callers on `idle` may resolve early. Documented, not fixed |
|
||||
| Limit pause | if the session is paused on a usage limit, nothing will fire until the reset. The wait times out honestly. Consider surfacing `limitPaused: true` in the response so the caller can back off |
|
||||
|
||||
### 3.7 Tests
|
||||
|
||||
- `test/session-wait-registry.test.ts` (pure): immediate resolve, transition-required, multi-signal
|
||||
first-wins, timeout, cap exceeded, cancel on session end, no listener leak after resolve,
|
||||
chunk-straddling output match, case-insensitive match.
|
||||
- `test/routes/session-wait-routes.test.ts` (`app.inject()`, no port): all three endpoints against
|
||||
a `MockSession`, including the 200-with-`timedOut` contract and the ownership 404.
|
||||
- `test/routes/session-input-wait.test.ts`: the send-and-wait race, plus proof that the non-wait
|
||||
path is unchanged (still returns before `writeViaMux` settles).
|
||||
- Live verification on a throwaway session before COM, per the always-end-to-end-test rule.
|
||||
|
||||
### 3.8 Files touched
|
||||
|
||||
- `src/config/agent-wait.ts` (new)
|
||||
- `src/web/session-wait-registry.ts` (new)
|
||||
- `src/web/session-listener-wiring.ts` (notify on idle/working/exit)
|
||||
- `src/web/routes/hook-event-routes.ts` (notify on stop/blocked)
|
||||
- `src/web/routes/session-routes.ts` (two new routes, `wait` fields on input)
|
||||
- `src/web/schemas.ts` (`SessionWaitQuerySchema`, `SessionWaitOutputQuerySchema`, extend
|
||||
`SessionInputWithLimitSchema`. Note: `.optional()` rejects `null`, so the frontend and any
|
||||
generated client must send `undefined`, never `null`)
|
||||
- `docs/api-reference.md`, `docs/extending-codeman.md`, README API table
|
||||
- `skills/codeman/SKILL.md` recipes (Part 1 depends on this)
|
||||
|
||||
---
|
||||
|
||||
## 4. Deferred: parts 3 to 5
|
||||
|
||||
Not in scope now, kept here so they are not lost.
|
||||
|
||||
### Part 3: promote `blocked` to a first-class state
|
||||
|
||||
`SessionStatus` is `'idle' | 'busy' | 'stopped' | 'error'`. "Needs you" exists three times over:
|
||||
hook events, the `tab-alert-action` CSS class, and the phone overview NEEDS YOU section, each
|
||||
re-deriving it. herdr makes `blocked` a real state that rolls up.
|
||||
|
||||
Add `blocked` (and possibly `done`) to `SessionStatus`, set it from the same hook events that
|
||||
Part 2 uses as wait signals, and clear it on the next `working`/`stop`. Then the tab strip, the
|
||||
mobile overview, the wait endpoints, and any external agent read one field.
|
||||
|
||||
Cost: `SessionStatus` is a widely-consumed union, so every exhaustive `switch` (the codebase has
|
||||
`assertNever` and `noFallthroughCasesInSwitch`) will need a branch. That is a feature, it makes
|
||||
the compiler find every site. This is a **minor** bump, not a patch: it widens a public type in
|
||||
the HTTP contract.
|
||||
|
||||
### Part 4: `GET /api/schema`
|
||||
|
||||
herdr ships `herdr api schema`. Every Codeman route is already Zod-validated, so
|
||||
`zod-to-json-schema` over `schemas.ts` gives a self-describing API almost free. Value: third-party
|
||||
tools and the skill stop drifting from hand-written docs. Open question: whether to emit full
|
||||
OpenAPI (`@fastify/swagger` would need per-route schema registration, which is a much larger
|
||||
change) or just dump the Zod schemas keyed by name (cheap, 80% of the value).
|
||||
|
||||
### Part 5: detection manifests instead of hardcoded patterns
|
||||
|
||||
CLI-specific readiness, blocked and usage-limit patterns live in code across
|
||||
`usage-limit-patterns.ts`, the respawn pattern helpers and `regex-patterns.ts`. Externalizing the
|
||||
per-CLI ones into data files would make adding a sixth CLI a data change instead of a code change.
|
||||
|
||||
**Do not copy the remote-update part.** herdr auto-fetches manifest updates from herdr.dev.
|
||||
Codeman auto-pulling behavioral rules from a vendor server contradicts its security posture.
|
||||
Bundled manifests plus local override only, no network.
|
||||
|
||||
### Explicit non-goals
|
||||
|
||||
- **Plugin runtime and marketplace.** `docs/extending-codeman.md` already argues this: a plugin
|
||||
runtime means third-party code inside a process that spawns agents with your credentials, on a
|
||||
server people expose over a tunnel. The reasoning still holds. If the marketplace _pattern_ is
|
||||
wanted, apply it to data (web tabs, case templates, cron recipes), never to executable code.
|
||||
- **Live PTY handoff on restart.** herdr needs it because it owns the terminals. Codeman
|
||||
delegates to tmux, so PTYs already survive a self-update restart.
|
||||
- **Socket API.** HTTP plus SSE is the existing, documented, stable contract. A second transport
|
||||
would double the surface for no capability gain.
|
||||
|
||||
---
|
||||
|
||||
## 5. Sequencing
|
||||
|
||||
| Step | Work | Gate |
|
||||
| ---- | ------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| 1 ✅ | `src/config/agent-wait.ts` + `session-wait-registry.ts` + unit tests | 48 tests green |
|
||||
| 2 ✅ | `GET .../wait` + wiring in listener-wiring, hook-event-routes, server teardown | 15 route tests green; live-verified on an isolated `CODEMAN_INSTANCE=waittest` instance (immediate resolve, 400 on a bad signal, 200+`timedOut` on timeout, hook `stop` and `permission_prompt`→`blocked` waking an in-flight wait, delete delivering `exit`, SIGTERM not blocked); full `test:ci` sweep green |
|
||||
| 3 ✅ | `GET .../wait-output` | 16 route tests green; live-verified on real PTY bytes (`echo MARKER` waking a blocked request in ~1s, `from=buffer` immediate hit, never-seen marker timing out at exactly 2001ms, nocase, `regex` refused with a 400); full `test:ci` sweep green |
|
||||
| 4 ✅ | `wait` field on `POST .../input`, non-wait path proven unchanged | 16 route tests green; live-verified (no-wait returns in 26ms with the historical bare body; an idle session did NOT satisfy a `wait` request, blocking the full 2001ms, which is the race the endpoint exists to close; the stop hook resolved a send-and-wait at 1510ms and the input was confirmed in the tmux pane; `wait:null` accepted) |
|
||||
| 5 ✅ | `skills/codeman/SKILL.md` + reference files + `.claude/skills` symlink | live dogfood: a real session orchestrates a worker end to end |
|
||||
| 6 ✅ | `codeman skill install` CLI + `applyAgentSkill()` + `agentSkillEnabled` setting | 10 unit tests (`test/agent-skill.test.ts`) + real-server case-creation tests (`test/quick-start.test.ts`, incl. the settings PUT accepting the key) green; CLI verified live (install/uninstall, global + `--case`, foreign/symlink refusals) |
|
||||
| 7 ✅ | Docs: api-reference, extending-codeman, README | plus `architecture-invariants.md` (§agent-wait-primitives), `CLAUDE.md` and the API reference's per-mode signal table |
|
||||
| 8 ✅ | COM (minor bump: new endpoints, new setting, new optional fields) | released as 1.13.0 (wait primitives + skill); step 6 followed in 1.14.1 and was republished as 1.14.2 after live-testing the packaged skill |
|
||||
|
||||
Parts 1 and 2 are independent enough to land separately, but the skill is much less useful
|
||||
without the wait endpoints, so the wait work goes first.
|
||||
|
||||
## 6. Open questions for the owner
|
||||
|
||||
1. ✅ `skills/` at the repo root: accepted (built that way; the install one-liner depends on it).
|
||||
2. ✅ `agentSkillEnabled` default: **OFF** for the first release, per §2.2's rationale (skills
|
||||
cost context on every turn; measure before defaulting on). Flip later if dogfooding earns it.
|
||||
3. ✅ Both: global install via `npx skills add` / `codeman skill install`, AND per-case
|
||||
auto-injection behind the (default-off) setting. Injection is add-only at session create and
|
||||
marker-guarded, so a user-authored copy is never touched.
|
||||
4. Is `X-Codeman-Caller-Session` self-protection worth the 10 lines, given it is a footgun guard
|
||||
and not a security boundary? (Still open, not built with step 6.)
|
||||
5. ✅ Regex support in `wait-output`: literal-only shipped, and a `regex` query param is
|
||||
rejected with a 400 rather than ignored, so an agent that assumed otherwise cannot
|
||||
silently wait on the wrong thing.
|
||||
|
||||
---
|
||||
|
||||
## 7. Build log: what actually happened
|
||||
|
||||
Written at the end of the build so the next person inherits the reasoning, not just the
|
||||
diff. Process artifacts (per-agent briefs, findings, reports) live in the gitignored
|
||||
`tmp/agent-wait-review/`; this section is the part worth keeping.
|
||||
|
||||
### What shipped
|
||||
|
||||
| Piece | Files |
|
||||
| ------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Bounds + clamping | `src/config/agent-wait.ts` (new) |
|
||||
| Blocking-wait registry | `src/web/session-wait-registry.ts` (new, IO-free, unit-tested) |
|
||||
| `GET .../wait`, `GET .../wait-output`, `wait`/`waitTimeout` on `POST .../input` | `src/web/routes/session-routes.ts` |
|
||||
| Signal wiring | `session-listener-wiring.ts` (idle/working/exit + output), `hook-event-routes.ts` (stop/blocked), `server.ts` (teardown, shutdown) |
|
||||
| Agent skill | `skills/codeman/SKILL.md` + `reference/`, `.claude/skills/codeman` symlink, `package.json` `files` |
|
||||
| Docs | `api-reference.md`, `extending-codeman.md`, `architecture-invariants.md`, `README.md`, `CLAUDE.md` |
|
||||
| Tests | `test/session-wait-registry.test.ts`, three `test/routes/session-*wait*.test.ts`, `http-contract.test.ts`, `mock-session.ts` |
|
||||
|
||||
### Bugs found in ADJACENT code, not in the new feature
|
||||
|
||||
These are the highest-value output of the exercise and none were on the plan:
|
||||
|
||||
1. **Every Codeman hook was dead on HTTPS installs.** `hooks-config.ts` built the hook
|
||||
curl as `curl -s` with no `-k` while the statusline exporter 300 lines below used
|
||||
`curl -sk` and documented why. Proven with the real hook command: `curl exit=60`
|
||||
without the flag, success with it, and the failure swallowed by the hook's own
|
||||
`2>/dev/null || true`. This silently killed `stop`, `permission_prompt`,
|
||||
`elicitation_dialog`, `idle_prompt`, `teammate_idle` and `task_completed`, taking
|
||||
respawn's definitive idle signals with them. Fixed, **plus** a staleness detector in
|
||||
`refreshStaleCodemanHooks` that regenerates the on-disk config of already-created
|
||||
cases (23 of 26 local cases carried the broken form; fixing the generator alone would
|
||||
have left every one of them broken).
|
||||
2. **`buildEnvExports()` exported a wrong-scheme `CODEMAN_API_URL`** (`http://` fallback
|
||||
on an HTTPS install). Now omitted rather than guessed, so in-session guards fail closed.
|
||||
3. **Programmatic input is only submitted when it contains `\r`.** `sendInput` sends Enter
|
||||
only if the payload has a carriage return; without it the text sits in the composer
|
||||
forever. Bit this build repeatedly before it was diagnosed, and had leaked into the
|
||||
docs' own examples.
|
||||
|
||||
### Design decisions worth not re-litigating
|
||||
|
||||
- **A timeout is HTTP 200** with `wait.timedOut`, never a 4xx: callers loop over short
|
||||
waits because tunnels cut idle connections, and every poll boundary would otherwise be
|
||||
indistinguishable from failure.
|
||||
- **Send-and-wait must be one endpoint.** A separate POST-then-wait races: between the
|
||||
write and the flip to `working`, a wait sees the stale `idle` and reports the PREVIOUS
|
||||
turn as this one. The waiter is registered before the write.
|
||||
- **`stop`/`blocked` exist for `claude` mode only.** They come from Claude Code hooks;
|
||||
`shell` installs none either, so keying off `isExternalCliMode()` was wrong.
|
||||
- **Literal matching only, never regex.** JS `RegExp` backtracks; herdr can offer
|
||||
`--regex` because Rust's regex crate is linear-time.
|
||||
- **Client-hangup abort listens on `reply.raw` guarded by `writableFinished`.** On
|
||||
`req.raw`, `close` fires when the request BODY ends, which on a POST killed every
|
||||
send-and-wait instantly, and no `app.inject()` test can see it (inject never emits
|
||||
`close`).
|
||||
- **Liveness cannot come from `session.pid`.** For a tmux session that is the local
|
||||
`tmux attach` client, not the worker: a worker exiting inside its pane leaves
|
||||
`pane_dead=1` with the client alive, so `pid` never goes null. Liveness is probed at
|
||||
the mux layer, cached (~750 ms) and only on blocking waits, never on the input hot path.
|
||||
|
||||
### Verification rounds
|
||||
|
||||
Six agents across three rounds, each verifying the previous round's work rather than its
|
||||
own. Findings that mattered, in order of severity, were: the dead-pane liveness gap; the
|
||||
`reply.raw` abort regression; abandoned long-polls leaking waiter slots; a crashed session
|
||||
reporting `idle`; `shell` accepting `until=stop`; and a documented recipe that reported
|
||||
success without running its task. Two traps recurred often enough to name:
|
||||
|
||||
- **Vacuous passes.** `app.inject()` never emits `close`; a latched `cancelEverything()`
|
||||
in `afterEach` silently killed the registry for every later test in a file; three test
|
||||
files sharing one session id against the process-wide registry let one file's leftover
|
||||
waiter fail another's assertion. Any new wait test needs care on all three.
|
||||
- **HTTP-only test instances.** Every isolated instance used during the build was plain
|
||||
HTTP, which is exactly why the HTTPS hook bug survived so long. Test the transport the
|
||||
user actually runs.
|
||||
|
||||
### Resolved at wrap-up (2026-08-08, conclusion pass)
|
||||
|
||||
- **R2-A**: the fire-and-forget-then-gather-sequentially pattern was **removed from
|
||||
the skill** rather than patched. Signals are edge-triggered with no history, so a
|
||||
`stop` that fires before its waiter registers is unobservable afterwards; a
|
||||
`fresh=0` gather was rejected because the only `until` set that current state can
|
||||
satisfy answers `idle` for a prompt that never submitted, resurrecting the exact
|
||||
false-success failure R2-B had just closed. Flow 3b's pattern B now gathers on
|
||||
latched `wait-output` markers (`from=buffer`), the same mechanism that makes the
|
||||
shell flows reliable; the limitation is recorded in
|
||||
`architecture-invariants#agent-wait-primitives` and `endpoints.md`. The durable
|
||||
fix, a latched last-signal-per-turn on the server, stays with deferred Part 3.
|
||||
- Docs F7/F8, F4 and the false-`idle` attribution: `api-reference.md`,
|
||||
`extending-codeman.md` and `architecture-invariants.md` rewritten to the post-fix
|
||||
matcher (one normalized stream, chunk-straddling found, snippet as a rendering of
|
||||
the matched window), the real no-PTY answer (`ended:true`, `aborted:false`,
|
||||
`delivered:false`), and the startup-idle mechanism (a session parked on the trust
|
||||
dialog emits no further `idle`; the false success is the startup transition).
|
||||
- Orchestrate #12, #5/R2-B, #6, and R2-C..R2-E: fire-and-forget's empty `data`
|
||||
documented; every send-and-wait retry loop now treats `duplicate:true` +
|
||||
`immediate:true` as "no new turn ran" and reads the terminal before believing it;
|
||||
claude fan-out is pattern A (backgrounded send-and-waits) or the marker gather;
|
||||
readiness budgets rebalanced (5 s stage 1, 45 s stage 3) with the virgin-case
|
||||
floor named; the auth fallback now also reads the supervisor definition
|
||||
(`codeman-web.service` / launchd plist) and accepts `export`-prefixed `.env`
|
||||
lines; `pid != null` is documented as startup-only, never liveness.
|
||||
- Both public readiness recipes (extending-codeman.md, README) are bypass-first with
|
||||
the trust probe as the bounded fallback; the worked recipe carries `-k` and fails
|
||||
loudly on an empty SID; the hook `-k`/self-heal fix appears in every
|
||||
"hooks go missing" list; the multi-word-TUI claim is "unreliable", not "never".
|
||||
|
||||
### Still open
|
||||
|
||||
Both release-checklist items that used to sit here are done: `skills/` is tracked and
|
||||
ships through `package.json` `files` (published with 1.13.0, republished with 1.14.2),
|
||||
and the changeset was consumed, committed and deployed. What is left:
|
||||
|
||||
- Deferred with Part 3: the latched last-signal-per-turn. Nice-to-haves from the
|
||||
reviews: N2 (create the death-watcher inside its `try`, still built one line above
|
||||
it in `GET .../wait`) and converting timeout-shaped test detections into fast
|
||||
assertions.
|
||||
- §2.4's `X-Codeman-Caller-Session` footgun guard: still not built (open question 4).
|
||||
|
||||
### Step 6 (2026-08-09): install command, per-case injection, the setting
|
||||
|
||||
Built to the §2.6 file list, mirroring the statusLine mechanism throughout:
|
||||
|
||||
| Piece | Where |
|
||||
| ----- | ----- |
|
||||
| `applyAgentSkill(casePath, enabled)` + `installAgentSkillInto` / `removeAgentSkillFrom` | `src/hooks-config.ts` |
|
||||
| `codeman skill install` / `skill uninstall` (`--global` default, `--case <name>`) | `src/cli.ts` |
|
||||
| `agentSkillEnabled` (SYNCED, default OFF) | `schemas.ts` (`SettingsUpdateSchema`), `getAgentSkillEnabled()` on `ConfigPort`/`server.ts`, checkbox in `index.html` + `settings-ui.js` |
|
||||
| Injection call sites (Claude mode only) | `POST /api/sessions` next to `refreshStaleCodemanHooks`; `POST /api/quick-start` after the case-create/self-heal blocks (local + docker cases; remote skipped, its path lives on another host) |
|
||||
| Tests | `test/agent-skill.test.ts` (10 unit), `test/quick-start.test.ts` (real server: default-off, PUT accepts key, injection on create, shell-mode skipped) |
|
||||
|
||||
Decisions worth keeping:
|
||||
|
||||
- **Ownership marker, prefix-matched.** The injected SKILL.md ends with
|
||||
`<!-- codeman-managed-agent-skill: … -->`; install/refresh/remove all refuse a copy
|
||||
without the marker (a user's own skill) and match on the PREFIX so a wording change
|
||||
cannot disown older injected copies (the `BACKGROUND_WAKE_MARKER_PREFIX` pattern).
|
||||
- **Symlink refusal.** This repo's own dogfooding layout
|
||||
(`.claude/skills/codeman -> ../../skills/codeman`) means the injector must `lstat`
|
||||
the skill dir AND its `skills/` parent and bail on a symlink, or enabling the
|
||||
setting in the Codeman repo itself would overwrite the skill source through the link.
|
||||
- **ADD-ONLY at session create**, same shared-`.claude` rationale as the statusLine:
|
||||
a create while the setting is off must not yank the skill out from under other live
|
||||
sessions in the repo. The remove path exists (CLI `skill uninstall`, tests); no
|
||||
automatic sweep removes on toggle-off.
|
||||
- **Removal is manifest-based, never `rm -rf`**: only files the packaged source would
|
||||
have written are deleted, directories are pruned bottom-up only if they emptied, so
|
||||
a user's extra notes in `reference/` survive an uninstall.
|
||||
- **Source resolution**: `join(moduleDir, '..', 'skills', 'codeman')` works from
|
||||
`src/` (tsx), `dist/` (tsc build), and the npm tarball alike, because all three sit
|
||||
one level below the package root and `files` ships `skills/`.
|
||||
- **Nothing acts on the setting at PUT time**: injection reads the merged persisted
|
||||
settings at session create (`readSettings`, ~2s cache), so the partial-PUT invariant
|
||||
(`toggleService` reading `merged`) is untouched by construction.
|
||||
|
||||
### 2026-08-09 addendum: cross-session messaging folded into the skill
|
||||
|
||||
Claude Code 2.1.224+ ships cross-session messaging: `ListAgents`/`SendMessage`
|
||||
tools, a per-session Unix inbox socket, and a registry in
|
||||
`~/.claude/sessions/<pid>.json`. Codeman's claude workers are ordinary local Claude
|
||||
Code sessions, so the skill now routes task delivery and result collection over it
|
||||
when available, while the HTTP primitives keep spawn, readiness, synchronization,
|
||||
liveness and delete. New `skills/codeman/reference/messaging.md` (ships with zero
|
||||
installer changes: `readAgentSkillSource()` enumerates `reference/*.md` from disk),
|
||||
Flow 5 in recipes.md, and §4 in SKILL.md.
|
||||
|
||||
Verified live (claude-cli 2.1.226, Linux):
|
||||
|
||||
- A message to an idle worker starts a turn and that turn fires the normal `stop`
|
||||
hook (8.3 s send-to-stop measured), so the HTTP wait primitives compose with
|
||||
messaging unchanged; delivery to a busy session lands between tool calls.
|
||||
- First contact needs the `name [ref]` form; the bare name errors with the exact
|
||||
string to resend. The `uds:` reply address of an inbound message works as a `to`.
|
||||
- The `tmux codeman-<id8>` column in `ListAgents` (and the registry's `tmux` field)
|
||||
is the join key to Codeman session ids. The registry's `sessionId` field starts as
|
||||
the Codeman id (we spawn `claude --session-id <id>`) but drifts after `/clear` or
|
||||
resume, so it must never be the join key.
|
||||
- The feature is flag-gated beyond the version: two 2.1.226 sessions on one machine,
|
||||
one with an inbox socket and one without. Absence is a fallback case, not an error.
|
||||
- Codeman's default `--dangerously-skip-permissions` spawn puts both ends in the
|
||||
bypassing class, which delivers; mixed classes hold behind an approval dialog that
|
||||
expires unattended (upstream default 5 min), which on a headless worker means the
|
||||
message silently dies. The skill's backstop covers it.
|
||||
|
||||
Follow-up, landed in the same PR: local claude spawns now pass
|
||||
`--name <session name>` so peers carry Codeman session names. The gate is
|
||||
`buildNameCliArgs()` (session-cli-builder.ts), fail-closed at
|
||||
`CLAUDE_NAME_FLAG_MIN_VERSION = 2.1.224`: that is the messaging release, the flag's
|
||||
presence there was verified against the installed 2.1.224 binary, and the version
|
||||
comes from `getClaudeCliVersion()` (null on probe failure and under vitest), so an
|
||||
older or unknown CLI gets a command byte-identical to before. That matters because
|
||||
claude aborts startup on an unknown option, which would kill every session spawn.
|
||||
The value is allowlist-sanitized (Unicode letters/digits plus ` ._:-`, leading
|
||||
dashes stripped so it cannot parse as another option, 64-char cap, empty result =
|
||||
flag omitted) before the double-quoted interpolation in `buildSpawnCommand`, and
|
||||
only the LOCAL command carries it: the docker/remote builders never see it, since
|
||||
their CLI is not the binary the probe measured. E2E on an isolated instance
|
||||
(`CODEMAN_INSTANCE`): process cmdline `claude ... --name w9-msgtest`, registry
|
||||
`name: "w9-msgtest"`, `ListAgents` lists it under that name, a message round-trip
|
||||
works, and its replies arrive tagged `from-name="w9-msgtest"` (a derived-name
|
||||
worker's replies carry no `from-name`). A quick-start without `sessionName` has an
|
||||
empty Codeman name, so the peer name stays derived: agents should name their
|
||||
workers. Tests: `test/name-flag-injection.test.ts`.
|
||||
@@ -46,6 +46,20 @@ payload return `{ "success": true, "data": {} }`.
|
||||
> `GET /api/screenshots/:name`, `GET /q/:code` (QR redirect), and the
|
||||
> `GET /ws/sessions/:id/terminal` WebSocket upgrade.
|
||||
|
||||
> The [agent wait endpoints](#long-polling-agent-wait) use the normal envelope but
|
||||
> are the only JSON endpoints that deliberately **hold the connection open**, for up
|
||||
> to 600 s. Proxy operators and HTTP clients with a global read timeout need to know
|
||||
> that before pointing them at Codeman.
|
||||
|
||||
⚠️ **A `401` is the one status that is not an envelope.** Authentication is rejected
|
||||
in a request hook, before any handler runs, and it replies with the bare string
|
||||
`Unauthorized` (`Unauthorized: hook secret required` on the hook path) plus
|
||||
`WWW-Authenticate: Basic realm="Codeman"`. There is no `success`, no `error`, and no
|
||||
`errorCode`, because the wrapping hook only wraps object payloads. So a client that
|
||||
pipes every response straight into a JSON parser dies with a parse error rather than
|
||||
reporting an auth failure, which is a confusing way to discover that a password is
|
||||
set. Branch on the HTTP status **before** parsing.
|
||||
|
||||
## Error codes → HTTP status
|
||||
|
||||
The single source of truth is `ErrorStatus` / `httpStatusForErrorCode()` in
|
||||
@@ -66,6 +80,459 @@ the HTTP status.
|
||||
|
||||
Adding a new error code is non-breaking; removing or renaming one is a major change.
|
||||
|
||||
## Long-polling (agent wait)
|
||||
|
||||
Three calls block until something happens instead of answering immediately. They
|
||||
exist because SSE is Codeman's only other "tell me when" channel, and an agent
|
||||
driving the API from a shell tool cannot practically hold a stream and parse
|
||||
events inline.
|
||||
|
||||
| Call | Blocks until |
|
||||
|------|--------------|
|
||||
| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires |
|
||||
| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output |
|
||||
| `POST /api/v1/sessions/:id/input` with `wait` | the input is delivered **and then** a signal fires |
|
||||
|
||||
`POST .../input` with `wait` is not the same as a `POST` followed by a separate
|
||||
`GET .../wait`. It registers the waiter **before** writing, which closes the window
|
||||
in which a separate wait sees the session still idle from the previous turn and
|
||||
answers instantly with the wrong turn's result. Use it whenever you send a prompt
|
||||
and want to know when that prompt is done.
|
||||
|
||||
### Three semantics that break callers who assume otherwise
|
||||
|
||||
**1. A timeout is HTTP `200`, not an error.** A wait that ends without its signal
|
||||
returns `{"success":true, ...,"wait":{"timedOut":true,"signal":null}}`. The
|
||||
intended pattern is a client-side loop over short waits, because `tailscale serve`
|
||||
and cloudflared can both cut an idle connection, and turning every poll boundary
|
||||
into a `4xx` would make that loop indistinguishable from a real failure. `408` is
|
||||
auto-retried by several clients (silently doubling the polling load), `504` is what
|
||||
a genuine tunnel failure looks like, and `204` cannot carry `waitedMs` / `status` /
|
||||
`limitPaused`. Reserve error handling for the four codes in the table below.
|
||||
|
||||
**2. `stop` and `blocked` fire only for `claude` sessions.** Both come from Claude
|
||||
Code hooks, and no other mode installs them: `shell` runs no agent, and the external
|
||||
CLIs (`opencode`, `codex`, `gemini`, `antigravity`, `pi`) render their own TUIs and post
|
||||
no hooks. For every non-`claude` mode only `idle`, `working` and `exit` are
|
||||
accepted, and of those only `exit` is dependable: see the caveats under
|
||||
[Signals](#signals) before building on `idle`. Requesting `stop` or `blocked`
|
||||
**explicitly** on such a session is a
|
||||
`400`; omitting `until` never fails, the server just drops them from the default set
|
||||
and echoes the narrowed set back as `wait.until`. Three more places hooks can go
|
||||
missing even in `claude` mode: a **Docker case** needs
|
||||
`CODEMAN_DOCKER_BRIDGE_HOOKS=1`, since a container cannot reach a loopback-bound
|
||||
Codeman (without it, only `idle` / `working` / `exit` work); a **remote-SSH
|
||||
case** runs the agent on another host, whose hooks may never reach this server at
|
||||
all; and a case whose hook config was written by **Codeman < 1.13.0 against an
|
||||
`--https` install** carries hook curls without `-k`, which TLS-fail silently (the
|
||||
hook line ends in `|| true`). Codeman now writes `curl -sk` and repairs a stale
|
||||
case config the next time a session starts in that case. When in doubt, ask for
|
||||
`stop,idle,exit` so a session without hooks still resolves on the heuristic
|
||||
signal.
|
||||
|
||||
**3. `from=now` does not mean "printed after you asked".** tmux repaints the visible
|
||||
screen on attach, on resize, and on any TUI redraw, and a repaint arrives as
|
||||
ordinary output, so text that was already on screen can satisfy a fresh wait. This
|
||||
was observed live: a marker echoed a minute earlier matched instantly on a new
|
||||
`from=now` wait. It is inherent to running the agent under a multiplexer, so the
|
||||
contract is a **marker unique to each call** (`MARK="DONE_$RANDOM"`, send
|
||||
`echo $MARK`, then wait on `$MARK`), never a generic string like `BUILD OK`.
|
||||
|
||||
### Signals
|
||||
|
||||
| Signal | Source | Actually fires for |
|
||||
|--------|--------|--------------------|
|
||||
| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) |
|
||||
| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) |
|
||||
| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only |
|
||||
| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below |
|
||||
| `exit` | no process is behind the session | every mode |
|
||||
|
||||
`stop` is the signal to orchestrate on where it exists; `idle` is a heuristic
|
||||
fallback that can flap mid-turn when a spinner pauses. The default set when `until`
|
||||
is omitted is `stop,idle,exit` (`exit` is in there so a worker that crashes resolves
|
||||
the wait promptly instead of burning the caller's whole timeout on something that
|
||||
can no longer happen). On a `claude` worker, prefer an explicit `until=stop,exit`
|
||||
once the session is up: the default set's `idle` also resolves on a spinner pause,
|
||||
and on a fresh session the **startup** `idle` (emitted when the CLI first comes up)
|
||||
can land inside your first wait window and report a turn that never ran. Measured:
|
||||
a session parked on the trust dialog emits no *further* `idle`, so it is the
|
||||
startup transition, not the dialog, that produces the false success below.
|
||||
|
||||
⚠️ **`exit` means "nothing is running", which includes "not started yet".** The
|
||||
server answers from `pid === null` plus a mux-layer pane-death probe, and that
|
||||
covers a session that exited — including a worker that died *inside* its tmux pane
|
||||
while the local attach client (and therefore `pid`) lives on — one that was
|
||||
detached, and one that was **created but never started**. So the first wait
|
||||
after `POST /api/v1/sessions` returns `{"signal":"exit","immediate":true}` in
|
||||
milliseconds, and reading that as "the worker died" is wrong: it means start it, or
|
||||
wait for it to come up. `status` is carried alongside so nothing is hidden. The
|
||||
alternative (trusting `status`) is worse, because a dead PTY parks the session at
|
||||
`status: "idle"`, which would answer the default wait with `immediate: true` for a
|
||||
worker that has crashed. A worker dying while a wait is parked resolves it within
|
||||
a few seconds (a background death-watcher), not at the timeout.
|
||||
|
||||
⚠️ **`blocked` is reachable less often than the table suggests.** It fires on two
|
||||
hooks, and the default configuration suppresses one of them: Codeman spawns claude
|
||||
with `--dangerously-skip-permissions`, so permission prompts do not happen unless the
|
||||
instance is switched to the `auto` Claude mode (App Settings), or the caller is a
|
||||
multi-user account without the bypass grant, which is forced to `--permission-mode
|
||||
auto`. What does still fire under the default is `elicitation_dialog`, the agent
|
||||
asking the user a question. So `until=stop,blocked,exit` is a reasonable belt on a
|
||||
long turn, but a worker that never comes back is far more likely to be working than
|
||||
blocked, and polling `blocked` alone will sit at its timeout.
|
||||
|
||||
⚠️ **On a `shell` session, only `exit` and marker-matching are dependable.** A shell
|
||||
session emits its one `idle` at startup and then stays `status: "idle"` forever,
|
||||
whatever the pane is doing, so it never emits a *transition*. Since send-and-wait
|
||||
requires a transition (and so does `fresh=1`), both can only time out there:
|
||||
a documented default `wait` on a shell worker running `sleep 4` times out at the
|
||||
full 25 s. Synchronize hook-less sessions with `wait-output` and a unique marker
|
||||
instead. The same caution applies to the external CLIs.
|
||||
|
||||
### Readiness is not a signal
|
||||
|
||||
Nothing here reports "the agent is ready for a prompt", and no combination of
|
||||
`until`/`fresh` synthesizes one. A freshly created session reads as `exit` (above),
|
||||
and a `claude` worker in a brand-new case comes up on the CLI's **trust dialog**,
|
||||
which contains a ❯ prompt of its own. Send-and-wait posted at that moment types the
|
||||
prompt into the dialog, where the `\r` never gets past it, while the session's
|
||||
startup `idle` lands inside the wait window: the wait resolves on `idle` in a
|
||||
couple of seconds with `timedOut: false`, which looks exactly like a completed
|
||||
turn.
|
||||
|
||||
The reliable sequence is: poll `GET /api/v1/sessions/:id` until `.data.pid` is
|
||||
non-null, then `wait-output` for the composer's own marker (`bypass`, the status
|
||||
bar of a CLI spawned in bypass mode) with a short timeout, handling the trust
|
||||
dialog only as the bounded fallback (`trust` matched → send `\r` → wait for
|
||||
`bypass` again). Do not probe `trust` first and Enter blindly: the dialog text
|
||||
stays in the terminal buffer for the life of the session, so a `trust` probe with
|
||||
`from=buffer` keeps matching on every later run and the Enter lands in a ready
|
||||
composer. A worked version is in
|
||||
[`extending-codeman.md`](extending-codeman.md#seam-3-http-api-and-cli).
|
||||
|
||||
### `GET /api/v1/sessions/:id/wait`
|
||||
|
||||
| Param | Type | Default | Notes |
|
||||
|-------|------|---------|-------|
|
||||
| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback |
|
||||
| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` |
|
||||
| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time |
|
||||
|
||||
```bash
|
||||
curl -s "$API/api/v1/sessions/$SID/wait?until=stop,exit&timeout=60000"
|
||||
```
|
||||
|
||||
Both GET wait routes answer with `Cache-Control: no-store`, because the documented
|
||||
pattern polls one identical URL in a loop and a cached `{"timedOut":true}` would
|
||||
turn that loop into a busy spin. `POST .../input` sends no cache header (it is a
|
||||
POST, which is not heuristically cacheable).
|
||||
|
||||
⚠️ **Unknown query parameters are ignored, not rejected**, with one exception
|
||||
(`regex`, below). In particular `match=` on `/wait` is silently dropped and you get
|
||||
a plain signal wait, so check the endpoint path before blaming the parameters.
|
||||
|
||||
### `GET /api/v1/sessions/:id/wait-output`
|
||||
|
||||
| Param | Type | Default | Notes |
|
||||
|-------|------|---------|-------|
|
||||
| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found |
|
||||
| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing |
|
||||
| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking |
|
||||
| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` |
|
||||
|
||||
**Matching is literal, never a pattern.** A `regex` parameter is rejected with a
|
||||
`400` rather than ignored, so a caller that assumed otherwise finds out immediately
|
||||
instead of waiting on the wrong thing. The reasoning is in
|
||||
[`architecture-invariants.md`](architecture-invariants.md#agent-wait-primitives).
|
||||
|
||||
#### What the matcher actually sees
|
||||
|
||||
The matcher scans the raw PTY stream, **normalized**: ANSI escape sequences are
|
||||
stripped — CSI, OSC, and the charset-designation escapes a stock bash prompt emits
|
||||
on every line (`ESC ( B`), so `match=tnode:` matches a prompt that renders
|
||||
`…@tnode:` — a partial escape arriving at a chunk boundary is held back until its
|
||||
tail arrives, and a match may straddle PTY chunks: `printf STRAD; sleep 1; printf
|
||||
DLEQQ` is matchable as `STRADDLEQQ` (all measured live). Three caveats remain:
|
||||
|
||||
⚠️ **It is still the byte stream, not the rendered pane.** `GET .../terminal`
|
||||
answers from a tmux screen capture (`data.source: "mux-visible"`), the finished
|
||||
picture; the matcher sees the stream that painted it. For linear output the two
|
||||
agree once escapes are stripped, but a full-screen TUI composes its picture with
|
||||
cursor positioning, so what the pane shows and what the stream carries can differ.
|
||||
Seeing your string in `terminal?tail=` makes a match likely, not guaranteed.
|
||||
|
||||
⚠️ **A TUI's text can arrive without its spaces.** Claude Code positions words
|
||||
with cursor moves rather than printing spaces, so screen text can reach the
|
||||
matcher as `Quicksafetycheck:Isthisaprojectyoucreated...`. Whether a given phrase
|
||||
keeps its spaces depends on how the TUI happened to draw it (measured: `I trust
|
||||
this folder` matched, `Quick safety check` did not), so a multi-word `match`
|
||||
against a TUI pane is unreliable rather than impossible. Match a **single
|
||||
space-free token**, ideally one you printed yourself. Plain command output (a
|
||||
shell worker, an `echo`) keeps its spaces.
|
||||
|
||||
⚠️ **The returned `snippet` is a rendering of the matched text, not a quotation of
|
||||
it.** It is cut from the same normalized stream the match ran against, then
|
||||
cleaned for display: remaining raw control bytes are removed (an agent pipes the
|
||||
snippet into its own terminal, so a worker's bytes must not be able to reset that
|
||||
display) and blank runs are collapsed. A printable needle that matched will appear
|
||||
in it; a needle containing control bytes or a blank run may not survive verbatim.
|
||||
|
||||
```bash
|
||||
MARK="DONE_$RANDOM"
|
||||
curl -sG "$API/api/v1/sessions/$SID/wait-output" \
|
||||
--data-urlencode "match=$MARK" --data-urlencode 'timeout=120000'
|
||||
```
|
||||
|
||||
Build the query with `-G --data-urlencode` rather than by hand: a `+` in a
|
||||
hand-written query string decodes to a space.
|
||||
|
||||
### `POST /api/v1/sessions/:id/input` with `wait`
|
||||
|
||||
Two optional fields on the existing endpoint:
|
||||
|
||||
| Field | Type | Notes |
|
||||
|-------|------|-------|
|
||||
| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" |
|
||||
| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` |
|
||||
|
||||
Both are `nullish`, so an explicit `null` from `JSON.stringify` is accepted as
|
||||
"absent" rather than failing validation. That is deliberate: `.optional()` would
|
||||
reject it, which has shipped as a real bug twice.
|
||||
|
||||
The input must end with `\r` (a real carriage return in the JSON string): Enter is
|
||||
sent only when the input contains one, so text without it is typed onto the
|
||||
worker's prompt but never submitted, and the wait then runs its full timeout on a
|
||||
turn that never started. Verified live; this is the most common silent failure on
|
||||
this endpoint.
|
||||
|
||||
```bash
|
||||
curl -s -X POST "$API/api/v1/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"run the tests\r","useMux":true,"clientId":"agent-1","seq":1,
|
||||
"wait":"stop","waitTimeout":600000}'
|
||||
```
|
||||
|
||||
A **tagged duplicate** (a `clientId` + `seq` pair the server has already applied)
|
||||
still honors `wait`, because the caller's question is unanswered, but it answers
|
||||
from the session's current state rather than requiring a new transition: the
|
||||
original turn may be long over. It comes back as
|
||||
`"delivered": false, "duplicate": true`.
|
||||
|
||||
### Response
|
||||
|
||||
All three nest the wait result under `data.wait`, so one client helper works against
|
||||
any of them:
|
||||
|
||||
```json
|
||||
{ "success": true, "data": {
|
||||
"sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464",
|
||||
"status": "idle",
|
||||
"limitPaused": false,
|
||||
"wait": {
|
||||
"signal": "stop", "until": ["stop", "idle", "exit"],
|
||||
"timedOut": false, "immediate": false, "ended": false, "aborted": false,
|
||||
"waitedMs": 8421, "timeoutMs": 60000
|
||||
}
|
||||
}}
|
||||
```
|
||||
|
||||
`POST .../input` returns the same `wait` object alongside `delivered`, `duplicate`,
|
||||
`status` and `limitPaused`. `POST .../input` **without** `wait` is unchanged and
|
||||
still returns `{"success": true, "data": {}}`.
|
||||
|
||||
⚠️ `delivered: false` has **two** meanings, and they must be told apart by
|
||||
`duplicate`: with `duplicate: true` the input was suppressed as an already-applied
|
||||
redelivery (harmless, the turn it refers to may be long over), while with
|
||||
`duplicate: false` the **write failed** (typically no PTY behind the session). A
|
||||
client that reads `delivered === false` as "duplicate" silently treats a failed send
|
||||
as a success.
|
||||
|
||||
| Field | Type | Meaning |
|
||||
|-------|------|---------|
|
||||
| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) |
|
||||
| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) |
|
||||
| `wait.matched` | boolean | the string appeared (`/wait-output` only) |
|
||||
| `wait.match` | string | the literal that was searched for (`/wait-output` only) |
|
||||
| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) |
|
||||
| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` |
|
||||
| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) |
|
||||
| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met |
|
||||
| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome |
|
||||
| `wait.waitedMs` | number | wall-clock ms actually spent waiting |
|
||||
| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied |
|
||||
| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand |
|
||||
| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard |
|
||||
|
||||
Read the outcome by discriminator, in this order:
|
||||
|
||||
1. `wait.signal !== null` (or `wait.matched === true`): the thing happened.
|
||||
2. `wait.timedOut`: a poll boundary. Loop again.
|
||||
3. `wait.ended` or `wait.aborted`: the wait answered nothing, because the session is
|
||||
gone or was never running. Re-check the session instead of looping.
|
||||
|
||||
`wait.immediate` is not a fourth outcome: it rides along with the first one and
|
||||
means the condition already held at call time, so nothing was actually waited for.
|
||||
If that is not what you meant, you wanted `fresh=1` or the send-and-wait form. Note
|
||||
that `{"signal":"exit","immediate":true}` on a session you just created is the
|
||||
not-started-yet case, not a crash.
|
||||
|
||||
**The timeout is clamped, so read it back.** A request for 1800000 ms is silently
|
||||
reduced to the server's ceiling (600000 ms by default, operator-tunable), and a
|
||||
request for 1 ms is raised to 1000 ms. `wait.timeoutMs` is the value that was
|
||||
applied. Without checking it, a caller that asked for 30 minutes and got 10 will
|
||||
read the timeout as "the worker is wedged" and kill a session that was working fine.
|
||||
|
||||
### Errors
|
||||
|
||||
| `errorCode` | HTTP | When |
|
||||
|-------------|------|------|
|
||||
| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` |
|
||||
| `NOT_FOUND` | 404 | no such session, or one this caller does not own |
|
||||
| `SESSION_BUSY` | 409 | this session's waiter cap is full |
|
||||
| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem |
|
||||
|
||||
The two capacity codes are deliberately different. A process-wide cap reported as
|
||||
`SESSION_BUSY` would tell the caller to switch sessions, which cannot help. The
|
||||
error message names the cap that was hit.
|
||||
|
||||
⚠️ A `401` is **not** in this table and is not an envelope at all (see
|
||||
[Response envelope](#response-envelope)). It matters most here: a polling loop that
|
||||
pipes each wait straight into `jq` fails with a parse error on every iteration
|
||||
against a password-protected server, which reads as "the wait endpoints are broken".
|
||||
Check the status first.
|
||||
|
||||
The per-session cap is a **combined** budget: signal waiters and output waiters
|
||||
count against the same 16, not 16 of each. An abandoned request no longer holds its
|
||||
slot, because the routes release the waiter when the client disconnects, but a
|
||||
client that opens many concurrent waits against one session will still hit the cap.
|
||||
|
||||
## Session lineage (`parentSessionId`)
|
||||
|
||||
A create request may name the session that spawned it, which the web UI draws as a
|
||||
line between the two tabs. Accepted on `POST /api/v1/sessions` and
|
||||
`POST /api/v1/quick-start`, either way:
|
||||
|
||||
```bash
|
||||
# as a body field
|
||||
-d '{"caseName":"worker-1","mode":"claude","parentSessionId":"'"$CODEMAN_SESSION_ID"'"}'
|
||||
|
||||
# or as a header, which is what an agent driving many spawns should use: set it once
|
||||
# on the curl invocation and every spawn call carries it
|
||||
-H "X-Codeman-Parent-Session: $CODEMAN_SESSION_ID"
|
||||
```
|
||||
|
||||
The body field wins if both are present. The value is resolved against live sessions
|
||||
(exact id, or a unique prefix of at least 8 characters) and must belong to the same
|
||||
owner as the session being created.
|
||||
|
||||
**It cannot fail your spawn.** An unknown, stale, foreign or malformed value is
|
||||
silently dropped and the session is created without lineage — never a `400`. It is
|
||||
also pure decoration: it confers no permission, and a child is unaffected by its
|
||||
parent exiting. It appears on session state as `parentSessionId` (absent when
|
||||
unresolved) and survives a server restart.
|
||||
|
||||
## Approvals Inbox
|
||||
|
||||
Cross-session queue of prompts waiting on a human (permission dialogs,
|
||||
AskUserQuestion questions, idle prompts). Claude-mode sessions only; items are
|
||||
in-memory (a server restart drops them; the next prompt re-fires the hook).
|
||||
Design: [`approvals-inbox-plan.md`](approvals-inbox-plan.md).
|
||||
|
||||
- `GET /api/v1/approvals` → `{ approvals: ApprovalItem[] }`, oldest first,
|
||||
ownership-scoped in multi-user mode. `ApprovalItem`: `{ id, sessionId,
|
||||
sessionName, kind: 'permission'|'question'|'idle', createdAt, toolName?,
|
||||
toolSummary?, message?, cwd?, context?, options?: {n, label}[],
|
||||
acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame;
|
||||
`options` is present only when the dialog's numbered choices parsed
|
||||
confidently; `acknowledgedAt` marks an item a human has already looked at
|
||||
(see `/viewed` below) and tells clients not to re-arm its tab alert. Listing
|
||||
also runs a staleness sweep over the caller's own items: the pane is
|
||||
re-captured, and an item whose dialog no longer parses is resolved as
|
||||
`resolved_in_terminal` instead of being returned (only items whose original
|
||||
frame parsed `options` can be dropped this way, so an unreadable capture
|
||||
keeps the item).
|
||||
- `POST /api/v1/approvals/:id/answer` with `{ action: 'approve' }` (sends the
|
||||
digit `1`), `{ action: 'deny' }` (sends Esc), `{ action: 'option', option: n }`
|
||||
(sends the digit; accepted only when `n` is among the item's parsed
|
||||
`options`), or `{ action: 'text', text }` (idle prompts only; submits the
|
||||
line as a prompt). `404 NOT_FOUND` when the item is no longer pending,
|
||||
`409 CONFLICT` when the dialog left the screen or another actor answered
|
||||
first, `422 OPERATION_FAILED` when the session refused input.
|
||||
- `POST /api/v1/approvals/:id/dismiss` removes the item without keystrokes.
|
||||
- `POST /api/v1/approvals/session/:sessionId/viewed` → `{ sessionId,
|
||||
acknowledged: itemId | null }`. Marks the session's pending **idle** item as
|
||||
seen by a human (the web UI calls it when you open the session's tab): the
|
||||
item stays pending and answerable, but stops arming the yellow tab alert on
|
||||
every client, including after a reload. Permission/question items are never
|
||||
acknowledged this way, since looking at a dialog does not answer it. `404`
|
||||
for an unknown or inaccessible session; acknowledging twice is a no-op
|
||||
(`acknowledged: null`).
|
||||
|
||||
SSE events: `approval:pending` (full item), `approval:updated` (context/options
|
||||
re-captured, or the item acknowledged), `approval:resolved` (`{ id, sessionId, kind, resolution }` with
|
||||
`resolution` one of `answered | resolved_in_terminal | superseded |
|
||||
session_ended | dismissed | expired`).
|
||||
|
||||
## Read My Mind intent profiles
|
||||
|
||||
Per-case profiles of what the user is trying to accomplish: user/agent-stated
|
||||
goals plus the user's recently submitted prompts, captured from the Claude
|
||||
session transcript while the opt-in `readMyMindEnabled` setting is on (default
|
||||
OFF). Keyed by owner + workingDir, so the profile survives `/clear`, respawns,
|
||||
and session churn. Stored in `~/.codeman/intents.json` (mode 0600); never fed
|
||||
into `/api/v1/search`. Design: [`readmymind-plan.md`](readmymind-plan.md);
|
||||
user guide: [`readmymind.md`](readmymind.md).
|
||||
|
||||
- `GET /api/v1/sessions/:id/intent` -> `{ intent: IntentProfile }` for the
|
||||
session's case. `IntentProfile`: `{ key, workingDir, updatedAt, goals,
|
||||
recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap
|
||||
50, each <= 500 chars). A case with nothing recorded answers an empty
|
||||
profile with `updatedAt: 0`; nothing is persisted by reads.
|
||||
- `PUT /api/v1/sessions/:id/intent` with `{ goals }` (<= 8192 chars, strict
|
||||
schema) replaces the goals text and answers the updated profile.
|
||||
`400 INVALID_INPUT` on over-long or unknown fields.
|
||||
- `DELETE /api/v1/sessions/:id/intent` -> `{ deleted: boolean }` forgets the
|
||||
case's profile entirely.
|
||||
- `POST /api/v1/sessions/:id/readmymind` predicts the user's next prompt:
|
||||
a one-shot model call over the intent profile plus live session signals
|
||||
(pending approval dialog, transcript tail, git state, run-summary events,
|
||||
sibling sessions). Body is optional; the rethink flow passes
|
||||
`{ steer?, rejected? }` (strict schema: `steer` <= 2000 chars, `rejected`
|
||||
up to 10 strings <= 1000 chars). Answers
|
||||
`{ suggestions: { prompt, why, kind }[], durationMs }` with 1-3 suggestions
|
||||
(`kind`: `continue` | `verify` | `redirect`; prompts are single-line).
|
||||
Claude-mode sessions only (`400 INVALID_INPUT` otherwise); one prediction in
|
||||
flight per session (`409 CONFLICT`); predictor failures answer
|
||||
`502 OPERATION_FAILED`. Takes 5-90 s and costs real tokens. Suggestions are
|
||||
only ever returned, never sent: submitting one is the caller's explicit act.
|
||||
|
||||
All four enforce session ownership in multi-user mode; a foreign session id
|
||||
answers `404 NOT_FOUND` (no existence leak), and profiles of two owners of the
|
||||
same directory are distinct by construction.
|
||||
|
||||
## Voice dictation
|
||||
|
||||
Browser dictation transcribed through this server's Claude Code login, i.e. the
|
||||
same speech-to-text service the CLI's own `/voice` mode uses. Gated on the synced
|
||||
`claudeVoiceEnabled` setting (default OFF). Design:
|
||||
[`claude-voice-plan.md`](claude-voice-plan.md).
|
||||
|
||||
- `GET /api/v1/voice/status` -> `{ available, reason?, subscriptionType?,
|
||||
expiresAt? }`. `reason` is `disabled` (setting off), `no-credentials` (nobody
|
||||
signed in to Claude Code on the server), `expired` (the access token elapsed;
|
||||
running any Claude session refreshes it) or `malformed`. The OAuth token
|
||||
itself is never returned by this or any other endpoint.
|
||||
- `GET /ws/voice/stream?language=&keyterms=` (WebSocket, not under `/api`)
|
||||
relays one dictation. Client sends binary frames of signed 16-bit
|
||||
little-endian PCM, 16 kHz mono (<= 64 KB per frame), plus JSON control frames
|
||||
`{"t":"finalize"}` (ask for the final transcript) and `{"t":"stop"}`. Server
|
||||
sends `{"t":"ready"}`, `{"t":"transcript","text","final"}` (each frame is the
|
||||
WHOLE running transcript, not a delta), `{"t":"error","message"}` and
|
||||
`{"t":"closed"}`. Close codes: `4003` disallowed Host/Origin, `4004`
|
||||
unavailable (reason in the close reason), `4008` too many concurrent streams.
|
||||
Streams are capped in count and length (`src/config/voice.ts`).
|
||||
|
||||
## Authentication
|
||||
|
||||
Optional HTTP Basic (`CODEMAN_USERNAME`/`CODEMAN_PASSWORD`) → opaque
|
||||
@@ -82,6 +549,29 @@ the stable contract — event names are not renamed without a major bump. An
|
||||
optional `?sessions=<id,...>` filter suppresses only the high-volume terminal
|
||||
stream; lifecycle/metadata events are delivered to all clients regardless.
|
||||
|
||||
### `sse:heartbeat` (liveness)
|
||||
|
||||
Every 15s the server writes a `sse:heartbeat` frame to every connected client:
|
||||
|
||||
```
|
||||
event: sse:heartbeat
|
||||
data: {"t":1755100000000}
|
||||
```
|
||||
|
||||
`t` is the server's epoch-ms timestamp at write time. The frame carries no
|
||||
application state and can be ignored for correctness. It exists so a client can
|
||||
tell a live stream from a dead one: an `EventSource` whose connection has been
|
||||
idle-closed by a proxy (or that resumed from sleep on a stale socket) keeps
|
||||
delivering nothing without ever firing `onerror`. Clients that care should treat
|
||||
silence longer than about three intervals as a dead stream and reconnect, which
|
||||
is what the bundled frontend does.
|
||||
|
||||
This replaced a `:keepalive` SSE **comment**, which served the same
|
||||
proxy-flushing purpose but is invisible to `EventSource` by spec and so could
|
||||
never be observed by a client. Consumers written against the old behavior are
|
||||
unaffected: `EventSource` dispatches only events that have a registered
|
||||
listener, so an unknown event name is dropped.
|
||||
|
||||
## Consuming from JavaScript
|
||||
|
||||
The bundled frontend reads responses through `_apiJson()`
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
# Approvals Inbox (design)
|
||||
|
||||
One cross-session inbox for every prompt that is waiting on a human: permission dialogs, questions (AskUserQuestion / elicitation), and idle prompts. Cards are answerable in place (option digits, Esc, or a typed prompt) from desktop, phone overview, and push notification action buttons. Inspired by Cloudflare OS's Gatekeeper approval queue (https://github.com/cloudflare/cloudflare-os, asynchronous human-in-the-loop approvals): with a fleet of sessions the human is the bottleneck, and today answering means finding the right tab.
|
||||
|
||||
## Problems this fixes (all real today)
|
||||
|
||||
1. **No cross-session surface.** Pending prompts exist only as per-tab alert colors (`tab-alert-action`/`tab-alert-idle`) and NEEDS YOU rows on the phone overview. Answering means switching to the session and typing.
|
||||
2. **Alerts die on reload.** `pendingHooks` lives only in `app.js` memory, fed by transient SSE `hook:*` events. A page reload (or a phone browser evicting the tab) silently loses every pending alert. There is no server-side record.
|
||||
3. **Push Approve/Deny buttons are dead.** `PUSH_EVENT_MAP` already attaches `approve`/`deny` actions to permission pushes, and `sw.js` forwards `event.action` to the page, but the `notification-click` handler in settings-ui.js ignores it (and when no tab is open, the action is dropped entirely). The buttons render on the lock screen and do nothing.
|
||||
4. **Card context is missing.** The frontend handlers read `data.question` / `data.message` / `data.tool`, but `sanitizeHookData` never forwards `message`, so notifications show generic fallback text.
|
||||
|
||||
## Scope
|
||||
|
||||
- Claude mode only (hooks fire only for `claude`; external CLIs keep their output-stabilization heuristics and get no inbox items). This mirrors the wait-primitive `stop`/`blocked` gating.
|
||||
- Permission prompts occur for sessions running `ClaudeMode` `normal` / `auto` / `allowedTools` (and the trust-folder dialog even under skip-permissions). Question and idle prompts occur in every mode including `dangerously-skip-permissions`.
|
||||
- In-memory store (plus the frontend seeding from it on load). Server restart drops items; hooks re-fire on the next prompt. No new state file in v1.
|
||||
|
||||
## Data model
|
||||
|
||||
At most **one active item per session**: the Claude TUI shows one dialog at a time, so a new prompt event supersedes the session's previous item (resolution `superseded`).
|
||||
|
||||
```ts
|
||||
interface ApprovalItem {
|
||||
id: string; // `${sessionId}:${seq}`
|
||||
sessionId: string;
|
||||
sessionName: string;
|
||||
kind: 'permission' | 'question' | 'idle';
|
||||
createdAt: number;
|
||||
toolName?: string; // from sanitized hook data
|
||||
toolSummary?: string; // command / file_path / description, already bounded
|
||||
message?: string; // Notification hook `message` (newly allowlisted)
|
||||
cwd?: string;
|
||||
context?: string; // ANSI-stripped visible pane frame tail, ≤ 4000 chars
|
||||
options?: { n: number; label: string }[]; // parsed from context when confident
|
||||
}
|
||||
```
|
||||
|
||||
Resolutions (server-emitted, item removed from pending): `answered` (via inbox), `resolved_in_terminal` (stop / elicitation_complete / elicitation_response / session went working), `superseded`, `session_ended`, `dismissed`, `expired` (12h TTL sweep).
|
||||
|
||||
## Backend
|
||||
|
||||
### Store: `src/approval-inbox.ts`
|
||||
|
||||
Module-level singleton in the style of `session-wait-registry.ts` (pure, no `Session` import, injected emit callback so there is no import cycle with the server):
|
||||
|
||||
- `notePrompt(info)` creates/supersedes the session's item; schedules ONE re-capture ~600ms later (the Notification hook can fire before the dialog finishes painting) which updates `context`/`options` and emits `approval:updated`.
|
||||
- `resolveForSession(sessionId, reason)`, `dismiss(id)`, `answerable(id)`, `listPending()`, `stop()` (clears timers; tests).
|
||||
- Option parsing (pure, unit-tested): consecutive `❯? N. label` lines, 2..6 options, labels ≤ 120 chars. Parsed options gate which digits the answer endpoint accepts; when parsing fails the card falls back to Approve(1)/Deny(Esc) only.
|
||||
- TTL: items expire after 12h (checked on read + a lazy sweep; no standing interval).
|
||||
|
||||
### Wiring
|
||||
|
||||
- `hook-event-routes.ts`: on `permission_prompt` / `elicitation_dialog` / `idle_prompt`, call `notePrompt` with sanitized data + a pane capture callback (`mux.capturePaneBuffer(muxName)` visible frame, ANSI-stripped via existing utils; fall back to `session.terminalBuffer` tail). On `stop` / `elicitation_complete` / `elicitation_response`, `resolveForSession(id, 'resolved_in_terminal')`.
|
||||
- `session-listener-wiring.ts`: `working` listener resolves **idle items only** (`working` is heuristic and can flap mid-turn, so it must never clear a pending permission/question dialog); `exit` resolves with `session_ended`. Same singleton-import pattern as `sessionWaits`.
|
||||
- Session delete route: resolve with `session_ended`.
|
||||
- **New hook matchers** `elicitation_complete` + `elicitation_response` added to `generateHooksConfig()`, `HookEventType`, `HookEventSchema`, and both SSE registries. `refreshStaleCodemanHooks` gets a staleness probe for them (`hooksJson.includes('elicitation_complete')`) so existing cases heal on next Claude spawn, exactly like the `-k`/secret/marker probes.
|
||||
- `sanitizeHookData`: allowlist `message` (bounded 500 chars). This also un-deadens the existing notification text paths.
|
||||
|
||||
### Routes: `src/web/routes/approval-routes.ts`
|
||||
|
||||
Normal authed API (NOT the hook-secret bypass), `ApiResponse` envelope, Zod schemas in `schemas.ts`:
|
||||
|
||||
- `GET /api/approvals` → pending items, multi-user filtered by `canAccessOwned` (same policy as session lists). Also sweeps the caller's own items for staleness through `verifyStillAnswerable()`: Claude Code fires no "permission answered" hook, so a dialog answered in the terminal used to sit pending until `stop` and re-arm a red tab alert on the next page load. Only items whose original frame parsed options can be dropped this way, so an unreadable capture keeps the alert.
|
||||
- `POST /api/approvals/:id/answer` body `{ action: 'approve' | 'deny' | 'option' | 'text', option?, text? }`:
|
||||
- `approve` → `writeViaMux('1')` (option 1 is always plain Yes; no Enter, menus react to the digit).
|
||||
- `deny` → `writeViaMux('\x1b')` (Esc is the official No/cancel; precedent: auto-resume sends Esc the same way).
|
||||
- `option` → digit `String(n)`; accepted only when `n` is within the item's parsed options (prevents blind digit-poking at an unparsed dialog).
|
||||
- `text` → `idle` items only: single line, embedded newlines stripped, sent as `text\r` (the `\r` discipline from CLAUDE.md).
|
||||
- Guards: item still pending (404 otherwise), session exists + ownership via `findSessionOrFail`, session mode installs hooks. **Answer-time re-capture**: for items whose frame parsed options, the pane is re-captured before sending; if the dialog no longer parses, the item resolves and the answer is refused with 409 (the keystroke would land in whatever now has focus). Marks `answered` BEFORE the write so a double-tap cannot double-send; rolls back to pending if the write fails.
|
||||
- `POST /api/approvals/:id/dismiss` → remove without keystrokes.
|
||||
- `POST /api/approvals/session/:sessionId/viewed` → acknowledge the session's pending **idle** item (`acknowledgedAt`, emitted as `approval:updated`). Added after the owner reported that a yellow tab clicked and checked went yellow again on reload: the view-clears-idle rule lived in one browser's memory, so the seed re-armed it and other devices never saw the clear. Acknowledgement is deliberately **not** resolution (the prompt is still unanswered, so it stays in the inbox and stays available as Read My Mind context), and deliberately **idle-only** (looking at a permission/question dialog does not answer it, so the red alert survives being viewed).
|
||||
|
||||
### SSE
|
||||
|
||||
`approval:pending`, `approval:updated`, `approval:resolved` in `sse-events.ts` + `SSE_EVENTS` in constants.js (the parity test pins the sync). Broadcasts carry `sessionId`, so multi-user SSE scoping applies unchanged.
|
||||
|
||||
### Push
|
||||
|
||||
- `sendPushNotifications` payload gains `approvalId` for the three hook events. Both `approvalId` and the Approve/Deny `actions` are **gated on the opt-in setting**: with it off, permission pushes carry no buttons at all (pre-inbox they rendered and did nothing, so stripping them is the honest shape).
|
||||
- `sw.js` `notificationclick`: when `event.action` is `approve`/`deny`, POST `/api/approvals/:id/answer` directly from the worker (same-origin, cookie credentials) so the buttons work **with no tab open**; on failure fall back to focusing/opening a tab. Non-action clicks keep today's behavior.
|
||||
- Page-side `notification-click` handler: honor `action` instead of dropping it (also setting-gated, for stale notifications sent before the toggle flipped).
|
||||
- Question/idle pushes keep no action buttons (options vary per dialog); tapping opens the inbox.
|
||||
|
||||
## Frontend
|
||||
|
||||
New module `approvals-ui.js` (@loadorder 11.2, after panels-ui.js), prettier-formatted (not added to `.prettierignore`).
|
||||
|
||||
- **Seed on connect**: `GET /api/approvals` on init and SSE reconnect; each pending item re-feeds `setPendingHook(...)` so tab alerts and the phone overview survive reload (fixes problem 2 with zero changes to the alert state machine). Items carrying `acknowledgedAt` are skipped, and `markIdleAlertSeen()` (app.js) is what sets it: viewing a session clears its yellow locally and POSTs `.../viewed`, so "I checked it" survives the reload and reaches the user's other devices through `approval:updated`.
|
||||
- **Desktop**: header bell `btn-approvals` with count badge. Ships default-hidden via marker class `btn-approvals--hidden` (same policy as the attachments button, so `test/mobile-header-buttons-policy.test.ts` excludes it from the default-visible enumeration); JS shows it only while count > 0. Click toggles a drawer of cards: session name + kind, tool/message summary, mono context block, buttons rendered from parsed options (else Approve/Deny), plus Dismiss and Open session. Esc closes; existing z-index layers respected.
|
||||
- **Phone**: header button stays hidden (`mobile.css`); the phone surface is the overview's NEEDS YOU section, whose rows gain inline ✓/✗ buttons for permission items (tap-through to the session remains the row's main action). Toolbar classes/status language rules from the mobile-overview section of CLAUDE.md apply.
|
||||
- **i18n**: new strings registered in i18n.js (en + zh-CN); status words carry `data-i18n-skip` where they would collide (mirroring the overview pills).
|
||||
- **Setting**: `approvalsInboxEnabled`, synced (in `SettingsUpdateSchema`), **default OFF** (owner decision: the entire feature is opt-in, meaning no bell, no drawer, no overview strips, no seeding, and no push action buttons until enabled in App Settings → Panels). Only the store and answer endpoints keep running regardless, so flipping the toggle ON surfaces anything already pending immediately, with no restart.
|
||||
|
||||
## Race honesty
|
||||
|
||||
The prompt can be answered in the terminal a moment before an inbox answer lands; then the keystroke would hit whatever now has focus (worst case: a digit typed into the composer, not submitted, since no `\r` is ever sent for menu answers). Mitigations, in order: answer-time re-capture (the dialog must still parse on screen or the answer is refused), answered-before-write marking, digit-only/Esc-only writes for menus, and the card's context block showing what the pane looked like when captured. This is the same class of risk `writeViaMux` automation (auto-resume, respawn) already accepts.
|
||||
|
||||
## Tests
|
||||
|
||||
- `test/approval-inbox.test.ts`: supersede per session, every resolution path, TTL, option parsing fixtures (2-option, 3-option with ❯, unparseable frame), re-capture update.
|
||||
- `test/routes/approval-routes.test.ts` (`app.inject`, no port): list; hook event creates item; answer approve/deny/option writes the exact bytes (test-PTY echo asserts them); text answers restricted to idle; 404 unknown id; 409 answered twice; option out of range rejected; multi-user scoping.
|
||||
- Existing suites extended: hook-event schema accepts the two new events; `sanitizeHookData` forwards bounded `message`; SSE parity + mobile-header policy pass as-is by construction.
|
||||
|
||||
## Docs
|
||||
|
||||
- CLAUDE.md: Key Patterns entry + SSE/route counts + frontend load order.
|
||||
- `docs/api-reference.md`: the two endpoints + three SSE events (additive, fine under the 0.9.x contract).
|
||||
+139
-17
File diff suppressed because one or more lines are too long
@@ -300,6 +300,11 @@ For reference when writing browser tests:
|
||||
.xterm // Terminal container
|
||||
#helpModal // Help modal
|
||||
#appSettingsModal // Settings modal
|
||||
#sessionOptionsModal // Session Options (same set-* surface)
|
||||
#createCaseModal // Add Case (same set-* surface)
|
||||
.set-rail-item // Rail entry: scrolls in App Settings, switches in the other two
|
||||
.set-section // A settings section (`.hidden` on the inactive ones outside App Settings)
|
||||
.set-row // One setting: label + description left, control right
|
||||
.modal-content // Modal content
|
||||
.modal-close // Modal close button
|
||||
.header-brand .logo // Logo text
|
||||
|
||||
@@ -149,9 +149,17 @@ to Claude as a system reminder. This implies `"async": true`; ordinary async
|
||||
hooks do not wake an idle turn, and their output waits for the next interaction.
|
||||
|
||||
Codeman uses this on `PostToolUse(Bash)`: a self-contained Node helper extracts
|
||||
the background task ID from the Bash result, watches the session transcript for
|
||||
the matching completion notification, and exits 2. It does not send terminal
|
||||
input, so it cannot submit a user's partially written prompt.
|
||||
the background task ID from the Bash result, watches the originating transcript
|
||||
and, for subagents, the top-level parent transcript for the matching completion
|
||||
notification, and exits 2. Claude records a subagent's Bash result in its
|
||||
`subagents/agent-*.jsonl` file but queues completion in the lead session JSONL.
|
||||
The task ID keeps each wake targeted. The helper does not send terminal input,
|
||||
so it cannot submit a user's partially written prompt.
|
||||
|
||||
For script-dispatched Codex work, `codex-run.sh` writes the final response
|
||||
between `CODEMAN_RESULT_BEGIN/END` markers in the background task output. The
|
||||
rewake helper includes a maximum of 64 KiB of that report in its feedback. UI
|
||||
subagent discovery and dispatcher result delivery are separate contracts.
|
||||
|
||||
### Notification
|
||||
|
||||
@@ -219,6 +227,16 @@ Or to allow exit:
|
||||
|
||||
**Use Cases**: Control nested loops, verify subagent output.
|
||||
|
||||
The hook input includes `agent_id`, `agent_transcript_path`, and
|
||||
`last_assistant_message`. Like `Stop`, a command hook can return
|
||||
`{"decision":"block","reason":"..."}` to keep the subagent running and feed
|
||||
the reason back to it.
|
||||
|
||||
Codeman uses this to prevent premature reports from workers that still own live
|
||||
Monitor or background-Bash processes. It derives candidate task IDs from the
|
||||
subagent transcript, but requires a matching live Linux process descriptor for
|
||||
`tasks/<id>.output`; historical task text by itself is not treated as active.
|
||||
|
||||
### TeammateIdle
|
||||
|
||||
**When**: When an agent-team teammate is about to go idle.
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
# Claude voice dictation in Codeman
|
||||
|
||||
Wire Codeman's existing mic button to the same speech-to-text service Claude Code's own
|
||||
`/voice` mode uses, so dictation works with **no third-party API key** for anyone already
|
||||
signed in to Claude Code on the server.
|
||||
|
||||
## Why the CLI's own voice mode cannot be reused directly
|
||||
|
||||
Claude Code 2.1.x ships voice input: `/voice hold|tap|off` arms it, the CLI opens the
|
||||
**host's** microphone (native `audio-capture-napi`, falling back to `sox`/`arecord` on Linux
|
||||
after probing `/proc/asound/cards`), streams PCM upstream and types the transcript into its
|
||||
own composer.
|
||||
|
||||
Every part of that is on the wrong machine for Codeman. The CLI runs inside a tmux pane on
|
||||
the server, which is typically headless and has no sound card at all, while the human is in
|
||||
a browser on a phone somewhere else. Toggling `/voice` in the pane from Codeman would arm a
|
||||
microphone nobody is sitting in front of. So Codeman keeps capturing audio in the browser,
|
||||
where the user actually is, and only borrows the CLI's **transcription backend**.
|
||||
|
||||
## The backend, as the CLI uses it
|
||||
|
||||
Extracted from the 2.1.226 binary (`connectVoiceStream`):
|
||||
|
||||
| | |
|
||||
| --- | --- |
|
||||
| URL | `wss://api.anthropic.com/api/ws/speech_to_text/voice_stream` |
|
||||
| Query | `encoding=linear16`, `sample_rate=16000`, `channels=1`, `endpointing_ms=300`, `utterance_end_ms=1000`, `language=<lang>`, `use_conversation_engine=true`, `stt_provider=deepgram-nova3` |
|
||||
| Headers | `Authorization: Bearer <Claude Code OAuth access token>`, `User-Agent`, `x-app: cli`, `anthropic-client-platform`, optional `x-config-keyterms` |
|
||||
| Audio | raw binary frames, PCM signed 16-bit little-endian, 16 kHz, mono |
|
||||
| Keepalive | `{"type":"KeepAlive"}` on open, then every 8 s |
|
||||
| Finalize | `{"type":"CloseStream"}`, then wait for the endpoint frame |
|
||||
| Downstream | `{"type":"TranscriptText"\|"TranscriptInterim","data":"…"}` (running interim), `{"type":"TranscriptEndpoint"}` (promotes the pending interim to final), `{"type":"TranscriptError",…}`, `{"type":"error","message":…}` |
|
||||
|
||||
Deepgram Nova-3 runs server-side, so the Deepgram-quality result arrives without a Deepgram
|
||||
account. Verified against the live endpoint before this design was written: connect, stream
|
||||
PCM, receive interims and an endpoint frame.
|
||||
|
||||
## Architecture
|
||||
|
||||
The browser cannot call that endpoint itself: it would need the OAuth bearer token in page
|
||||
JavaScript (and CORS would refuse anyway). So the audio goes browser → Codeman → Anthropic,
|
||||
and Codeman is the only thing that ever touches the token.
|
||||
|
||||
```
|
||||
mic → AudioWorklet (Float32 → PCM16 @16 kHz)
|
||||
→ wss://<codeman>/ws/voice/stream [cookie/basic auth, Origin+Host guarded]
|
||||
→ VoiceStreamRelay (reads ~/.claude/.credentials.json per connect)
|
||||
→ wss://api.anthropic.com/api/ws/speech_to_text/voice_stream
|
||||
← {"t":"transcript","text":…,"final":…} → existing _insertText() path
|
||||
```
|
||||
|
||||
Nothing about the insert path changes: the transcript lands in the same preview overlay,
|
||||
the same direct/compose insert modes, the same green Send button.
|
||||
|
||||
### Server pieces
|
||||
|
||||
- **`src/claude-credentials.ts`** — locate and parse the Claude Code OAuth credentials.
|
||||
`parseClaudeCredentials()` is pure (JSON string + `now` → status) and unit-tested;
|
||||
`readClaudeOAuthToken()` wraps it with IO: `$CLAUDE_CONFIG_DIR/.credentials.json` or
|
||||
`~/.claude/.credentials.json`, and on macOS the login keychain
|
||||
(`security find-generic-password -s "Claude Code-credentials"`).
|
||||
**Read-only, always.** Codeman never writes credentials and never refreshes the token: a
|
||||
refresh rotates the refresh token, and racing Claude Code's own refresh could sign the
|
||||
user out of their CLI. An expired token surfaces as a plain "run a Claude session to
|
||||
refresh" error instead.
|
||||
The token is never logged, never returned by any endpoint, and never sent to the browser.
|
||||
|
||||
- **`src/web/voice-stream.ts`** — pure `buildVoiceStreamUrl()` / `buildVoiceStreamHeaders()` /
|
||||
`sanitizeKeyterms()` (ASCII-only, deduped, 1024-char cap, mirroring the CLI), plus
|
||||
`VoiceStreamRelay`, which owns one upstream socket: keepalive timer, audio passthrough,
|
||||
transcript translation, finalize, and the caps below.
|
||||
|
||||
- **`src/web/routes/voice-routes.ts`**
|
||||
- `GET /api/voice/status` → `{ available, reason, subscriptionType?, expiresAt? }`. Never
|
||||
the token. `available:false` with a machine-readable `reason` (`disabled`, `no-credentials`,
|
||||
`expired`) is what the settings row and the provider resolver read.
|
||||
- `GET /ws/voice/stream?language=&keyterms=` → the relay. Same upgrade guard as
|
||||
`/ws/sessions/:id/terminal`: allowed Host, same-site Origin, and the global auth hook has
|
||||
already run on the handshake.
|
||||
|
||||
Caps, because an open mic is an open pipe: one stream per connection, `MAX_VOICE_STREAMS`
|
||||
concurrent server-wide, a hard `MAX_STREAM_MS` per stream, and a per-frame size cap. A tab
|
||||
left recording cannot bill an unbounded amount of upstream audio.
|
||||
|
||||
### Frontend pieces
|
||||
|
||||
- **`voice-pcm-worklet.js`** — an `AudioWorkletProcessor` converting Float32 blocks to PCM16
|
||||
and posting ~256 ms frames back. `MediaRecorder` cannot produce raw PCM, which is why the
|
||||
existing Deepgram path (container audio, auto-detected) cannot be reused as-is. Falls back
|
||||
to `ScriptProcessorNode` where AudioWorklet is unavailable.
|
||||
- **`ClaudeVoiceProvider`** in `voice-input.js` — mirrors `DeepgramProvider`'s shape
|
||||
(`start({language, keyterms, onStream, onResult, onError, onEnd})`) so `VoiceInput` treats
|
||||
the three providers uniformly.
|
||||
- **Provider resolution** — new `voiceSettings.provider`: `auto` (default) | `claude` |
|
||||
`deepgram` | `webspeech`. `auto` picks Claude when `/api/voice/status` reports it
|
||||
available, else Deepgram when a key is set, else Web Speech. Pinning a provider always
|
||||
wins, so an existing Deepgram user can keep exactly what they have.
|
||||
|
||||
### Settings
|
||||
|
||||
- `claudeVoiceEnabled` — synced, **default OFF**, gating the whole server side. Off is the
|
||||
honest default: turning it on means this machine's Claude subscription starts paying for
|
||||
transcription for whoever can reach the UI, and the audio goes to Anthropic rather than to
|
||||
wherever it went before. One switch in Settings → Voice, and the mic works with no key.
|
||||
- `voiceSettings.provider` — per the resolution table above; joins the existing synced
|
||||
`voiceSettings` object.
|
||||
|
||||
## Things worth knowing
|
||||
|
||||
- **This uses an undocumented endpoint with subscription credentials.** It is the user's own
|
||||
token, on the user's own machine, driving the user's own Claude Code install, but it is not
|
||||
a published API and Anthropic can change or restrict it. Default-OFF is deliberate; the
|
||||
Deepgram and Web Speech paths stay untouched as the supported fallbacks.
|
||||
- **Multi-user mode**: every user's dictation would run on the server owner's Claude
|
||||
credentials, exactly as every user's *sessions* already run on them. Consistent, but worth
|
||||
stating out loud in the settings copy.
|
||||
- **Token lifetime** is about 8 hours, refreshed by Claude Code itself whenever it runs. The
|
||||
relay re-reads the file on every connect rather than caching, so a refresh is picked up on
|
||||
the next press of the mic.
|
||||
- **HTTPS or localhost**: `getUserMedia` needs a secure context. Prod is HTTPS behind
|
||||
`tailscale serve`, so this is already satisfied; the existing error copy covers the rest.
|
||||
@@ -44,9 +44,9 @@ records), kept distinct from the existing `ScheduledRun`.
|
||||
|
||||
## 2. Where agent/session types are defined
|
||||
|
||||
- `type SessionMode = 'claude' | 'shell' | 'opencode' | 'codex' | 'gemini' | 'antigravity'`
|
||||
- `type SessionMode = 'claude' | 'shell' | 'opencode' | 'codex' | 'gemini' | 'antigravity' | 'pi'`
|
||||
(`src/types/session.ts:43-44`). `shell` covers the brief's "Terminal/custom".
|
||||
- CLI availability resolvers in `src/utils/{claude,codex,gemini,antigravity,opencode}-cli-resolver.ts`.
|
||||
- CLI availability resolvers in `src/utils/{claude,codex,gemini,antigravity,opencode,pi}-cli-resolver.ts`.
|
||||
- **Integration point:** the job's `agentType` reuses `SessionMode` verbatim.
|
||||
|
||||
## 3. Where input is sent into a session
|
||||
|
||||
+2
-2
@@ -1,7 +1,7 @@
|
||||
# Cron Jobs — User & Operator Guide
|
||||
|
||||
Codeman's **Cron** feature lets you save named, recurring jobs that automatically
|
||||
spin up a Claude (or shell / OpenCode / Codex / Antigravity / Gemini) session on a schedule and
|
||||
spin up a Claude (or shell / OpenCode / Codex / Antigravity / Gemini / Pi) session on a schedule and
|
||||
feed it a prompt. Think "cron for agent sessions": _"every weekday at 3am, open a
|
||||
Claude session in `~/proj` and tell it to update dependencies and open a PR."_
|
||||
|
||||
@@ -91,7 +91,7 @@ These map 1:1 to `CronJobSchema` (`src/web/schemas.ts`) and the `CronJob` type
|
||||
| Field | Required | Values / limits | Notes |
|
||||
| -------------------------- | ----------- | -------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
||||
| `name` | ✅ | 1–200 chars | Display name; also used as the created session's name. |
|
||||
| `agentType` | ✅ | `claude` \| `shell` \| `opencode` \| `codex` \| `gemini` \| `antigravity` | Reuses Codeman's `SessionMode`. `shell` = a plain terminal. |
|
||||
| `agentType` | ✅ | `claude` \| `shell` \| `opencode` \| `codex` \| `gemini` \| `antigravity` \| `pi` | Reuses Codeman's `SessionMode`. `shell` = a plain terminal. ⚠️ A `pi` job's readiness poll looks for `❯`/a token count, neither of which pi prints, so it burns the poll budget and then sends the prompt anyway (slower start, still works). |
|
||||
| `workingDir` | ✅ | valid path (allowlist-validated) | Validated at **create/update** (must exist, be a directory, and not resolve into a blocked tree — `/etc`, `/root`, `/proc`, `/sys`, `/dev`, or `/` itself) and again **at fire time**. |
|
||||
| `launchCommand` | — | ≤ 2000 chars, single line | `shell` mode only: sent as the **first input line** once the shell is up, before the prompt. Ignored for other agent types. |
|
||||
| `promptMode` | ✅ | `inline_text` \| `prompt_file_path` | See §5. |
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
Run a case inside an **isolated Docker container** instead of directly on the host. Any number of Codeman sessions can share one container (it is scoped to the case, not the session), so a whole project lives in a sandbox with its own network, resource caps, and filesystem, and you can **export the container to move it to another machine**.
|
||||
|
||||
Docker mode is a **location overlay on cases**, the direct analog of [remote SSH cases](./remote-hosts.md): where a remote case runs a local tmux pane doing `ssh host` into a durable remote tmux server, a docker case runs a local tmux pane doing `docker exec -it` into a durable **in-container** tmux server. It is not a separate `SessionMode`, so `claude` / `shell` / `opencode` / `codex` / `gemini` / `antigravity` all work inside the container.
|
||||
Docker mode is a **location overlay on cases**, the direct analog of [remote SSH cases](./remote-hosts.md): where a remote case runs a local tmux pane doing `ssh host` into a durable remote tmux server, a docker case runs a local tmux pane doing `docker exec -it` into a durable **in-container** tmux server. It is not a separate `SessionMode`, so `claude` / `shell` / `opencode` / `codex` / `gemini` / `antigravity` / `pi` all work inside the container.
|
||||
|
||||
## One-time setup: build the base image
|
||||
|
||||
@@ -25,10 +25,12 @@ A zero exit code only proves the layers ran, not that the toolchain works. Verif
|
||||
|
||||
```bash
|
||||
docker run --rm codeman/agent:base bash -lc \
|
||||
'for c in claude codex gemini opencode agy; do printf "%-9s " $c; $c --version 2>&1 | head -1; done'
|
||||
'for c in claude codex gemini opencode agy pi; do printf "%-9s " $c; $c --version 2>&1 | head -1; done'
|
||||
```
|
||||
|
||||
Antigravity (`agy`) is the one CLI not installed from npm (Google ships a standalone binary), so it has its own Dockerfile step and adds roughly 190MB; a full image lands near 1.6GB.
|
||||
Antigravity (`agy`) is the one CLI not installed from npm (Google ships a standalone binary), so it has its own Dockerfile step and adds roughly 190MB; a full image lands near 1.6GB. Pi also gets its own step, because upstream documents installing it with `--ignore-scripts` and that flag must not silently change how the other four npm CLIs install.
|
||||
|
||||
Pi's credentials are seeded per-FILE rather than as a whole directory (`auth.json`, `settings.json`, `trust.json`, `models.json`, `models-store.json` out of `~/.pi/agent`), because that directory also holds `sessions/`, `extensions/`, `skills/` and the installed package trees — gigabytes on an active host. Consequence: in-container pi sessions are invisible host-side, so `pi -c` inside a Docker case only sees that container's own history. See [`pi-integration.md`](./pi-integration.md).
|
||||
|
||||
## Quickest path: one-click "Run in Docker"
|
||||
|
||||
|
||||
+169
-7
@@ -44,6 +44,10 @@ is in [`api-reference.md`](api-reference.md).
|
||||
the payload at the top level rather than under `data`. Read defensively with
|
||||
`body.data ?? body`.
|
||||
|
||||
⚠️ A `401` is not an envelope at all: auth is rejected in a request hook that
|
||||
replies with the bare string `Unauthorized`, so parsing it as JSON throws. Branch on
|
||||
the status code before you parse, or a missing password looks like a broken endpoint.
|
||||
|
||||
**Already driving Codeman from an agent?** The README's
|
||||
[Programmatic Guide](../README.md#driving-codeman-from-an-agent--programmatic-guide)
|
||||
covers the in-session case: the `CODEMAN_MUX`, `CODEMAN_API_URL`,
|
||||
@@ -152,10 +156,17 @@ for (;;) {
|
||||
|
||||
## Seam 3: HTTP API and CLI
|
||||
|
||||
Around 199 handlers across 21 route files cover sessions, cases, files, cron,
|
||||
Around 200 handlers across 21 route files cover sessions, cases, files, cron,
|
||||
respawn, Ralph, the orchestrator, search, and admin. Each route module carries an
|
||||
`@fileoverview` describing its endpoints.
|
||||
|
||||
If the caller is an agent running _inside_ a Codeman session, install the packaged
|
||||
agent skill instead of teaching it these calls by hand: `skills/codeman` in the repo
|
||||
(`npx skills add Ark0N/Codeman --skill codeman -g`, or `codeman skill install
|
||||
[--case <name>]`, or the synced `agentSkillEnabled` App Setting for automatic
|
||||
per-case injection on Claude session create). The skill carries the guard, the
|
||||
safety rules, and verified wait/orchestration recipes.
|
||||
|
||||
The common ones:
|
||||
|
||||
```bash
|
||||
@@ -167,10 +178,12 @@ curl -u admin:$PASS -X POST http://127.0.0.1:3000/api/v1/sessions \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"workingDir":"/home/me/project","mode":"claude"}'
|
||||
|
||||
# Send a prompt (single-line only)
|
||||
# Send a prompt (single-line only, and it must end with \r: Enter is sent only
|
||||
# when the input contains a carriage return; without it the text sits on the
|
||||
# session's prompt unsubmitted)
|
||||
curl -u admin:$PASS -X POST http://127.0.0.1:3000/api/v1/sessions/$ID/input \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"run the tests","useMux":true}'
|
||||
-d '{"input":"run the tests\r","useMux":true}'
|
||||
```
|
||||
|
||||
`POST .../input` also accepts `clientId` (stable per client, max 128 chars) and
|
||||
@@ -178,6 +191,124 @@ curl -u admin:$PASS -X POST http://127.0.0.1:3000/api/v1/sessions/$ID/input \
|
||||
at-most-once, so retrying after a dropped connection cannot type the prompt
|
||||
twice. Omit them entirely rather than sending `null`.
|
||||
|
||||
It also accepts `wait` and `waitTimeout`, which hold the response open until the
|
||||
session finishes the turn you just started. `wait` is `true` (the default signal
|
||||
set) or a comma list of `idle,working,stop,blocked,exit`; the result comes back
|
||||
under `data.wait`. Sending them changes nothing for callers that do not: without
|
||||
`wait` the response is still `{"success": true, "data": {}}` and the write is still
|
||||
fire-and-forget. The two interact with `clientId` / `seq` in one way worth knowing:
|
||||
a **tagged duplicate** (a pair the server already applied) skips the write but still
|
||||
waits, answering from the session's current state rather than blocking for a
|
||||
transition that already happened. It reports `"delivered": false, "duplicate": true`.
|
||||
|
||||
### Waiting instead of polling
|
||||
|
||||
Three calls block until something happens: `GET /api/v1/sessions/:id/wait` (a
|
||||
lifecycle signal), `GET /api/v1/sessions/:id/wait-output` (a literal string in the
|
||||
output), and the `wait` field above. Full parameter and response tables are in
|
||||
[`api-reference.md`](api-reference.md#long-polling-agent-wait). Four things decide
|
||||
whether your integration works, and the last one is what actually bites:
|
||||
|
||||
- **A timeout is a `200` with `wait.timedOut: true`**, not an error. Loop over short
|
||||
waits rather than issuing one long one, because `tailscale serve` and cloudflared
|
||||
both cut idle connections and a single 10-minute call is the pattern most likely
|
||||
to die in the field.
|
||||
- **`wait.timeoutMs`** is the timeout after server-side clamping (600 s ceiling by
|
||||
default). Read it rather than assuming you got what you asked for.
|
||||
- **`stop` and `blocked` only exist for `claude` sessions**, and on a `shell` session
|
||||
even `idle` fires only once at startup, so send-and-wait there can only time out.
|
||||
See the Gotchas below.
|
||||
|
||||
⚠️ **There is no readiness signal, and skipping readiness is the failure that looks
|
||||
like success.** A session reports `idle` before its CLI has spawned, and a `claude`
|
||||
worker in a brand-new case comes up on the CLI's **trust dialog**, which has a ❯
|
||||
prompt of its own. Prompt it at that moment and the text lands in the dialog, the
|
||||
`\r` does not get past it, and the session's startup `idle` lands inside the wait
|
||||
window: the wait resolves on `idle` in a couple of seconds with `timedOut: false`,
|
||||
indistinguishable from a finished turn. Wait for the pid, then wait for the
|
||||
composer, answering the dialog only as the bounded fallback.
|
||||
|
||||
A worked orchestration: start a worker, get it ready, prompt it, wait, clean up.
|
||||
|
||||
```bash
|
||||
API="${CODEMAN_API_URL:-http://127.0.0.1:3000}" # auto-set in-session, correct scheme included
|
||||
AUTH=(-u "admin:$CODEMAN_PASSWORD") # omit entirely if no password is set
|
||||
CURL=(curl -sk "${AUTH[@]}") # -k: harmless on http, required on --https installs (self-signed cert)
|
||||
|
||||
# 1. Start a worker session (creates the case if it does not exist yet).
|
||||
# The guard matters: a TLS or auth failure otherwise leaves SID empty and every
|
||||
# later step "succeeds" against nothing.
|
||||
SID=$("${CURL[@]}" -X POST "$API/api/v1/quick-start" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"caseName":"worker-1","mode":"claude"}' | jq -r '.data.sessionId')
|
||||
[ -n "$SID" ] && [ "$SID" != null ] || { echo "quick-start failed"; exit 1; }
|
||||
|
||||
# 2. READINESS: composer marker first, trust dialog only as the bounded fallback.
|
||||
# Skip this and step 3 reports a turn that never ran. Do NOT probe trust first
|
||||
# and Enter blindly: the dialog text stays in the buffer for the life of the
|
||||
# session, so on every later run that probe matches stale text and the Enter
|
||||
# lands in a ready composer. Match single tokens only: TUI text can arrive
|
||||
# without its spaces. Stage 1 is short on purpose (an already-trusted case
|
||||
# matches in <1 s; a first-run case can never pass it and pays it in full).
|
||||
until [ "$("${CURL[@]}" "$API/api/v1/sessions/$SID" | jq '.data.pid')" != null ]
|
||||
do sleep 1; done
|
||||
R=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \
|
||||
--data-urlencode 'match=bypass' --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=5000') # composer's status bar = ready
|
||||
if ! jq -e '.data.wait.matched' <<<"$R" >/dev/null; then
|
||||
T=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \
|
||||
--data-urlencode 'match=trust' --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=2000')
|
||||
jq -e '.data.wait.matched' <<<"$T" >/dev/null && \
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' -d '{"input":"\r","useMux":true}' >/dev/null
|
||||
"${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \
|
||||
--data-urlencode 'match=bypass' --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=45000' >/dev/null
|
||||
fi
|
||||
|
||||
# 3. Send the prompt AND register the wait in one call, so the answer cannot be
|
||||
# the previous turn's idle state. Single line only, ending in \r (otherwise
|
||||
# Enter is never sent and this wait times out on a turn that never started).
|
||||
W=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"Run the test suite and summarize the failures\r","useMux":true,
|
||||
"clientId":"orchestrator","seq":1,"wait":"stop,exit","waitTimeout":60000}' \
|
||||
| jq -c '.data.wait')
|
||||
|
||||
# 4. That first wait probably timed out (60 s). Keep going in SHORT waits.
|
||||
for _ in $(seq 1 30); do
|
||||
[ "$(jq -r '.timedOut' <<<"$W")" = 'true' ] || break # signal fired, or wait ended
|
||||
W=$("${CURL[@]}" \
|
||||
"$API/api/v1/sessions/$SID/wait?until=stop,exit&timeout=60000" | jq -c '.data.wait')
|
||||
done
|
||||
jq -r 'if .ended or .aborted then "worker is not running"
|
||||
elif .timedOut then "still working after 30 waits"
|
||||
else "signal: \(.signal)" end' <<<"$W"
|
||||
|
||||
# 5. Read what it produced, then delete the session YOU created, by exact id.
|
||||
# ⚠️ NOT /output: its textOutput is empty for every tmux-backed session.
|
||||
# `tail` counts BYTES, and the payload is terminal data with ANSI in it.
|
||||
"${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer'
|
||||
"${CURL[@]}" -X DELETE "$API/api/v1/sessions/$SID"
|
||||
```
|
||||
|
||||
Waiting on a marker instead of a signal is the form that works in **every** mode,
|
||||
and the only one that works on a `shell` session:
|
||||
|
||||
```bash
|
||||
# ⚠️ Split the marker so the typed line never contains it: your own keystrokes echo
|
||||
# into the output stream, so an unsplit marker matches before the command has run.
|
||||
# `from=buffer` also catches a marker that printed before the wait registered.
|
||||
N=$RANDOM
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d "{\"input\":\"M=DONE; npm test; echo \${M}_$N rc=\$?\r\",\"useMux\":true}"
|
||||
"${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \
|
||||
--data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=60000' | jq '.data.wait'
|
||||
```
|
||||
|
||||
For shell scripting, the `codeman` CLI is the same surface without the HTTP
|
||||
plumbing:
|
||||
|
||||
@@ -221,10 +352,41 @@ Every one of these has cost somebody real time.
|
||||
shipped bugs more than once.
|
||||
- **`text/plain` bodies stay raw.** Auto-parsing them as JSON enabled
|
||||
simple-request CSRF, so it is deliberate. Send `application/json`.
|
||||
- **Prompts are single-line.** With `useMux: true` the server delivers your text
|
||||
and then Enter as two separate writes, so you do not append `\r` yourself. A
|
||||
multi-line string breaks the agent's Ink-based input handling: send one line,
|
||||
or split it across calls.
|
||||
- **Prompts are single-line and must end with `\r`.** The server splits your text
|
||||
and Enter into two separate tmux writes (Ink needs them apart), but it sends the
|
||||
Enter **only when the input contains a carriage return**. Without it your text
|
||||
sits on the prompt unsubmitted, which is the single most common "the wait
|
||||
endpoints don't work" report: the wait runs its full timeout on a turn that never
|
||||
started. Newlines inside the string are stripped rather than rejected, so
|
||||
`"echo A\necho B\r"` runs the single joined command `echo Aecho B`: send one line
|
||||
per call.
|
||||
- **`wait-output`'s `from=now` is not "printed after you asked".** tmux repaints
|
||||
the visible screen on attach, on resize, and on any TUI redraw, and a repaint
|
||||
arrives as ordinary output, so text already on screen can satisfy a fresh wait.
|
||||
Observed live: a marker echoed a minute earlier matched instantly. Use a marker
|
||||
unique to each call, and build it so the typed line never contains it (your own
|
||||
keystrokes echo into the stream). Matching is a literal substring, so `regex=` is
|
||||
rejected with a `400` rather than ignored.
|
||||
- **`wait-output` matches the normalized PTY stream, not the screen.** ANSI escape
|
||||
sequences are stripped (the `ESC ( B` charset escape a bash prompt emits on every
|
||||
line included), a partial escape at a chunk boundary is held back until its tail
|
||||
arrives, and a match may straddle PTY chunks, so text you printed yourself
|
||||
matches reliably (`printf STRAD; sleep 1; printf DLEQQ` is matchable as
|
||||
`STRADDLEQQ`). What can still fail is TUI output: a full-screen TUI positions
|
||||
words with cursor moves, so its text can reach the matcher **without spaces** and
|
||||
a multi-word match is unreliable there. Match one short space-free token, ideally
|
||||
one you printed yourself, and keep it out of the typed line (your own keystrokes
|
||||
echo into the stream).
|
||||
- **`stop` and `blocked` never fire for `shell`, `opencode`, `codex`, `gemini`,
|
||||
`antigravity` or `pi` sessions.** They come from Claude Code hooks, which no other mode
|
||||
installs, so only `idle`, `working` and `exit` exist there. Asking for them
|
||||
explicitly is a `400`; omitting `until` is safe, since the server drops them from
|
||||
the default set and echoes what it actually waited on as `wait.until`. Even in
|
||||
`claude` mode, a Docker case needs `CODEMAN_DOCKER_BRIDGE_HOOKS=1` for hooks to
|
||||
reach the server at all, a remote-SSH case's hooks may never arrive, and a case
|
||||
written by Codeman < 1.13.0 against an `--https` install carries hook curls
|
||||
without `-k` that TLS-fail silently — a 1.13.0+ server rewrites them the next
|
||||
time a session starts in that case.
|
||||
- **Unwrap the envelope** before reading fields. `data` is not the response body.
|
||||
|
||||
## Publishing your integration
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 34 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 207 KiB |
@@ -0,0 +1,681 @@
|
||||
# Pi (pi.dev) Run Mode: Implementation Plan
|
||||
|
||||
Tracking issue: [#206 "Plans to support pi.dev?"](https://github.com/Ark0N/Codeman/issues/206)
|
||||
|
||||
Status: **IMPLEMENTED 2026-08-13** (see `docs/pi-integration.md` for the user-facing
|
||||
guide). Everything below is the design record; the open questions were resolved
|
||||
empirically against pi 0.84.1 and the answers are recorded inline as **RESULT**
|
||||
notes. Originally reworked 2026-08-06; **rechecked 2026-08-13 against master @
|
||||
`f39beb3` (v1.17.0)**, and every line anchor below was re-verified at that commit (the 1.11.2-era
|
||||
anchors drifted heavily: six releases landed in between, including the settings-surface overhaul and
|
||||
the codex predictive-echo work, both of which added new pi touchpoints, §2.10 and the Brain picker in
|
||||
Phase 3). Upstream facts verified against `@earendil-works/pi-coding-agent` **v0.84.1** (npm latest,
|
||||
published 2026-08-07) and the [`earendil-works/pi`](https://github.com/earendil-works/pi) repo (cite
|
||||
that name: upstream docs still contain stale `pi-mono` links from a repo rename). Line numbers are
|
||||
anchors for orientation, not contracts; they drift.
|
||||
|
||||
---
|
||||
|
||||
## 1. What Pi is
|
||||
|
||||
[Pi](https://pi.dev) (MIT) is a minimal, extensible coding-agent harness. Facts below are verified
|
||||
against the upstream docs in `packages/coding-agent/docs/`.
|
||||
|
||||
| Property | Value |
|
||||
| ---------------- | -------------------------------------------------------------------------------------------------- |
|
||||
| Binary | `pi` (`bin: { pi: 'dist/cli.js' }`) |
|
||||
| npm package | `@earendil-works/pi-coding-agent`, latest **0.84.1** (2026-08-07; 0.84.0 was 2026-08-06); `legacy-node20` dist-tag at 0.74.2 |
|
||||
| Install | `npm install -g --ignore-scripts @earendil-works/pi-coding-agent`, or `curl -fsSL https://pi.dev/install.sh \| sh` (the curl installer also goes through global npm, so both uninstall via npm) |
|
||||
| Config dir | `~/.pi/agent` (override: `PI_CODING_AGENT_DIR`). Holds `auth.json`, `trust.json`, `settings.json`, `models.json` (user-defined providers), `models-store.json` (cached catalogs), `keybindings.json`, `extensions/`, `skills/`, `prompts/`, `themes/`, `AGENTS.md`, `SYSTEM.md`, and the package trees `npm/` + `git/` |
|
||||
| Sessions | `~/.pi/agent/sessions/--<cwd with / replaced by ->--/<timestamp>_<uuid>.jsonl`, tree-structured (`id`/`parentId`), format v3. Overrides: `PI_CODING_AGENT_SESSION_DIR`, `--session-dir` |
|
||||
| Credentials | `~/.pi/agent/auth.json` (OAuth subscriptions + API keys, auto-refresh), plus ~34 provider env vars with **no common prefix**. 0.84.1 adds `pi auth check` (auth preflight with optional credential output) |
|
||||
| TUI | Default: **main screen with terminal-owned scrollback**. Since **0.84.0** an experimental fullscreen mode exists, selectable via `--tui-mode fullscreen` **or at runtime through `/settings`**; the default remains the main-screen mode |
|
||||
| Providers | 15+ (Anthropic, OpenAI, Google, Azure, Bedrock, Mistral, Groq, xAI, OpenRouter, Copilot, Baseten since 0.84.0, ...). OAuth subscription login via `/login` for six: ChatGPT Plus/Pro, Claude Pro/Max, GitHub Copilot, xAI, OpenRouter, Radius |
|
||||
| Permission model | **No permission prompts at all.** No built-in sandbox, no MCP (none planned), no sub-agents, no plan mode, no to-dos, no background bash. Tools run with the user's own permissions |
|
||||
| Trust model | "Project trust" gates **loading** of project-local `.pi/` config/extensions/skills and **installing missing project packages**, not tool execution. Triggered only when the cwd (or an ancestor) contains `.pi/settings.json`, `.pi/extensions\|skills\|prompts\|themes`, `.pi/SYSTEM.md`/`.pi/APPEND_SYSTEM.md`, or `.agents/skills`; a bare `.pi/` directory does NOT prompt. Global `defaultProjectTrust`: `ask` (default) / `always` / `never` |
|
||||
|
||||
Three consequences shape the whole integration:
|
||||
|
||||
1. **There is no `--dangerously-skip-permissions` analog and none is needed.** Pi never prompts for
|
||||
tool approval. The Claude/Codex/Gemini/Antigravity pattern of "send the bypass flag so the session
|
||||
is not stuck on a modal" does not apply. Codeman must not invent a flag here.
|
||||
2. **The one privileged knob is `--approve` / `-a`** (trust project-local files for this run), which
|
||||
makes pi load and execute project `.pi/extensions` TypeScript **and run an npm install of missing
|
||||
project packages**. That is the field the multi-user clamp has to cover. Its explicit inverse
|
||||
`-na` / `--no-approve` exists, which lets the clamp force-deny rather than merely omit (§3, §5.2).
|
||||
3. **Provider keys cannot ride the env allowlist.** Pi's provider key vars (`ANTHROPIC_API_KEY`,
|
||||
`OPENAI_API_KEY`, `DEEPSEEK_API_KEY`, `HF_TOKEN`, `BASETEN_API_KEY`, ...) share no prefix, so
|
||||
there is no way to admit them through `ALLOWED_ENV_PREFIXES` without widening the list for every
|
||||
mode (§2.4).
|
||||
|
||||
---
|
||||
|
||||
## 2. Design decisions
|
||||
|
||||
### 2.1 Mode identity
|
||||
|
||||
`SessionMode` gains `'pi'`. Not a location overlay (unlike Docker/remote-SSH cases), not a web tab:
|
||||
a real sixth CLI backend with its own PTY, tmux session and respawn behaviour, exactly like
|
||||
`antigravity`. Append `pi` after `antigravity` in every enum/list to keep ordering consistent.
|
||||
|
||||
| Surface | Value |
|
||||
| ---------------- | --------------------------------------------------------------------- |
|
||||
| `SessionMode` | `'pi'` |
|
||||
| Display label | `Pi` |
|
||||
| Tab badge | `pi` (two-letter lowercase, like `sh`/`oc`/`cx`/`gm`/`ag`) |
|
||||
| Run button label | `Run PI` (short-label ternary in `_applyRunMode`, pattern `Run AG`) |
|
||||
| Kill-menu label | `Kill Tmux & Pi` |
|
||||
| Identity color | **`#f472b6` (rose-400)**. Verified free: live computed values on the default skin are claude `#38b6f0`, opencode `#44b993`, codex `#2b8fd9`, gemini `#8ab4f8`, antigravity `#22d3ee`, shell `#98a2b1`, web `#38bdf8`; purple is codex's base hex and amber reads as the shell tab badge, so pink/rose (or orange `#fb923c`) are the only genuinely free hues. No `pi` CSS identifier collides anywhere (`mode-pi`, `.tab-mode.pi`, `.run-mode-dot.pi` all grep clean, re-checked at f39beb3) |
|
||||
| Env prefix | `PI_` |
|
||||
| Dependency id | `pi` |
|
||||
| Status endpoint | `GET /api/pi/status` |
|
||||
|
||||
### 2.2 `isExternalCliMode()` yes, `isAltScreenStripMode()` no
|
||||
|
||||
Pi joins `isExternalCliMode()` (`session.ts:164-167`): its own TUI, its own output format, so the
|
||||
Ralph tracker, `BashToolParser`, token/CLI-info scraping and the `❯` readiness probe all stay off
|
||||
(gates at `session.ts:1100`, `:1701`, `:2000`, `:2103`), and readiness falls back to the output
|
||||
stabilization used by the other external CLIs.
|
||||
|
||||
Pi stays **out** of `isAltScreenStripMode()` (`session.ts:197-199`, currently codex/claude/gemini;
|
||||
antigravity and opencode are deliberately excluded). Pi's default TUI renders into the main screen
|
||||
with terminal-owned scrollback, so there is nothing to strip. The fullscreen mode **shipped in
|
||||
0.84.0 and is runtime-switchable via `/settings`**, so Codeman cannot assume a pi session stays
|
||||
main-screen for its lifetime; staying out of the strip list is exactly what makes that safe (the alt
|
||||
screen is load-bearing when the user flips to fullscreen, as it is for `opencode`). Putting pi IN
|
||||
the strip list would corrupt fullscreen sessions. Three mirrors must stay consistent (all unchanged
|
||||
for pi, i.e. pi appears in none of them): the replay-side strip in `session-routes.ts:2275`, the
|
||||
live-stream twin in `session.ts`, and the frontend `_sessionUsesServerMouseStrip()` in
|
||||
`terminal-ui.js` (usages `:3432`, `:3697`).
|
||||
|
||||
### 2.3 tmux required, no direct-PTY fallback, no per-mode configurator
|
||||
|
||||
Same rule as the other external CLIs: `pi` mode throws if tmux is unavailable. Add a fourth block to
|
||||
the guard chain at `session.ts:1751-1768` (antigravity's is `:1765-1768`).
|
||||
|
||||
**No `_configurePi()` is needed.** Opencode/codex/gemini each have a tmux-`setenv` configurator
|
||||
(`tmux-manager.ts:1709-1727`), but antigravity has none: it relies entirely on the generic
|
||||
`applyEnvOverrides()` (`tmux-manager.ts:1643`, `VALID_KEY = /^[A-Z_][A-Z0-9_]*$/`), which runs for
|
||||
every mode in both create (`:1880`) and respawn (`:2107`) and injects via socket-scoped
|
||||
`tmux setenv`, never the spawn command line. Pi follows the antigravity precedent: `PI_*` overrides
|
||||
flow through `applyEnvOverrides()` and nothing else.
|
||||
|
||||
Pi joins the truecolor branches: `buildEnvExports()` (`tmux-manager.ts:1604-1609`,
|
||||
`export COLORTERM=truecolor` + `unset NO_COLOR` for codex/gemini/antigravity) and the attach-env
|
||||
condition at `session.ts:1400-1402` (`buildMuxAttachEnv(...)`, whose comment says it must mirror
|
||||
`buildEnvExports`). Add `|| mode === 'pi'` to both, or the tmux session and the attach client
|
||||
disagree about color depth.
|
||||
|
||||
### 2.4 Env prefix: `PI_` only
|
||||
|
||||
Add `'PI_'` to `ALLOWED_ENV_PREFIXES` (`schemas.ts:125`) and to the prose error message at `:163`
|
||||
(two edits: the message hardcodes the list, and since 1.12+ it also names the exact-key allowlist,
|
||||
currently `...ANTIGRAVITY_* keys and CLAUDE_CONFIG_DIR are allowed.`; there is now a separate
|
||||
`ALLOWED_ENV_KEYS` exact-key set alongside the prefix list, which pi does not need to touch). That
|
||||
covers every documented variable pi reads: `PI_CODING_AGENT_DIR`, `PI_CODING_AGENT_SESSION_DIR`,
|
||||
`PI_PACKAGE_DIR`, `PI_OFFLINE`, `PI_SKIP_VERSION_CHECK`, `PI_TELEMETRY`, `PI_CACHE_RETENTION`,
|
||||
`PI_SHARE_VIEWER_URL`, `PI_HARDWARE_CURSOR`, `PI_EXPERIMENTAL` (whose meaning 0.84.0 extended to
|
||||
strict JSON-schema tool sampling). (Pi also *sets* `PI_CODING_AGENT=true` and `AI_AGENT=pi` in child
|
||||
processes; those are output markers, not inputs, and need nothing from us.)
|
||||
|
||||
**Deliberately not added:** `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `GEMINI_API_KEY`, `XAI_API_KEY`,
|
||||
`GROQ_API_KEY`, `MISTRAL_API_KEY` and the other ~28 provider keys. `ALLOWED_ENV_PREFIXES` is a
|
||||
single global list applied by one Zod refine with no mode context (`safeEnvOverridesSchema`,
|
||||
`schemas.ts:153-165`), so allowlisting bare provider keys for pi would widen the allowlist for
|
||||
**every** mode at once, violating the multi-CLI prefix discipline in CLAUDE.md. Users authenticate
|
||||
pi through `/login` (stored in `~/.pi/agent/auth.json`, auto-refreshed) or by exporting the key in
|
||||
the Codeman server process's own environment.
|
||||
|
||||
Making the allowlist mode-aware is the clean fix, listed as a follow-up in §9. Do not smuggle it
|
||||
into this change.
|
||||
|
||||
### 2.5 Docker credential policy: seed files, not the whole dir
|
||||
|
||||
`CRED_STORES` (`docker-hosts.ts:597-605`; file unchanged since the 2026-08-06 verification) gets a
|
||||
`.pi/agent` entry. Nested `rel` paths already work (`.config/gcloud` maps to seed name
|
||||
`.config-gcloud` via the `replace(/\//g, '-')` at `:620`). Unlike antigravity, which needed **no**
|
||||
entry (`agy` nests all state under `~/.gemini/antigravity-cli/`, already covered by the `.gemini`
|
||||
policy, per the comment at `:599-602`), pi has its own top-level dir and needs its own entry. Use
|
||||
`seedFiles`, **not** `seedWhole`:
|
||||
|
||||
```ts
|
||||
{ rel: '.pi/agent', seedFiles: ['auth.json', 'settings.json', 'trust.json', 'models.json', 'models-store.json'] },
|
||||
```
|
||||
|
||||
Rationale: `~/.pi/agent` also contains `sessions/`, `extensions/`, `skills/` and the installed
|
||||
package trees (`npm/`, `git/`), which on an active host is easily gigabytes; `seedWhole` would
|
||||
`cp -a` all of it into every container start. The five seeded files are what pi needs to
|
||||
authenticate and behave consistently: `models.json` is in the list because it holds user-defined
|
||||
custom providers, and omitting it would silently strip those inside containers. Seeding (RO mount
|
||||
then copy) also means the in-container pi never writes refreshed OAuth tokens back to the host,
|
||||
which is the whole point of the seeding policy, and bind mounts stay excluded from `docker commit`
|
||||
so exports remain secret-free.
|
||||
|
||||
Trade-off to accept and document: in-container pi sessions are not visible host-side, so `pi -c`
|
||||
inside a Docker case only sees that container's own history. Codex shares `sessions/` RW precisely
|
||||
because Codeman reads it host-side for the response viewer; there is no such reader for pi yet
|
||||
(the response-viewer follow-up in §9 would justify flipping this).
|
||||
|
||||
### 2.6 The `pi` binary name is generic
|
||||
|
||||
Unlike `agy`/`codex`/`gemini`, `pi` is a short, common name (Raspberry Pi tooling, personal scripts,
|
||||
`$PATH` accidents). The resolver must not blindly trust a hit. None of the existing external-CLI
|
||||
resolvers execute their binary (only `claude-cli-resolver.ts` does, via the cached
|
||||
`getClaudeCliVersion()`, skipped under vitest), so the sanity check is new ground: model it on
|
||||
`getClaudeCliVersion()`. Run `pi --version` once via `execFileSync`, cache the result module-level,
|
||||
skip under `VITEST`, and require output matching `/^\d+\.\d+\.\d+/`; on mismatch treat the binary as
|
||||
unavailable and log the rejected path. Surface `{ available, path, version }` from
|
||||
`GET /api/pi/status` so a misresolution is diagnosable from the UI (additive relative to the sibling
|
||||
endpoints' `{ available, path }`). The `dependency-registry` entry carries `versionArg: '--version'`
|
||||
for `codeman doctor`.
|
||||
|
||||
### 2.7 tmux extended keys (a real pi-specific footgun)
|
||||
|
||||
Pi documents (`docs/tmux.md`, verified verbatim) that without
|
||||
|
||||
```tmux
|
||||
set -g extended-keys on
|
||||
set -g extended-keys-format csi-u
|
||||
```
|
||||
|
||||
tmux collapses `Shift+Enter` and `Ctrl+Enter` into a plain `\r` (and `Alt+Enter` into `\x1b\r`), and
|
||||
pi's editor uses those for newline vs submit. `extended-keys-format` requires tmux 3.5+; tmux
|
||||
3.2-3.4 works with `extended-keys on` alone (pi then falls back to xterm `modifyOtherKeys`).
|
||||
Codeman's own browser input path sends `\r` for submit, so basic use works unconfigured, but
|
||||
newline-in-editor is degraded both for a user typing in an attached terminal (`sc`) and potentially
|
||||
for the browser Shift+Enter path.
|
||||
|
||||
Upstream recommends `~/.tmux.conf` and notes the setting may need a full `tmux kill-server` restart
|
||||
to take effect. **Codeman must NEVER run `kill-server` on its socket** (it would kill every live
|
||||
session, including `w1`/`w2`/`w3`). Action: attempt to set both options **server-scoped on
|
||||
Codeman's own socket only** (`tmux -L codeman set -s ...`, never `-g` on the user's default socket)
|
||||
at the point the tmux server is first started, verify with `tmux -L codeman show-options -s` and an
|
||||
empirical Shift+Enter test which scope actually takes for the installed tmux version, and fall back
|
||||
to a documented manual step in `docs/pi-integration.md` (a `~/.tmux.conf` snippet plus the
|
||||
kill-server caveat) if it cannot be applied safely to an already-running server. Upstream does not
|
||||
discuss socket- or server-scoped configuration at all, so this verification is original work, not a
|
||||
doc lookup.
|
||||
|
||||
**RESULT (measured, tmux 3.4 + pi 0.84.1):** `tmux -L <socket> set -s extended-keys on` takes effect
|
||||
on an **already-running** server with **no `kill-server`** — pi's own startup warning
|
||||
(`Warning: tmux extended-keys is off…`, a convenient in-band probe) disappears for the next session
|
||||
started afterwards. `extended-keys-format` does **not exist on tmux 3.4** and errors with
|
||||
`invalid option: extended-keys-format`, so the two options must be issued independently rather than
|
||||
chained. Decision: Codeman does **not** set this itself — it is a server-wide tmux option affecting
|
||||
every session of every backend, so silently changing key encoding is not Codeman's call. It is
|
||||
documented as a user step in `docs/pi-integration.md` instead, carrying the measured facts.
|
||||
|
||||
### 2.8 The completeness trap: which mode tables fail loud vs silent
|
||||
|
||||
Adding `'pi'` to the `SessionMode` union makes some omissions compile errors and leaves others
|
||||
silent. The plan calls this out so review can focus on the silent ones.
|
||||
|
||||
**Loud (typecheck fails until edited):** `getModeLabel()` (`session.ts:168-183`, exhaustive switch
|
||||
with no default), `defaultDockerCommandForMode` and `defaultRemoteCommandForMode` (both typed
|
||||
`Record<...CommandMode, string>`), **but only after** `RemoteCommandMode` (`types/session.ts:48-51`)
|
||||
and `DockerCommandMode` (`:157-161`) are widened: both are `Extract<SessionMode, '...'>` with every
|
||||
member spelled out, so forgetting the `Extract` lists keeps `tsc` green while docker/remote pi cases
|
||||
silently fall back to `exec bash -l` via the `|| commands.shell` on the lookup. Edit union + both
|
||||
`Extract` lists + both `Record` literals together.
|
||||
|
||||
**Silent (compiles clean, mode just doesn't work):**
|
||||
|
||||
- `appendResumeFlag()` (`tmux-manager.ts:1030-1042`) has a `default:` arm; a missing `case 'pi'`
|
||||
silently drops docker resume.
|
||||
- `buildSpawnCommand()` (`:770-825`) and `buildPathExport()` (`:1680-1707`) are if-chains with
|
||||
fallthrough returns; a missing branch spawns pi as a login shell / with no PATH augmentation.
|
||||
- `isExternalCliMode()` / `isAltScreenStripMode()` are boolean chains.
|
||||
- The `runMode` accessor's **setter whitelist** (`session-ui.js:2949-2960`) coerces any unknown mode
|
||||
to `'claude'`. Omitting `pi` there makes the mode **unselectable while every other edit appears to
|
||||
work**: this is the single most deceptive omission in the frontend.
|
||||
- `window.__codemanCliAvailable` (injected by `renderIndexHtml`, `server.ts:1375-1407`): the client
|
||||
treats a **missing key as available** (`isCliAvailable` in settings-ui.js), so forgetting the
|
||||
injection un-gates pi on boxes without the CLI instead of hiding it.
|
||||
|
||||
### 2.9 The Daylight skin cascade eats per-mode run-button colors
|
||||
|
||||
A finding that changes the CSS work (verified empirically with computed styles on the live
|
||||
instance, re-confirmed at f39beb3): `styles.css:13681` opens a nested skin block,
|
||||
`html:not([data-skin="og"]) { ... }`, and the **default skin is `daylight-blue`, not `og`**, so the
|
||||
block is live for every default-skin user. Inside it, `.btn-toolbar.btn-run` is re-declared
|
||||
generically and per-mode only for claude/opencode/codex (codex at `:13787`). CSS nesting adds the
|
||||
wrapper's specificity (the nested rules resolve to (0,3,1) vs (0,3,0) for
|
||||
`.btn-toolbar.btn-run.mode-X`), so **gemini's and antigravity's toolbar gradients are dead on the
|
||||
default skin**: both render the generic claude gradient today, still unfixed as of f39beb3. The
|
||||
base-sheet rules (gemini/antigravity at `:4406`/`:4420`) only ever render on the `og` skin. Since
|
||||
1.12+ styles.css itself documents this trap in comments (`:9214`, `:11091`), which confirms the
|
||||
mechanism.
|
||||
|
||||
Consequences for pi:
|
||||
|
||||
- The toolbar gradient needs **two** rules: one in the base sheet (`:4420` area, for `og`), and one
|
||||
**inside** the `13681` block next to codex's (`:13787` area), using the block's own idiom
|
||||
(or the color is invisible to the average user).
|
||||
- `mobile.css` phone-toolbar colors need `!important` on `background`/`border-color`/`color`,
|
||||
exactly as the CLAUDE.md gotcha prescribes. Antigravity's phone block (`mobile.css:895-910`,
|
||||
inside the `@media (max-width: 430px)` opened at `:338`) has no `!important` and is dead on the
|
||||
default skin; do not copy that mistake.
|
||||
- Three surfaces work from base rules alone (verified): run-mode **dots** (list at `:4506-4516`;
|
||||
the skin block overrides only claude/opencode/codex/shell dots, so a base-sheet
|
||||
`.run-mode-dot.pi` renders as authored), **tab badges**, and the **welcome button** (the skin
|
||||
block overrides only claude/opencode/tunnel welcome buttons).
|
||||
- Optional, separate cleanup (not this change): gemini/antigravity could get the same in-block
|
||||
treatment to resurrect their colors.
|
||||
|
||||
### 2.10 Local-echo policy: pi lands on the buffer overlay by default
|
||||
|
||||
New since the first draft of this plan: the codex predictive-echo work (1.13+) introduced a
|
||||
per-session echo policy in `_updateLocalEchoState()` (terminal-ui.js, `_localEchoPolicy` set at
|
||||
`:2837`): `codex → 'predict'` (write-through predictive echo), `shell → 'off'`, **everything else
|
||||
→ 'buffer'** (the `LocalEchoOverlay` that buffers typed text until Enter). Pi therefore gets the
|
||||
buffer overlay on touch devices with zero edits, via the fallthrough.
|
||||
|
||||
That default is a real open question, not a freebie: the codex history (issues #218/#219/#220/#222)
|
||||
shows that a composer which re-renders per keystroke (live-filtering slash picker, server-side
|
||||
cursor movement, wrap-as-you-type) is starved by buffer-until-Enter, and pi's editor is exactly
|
||||
such a composer. Decision for v1: ship with the default `'buffer'` policy but make phone-profile
|
||||
typing an explicit E2E gate (§7 step 4); if pi's editor mis-renders under the overlay, the cheap
|
||||
fallback is forcing `'off'` for pi (one branch in `_updateLocalEchoState`), and teaching the
|
||||
predict path pi's composer row is a follow-up, not a v1 requirement.
|
||||
`test/local-echo-codex-gating.test.ts` pins the per-mode policy via
|
||||
`it.each(['claude', 'gemini', 'opencode'])` lists (`:193`, `:376`); add `'pi'` to those lists once
|
||||
the buffer decision is confirmed (or pin the `'off'` branch if that is the outcome).
|
||||
|
||||
**RESULT (measured, pi 0.84.1, iPhone 14 Pro profile + a PTY-level A/B):** the buffer policy
|
||||
**holds**; codex's failure mode does **not** reproduce. Pi's slash picker re-filters on the **whole
|
||||
composer content**, not on per-keystroke deltas: a one-shot literal write of `/set` (what the overlay
|
||||
flush does) filters the picker to `settings` **identically** to sending `/ s e t` as five separate
|
||||
keystrokes, and the delayed `\r` then selects it and opens the settings menu. Prose prompts buffer
|
||||
correctly (`pendingText` right, nothing on the PTY before Enter), flush on Enter, and are accepted as
|
||||
a single prompt. `'pi'` was added to both `it.each` lists. The `'off'` fallback stays documented but
|
||||
unused.
|
||||
|
||||
---
|
||||
|
||||
## 3. Config surface: `PiConfig` to CLI flags
|
||||
|
||||
```ts
|
||||
/** Pi CLI session configuration */
|
||||
export interface PiConfig {
|
||||
/** Model pattern or ID. Supports `provider/id` and a `:<thinking>` suffix (e.g. `sonnet:high`). Passed via --model. */
|
||||
model?: string;
|
||||
/** Provider name (anthropic, openai, google, ...). Passed via --provider. */
|
||||
provider?: string;
|
||||
/** Reasoning level. Passed via --thinking. */
|
||||
thinking?: 'off' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'max';
|
||||
/** Continue the most recent session (-c). Per-cwd scoping is strongly implied upstream but not documented; treat as probable. */
|
||||
continueSession?: boolean;
|
||||
/** Resume a specific session by ID or partial UUID (--session). Codeman deliberately accepts ids only, never paths. */
|
||||
resumeSessionId?: string;
|
||||
/**
|
||||
* Tri-state project trust (repo-local `.pi/` settings/extensions/skills, plus installing
|
||||
* missing project packages):
|
||||
* true -> --approve (trust for this run; loads and EXECUTES repository TypeScript)
|
||||
* false -> --no-approve (force-deny; the trust prompt never appears)
|
||||
* absent -> pi's own defaultProjectTrust (ask).
|
||||
* Multi-user: MATERIALIZED to false for non-granted owners (§5.2).
|
||||
*/
|
||||
approveProjectTrust?: boolean;
|
||||
}
|
||||
```
|
||||
|
||||
Flag mapping in `buildPiCommand()` (new, `tmux-manager.ts`, directly after `buildAntigravityCommand`
|
||||
at `:718-736`; every builder there regex-allowlists each user value and silently drops failures
|
||||
because the result lands in a `bash -c "..."` string):
|
||||
|
||||
| Field | Flag | Validation |
|
||||
| --------------------- | ------------------------------- | --------------------------------------------------------------------------------- |
|
||||
| `approveProjectTrust` | `--approve` / `--no-approve` / nothing | tri-state boolean, clamped (§5.2) |
|
||||
| `model` | `--model <v>` | `/^[a-zA-Z0-9._\-/:]+$/` (`:` for `sonnet:high`, `/` for `openai/gpt-4o`) |
|
||||
| `provider` | `--provider <v>` | `/^[a-z0-9-]+$/` |
|
||||
| `thinking` | `--thinking <v>` | runtime allowlist of the 7 enum values (defense in depth beyond Zod) |
|
||||
| `resumeSessionId` | `--session <v>` | `/^[a-zA-Z0-9._-]+$/` (same shape as `RESUME_ID_SAFE`, `:1021`; excludes paths on purpose) |
|
||||
| `continueSession` | `-c` | boolean; **skipped when a valid `resumeSessionId` is present** (the two conflict) |
|
||||
|
||||
**Not** wired in v1, with reasons:
|
||||
|
||||
- `--api-key <key>`: ⚠️ **never wire this.** It puts a provider secret on the spawn command line,
|
||||
which is exactly what the socket-scoped `tmux setenv` discipline exists to prevent (visible in
|
||||
`ps`, tmux server state, and logs). Listed here so nobody "helpfully" adds it later.
|
||||
- `--tui-mode` (released in 0.84.0): never passed by Codeman. The main-screen default is the
|
||||
friendly case for the browser terminal, and fullscreen remains the user's own runtime choice via
|
||||
`/settings` (§2.2 is designed for that). `--use-theme` (still unreleased) likewise.
|
||||
- `--name <name>` (`-n`): nice for `/resume` readability, but names contain spaces and would be the
|
||||
first user-controlled value needing real shell quoting in `buildSpawnCommand`. Defer.
|
||||
- `--no-session`: ephemeral mode fights respawn/resume. Defer.
|
||||
- `-p`/`--print`, `--mode json`, `--mode rpc`: non-interactive transports, a different product shape
|
||||
(§9). Note upstream already shipped a breaking change to JSON-mode `message_update` framing, so
|
||||
any future consumer must assemble deltas.
|
||||
- `--tools` / `--exclude-tools` / `--no-tools` / `--no-builtin-tools` (`-t`/`-xt`/`-nt`/`-nbt`): a
|
||||
genuinely useful "read-only session" affordance (0.84.0 also added a `defaultTools` setting), but
|
||||
it needs UI design. Follow-up.
|
||||
- `-r`/`--resume` (interactive picker), `--fork`, `-e`/`--extension`, `--skill`, `--system-prompt`,
|
||||
`--append-system-prompt`, `--export`, `--models`, `--list-models`: not session-manager concerns in
|
||||
v1. (`-e` matters later: §9's extension follow-up notes CLI extensions load before trust
|
||||
resolution.)
|
||||
|
||||
---
|
||||
|
||||
## 4. Implementation phases
|
||||
|
||||
### Phase 1: Backend core
|
||||
|
||||
| File | Change |
|
||||
| ----------------------------------- | ------------------------------------------------------------------------------------------------------------ |
|
||||
| `src/utils/pi-cli-resolver.ts` | **New**, mirror `antigravity-cli-resolver.ts` (65 lines: search-dir list, module-level cache with `''` negative sentinel, `which pi` first). Search dirs: `~/.local/bin`, `/usr/local/bin`, `~/.bun/bin`, `~/.npm-global/bin`, `~/bin`. Add the `pi --version` sanity probe from §2.6 (execFileSync, cached, vitest-skipped). Export `resolvePiDir()`, `isPiAvailable()`, `getPiCliVersion()` |
|
||||
| `src/utils/index.ts` | Re-export the three (resolver block `:30-36`) |
|
||||
| `src/types/session.ts` | `SessionMode` union `:46`; **both `Extract` lists**: `RemoteCommandMode` `:48-51`, `DockerCommandMode` `:157-161` (§2.8); new `PiConfig` after `AntigravityConfig` (`:325-333`); `SessionState.piConfig` after `:486`; `@fileoverview` mode list `:11` + config list `:17` |
|
||||
| `src/mux-interface.ts` | `piConfig?: PiConfig` on `CreateSessionOptions` (config block ends `:78`) and `RespawnPaneOptions` (ends `:109`) |
|
||||
| `src/session.ts` | `isExternalCliMode()` `:164-167` (+pi); `getModeLabel()` `:168-183` (+`'Pi'`); `_piConfig` field decl `:466-470`; ctor option `:556-563` + apply `:652-654`; `toState()` `:1227-1230`; `_buildRespawnPaneOptions()` `:1466-1469` (single source of truth shared by `startInteractive` and `reattachRemote`); `startInteractive()` createSessionOptions `:1680-1683`; COLORTERM attach-env condition `:1400-1402` (+pi); requires-tmux guard chain `:1751-1768` (new block: "Pi sessions require tmux for env override injection via setenv") |
|
||||
| `src/tmux-manager.ts` | `buildPiCommand()` after `:736` per §3; `buildSpawnCommand()` signature `:770-779` + dispatch branch after `:822-825`; `appendResumeFlag()` `:1030-1042` (`case 'pi': return \`${modeCommand} --session ${resumeId}\`;`); `buildEnvExports()` truecolor branches `:1604-1609` (+pi); `buildPathExport()` `:1680-1707` (+pi branch calling `resolvePiDir()`); missing-CLI error chain in `createSession` `:1788-1806` (+pi, install hint `npm install -g --ignore-scripts @earendil-works/pi-coding-agent`; note `respawnPane` deliberately has no such check); `piConfig` threading at the four sites `:1748`, `:1817`, `:2041`, `:2080`. **No `_configurePi`** (§2.3) |
|
||||
| `src/config/dependency-registry.ts` | New entry after antigravity's (`:101-108`; file unchanged since 2026-08-06): `{ id: 'pi', label: 'Pi CLI', category: 'core', required: false, usedBy: ['Pi sessions'], resolvers: [{ match: ALL, resolver: { kind: 'path', bins: ['pi'], versionArg: '--version' } }] }` |
|
||||
| `src/docker-hosts.ts` | `defaultDockerCommandForMode` `:138-149`: `pi: 'exec pi'`. `CRED_STORES` `:597-605`: the `.pi/agent` seedFiles entry per §2.5 (nested `rel` already handled at `:613-645`). File unchanged since 2026-08-06 |
|
||||
| `src/remote-hosts.ts` | `defaultRemoteCommandForMode` `:92-118`: `pi: remoteLoginShellCommand('pi')` (`remoteLoginShellCommand` at `:88-90`). Login-shell routing is mandatory (the #209/e803186 lesson: ssh remote-command exec sees only sshd's minimal PATH, and npm's global bin is usually only on PATH via rc files) |
|
||||
|
||||
### Phase 2: Web layer
|
||||
|
||||
| File | Change |
|
||||
| ---------------------------------- | ----------------------------------------------------------------------------------------------- |
|
||||
| `src/web/schemas.ts` | `'PI_'` in `ALLOWED_ENV_PREFIXES` `:125` **and** the prose error message `:163` (which now also names `CLAUDE_CONFIG_DIR`; the `ALLOWED_ENV_KEYS` exact-key set needs no change); new `PiConfigSchema` after `AntigravityConfigSchema` (`:256-271`), mirroring §3's regexes, `.optional()`, not `.strict()`; `piConfig` on `CreateSessionSchema` (`:299` area) and `QuickStartSchema` (`:712` area); `'pi'` in all three mode enums (`:285`, `:708`, cron `agentType` `:1214`; they are byte-identical and there is no fourth); `pi` key in `RemoteCommandOverridesSchema` `:426-436` (it is `.strict()`, so an unknown key is a hard error today; one edit covers both remote `:501` and docker `:577` reuse) |
|
||||
| `src/web/routes/session-routes.ts` | Thread `piConfig` through create (`POST /api/sessions`): disk-strip exclusion chain `:705-712`, availability gate `:782-790` (+`isPiAvailable` with install-hint error), model resolution `:825-838` (`mode === 'pi' ? body.piConfig?.model : ...`), clamp call `:845`, Session ctor `:860` (`piConfig: mode === 'pi' ? gatedPiConfig : undefined`). Quick-start (`POST /api/quick-start`, handler `:2559`): remote-case config rejection `:2614-2621` and docker-case `:2645-2652` (+`piConfig`: per-CLI config does not cross ssh or the bind mount), hooks-scaffold exclusions `:2801`/`:2809`, availability gate `:2744-2752` (local-case branch only), env-strip chains `:2833`/`:2863`, model resolution `:2885`, clamp `:2897`, ctor `:2913`. **Extend `clampExternalCliBypassForOwner()`** (`:305-336`, doc comment above): fifth param + return field; pi joins the **materialize** branch per §5.2. Alt-screen replay-strip at `:2275` unchanged (pi not in it, §2.2) |
|
||||
| `src/web/routes/system-routes.ts` | `GET /api/pi/status` after the antigravity handler (`:418-426`; file unchanged since 2026-08-06), same shape plus `version` (§2.6); update the "CLI Integrations" prose comment `:377` |
|
||||
| `src/web/server.ts` | Restore path: `piConfig: muxSession.mode === 'pi' ? savedState?.piConfig : undefined` after `:2636`. **`renderIndexHtml` CLI-availability injection `:1375-1407`**: add `isPiAvailable` to the dynamic-import tuple (`:1382`) and a `pi` key to the injected object (`:1399`). Per §2.8 a missing key reads as *available*, so this is a correctness edit, not polish |
|
||||
|
||||
### Phase 3: Frontend
|
||||
|
||||
The antigravity touchpoints are the template. Since the first draft, the settings-surface overhaul
|
||||
moved most anchors and added one **new touchpoint** (the clone-repo Brain picker below).
|
||||
`constants.js`, `api-client.js`, `ralph-wizard.js`, `cron-ui.js`, `webview-tabs.js` and `sw.js`
|
||||
still need **no** changes (re-verified zero mode coupling at f39beb3; cron-ui reads the `<select>`
|
||||
generically and special-cases only `shell`).
|
||||
|
||||
| File | Change |
|
||||
| ------------------- | ------------------------------------------------------------------------------------------------------ |
|
||||
| `index.html` | Welcome button `welcomePiBtn` after Gemini's (antigravity's is `:347`; there is deliberately no codex welcome button), `display:none` default, `onclick="app.setRunMode('pi'); app.runPi()"`, text `Run Pi`; run-mode-option row with `.run-mode-dot.pi` after antigravity's (`:526-528`), before the `.run-mode-sep` `:529`; cron `<option value="pi">Pi</option>` after `:803`; **NEW: the clone-repo "Brain" picker** (`cloneCaseBrain`, `:2476-2486`): add `<option value="pi" data-cli="pi">Pi</option>` after the antigravity option `:2483` (gating is automatic: session-ui.js `:2107-2115` hides options whose `data-cli` fails `isCliAvailable`, and `:2250` reads the value at clone time); docker image hint `:2624` (`claude/codex/gemini/opencode/agy` + pi). No per-CLI remote-command override field needed (only codex has one, `:2559`) |
|
||||
| `session-ui.js` | `@fileoverview` mode list `:2`; `run()` dispatch branch after `:400-402`; `_refreshRunModeAvailability` list `:468` (+`'pi'` as a quoted literal, the static test in §6 demands it); short-label ternary `:565` (+`'Run PI'`); **the `runMode` setter whitelist `:2949-2960`** (§2.8, the deceptive one); new `runPi()` modeled on `runAntigravity()` `:1170-1219`: same remote/docker skip, same `_beginSessionLaunchStatus` frame, probes `/api/pi/status` reading `(await res.json()).data.available` (envelope!), **sends no `piConfig` at all** (no bypass exists and trust defaults are pi's own; envOverrides still sent for local cases), install-hint error text matching Phase 1's; `isAltMode` `:1233` and `isExternalCli` `:1263` four-way comparisons (+pi) |
|
||||
| `settings-ui.js` | `applyWelcomeCliVisibility()` `:1176-1191`: add `['welcomePiBtn', 'pi']` |
|
||||
| `app.js` | Response-viewer agent label `:1998-2009` (+pi -> `'Pi'`); tab badge ternary `:3884` (`<span class="tab-mode pi" aria-hidden="true">pi</span>`; claude stays badge-less); kill-title ternary `:5046-5057` (`Kill Tmux & Pi`) |
|
||||
| `panels-ui.js` | Command-palette `labels` map `:430` (+`pi: 'Pi'`; the `\|\| mode` fallback means this is cosmetic, not load-bearing) |
|
||||
| `mobile-overview.js`| `MOBILE_OVERVIEW_RUN_MODES` `:55-62`: `{ mode: 'pi', label: 'Pi', short: 'Pi' }` after antigravity `:60`, before the shell entry. Nothing else: the Run-button badge (`:499`) and menu builder (`:554-556`) consume the list generically, and the buttons carry `btn-toolbar btn-run mode-pi`, which is exactly why they inherit the §2.9 cascade problem and its fix |
|
||||
| `terminal-ui.js` | Badge-row comment `:1750` only (the badge itself is a raw `s.mode` passthrough, no list to extend). `_sessionUsesServerMouseStrip` unchanged (§2.2). `_updateLocalEchoState` unchanged for v1 (§2.10: pi lands on `'buffer'` via the fallthrough; only touch it if E2E forces the `'off'` fallback) |
|
||||
| `i18n.js` | `'Run Pi': '运行 Pi'` in the zh-CN table (`:102-107`, matches the welcome-button text; short labels like `Run PI` are deliberately untranslated, as are the other modes') |
|
||||
| `styles.css` | Tab badge `.session-tab .tab-mode.pi` after `:2157` (`background: rgba(244,114,182,0.2); color: #f472b6;`); add `.session-tab .tab-mode.pi` to the light-skin ink list `:325-336` (gemini + antigravity are its precedent, `:332`); welcome `.welcome-btn-pi` + `:hover` after antigravity's `:3366` block, rose family (e.g. base `linear-gradient(135deg, #33121f 0%, #9d174d 55%, #be185d 100%)`, border `rgba(244,114,182,0.4)`, text `#fce7f3`); toolbar gradient pair `.btn-toolbar.btn-run.mode-pi, .btn-toolbar.btn-run-gear.mode-pi` + `:hover` after `:4420`'s antigravity block; `.run-mode-dot.pi { background: #f472b6; }` in the dot list `:4506-4516`; **and the §2.9 rule inside the Daylight block** next to codex's `:13787` (e.g. `background: linear-gradient(135deg, #be185d, #f472b6); border-color: #be185d; color: #fff1f7;`). The dot needs no skin-block entry (the block overrides only claude/opencode/codex/shell dots; gemini/antigravity dots already fall through correctly) |
|
||||
| `mobile.css` | Phone toolbar block after `:910` inside the `@media (max-width: 430px)` opened at `:338`: `mode-pi` base + `:active`, **with `!important` on background/border-color/color** (§2.9; antigravity's block `:895-910` omits it and is dead); light-skin override entry after `:2985` with the same four-skin `html:is(...)` prefix as its siblings |
|
||||
|
||||
### Phase 4: Docker image and installer
|
||||
|
||||
Both files are unchanged since the 2026-08-06 verification; all anchors stand.
|
||||
|
||||
- `docker/agent.Dockerfile`: a **separate** `RUN` step after the antigravity block (`:38-45`), not a
|
||||
fifth line in the shared npm block (`:31-36`), because pi documents `--ignore-scripts` and that
|
||||
flag must not silently change how the other four install:
|
||||
|
||||
```dockerfile
|
||||
# Pi (pi.dev). Upstream documents --ignore-scripts (pi needs no lifecycle scripts);
|
||||
# kept out of the shared npm block above so the flag cannot affect the other CLIs.
|
||||
RUN npm install -g --ignore-scripts @earendil-works/pi-coding-agent \
|
||||
&& npm cache clean --force \
|
||||
&& pi --version
|
||||
```
|
||||
|
||||
Implementation checklist item: the gid-0 pre-created dirs at `:64-68` include `.claude/projects`
|
||||
and `.codex/sessions`; verify whether the cred-seed copy into `~/.pi/agent` creates its target
|
||||
dir in a fresh container or whether `.pi/agent` must join that `mkdir` line. Rebuild with
|
||||
`node scripts/build-agent-image.mjs --no-cache` (the script itself needs no change; nothing in it
|
||||
is CLI-specific). The cached npm layer has silently frozen a CLI at a broken version before; see
|
||||
`docs/docker-cases.md`.
|
||||
- `install.sh` (six edit sites, all verified): `PI_SEARCH_PATHS` block after `:125` (mirror the
|
||||
resolver's dirs); `check_pi` / `get_pi_path` pair inserted at `:531` (antigravity's pair spans
|
||||
`:504-530`); the satisfying-AI-CLI chain `:2032-2063` (`has_pi` local at `:2037` area, detect
|
||||
block after `:2059`, widen the five-way test at `:2061` and the warn text at `:2063`); the menu
|
||||
option-4 text `:2070`; the skip-path hints `:2115-2116` (add
|
||||
`npm install -g --ignore-scripts @earendil-works/pi-coding-agent (Pi)`); the final no-CLI
|
||||
reminder `:2416-2423` (add `check_pi` to the condition and a pi line to the echo block).
|
||||
Detection plus a hint only; do **not** add an auto-install path in this change.
|
||||
|
||||
### Phase 5: Docs
|
||||
|
||||
- `docs/pi-integration.md` (**new**, user-facing): install (both installers uninstall via npm), auth
|
||||
(`/login` OAuth for six providers vs API keys; `pi auth check` for preflight; Claude Pro/Max
|
||||
third-party harness usage bills as Anthropic "extra usage" per token, not plan limits; OpenRouter
|
||||
login supports pasting the redirect URL, which matters over remote SSH), what Codeman wires up
|
||||
and deliberately does not (§3, incl. never passing `--tui-mode`), the tmux extended-keys note
|
||||
from §2.7 with the manual `~/.tmux.conf` fallback, Docker/remote behaviour (in-container sessions
|
||||
invisible host-side), the trust model in §1 words, known gaps.
|
||||
- `CLAUDE.md`: tech-stack line (six CLIs + `SessionMode` union), the env-prefix gotcha bullet, the
|
||||
multi-CLI prefix-discipline bullet, the "External CLI modes" key-pattern paragraph (note it now
|
||||
also carries the codex predictive-echo block; pi's echo-policy decision from §2.10 belongs in the
|
||||
same paragraph), the `src/utils/` resolver list.
|
||||
- `docs/architecture-invariants.md`: the external-CLI-modes section. ⚠️ Its anchor was already
|
||||
renamed once to `#external-cli-modes-opencode-codex-gemini-antigravity` while CLAUDE.md's link
|
||||
text still shows the old name; when renaming again for pi, update every inbound link (CLAUDE.md
|
||||
and this file).
|
||||
- `docs/docker-cases.md` (cred-seeding table + supported modes + image contents),
|
||||
`docs/remote-sessions.md` (`RemoteCommandMode`), `docs/cron-guide.md` + `docs/cron-discovery.md`
|
||||
(`agentType` enum; note the readiness caveat from §6's cron paragraph),
|
||||
`docs/security-architecture.md` (env prefix allowlist row).
|
||||
- `README.md` + `README.zh-CN.md`: six CLIs.
|
||||
- `package.json` keywords: `pi`.
|
||||
- Update the issue #206 thread when it ships.
|
||||
|
||||
---
|
||||
|
||||
## 5. Security checklist
|
||||
|
||||
1. **Command injection.** Every `PiConfig` value is regex-validated in `buildPiCommand()` before
|
||||
entering the `bash -c "..."` string; anything failing validation is dropped, not escaped
|
||||
(matching the four existing builders). No user string reaches the spawn line unvalidated. Pinned
|
||||
by a "rejects unsafe values" test per field.
|
||||
2. **Multi-user clamp, materialize branch.** `approveProjectTrust` is the privilege-shaped field: it
|
||||
makes pi execute repository-supplied TypeScript and install project packages.
|
||||
`clampExternalCliBypassForOwner()` (`session-routes.ts:305-336`) has two branches, and pi belongs
|
||||
in the **gemini-style materialize branch**, not the codex/antigravity only-if-sent branch:
|
||||
pi's absent-config default is an *interactive trust prompt the session user can answer
|
||||
themselves in the terminal*, so merely omitting `--approve` is not a clamp. For a non-granted
|
||||
owner, materialize `{ ...(piConfig ?? {}), approveProjectTrust: false }` so `buildPiCommand`
|
||||
always emits `--no-approve` and the prompt never appears. Both call sites (`:845`, `:2897`)
|
||||
widen. This helper still has **zero test coverage** (re-confirmed at f39beb3); §6 adds the first
|
||||
tests.
|
||||
3. **Secrets stay off the command line.** `PI_*` overrides flow through `applyEnvOverrides()` /
|
||||
socket-scoped `tmux setenv`, never inlined into the spawn string. No `-e` at container create
|
||||
time. And `--api-key` is never wired (§3): it would put a provider secret into `ps`/tmux state.
|
||||
4. **Env allowlist not widened.** Only the `PI_` prefix is added; the provider keys stay out (§2.4)
|
||||
and `ALLOWED_ENV_KEYS` is untouched. Pinned by a test that `PI_OFFLINE` passes and
|
||||
`ANTHROPIC_API_KEY` still fails validation.
|
||||
5. **Docker seeding, not sharing.** Per §2.5: RO mount then copy, so refreshed OAuth tokens never
|
||||
write back to the host; bind mounts stay excluded from `docker commit` so exports remain
|
||||
secret-free.
|
||||
6. **Remote SSH.** `pi` mode goes through `defaultRemoteCommandForMode` and therefore
|
||||
`buildSshConnectionArgs()`. No hand-built ssh line anywhere.
|
||||
7. **No sandbox claims.** Pi documents that it has no sandbox and no permission prompts, and that
|
||||
extensions run with the user's full permissions. Codeman docs must say plainly that a pi session
|
||||
can read, write and execute anything the Codeman user can, and point at Docker cases as the
|
||||
isolation story. Do not imply the trust prompt is a safety boundary (upstream itself says it is
|
||||
not). Worth one doc sentence: `pi auth print-api-key` / `print-bearer-token` (0.83.0) and
|
||||
`pi auth check` (0.84.1) mean a pi session can print its own provider credentials by design;
|
||||
isolation, again, is Docker.
|
||||
8. **Loud-vs-silent audit.** Before review, walk §2.8's silent list and confirm each site has its
|
||||
pi branch; the loud ones the compiler already caught.
|
||||
|
||||
---
|
||||
|
||||
## 6. Test plan
|
||||
|
||||
- `test/pi-mode.test.ts` (**new**, modeled on `test/antigravity-mode.test.ts`, 125 lines, no port;
|
||||
file unchanged since 2026-08-06 so its structure remains the template):
|
||||
`CreateSessionSchema`/`QuickStartSchema` accept a pi config; unsafe `model`/`provider`/
|
||||
`resumeSessionId` values are rejected (`'pi; rm -rf /'` shapes); `buildSpawnCommand({ mode: 'pi', ... })`
|
||||
emits expected flags, drops invalid ones, emits `--no-approve` for `approveProjectTrust: false`
|
||||
and `--approve` for `true`, and skips `-c` when a `resumeSessionId` is present;
|
||||
`defaultDockerCommandForMode('pi') === 'exec pi'` and
|
||||
`defaultRemoteCommandForMode('pi') === 'exec "${SHELL:-/bin/sh}" -i -l -c \'pi\''`;
|
||||
`isExternalCliMode('pi') === true`, `isAltScreenStripMode('pi') === false`; the env pair
|
||||
(`PI_OFFLINE` accepted, `ANTHROPIC_API_KEY` rejected), mirroring antigravity-mode `:49-63`.
|
||||
- **First-ever coverage for `clampExternalCliBypassForOwner`** (still nothing in `test/` touches
|
||||
it): cover pi's materialize branch (absent config still yields `approveProjectTrust: false` for a
|
||||
non-granted owner; a sent `true` is forced to `false`; granted owner passes through) and, while
|
||||
there, pin the three existing modes' behavior. Prefer exporting the helper for direct unit tests
|
||||
over a heavier multi-user route fixture; either way it lives under `test/routes/`.
|
||||
- `test/run-mode-ui.test.ts`: extend `loadUi()`'s stub lists (welcome-button ids, mode buttons,
|
||||
`ALL_OFF`) and add pi welcome/dropdown gating cases; note the static parser test
|
||||
`'gates every mode the run-mode menu actually offers'` (`:433-456`) picks up the new
|
||||
`data-mode="pi"` from index.html automatically and **fails until** `_refreshRunModeAvailability`
|
||||
contains a quoted `'pi'`, which is exactly the regression it exists for. Add a
|
||||
`describe('Pi quick start')` modeled on the antigravity one (`:840`) driving `runPi()` against a
|
||||
stubbed `/api/pi/status` + `/api/quick-start`, asserting the posted body has `mode: 'pi'` and
|
||||
**no `piConfig`**, and that the envelope is unwrapped. (The short-label assertion pattern is at
|
||||
`:82`, `'Run AG'`.)
|
||||
- `test/render-index-html.test.ts` `:141`: the injected `window.__codemanCliAvailable` is asserted
|
||||
with an exact `toEqual` and now carries **seven** keys (claude, opencode, codex, gemini,
|
||||
antigravity, cloudflared, and since 1.12+ `git`), so it **must** gain the `pi` key (and the
|
||||
resolver mock an `isPiAvailable`); its comment explains why: a dropped key silently un-gates
|
||||
(§2.8).
|
||||
- `test/routes/system-routes.test.ts`: `GET /api/pi/status` shape, modeled on the antigravity
|
||||
describe (`:816-838`) + resolver mock (`:84-87`); file unchanged since 2026-08-06.
|
||||
- `test/mobile-overview.test.ts`: `:375` is an exact-array `toEqual` over the run-menu modes and
|
||||
**will fail until updated** to include `'pi'` (the second exact-array at `:366`,
|
||||
`['claude', 'shell']`, is a gating case and stays as-is); the sibling static parser then covers
|
||||
the new entry automatically. The no-hex-literals guard only scans `.mobile-overview*` rules, so
|
||||
pi's `mode-pi` colors in mobile.css do not trip it.
|
||||
- `test/local-echo-codex-gating.test.ts` (§2.10): once the buffer-policy decision is confirmed in
|
||||
E2E, add `'pi'` to the `it.each(['claude', 'gemini', 'opencode'])` lists (`:193`, `:376`) so the
|
||||
chosen policy is pinned.
|
||||
- `test/skin-themes.test.ts`: will NOT trip (it enumerates skins, not modes); run it anyway since
|
||||
styles.css is touched. `test/mobile-header-buttons-policy.test.ts`: trips only if a header
|
||||
button is added; pi adds none (welcome button and run-menu rows are outside `header-right`).
|
||||
- Cron: schema-level acceptance of `agentType: 'pi'` (the service consumes `SessionMode`
|
||||
generically; `src/cron/` is unchanged since the first draft). Known, documented degradation: the
|
||||
readiness poll (`cron-service.ts:515`) looks for `❯`/`tokens`, which pi never prints, so cron pi
|
||||
jobs burn the ready-poll attempts and then send anyway. Acceptable for v1; note it in
|
||||
`docs/cron-guide.md`.
|
||||
- Sweep with `npm run test:ci`. Never bare `npm test`. No new ports needed (all new/extended suites
|
||||
are portless).
|
||||
|
||||
---
|
||||
|
||||
## 7. End-to-end verification (required before COM)
|
||||
|
||||
Unit tests passing is not evidence the mode works (pi is not currently installed on the dev box, so
|
||||
step 1 is a real step). Before shipping:
|
||||
|
||||
1. Install pi (`npm install -g --ignore-scripts @earendil-works/pi-coding-agent`), authenticate once
|
||||
with `/login`.
|
||||
2. `curl -sk https://localhost:3000/api/pi/status | jq` reports `available: true`, the right path,
|
||||
and a sane `version`.
|
||||
3. Create a **throwaway** case, launch a pi session from the Run dropdown, send a prompt from the
|
||||
browser, confirm the reply renders and scrollback survives a tab switch. Do not touch
|
||||
`w1`/`w2`/`w3`.
|
||||
4. **Local-echo policy gate (§2.10):** on a phone profile, type into the pi editor through the
|
||||
buffer overlay (drive with `page.keyboard.type()`, never `app.sendInput()`, and force
|
||||
`app._localEchoEnabled = true`; headless Chromium reports touch as false) and confirm pi's
|
||||
composer renders the flushed text correctly on Enter. If it mis-renders, flip pi to the `'off'`
|
||||
branch in `_updateLocalEchoState` and pin that instead.
|
||||
5. Visual pass on the **default skin** (the §2.9 finding makes this the load-bearing check, not a
|
||||
formality): run-button gradient actually renders rose (not generic claude blue), dot, tab badge,
|
||||
welcome button, kill-menu label; then a phone profile (toolbar `!important` colors and light-skin
|
||||
overrides are the usual regressions).
|
||||
6. Kill and respawn the session; confirm `piConfig` round-trips through `state.json` and the pane
|
||||
comes back with the same flags. Then `/clear`-style respawn via the Respawn tab.
|
||||
7. Extended keys (§2.7): in an attached terminal, verify whether Shift+Enter inserts a newline in
|
||||
pi's editor with and without the socket-scoped options; record the outcome in
|
||||
`docs/pi-integration.md` either way. While attached, also flip `/settings` to the fullscreen TUI
|
||||
and back to confirm the no-strip decision holds (§2.2).
|
||||
8. Trust model: point a throwaway case at a repo containing `.pi/extensions`, confirm the trust
|
||||
prompt appears interactively and that a multi-user non-granted session instead launches with
|
||||
`--no-approve` (prompt never shown, extensions not loaded).
|
||||
9. **NOT RUN in this pass — an honest gap.** Docker case with `mode: 'pi'`: rebuild the agent image with `--no-cache`, confirm `pi --version`
|
||||
inside the container **as the `agent` user**, confirm seeded auth works and a session starts
|
||||
(this is exactly where the antigravity Docker path broke in 1.11.2: the CLI was never installed
|
||||
in the image).
|
||||
10. **NOT RUN in this pass — the other gap.** Remote SSH case with `mode: 'pi'`: confirm the
|
||||
login-shell wrapper resolves the npm global bin.
|
||||
11. Only then: changeset, `COM minor` (new capability, additive to the API surface).
|
||||
|
||||
**Verification actually performed** (2026-08-13, pi 0.84.1, isolated `CODEMAN_INSTANCE=pi-beta`
|
||||
server on :5055 with its own tmux socket and data dir): steps 1-8 pass. Highlights:
|
||||
`/api/pi/status` resolved through the **search-dir fallback** (pi installed to `~/.npm-global/bin`,
|
||||
deliberately not on PATH) and reported
|
||||
`{available:true, path:'/home/arkon/.npm-global/bin', version:'0.84.1'}`; the real spawn line came
|
||||
out as `… COLORTERM=truecolor … && pi --approve --provider anthropic --thinking high`; `piConfig`
|
||||
round-tripped through `state.json` across a **full server restart**; the trust prompt appeared for a
|
||||
case containing `.pi/extensions` + `.pi/settings.json`, and `--no-approve` suppressed it
|
||||
(`This project is not trusted. Project .pi resources and packages are ignored.`); on the **default
|
||||
`daylight-blue` skin** the toolbar Run button computed to
|
||||
`linear-gradient(135deg, rgb(190,24,93), rgb(244,114,182))` — genuinely rose and **distinct from
|
||||
claude's blue**, so the §2.9 cascade trap is avoided; and flipping `/settings` to the fullscreen TUI
|
||||
put the pane into the alt screen (`alternate_on=1`), **empirically confirming §2.2**: had pi been in
|
||||
the strip list, Codeman would have stripped that switch and corrupted the session. Steps 9-10 need a
|
||||
Docker daemon and a remote host respectively.
|
||||
|
||||
---
|
||||
|
||||
## 8. Effort estimate
|
||||
|
||||
Calibrated against the real antigravity history, which is the honest baseline: the feature commit
|
||||
`26cbbe0` was 24 files, +638/-63, and it then took **four follow-up commits** (`e803186` login-shell
|
||||
routing, `292ba2c` ownership helpers, `5d28999` CLI gating incl. tests, `0d0b772` docs/installer/UI
|
||||
propagation) totaling roughly +600/-170 across ~43 file-touches to make the mode actually
|
||||
first-class. Budgeting only the feature-commit shape under-scopes by ~40%. This plan folds all four
|
||||
follow-up surfaces in from the start (login-shell routing in Phase 1, availability gating in Phases
|
||||
2-3, installer/docs propagation in Phases 4-5), so expect the full footprint in one pass:
|
||||
|
||||
| Phase | Size |
|
||||
| --------------------- | -------------------------------------------------------------------------- |
|
||||
| 1. Backend core | ~260 lines across 9 files, one new file (resolver incl. version probe) |
|
||||
| 2. Web layer | ~110 lines across 4 files (incl. the clamp widening + availability inject) |
|
||||
| 3. Frontend | ~175 lines across 10 files (enumerations + CSS in two sheets + skin block + the Brain picker option) |
|
||||
| 4. Docker + installer | ~45 lines, plus one `--no-cache` image rebuild |
|
||||
| 5. Docs | one new doc, ~10 files touched |
|
||||
| 6. Tests | one new test file, 6 extended (2 of which fail loudly until updated), plus the first clamp coverage |
|
||||
|
||||
---
|
||||
|
||||
## 9. Out of scope, tracked as follow-ups
|
||||
|
||||
- **A Codeman pi extension for real idle/completion events (highest value, now fully de-risked).**
|
||||
Pi extensions are TypeScript modules with Node built-ins and npm deps available, so an HTTP POST
|
||||
to `/api/hook-event` is trivial. The **`agent_settled`** event **shipped in 0.84.0** and is
|
||||
documented for exactly this use case (fires only when pi will not continue on its own: after
|
||||
auto-retries, auto-compaction and queued follow-ups; `ctx.isIdle()` is true inside the handler).
|
||||
That is a genuine idle signal replacing output-silence heuristics, i.e. the same class of upgrade
|
||||
hooks give Claude sessions. The bash tool exposes five env vars (`PI_SESSION_ID`,
|
||||
`PI_SESSION_FILE`, `PI_PROVIDER`, `PI_MODEL`, `PI_REASONING_LEVEL`), injected per command. Bonus:
|
||||
an extension can own the **`project_trust`** event (first yes/no wins, and CLI `-e` extensions
|
||||
load *before* trust resolution), so Codeman could answer the trust prompt programmatically, a
|
||||
cleaner mechanism than the `--approve` flag for both the single-user convenience case and the
|
||||
multi-user deny case.
|
||||
- **Response viewer for pi.** Sessions are JSONL v3 under
|
||||
`~/.pi/agent/sessions/--<cwd-dashed>--/<timestamp>_<uuid>.jsonl` with an `id`/`parentId` tree and
|
||||
typed content blocks (text, image, thinking, toolCall); the cwd-derived dir name is trivially
|
||||
computable host-side. Feasible, and it would justify flipping the Docker cred policy to share
|
||||
`sessions/` RW like Codex.
|
||||
- **Mode-aware env allowlist.** Would let pi sessions accept provider keys without widening the
|
||||
global list. Needs `ALLOWED_ENV_PREFIXES` to become a per-mode map plus mode context inside the
|
||||
Zod refine.
|
||||
- **`--tools` / `--exclude-tools` / `--no-tools` / `--no-builtin-tools` read-only sessions** (plus
|
||||
the 0.84.0 `defaultTools` setting). Real product value, needs UI.
|
||||
- **Predictive echo for pi's composer** if the §2.10 buffer decision does not hold up in practice:
|
||||
teach `PredictiveEchoAddon` pi's composer row the way `isCodexComposerRow` handles codex's.
|
||||
- **`--mode json` / `--mode rpc`, and upstream's experimental remote-session client APIs**
|
||||
(transport-neutral `PiClient`, CBOR protocol, Unix-socket transport, `RemoteSession` controller,
|
||||
still unreleased as of 0.84.1). A potential non-PTY integration path, a different architecture
|
||||
from the tmux+PTY model. Note the already-shipped breaking change to `message_update` framing
|
||||
(delta-only): any consumer must assemble deltas between `message_start`/`message_end`.
|
||||
- **`--name` for session labels.** Blocked on shell-quoting a user string in `buildSpawnCommand`.
|
||||
|
||||
---
|
||||
|
||||
## 10. Risks
|
||||
|
||||
| Risk | Mitigation |
|
||||
| ------------------------------------------------------------------- | ------------------------------------------------------------------------------ |
|
||||
| `pi` resolves to an unrelated binary | `pi --version` + semver-shape check in the resolver (§2.6); path and version shown in `/api/pi/status` |
|
||||
| Pi's TUI repaints in a way the browser terminal handles badly | Test scrollback and repaint early (step 3 of §7); pi's default is main-screen with terminal-owned scrollback, which is the friendly case |
|
||||
| Fullscreen TUI mode (shipped 0.84.0, runtime-switchable) | Already designed for: pi stays OUT of the strip list, so a user flipping `/settings` to fullscreen gets opencode-like alt-screen behavior, not corruption. §7 step 7 tests the flip explicitly |
|
||||
| The buffer local-echo overlay fights pi's live composer | §2.10: explicit E2E gate (§7 step 4) with the one-line `'off'` fallback; predictive echo for pi is a tracked follow-up, not a v1 blocker |
|
||||
| Pi moves fast (pre-1.0; 9 releases in the 7 weeks before 0.84.1) | Keep the flag surface small; every flag validated and droppable; nothing pinned in the Dockerfile beyond the `--no-cache` rebuild cadence. Live example of the hazard: `--tui-mode` went from main-only docs to released between the two drafts of this plan |
|
||||
| Docker image grows | Pi is an npm package; the layer is modest next to the ~190MB `agy` binary |
|
||||
| Trust prompt blocks a session | Narrower than feared: only fires when `.pi/settings.json`, `.pi/extensions\|skills\|prompts\|themes`, `.pi/SYSTEM.md`/`APPEND_SYSTEM.md` or `.agents/skills` exists (bare `.pi/` does not). Documented; `approveProjectTrust` is the opt-in escape hatch; multi-user forces `--no-approve` (§5.2); the `project_trust` extension follow-up removes the prompt entirely |
|
||||
| Interactive `/login` OAuth can't complete headlessly | Document: authenticate once interactively (or seed `auth.json`); `pi auth check` verifies credentials preflight; OpenRouter's paste-the-redirect-URL flow covers remote SSH |
|
||||
| Provider auth is awkward without key prefixes in the allowlist | `/login` writes `~/.pi/agent/auth.json` once and Docker seeds it; the mode-aware allowlist follow-up removes the friction |
|
||||
| Cron pi jobs mis-detect readiness | Known degradation, documented in §6; readiness falls through after the poll budget and the prompt still sends |
|
||||
@@ -0,0 +1,235 @@
|
||||
# Pi (pi.dev) sessions
|
||||
|
||||
Codeman can drive [Pi](https://pi.dev) (`@earendil-works/pi-coding-agent`, MIT) as a
|
||||
session backend, alongside Claude Code, OpenCode, Codex, Gemini and Antigravity.
|
||||
`pi` is a sixth **run mode**: its own PTY, its own tmux session, its own tab colour
|
||||
(rose). It is not a location overlay like Docker or remote-SSH cases, and it is not
|
||||
a web tab.
|
||||
|
||||
Tracking issue: [#206](https://github.com/Ark0N/Codeman/issues/206). The design
|
||||
rationale behind each decision below lives in `docs/pi-integration-plan.md`.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
npm install -g --ignore-scripts @earendil-works/pi-coding-agent
|
||||
# or
|
||||
curl -fsSL https://pi.dev/install.sh | sh
|
||||
```
|
||||
|
||||
Both installers end up going through global npm, so either one uninstalls with
|
||||
`npm uninstall -g @earendil-works/pi-coding-agent`.
|
||||
|
||||
Codeman finds the binary via `which pi` and then the usual global-bin locations
|
||||
(`~/.local/bin`, `/usr/local/bin`, `~/.bun/bin`, `~/.npm-global/bin`, `~/bin`).
|
||||
|
||||
**`pi` is a short, generic name**, so unlike the other CLI resolvers Codeman does
|
||||
not trust a `which` hit on its own: it runs `pi --version` once and requires
|
||||
semver-shaped output. Anything else is rejected as "not installed" and the
|
||||
rejected path is logged. Check what it resolved:
|
||||
|
||||
```bash
|
||||
curl -s localhost:3000/api/pi/status | jq
|
||||
# { "available": true, "path": "/home/you/.local/bin", "version": "0.84.1" }
|
||||
```
|
||||
|
||||
That endpoint carries `version` on top of the shape the sibling `/api/*/status`
|
||||
endpoints return, precisely so a misresolution is visible rather than presenting
|
||||
as "the mode just doesn't work".
|
||||
|
||||
## Authenticate
|
||||
|
||||
Pi supports 15+ providers. Two ways in:
|
||||
|
||||
- **OAuth subscription login** — run `/login` inside a pi session. Six providers
|
||||
support it: ChatGPT Plus/Pro, Claude Pro/Max, GitHub Copilot, xAI, OpenRouter
|
||||
and Radius. Credentials land in `~/.pi/agent/auth.json` and pi refreshes them
|
||||
itself. OpenRouter's flow accepts a pasted redirect URL, which is what makes it
|
||||
workable over remote SSH.
|
||||
- **API keys** — exported in the environment of the **Codeman server process**.
|
||||
|
||||
⚠️ **Provider API keys cannot be sent as per-session `envOverrides`.** Pi reads
|
||||
about 34 provider variables (`ANTHROPIC_API_KEY`, `OPENAI_API_KEY`,
|
||||
`DEEPSEEK_API_KEY`, `HF_TOKEN`, `BASETEN_API_KEY`, …) that share no common prefix.
|
||||
Codeman's env allowlist is a single global list applied to every mode at once, so
|
||||
admitting bare provider keys for pi would widen the allowlist for Claude, Codex,
|
||||
Gemini and everything else too. Only the **`PI_*`** prefix was added, which covers
|
||||
every documented pi input: `PI_CODING_AGENT_DIR`, `PI_CODING_AGENT_SESSION_DIR`,
|
||||
`PI_PACKAGE_DIR`, `PI_OFFLINE`, `PI_SKIP_VERSION_CHECK`, `PI_TELEMETRY`,
|
||||
`PI_CACHE_RETENTION`, `PI_SHARE_VIEWER_URL`, `PI_HARDWARE_CURSOR`,
|
||||
`PI_EXPERIMENTAL`.
|
||||
|
||||
`pi auth check` verifies credentials before you start a long run.
|
||||
|
||||
Note if you authenticate with a Claude Pro/Max subscription: third-party harness
|
||||
usage bills as Anthropic "extra usage" per token rather than against plan limits.
|
||||
|
||||
## What Codeman wires up
|
||||
|
||||
`PiConfig` (per session, persisted in `state.json`, round-trips through respawn):
|
||||
|
||||
| Field | Flag | Notes |
|
||||
| --------------------- | -------------------------------------- | ---------------------------------------------------------------- |
|
||||
| `model` | `--model <v>` | Accepts `provider/id` and a `:<thinking>` suffix (`sonnet:high`) |
|
||||
| `provider` | `--provider <v>` | `anthropic`, `openai`, `google`, … |
|
||||
| `thinking` | `--thinking <v>` | `off`/`minimal`/`low`/`medium`/`high`/`xhigh`/`max` |
|
||||
| `continueSession` | `-c` | Skipped when `resumeSessionId` is set (the two conflict) |
|
||||
| `resumeSessionId` | `--session <v>` | Ids only, never paths |
|
||||
| `approveProjectTrust` | `--approve` / `--no-approve` / nothing | Tri-state, see below |
|
||||
|
||||
Every value is regex-validated and **dropped** (not escaped) if it fails, because
|
||||
the result is interpolated into the pane's `bash -c "…"` command.
|
||||
|
||||
The Run button sends **no `PiConfig` at all**: pi has no permission prompts to
|
||||
bypass, and project trust is a decision the person at the terminal makes.
|
||||
|
||||
## What Codeman deliberately does NOT wire up
|
||||
|
||||
- **`--api-key`.** Never. It would put a provider secret on the spawn command
|
||||
line, visible in `ps`, tmux server state and logs. `PI_*` overrides go through
|
||||
socket-scoped `tmux setenv` for exactly this reason.
|
||||
- **`--tui-mode`.** Pi's default main-screen TUI is the friendly case for a
|
||||
browser terminal. The fullscreen mode (0.84.0) stays your own runtime choice via
|
||||
`/settings`.
|
||||
- **`--name`, `--no-session`, `-p`/`--print`, `--mode json`, `--mode rpc`,
|
||||
`--tools`/`--exclude-tools`, `-e`/`--extension`, `--skill`,
|
||||
`--system-prompt`.** Tracked as follow-ups in the plan doc.
|
||||
|
||||
## Permission and trust model — read this
|
||||
|
||||
**Pi has no permission prompts and no sandbox.** There is no
|
||||
`--dangerously-skip-permissions` analog and none is needed: tools run with the
|
||||
user's own permissions, always. A pi session can read, write and execute anything
|
||||
the Codeman user can. If you need isolation, use a **Docker case** — that is the
|
||||
isolation story, here as everywhere else in Codeman.
|
||||
|
||||
Pi's "project trust" prompt is **not** a safety boundary (upstream says so too).
|
||||
It gates *loading* repo-local `.pi/` config, extensions and skills, and
|
||||
*installing* missing project packages. It only appears when the cwd or an ancestor
|
||||
contains `.pi/settings.json`, `.pi/extensions|skills|prompts|themes`,
|
||||
`.pi/SYSTEM.md`/`.pi/APPEND_SYSTEM.md`, or `.agents/skills`. A bare `.pi/`
|
||||
directory does not trigger it.
|
||||
|
||||
`approveProjectTrust: true` answers it with `--approve`, which means pi **loads
|
||||
and executes repository-supplied TypeScript** and runs an npm install for missing
|
||||
project packages. Treat it exactly as seriously as that sounds.
|
||||
|
||||
**Multi-user mode:** for an owner without the privileged-command grant, Codeman
|
||||
materializes `approveProjectTrust: false` so the pane launches with
|
||||
`--no-approve` and the prompt never appears. Merely *omitting* `--approve` would
|
||||
not be a clamp, since pi's own default is to ask and the session user could just
|
||||
answer yes.
|
||||
|
||||
Also worth knowing: `pi auth print-api-key` / `print-bearer-token` and
|
||||
`pi auth check` mean a pi session can print its own provider credentials by
|
||||
design. Isolation is Docker.
|
||||
|
||||
## tmux extended keys (Shift+Enter)
|
||||
|
||||
Pi's editor uses `Shift+Enter` / `Ctrl+Enter` for newline-vs-submit. Without
|
||||
extended keys, tmux collapses both into a plain `\r`. Upstream recommends:
|
||||
|
||||
```tmux
|
||||
set -g extended-keys on
|
||||
set -g extended-keys-format csi-u
|
||||
```
|
||||
|
||||
`extended-keys-format` needs tmux 3.5+; on 3.2–3.4 `extended-keys on` alone works
|
||||
(pi falls back to xterm `modifyOtherKeys`).
|
||||
|
||||
Codeman's browser input path sends `\r` for submit, so basic use works
|
||||
unconfigured — what degrades is newline-in-editor, mostly when you attach to the
|
||||
pane directly (`sc`).
|
||||
|
||||
⚠️ Upstream notes the setting may need a full `tmux kill-server` to take effect.
|
||||
**Never run `tmux kill-server` on Codeman's socket** — it would kill every live
|
||||
session, `w1`/`w2`/`w3` included.
|
||||
|
||||
**Measured (tmux 3.4, pi 0.84.1): no `kill-server` is needed.** Setting the option
|
||||
server-scoped on Codeman's own socket takes effect on the ALREADY-RUNNING server;
|
||||
the next pi session starts without the warning. Existing sessions keep the old
|
||||
setting until they respawn.
|
||||
|
||||
```bash
|
||||
tmux -L codeman set -s extended-keys on
|
||||
tmux -L codeman set -s extended-keys-format csi-u # tmux 3.5+ only, see below
|
||||
tmux -L codeman show-options -s | grep extended # verify
|
||||
```
|
||||
|
||||
On **tmux 3.4 and older, `extended-keys-format` does not exist** and the second
|
||||
line fails with `invalid option: extended-keys-format`. That is harmless — pi
|
||||
falls back to xterm `modifyOtherKeys` and `extended-keys on` alone silences the
|
||||
warning. Run the two lines independently rather than chained.
|
||||
|
||||
Pi tells you which state it is in: an unconfigured session prints
|
||||
`Warning: tmux extended-keys is off. Modified Enter keys may not work.` in its
|
||||
startup banner, so you can verify the change by starting a new pi session.
|
||||
|
||||
⚠️ Use `-L <socket>` and `-s`, never `-g` on your default socket, and never
|
||||
`kill-server`. Codeman does not set this for you: it is a server-wide tmux option
|
||||
and silently changing key encoding for every session of every backend is not
|
||||
Codeman's call to make.
|
||||
|
||||
## Typing from the browser (local echo)
|
||||
|
||||
On touch devices Codeman buffers typed characters in the `LocalEchoOverlay` and
|
||||
flushes them to the PTY on Enter. Pi gets that `'buffer'` policy, the same as
|
||||
Claude, Gemini and OpenCode.
|
||||
|
||||
This was an explicit open question, because that policy is exactly what broke
|
||||
Codex (issues #218/#219/#220/#222): Codex's composer reacts per keystroke, so
|
||||
buffer-until-Enter starved it. **Measured against pi 0.84.1: it does not
|
||||
reproduce.** Pi's slash-command picker re-filters on the whole composer content
|
||||
rather than on per-keystroke deltas, so a one-shot flush of `/set` filters the
|
||||
picker down to `settings` identically to typing it character by character, and
|
||||
the delayed `\r` then selects it. Prose prompts flush and submit correctly too.
|
||||
|
||||
If a future pi release changes that, the cheap fallback is one `'off'` branch in
|
||||
`_updateLocalEchoState` (terminal-ui.js); teaching `PredictiveEchoAddon` pi's
|
||||
composer row is the larger follow-up.
|
||||
|
||||
## Docker cases
|
||||
|
||||
The agent image (`docker/agent.Dockerfile`) installs pi in its own `RUN` step with
|
||||
`--ignore-scripts`, kept out of the shared npm block so the flag cannot change how
|
||||
the other four CLIs install. Rebuild with:
|
||||
|
||||
```bash
|
||||
node scripts/build-agent-image.mjs --no-cache # --no-cache is mandatory
|
||||
```
|
||||
|
||||
Credentials are **seeded**, not shared: `~/.pi/agent/auth.json`, `settings.json`,
|
||||
`trust.json`, `models.json` and `models-store.json` are mounted read-only and
|
||||
copied into the container's own `~/.pi/agent`. So an in-container pi never writes
|
||||
refreshed OAuth tokens back to the host, and `docker commit` exports stay
|
||||
secret-free. `models.json` is in the list because it holds user-defined custom
|
||||
providers, which would otherwise silently vanish inside containers.
|
||||
|
||||
Only those five files are seeded because `~/.pi/agent` also holds `sessions/`,
|
||||
`extensions/`, `skills/` and the installed package trees (`npm/`, `git/`), which
|
||||
on an active host is easily gigabytes.
|
||||
|
||||
**Trade-off:** in-container pi sessions are invisible host-side, so `pi -c` inside
|
||||
a Docker case only sees that container's own history.
|
||||
|
||||
## Remote SSH cases
|
||||
|
||||
`pi` mode is routed through an interactive login shell
|
||||
(`exec "$SHELL" -i -l -c 'pi'`), because sshd's remote-command PATH does not
|
||||
include npm's global bin on most hosts. Per-session config and `envOverrides` do
|
||||
not cross ssh and are rejected rather than silently ignored; use the per-host
|
||||
command override instead.
|
||||
|
||||
## Known gaps
|
||||
|
||||
- **No idle/completion hook.** Pi has no hook system Codeman can install into, so
|
||||
idle detection falls back to output-stabilization like the other external CLIs.
|
||||
Pi 0.84.0 shipped an `agent_settled` extension event that is a genuine idle
|
||||
signal; a Codeman pi extension using it is the highest-value follow-up.
|
||||
- **No response viewer.** Pi writes JSONL v3 session files under
|
||||
`~/.pi/agent/sessions/`; nothing reads them yet.
|
||||
- **Cron jobs mis-detect readiness.** The cron readiness poll looks for `❯` or a
|
||||
token count, neither of which pi prints, so a pi cron job burns its poll budget
|
||||
and then sends the prompt anyway. It works; it is just slower to start.
|
||||
- **Ralph, respawn heuristics, token/CLI-info parsing and the `❯` readiness probe
|
||||
are off** for pi, as for every external CLI.
|
||||
@@ -0,0 +1,143 @@
|
||||
# Predictive write-through echo for codex
|
||||
|
||||
Zero-lag local echo for codex sessions via a second, mosh-style mode in the
|
||||
`xterm-zerolag-input` package: every keystroke goes to the PTY exactly as the
|
||||
1.12.2 overlay-disabled path did (byte-identical wire behavior), while a
|
||||
`PredictiveEchoAddon` simultaneously paints the predicted glyph at the predicted
|
||||
cell. When the real echo lands, the prediction is confirmed and its span removed
|
||||
(invisible swap: identical glyph beneath). Mispredictions drop via a mismatch
|
||||
cascade + TTL. Visual-only, self-healing.
|
||||
|
||||
## Why this exists
|
||||
|
||||
Issues #218/#219/#220/#222 (one root cause) forced 1.12.2 to disable the
|
||||
LocalEchoOverlay for codex: buffer-until-Enter starves codex's per-keystroke TUI
|
||||
(live slash picker, arrows editing server-side composer state, composer
|
||||
rewrap/growth, paste_burst classification). Buffer mode is structurally
|
||||
incompatible with codex; write-through prediction is the only echo mode that
|
||||
can coexist with it.
|
||||
|
||||
## The reconciliation lesson (do not regress this)
|
||||
|
||||
`docs/local-echo-overlay-plan.md` ("What NOT to Do") documented that matching
|
||||
predictions against the raw output STREAM fails against Ink/TUI full-line
|
||||
redraws. This design reads the parsed terminal BUFFER instead (cells after
|
||||
xterm's parser ran), which converges to the same cells no matter how the bytes
|
||||
arrived. The Phase 0 recordings prove the point twice over: tmux converts
|
||||
codex's full-line redraws into minimal in-place deltas (an echo arrives as
|
||||
`e\x1b[K\x1b[20;80H...`), and codex itself paints word gaps with ECH+cursor-forward
|
||||
instead of spaces. Stream matching can never survive that; buffer diffing does
|
||||
not care.
|
||||
|
||||
## Phase 0 measurements (codex-cli 0.147.0 via tmux, 100x30, 2026-08-09)
|
||||
|
||||
Recorded with `scripts/dev/record-codex-frames.mjs` (production pipeline:
|
||||
codex inside tmux `status off`, chunks passed through the same full strip
|
||||
`session.ts _handleTerminalOutput()` applies to codex mode). Fixtures in
|
||||
`packages/xterm-zerolag-input/test/fixtures/codex/`; replay/measure with
|
||||
`scripts/dev/analyze-codex-frames.mjs <fixture>`.
|
||||
|
||||
| Question | Measured answer |
|
||||
| --------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Composer signature | Cursor row starts `"› "` (U+203A + space), text begins col 2. Present when empty (placeholder), while typing, and while the slash picker filters. `CODEX_COMPOSER_ROW_RE = /^› /` |
|
||||
| Composer text color | Plain default foreground, zero SGR around echoed chars. Span `foregroundColor` default (theme fg) is an exact match |
|
||||
| Placeholder | Cycling hint text ("Use /skills...", "Improve documentation in @filename", ...) rendered AT the cursor cell. First prediction lands over placeholder glyphs: covered by the snapshot + cursor-advance rules |
|
||||
| Wrap | Word-wrap near `cols - 2`; continuation rows are indented 2 spaces WITHOUT `› `. The gate therefore suppresses predictions on wrapped lines: deliberate fallback to real echo, wrap was the #220 ghost zone. `edgeMarginCells = 4` |
|
||||
| Modal (trust dialog) | Cursor parks on `" Press enter to continue"`: no `› ` prefix, gate false, zero predictions painted while keystrokes still reach the PTY (the ghost eliminator) |
|
||||
| Streaming | Error/reconnect bursts render above a re-rendered composer that keeps the `› ` signature; end-of-frame cursor parks at the insertion point (col 2 of the composer row). Confirms the cursor-advance confirm rule and the no-drop-on-baseY rule |
|
||||
| Echo shape under tmux | tmux emits minimal deltas for simple echoes and full repaints for busy frames; both converge in the parsed buffer |
|
||||
| Slash picker | Picker rows render below; the cursor row keeps the composer signature and advances per filter char, so predictions stay active while filtering (#222 surface) |
|
||||
|
||||
Constants decided at the Phase 0 gate: `CODEX_COMPOSER_ROW_RE = /^› /`,
|
||||
`ttlMs = 1000`, `maxPending = 32`, `cursorGraceMs = 150`, `edgeMarginCells = 4`,
|
||||
span colors = theme defaults, `underlinePredictions = false`.
|
||||
|
||||
## Algorithm
|
||||
|
||||
See `PredictiveEchoAddon` in
|
||||
`packages/xterm-zerolag-input/src/predictive-echo-addon.ts`. Summary of the
|
||||
rules and why each exists:
|
||||
|
||||
- **State**: ordered `PredictionRecord[]` (`seq`, `char`, `width`, cumulative
|
||||
`offsetCells`, `snapshot` of the cell at predict time, `sentAt`,
|
||||
`mismatches`), plus a run `_anchor {row, col}` captured when the outstanding
|
||||
count goes 0 -> 1. Positions are FIXED at predict time; confirmation deletes
|
||||
spans and never re-lays-out, so partial confirmation causes zero jitter.
|
||||
- **predictChar(ch)** runs an inline reconcile first and re-anchors whenever
|
||||
outstanding drains to zero (absorbs the echo-landed-between-keystrokes race).
|
||||
Guards: dims present, cursor numbers present, `viewportY === baseY`,
|
||||
`predictWhen` gate, single codepoint >= 0x20 (not 0x7f), width <= 2,
|
||||
`maxPending`, edge margin. Returns false = suppressed; the consumer sends the
|
||||
keystroke regardless.
|
||||
- **Coordinate base is `baseY`**: xterm's `cursorY` is baseY-relative, so
|
||||
absolute buffer line = `baseY + row`. `viewportY` would only coincide while
|
||||
the scrolled-to-bottom guards hold; the addon never relies on that.
|
||||
- **reconcile()** (debounced `onWriteParsed` microtask, inline in predictChar,
|
||||
TTL timer): clears everything when scrolled up; off-anchor-row cursor
|
||||
tolerated for `cursorGraceMs` then clears; PREFIX-ONLY confirm loop requiring
|
||||
cell match AND cursor advanced past the record (prevents false confirms
|
||||
against placeholder glyphs and makes identical in-place tmux repaints a
|
||||
no-op); TWO-PASS mismatch rule (a cell that is neither snapshot nor predicted
|
||||
char must persist across two passes before cascading the drop: a half-parsed
|
||||
row on pass N is fully redrawn a few ms later); TTL drop of the stale suffix.
|
||||
- **No drop on baseY change**: codex streams push lines to history while the
|
||||
composer stays viewport-pinned; predictions are row-relative to the pinned
|
||||
composer and remain valid (measured above).
|
||||
- **Anchor hold** (added by the independent post-build review): after any wire
|
||||
input whose cursor effect the display has not shown yet (backspace with
|
||||
nothing outstanding = deleting echoed text, every 'clear'-classified input,
|
||||
an IME/plain-paste 'text' commit, and the bypass send paths), new
|
||||
predictions are suppressed until the next PARSED write. Anchoring on the
|
||||
stale cursor painted ghosts one cell off ("tehh" on backspace-then-retype
|
||||
within RTT), blank-neutral and therefore TTL-lived. Worst case is exactly
|
||||
one unpredicted keystroke: its own echo is a write, which releases the hold.
|
||||
- **predictBackspace()** pops the newest outstanding record (informational
|
||||
return; the consumer forwards `\x7f` unconditionally). Deleting already-echoed
|
||||
text renders at RTT in v1.
|
||||
- **CJK/wide**: 2-cell spans, stacking by cumulative visual width, leading-cell
|
||||
confirm. In Codeman, IME input never reaches the hook (`window.cjkActive`
|
||||
returns from onData first); package support exists for other consumers.
|
||||
|
||||
## Integration map (Codeman)
|
||||
|
||||
- Policy: `_localEchoPolicy` (`'buffer' | 'predict' | 'off'`) computed at the
|
||||
end of `_updateLocalEchoState()`; codex + `localEchoEnabled` -> `'predict'`
|
||||
while `_localEchoEnabled` stays false (every 1.12.2 consumer unchanged).
|
||||
- onData hook sits between the buffer block and Normal Mode, classifies via
|
||||
`classifyPredictInput()` (pure, on `window.CodemanTerminalInput`), never
|
||||
returns, try/catch-wrapped: the wire path below is byte-identical with the
|
||||
predictor active, absent, or throwing.
|
||||
- Composer gate: `isCodexComposerRow()` set via `setPredictWhen()` at
|
||||
construction (the vendor footer stays package-agnostic).
|
||||
- Second vendor bundle `vendor/xterm-predictive-echo.js` (postinstall + build);
|
||||
the zerolag bundle build command is untouched and its output byte-identical.
|
||||
Missing/broken bundle = plain 1.12.2 echo (`typeof PredictiveEchoOverlay ===
|
||||
'undefined'` guard).
|
||||
- Prediction clears on: tab switch, SSE reconnect init, `insertTerminalText`,
|
||||
`clearTerminalInput`, voice send, keyboard-accessory `sendKey`, resize, skin
|
||||
and font changes re-read style via `refreshFont()`.
|
||||
|
||||
## Risk register
|
||||
|
||||
Eliminated structurally: other-mode regression (zero edits to buffer
|
||||
addon/branches, byte-identical existing bundle, policy-matrix + byte-identity
|
||||
tests); bundle breakage (separate bundle, graceful degradation); wire
|
||||
corruption (no-return fall-through + try/catch + byte-identity pins at vm and
|
||||
E2E level); modal ghosts (measured predictWhen gate); false confirms
|
||||
(cursor-advance rule); mid-parse flicker drops (two-pass rule); wrap
|
||||
misplacement (edge margin + continuation-row gate fallback + off-row grace).
|
||||
|
||||
Accepted residuals (visual-only, self-healing <= ttlMs, kill-switchable via
|
||||
`localEchoEnabled` per device): no predictions on wrapped continuation lines
|
||||
(gate false there, deliberate); brief dropout during composer growth; DOM-span
|
||||
vs WebGL glyph rendering can differ subtly (same trade-off as the buffer
|
||||
overlay, same font recipe); typing during an unsynchronized half-frame can
|
||||
mis-anchor one run (mismatch/TTL cleans within 1s).
|
||||
|
||||
## Future work
|
||||
|
||||
RTT-adaptive TTL; mosh-style confidence gating (paint only after the link
|
||||
proves laggy); predicted backspace into echoed text; predict mode for shell
|
||||
prompts; unifying the small font/container duplication between the two addons
|
||||
once predict mode has proven out; continuation-line prediction behind a
|
||||
smarter composer-extent detector.
|
||||
@@ -0,0 +1,140 @@
|
||||
# Read My Mind (design)
|
||||
|
||||
A 🧠 button that predicts the prompt you were about to type. Codeman keeps a per-case **intent profile** (your stated goals plus the real prompts you recently sent), feeds it and the live pane tail to a one-shot `claude -p`, and shows the predicted next prompt in a plan-mode-style approval dialog: **Send** / **Rethink** (with an optional steer note) / **Insert** (drop it on the composer to edit) / **Dismiss**. It is also a skill surface: the agent can read the intent profile, record intentions, and request a prediction over the HTTP API. Suggestions are **never auto-sent**; the human click is the boundary.
|
||||
|
||||
## UX flow
|
||||
|
||||
1. User hits 🧠 (desktop header button; phone: keyboard-accessory key).
|
||||
2. Modal opens with a spinner, then the top suggestion in an editable single-line field, rationale below it, up to 2 alternates as tappable rows.
|
||||
3. Buttons: **Send** (submits with `\r`), **Insert** (sends without `\r`, so the text sits unsubmitted on the CLI composer for editing, a documented mechanism), **Rethink** (optional free-text steer, e.g. "no, I meant the mobile bug", re-runs with the rejected suggestions included), **Dismiss**.
|
||||
4. Accepted prompts flow back into the intent history like any other sent prompt, so the profile self-corrects.
|
||||
|
||||
## Scope (v1)
|
||||
|
||||
- Claude mode only (capture rides Claude transcripts; external CLIs have no transcript watcher). Mirrors the approvals-inbox scoping.
|
||||
- Opt-in: `readMyMindEnabled`, synced, default **OFF**. While OFF: no capture, no UI surfaces. Privacy first, and every press costs real tokens.
|
||||
- One prediction in flight per session; the button disables while checking.
|
||||
- Sync request/response (the predictor takes 5-30s; agent-wait long-polls already hold requests longer). No new SSE events in v1.
|
||||
|
||||
## Data model
|
||||
|
||||
Per case, not per session: intentions outlive `/clear` and respawns.
|
||||
|
||||
```ts
|
||||
interface IntentProfile {
|
||||
key: string; // sha256(owner + ':' + realpath(workingDir)).slice(0, 16)
|
||||
workingDir: string;
|
||||
updatedAt: number;
|
||||
goals: string; // freeform markdown, user/agent editable, ≤ 8 KB
|
||||
recentPrompts: { ts: number; sessionId: string; text: string }[]; // FIFO cap 50, each ≤ 500 chars
|
||||
}
|
||||
```
|
||||
|
||||
Storage: `dataPath('intents.json')`, written mode 0600 (prompts can contain secrets; same posture as `users.json`). Never enters the `/api/search` index. Add to the CLAUDE.md State Files list.
|
||||
|
||||
## Intent capture
|
||||
|
||||
**Source: the session transcript, not the input paths.** `POST /api/sessions/:id/input` sees only programmatic input, and the WS channel delivers raw keystrokes (`session.write(msg.d)`), so neither yields clean submitted prompts. Claude's own JSONL transcript records every user turn as structured text, and `transcript-watcher.ts` already tails it. Add a `userPrompt` event there:
|
||||
|
||||
- Emit for `type: 'user'` entries whose content is a string or contains a text block; skip entries that are only `tool_result` blocks (tool results are wrapped as user messages).
|
||||
- Skip `<command-name>` / `<local-command-stdout>` tagged entries (local slash-command echo, not intent).
|
||||
- Skip texts < 3 chars (menu digits, Esc artifacts), truncate to 500, drop consecutive duplicates ("continue" spam from auto-resume stays but dedupes).
|
||||
|
||||
`IntentStore` (new `src/intent-store.ts`, pure core + IO wrapper, in the style of `session-order.ts`) subscribes via session wiring, gated on the setting resolved from **merged** settings per the partial-PUT rule.
|
||||
|
||||
## Context assembly (how the mind reading actually works)
|
||||
|
||||
The quality of the suggestion is decided before the model ever runs, by what we put in front of it. A new pure function `buildPredictionContext()` (in `src/readmymind-context.ts`, unit-testable with fixtures, no IO of its own; collectors inject their data) assembles a budgeted, priority-ordered prompt from every signal Codeman already has:
|
||||
|
||||
| # | Source | What it contributes | Cap |
|
||||
| - | ------ | ------------------- | --- |
|
||||
| 1 | **Pending dialog** (approvals-inbox store, when present) | If the session is sitting on an AskUserQuestion / permission / idle prompt, the honest "next prompt" is an *answer*. The dialog text + parsed options go in first and the model is told to answer it. | 2 KB |
|
||||
| 2 | **User goals** (`goals` from the intent profile) | The only fully-trusted statement of what the user wants. Highest authority in the trust ranking below. | 8 KB |
|
||||
| 3 | **Last assistant turn** (transcript, not the pane) | Assistant replies usually *end* with the fork in the road ("Want me to X?", "Next steps: ..."), so keep the **tail** when truncating. The transcript has the full message; the pane is a repaint window full of spinner junk. | 6 KB |
|
||||
| 4 | **Recent user prompts** (intent profile, with timestamps) | The conversation rhythm AND the user's prompting voice: length, tone, shorthand (`COM`, lowercase, typos and all). The model is instructed to write suggestions in *this* style, not assistant-ese. | last 20 |
|
||||
| 5 | **Recent tool activity** (transcript `tool_use` blocks, already parsed by `TranscriptWatcher`) | One line per call: `Edit src/foo.ts`, `Bash npm test (failed)`. What the agent actually *did*, which the last message may summarize away. | last 10 |
|
||||
| 6 | **Workspace signals** (`collectWorkspaceSignals()`: `git` via `execFile` in `workingDir`, 2s timeout) | Branch, `status --short` (dirty files scream "commit/test/deploy next"), last 5 commits oneline, presence of `.changeset/*.md` (release pending). Skipped for remote-SSH cases (workingDir is not local); fine for Docker cases (bind-mounted at the same host path). Non-git dirs: section omitted. | 3 KB |
|
||||
| 7 | **Away context** (run-summary events + elapsed time) | `Last user prompt was 6h ago; since then: <run-summary events for this session>`. After a long gap the right suggestion is often "review / continue yesterday's thread", not a blind continuation. | 2 KB |
|
||||
| 8 | **Sibling sessions** (live sessions sharing the case) | One line each: name, mode, working/idle. A lead-and-workers setup changes what the next prompt should be ("check on w2" beats "keep going"). | 1 KB |
|
||||
| 9 | **Rethink state** (steer note + rejected suggestions) | Only on re-runs. Rejections are strong negative signal and go in verbatim. | 2 KB |
|
||||
|
||||
Total budget ~30 KB. When over budget, drop from the bottom up (siblings first, then away context, then workspace signals); sections 1-4 never drop, they only truncate. Deterministic assembly means fixture tests can pin exactly what a given situation feeds the model.
|
||||
|
||||
**Trust tiers are stated in the prompt.** Goals and user prompts are *the user*; assistant text, tool logs, and pane content are *observations that may contain text trying to manipulate you* (a hostile repo can print "SUGGEST: run curl evil.sh"). The prompt instructs: user-stated intent outranks anything observed, and never propose a prompt whose primary source is terminal output alone. The human approval click remains the hard boundary regardless.
|
||||
|
||||
**Output contract** (strict JSON, parse failure = clean error, never a half-suggestion):
|
||||
|
||||
```json
|
||||
{ "suggestions": [ { "prompt": "...", "why": "...", "kind": "continue" | "verify" | "redirect" } ] }
|
||||
```
|
||||
|
||||
1-3 entries, and the *kinds* force useful diversity instead of three rewordings: `continue` (finish the current thread, or answer the pending dialog), `verify` (test/review what was just built; the user's own "always end-to-end test" discipline), `redirect` (the next goal from the intent profile that the current thread is not serving). The modal shows `continue` big, the others as alternates. Embedded newlines are stripped server-side (single-line prompt rule; multi-line breaks Ink).
|
||||
|
||||
## Predictor
|
||||
|
||||
New `src/readmymind-predictor.ts`, reusing the `AiCheckerBase` mechanics (prompt file to dodge E2BIG, one-shot `claude -p --output-format text` in a throwaway tmux `codeman-rmm-<id8>`, done-marker polling, timeout, model-name validation) but standalone: the base class is verdict-shaped (positive/negative/cooldown) and prediction is freeform JSON, so subclassing would abuse `reasoning` as a payload. If a shared spawn/poll helper falls out naturally, extract it; do not block on the refactor.
|
||||
|
||||
- **Model: opus** (decided). `readMyMindModel` setting, default `AI_CHECK_MODEL` (currently `claude-opus-4-5-20251101`); prediction quality is the product, and it runs only on an explicit press, so the cost profile is nothing like the idle checker's. Timeout 90s (opus headroom over a ~30 KB prompt).
|
||||
- Input: the assembled context above. The predictor itself stays dumb: text in, JSON out; all intelligence about *what to include* lives in the testable assembler.
|
||||
|
||||
## API (new `src/web/routes/readmymind-routes.ts`)
|
||||
|
||||
Normal authed API, `ApiResponse` envelope, Zod schemas in `schemas.ts`, ownership via `findSessionOrFail` (the profile key derives from the session's owner + workingDir, so multi-user scoping is structural):
|
||||
|
||||
- `GET /api/sessions/:id/intent` → the session's `IntentProfile`.
|
||||
- `PUT /api/sessions/:id/intent` body `{ goals }` (bounded) → update goals. Used by the modal's edit view and by the agent skill ("record that the user is working toward X").
|
||||
- `DELETE /api/sessions/:id/intent` → forget everything for this case (the modal's "Forget" affordance).
|
||||
- `POST /api/sessions/:id/readmymind` body `{ steer?, rejected? }` → `{ suggestions }`. 409 `INVALID_STATE` while a prediction is already running for the session; claude-mode sessions only (400 otherwise, mirroring wait-signal gating).
|
||||
|
||||
## Frontend
|
||||
|
||||
New module `readmymind-ui.js` (@loadorder 11.3, after panels-ui.js), prettier-formatted.
|
||||
|
||||
- **Desktop**: header button `btn-readmymind`, default-hidden via marker class `btn-readmymind--hidden` (the `!important` display rules require the marker-class pattern), shown by `applyHeaderVisibilitySettings()` when the setting is ON. Off phones per `test/mobile-header-buttons-policy.test.ts`.
|
||||
- **Phone**: a 🧠 key on the keyboard accessory bar (that bar is where input helpers live, and phones are where typing hurts most). Opens the same modal. Modal z-index respects the ≤768px layer rules (1300+).
|
||||
- **Send** goes server-side: `POST /api/sessions/:id/input` with `\r` appended. Deliberately NOT the browser keystroke path, so the `sendEnterKey` / local-echo-overlay trap never applies (the modal is UI chrome, not terminal typing). **Insert** is the same POST without `\r`.
|
||||
- i18n strings registered (en + zh-CN); suggestion text itself carries `data-i18n-skip`.
|
||||
|
||||
## Skill integration
|
||||
|
||||
The user-facing promise: the button is also a skill. Extend `skills/codeman`:
|
||||
|
||||
- New section "Read My Mind: intent + prediction" with the three intent verbs (read profile, append/replace goals, predict) and the guard notes (single-line prompts, never auto-send to another session without the user asking).
|
||||
- Update `reference/endpoints.md` (the endpoints.md drift test pins this).
|
||||
- The auto-injected case copy heals via the existing marker-owned `applyAgentSkill` mechanism; nothing new needed there.
|
||||
|
||||
Agent use cases this unlocks: a lead session records intentions as the user states them ("remember: shipping 1.16 is the goal"), and a returning user gets a prediction grounded in what the agent knew, not just raw prompt history.
|
||||
|
||||
## Security / privacy
|
||||
|
||||
- **The human gate is the injection mitigation**: pane output (attacker-influenceable) flows into the predictor, so its output is only ever *proposed*, rendered as text (`textContent`), and sent solely by an explicit user click. No auto-send path exists, including for the skill.
|
||||
- Intent data: 0600 file, bounded fields, per-owner keys, endpoints ownership-checked, excluded from search, cleared via DELETE.
|
||||
- Predictor spawns with the user's own credentials exactly like the AI idle/plan checkers; model name shell-validated the same way.
|
||||
- Setting OFF stops capture immediately; existing data stays until DELETE (explicit, not silent).
|
||||
|
||||
## Tests
|
||||
|
||||
- `test/intent-store.test.ts`: key derivation, caps/FIFO, consecutive-dupe skip, tag/tool_result filtering fixtures, 0600 mode, multi-user key separation.
|
||||
- `test/readmymind-context.test.ts`: fixture scenarios pinning the assembled prompt: pending-dialog-first ordering, tail-keeping truncation of the assistant turn, budget drop order (siblings before workspace signals), remote-case git skip, trust-tier framing present, rejected suggestions included only on rethink.
|
||||
- `test/readmymind-predictor.test.ts`: strict JSON parse, garbage output → error result, newline stripping, `kind` validation, rejected-suggestions threading into the prompt.
|
||||
- `test/routes/readmymind-routes.test.ts` (`app.inject`): CRUD round-trip, predict with a stubbed predictor, 409 while in flight, non-claude 400, ownership 404, Send/Insert byte assertions via the test-PTY echo (`\r` present vs absent).
|
||||
- Transcript capture: extend the transcript-watcher fixtures with user-turn entries.
|
||||
|
||||
## Phases
|
||||
|
||||
1. **Intent store + capture + intent endpoints + skill docs.** Immediately useful to agents even before any UI exists.
|
||||
2. **Context assembler + predictor + predict endpoint + desktop button/modal.** The feature as pitched. The assembler ships with all collectors it can serve from day one (transcript, intent, git, run-summary, siblings); the approvals collector activates when PR #245 lands.
|
||||
3. **Phone accessory key, rethink steering, alternates row.** Part 1 (shipped): the alternates row (tappable, swap into the field without losing edits; Rethink rejects the whole shown set), the phone 🧠 keyboard-accessory key (both bar templates, `rmm-enabled` marker class on the bar), and a phone-sized modal (small dialog, not full-screen). Part 2 (shipped): rethink steering, the free-text steer note under the suggestions, sent as `steer`, visible whenever Rethink is live (ready and empty-result phases), cleared on each open; the empty-result copy points at the note, and the footer buttons moved to the styled `btn-toolbar` convention (the bare `btn btn-*` classes they shipped with match no CSS in this codebase and rendered as unstyled UA buttons).
|
||||
4. Explicitly later: proactive predict-on-idle (ghost suggestion chip), auto-compaction of `recentPrompts` into `goals` via a cheap model, codex/gemini capture, cross-case "global" intent.
|
||||
|
||||
## Open questions
|
||||
|
||||
- Should Rethink's rejected-suggestion memory persist across modal closes, or reset each open?
|
||||
- Is a composer-adjacent placement (next to the toolbar Run controls) better than the header for discoverability?
|
||||
- Pending-dialog input (source #1) consumes the approvals-inbox store (PR #245, merged): the phase-2 collector reads pending items directly from `src/approval-inbox.ts`.
|
||||
|
||||
## Docs
|
||||
|
||||
- CLAUDE.md: Key Patterns entry, State Files (`intents.json`), frontend load order, route count.
|
||||
- `docs/api-reference.md`: four endpoints (additive under the 0.9.x contract).
|
||||
- `skills/codeman/reference/endpoints.md`: new rows (drift-test enforced).
|
||||
@@ -0,0 +1,108 @@
|
||||
# Read My Mind
|
||||
|
||||
Codeman's per-case memory of what you are trying to accomplish, and the 🧠 button that turns it into a predicted next prompt. Each case gets an **intent profile**: a freeform `goals` text (written by you or your agent) plus the prompts you actually submitted, captured automatically while the feature is on. Pressing 🧠 feeds that profile and the live session signals to a one-shot model call and shows the predicted prompt for you to send, edit, or rethink. Nothing is ever sent to a session automatically. Design doc: [`readmymind-plan.md`](readmymind-plan.md).
|
||||
|
||||
## What it does
|
||||
|
||||
- Captures the prompts you submit in Claude sessions into a per-case history (50 most recent, bounded).
|
||||
- Lets you (or your agent) record explicit goals per case.
|
||||
- Predicts your next prompt on demand (the 🧠 header button, or `POST .../readmymind` for agents): the suggestion arrives in a modal with Send / Insert / Rethink / Dismiss.
|
||||
- Exposes the profile over the HTTP API, and to agents through the `codeman` skill, so an agent can ground its work in what you actually want instead of guessing from the last screenful.
|
||||
|
||||
## Turning it on
|
||||
|
||||
App Settings → Header & Panels → Cross-session features → **Read My Mind** (synced setting `readMyMindEnabled`, default **OFF**). It gates everything: capture, the header button, and nothing shows anywhere while it is off. The API equivalent:
|
||||
|
||||
```bash
|
||||
curl -sk -X PUT https://localhost:3000/api/settings \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"readMyMindEnabled": true}'
|
||||
```
|
||||
|
||||
Add `-u user:password` if your install has `CODEMAN_PASSWORD` set, and drop `-k`/use `http://` for a plain-HTTP dev server. Turning it OFF stops capture immediately; existing profiles stay until you delete them (below).
|
||||
|
||||
## The 🧠 button
|
||||
|
||||
On a Claude session, press the brain button in the header (desktop) or the 🧠 key on the keyboard accessory bar (phones and tablets; it appears when the setting is on). Codeman assembles everything it already knows: your goals, your recent prompts (with your voice: length, tone, shorthand), the tail of the last assistant reply, recent tool activity, git state (branch, dirty files, pending changesets), how long you have been away and what happened meanwhile, sibling sessions in the same case, and any dialog the session is currently waiting on. A one-shot model call (opus by default, `readMyMindModel` to override) turns that into 1-3 suggestions; the top one lands in an editable field with its rationale, and the others render as tappable alternate rows: tap one to swap it into the field (edits you already made are kept on the row you leave).
|
||||
|
||||
- **Send** submits it to the session (with Enter).
|
||||
- **Insert** drops it on the CLI composer *without* Enter, so you can edit it in the terminal before sending.
|
||||
- **Rethink** re-runs with everything shown (the field and the alternates) recorded as rejected. An optional steer note below the suggestions ("no, I meant the mobile bug") rides along as your own words, the highest-authority signal the predictor gets; it stays in the field across re-runs until you clear it or reopen the modal.
|
||||
- **Dismiss** closes; nothing happens.
|
||||
|
||||
A prediction takes 5-90 seconds and costs real tokens; one runs per session at a time. If the session is sitting on a permission/question dialog, the suggestion is usually an answer to that dialog: that is intentional.
|
||||
|
||||
**Security note**: the prediction reads observable content (assistant output, tool logs, git output) which a hostile repo could try to steer. The predictor is told user-stated intent outranks anything observed, and, more importantly, a suggestion is only ever *proposed*: your click is the boundary. No auto-send path exists, including for agents.
|
||||
|
||||
## What gets captured, exactly
|
||||
|
||||
Capture reads the Claude session transcript, not your keystrokes: when a user turn lands in the transcript, its text is folded into the case's profile. Filters applied on the way in:
|
||||
|
||||
- **Claude-mode sessions only.** Shell, OpenCode, Codex, Gemini, Antigravity, and Pi sessions are never captured (they have no transcript watcher).
|
||||
- Tool results, local slash-command echo (`/model` and friends), system wrappers, and interrupt markers are skipped.
|
||||
- Entries shorter than 3 characters are skipped (menu digits, Esc artifacts).
|
||||
- Consecutive duplicates collapse (auto-resume's "continue" spam counts once per run).
|
||||
- Each prompt is stored as one line, truncated to 500 characters; the history caps at 50 prompts FIFO.
|
||||
|
||||
Because the transcript path arrives via Claude Code hooks, capture needs hooks to reach the server, the same condition as hook-based idle detection. Docker cases against a loopback-only server need `CODEMAN_DOCKER_BRIDGE_HOOKS=1`; remote-SSH cases do not capture.
|
||||
|
||||
## What is never captured
|
||||
|
||||
- Anything while `readMyMindEnabled` is OFF (capture is not retroactive).
|
||||
- Terminal output, keystrokes, passwords typed into shells: only submitted Claude prompts are read.
|
||||
- Nothing leaves the machine beyond the model call you explicitly trigger, and profiles are never fed into `/api/search`.
|
||||
|
||||
## Where it lives, and how to wipe it
|
||||
|
||||
Profiles live in `~/.codeman/intents.json`, written atomically at mode 0600 (captured prompts can contain secrets). The file is per Codeman instance. Keys derive from owner + the case's resolved working directory, so profiles survive `/clear`, respawn cycles, and session churn, and in multi-user mode two owners of the same directory get separate profiles.
|
||||
|
||||
Forget one case: `DELETE /api/sessions/:id/intent` (below). Forget everything: stop the server and delete `~/.codeman/intents.json`.
|
||||
|
||||
## The API
|
||||
|
||||
Four endpoints, session-scoped so ownership is enforced by the session itself (`/api/v1/` aliases work too; full spec in [`api-reference.md`](api-reference.md)):
|
||||
|
||||
```bash
|
||||
# Read the profile for a session's case
|
||||
curl -sk https://localhost:3000/api/sessions/$SID/intent | jq '.data.intent'
|
||||
|
||||
# Record goals (REPLACES the text: read + merge if you want to append)
|
||||
curl -sk -X PUT https://localhost:3000/api/sessions/$SID/intent \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"goals":"ship 1.17; then mobile polish"}'
|
||||
|
||||
# Forget the case
|
||||
curl -sk -X DELETE https://localhost:3000/api/sessions/$SID/intent
|
||||
|
||||
# Predict the next prompt (claude-mode only; takes 5-90 s)
|
||||
curl -sk -X POST https://localhost:3000/api/sessions/$SID/readmymind \
|
||||
-H 'Content-Type: application/json' -d '{}' | jq '.data.suggestions'
|
||||
```
|
||||
|
||||
A case with nothing recorded answers an empty profile with `updatedAt: 0`; reads never persist anything. Goals cap at 8192 characters and the schema is strict, so unknown fields or over-long goals answer `400 INVALID_INPUT`. A session you do not own answers `404 NOT_FOUND`, indistinguishable from a nonexistent one. Predict answers `{ suggestions: [{ prompt, why, kind }], durationMs }` (`kind`: `continue` / `verify` / `redirect`), `409 CONFLICT` while one is already running, `400 INVALID_INPUT` on non-claude sessions, and `502 OPERATION_FAILED` when the model produced no usable JSON. The rethink flow passes `{"steer":"…","rejected":["…"]}`.
|
||||
|
||||
## For agents (the skill)
|
||||
|
||||
The `codeman` agent skill documents the same verbs (SKILL.md §3 plus `reference/endpoints.md`), with the ground rules: read the profile to understand what the user wants, record goals the user actually stated, merge instead of blind-writing (PUT replaces), never delete a profile unprompted, and never send a predicted suggestion into a session unless the user asked. It is the user's memory, not the agent's.
|
||||
|
||||
## What comes next
|
||||
|
||||
Explicitly later: proactive predict-on-idle, auto-compaction of the prompt history into goals, non-Claude capture. See the phases section of [`readmymind-plan.md`](readmymind-plan.md).
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
| Symptom | Cause / fix |
|
||||
| ------- | ----------- |
|
||||
| No 🧠 button in the header | `readMyMindEnabled` is OFF (App Settings → Header & Panels → Cross-session features), you are on a phone (there it is a key on the keyboard accessory bar instead, visible while typing), or the active session is not claude-mode |
|
||||
| Prediction feels generic | The profile is thin: record goals (PUT or ask your agent to), and let capture accumulate a few real prompts first |
|
||||
| "A prediction is already running" (409) | One per session at a time; wait for the current one (up to 90 s) |
|
||||
| Prediction fails (502) | The model returned no usable JSON, or the CLI could not start; retry. Check `readMyMindModel` if you overrode it |
|
||||
| Profile stays empty although I am prompting | `readMyMindEnabled` was OFF at the time (capture is not retroactive), the session is not claude-mode, or hooks are not reaching the server (Docker case on a loopback bind without `CODEMAN_DOCKER_BRIDGE_HOOKS=1`, or a remote-SSH case) |
|
||||
| Short answers I typed are missing | Entries under 3 characters are filtered by design (menu digits, Esc artifacts) |
|
||||
| My goals text vanished after an agent wrote to it | PUT replaces the whole text; the skill tells agents to read + merge, but a blind write wins. Re-state the goals; consider phrasing them in the session so capture keeps the evidence |
|
||||
| Two profiles for what I think is one case | Different owners in multi-user mode, or genuinely different directories; paths are realpath-resolved, so symlink spellings converge but distinct checkouts do not |
|
||||
| `400 INVALID_INPUT` on PUT | Goals over 8192 chars, or an extra field in the body (strict schema) |
|
||||
|
||||
## Where the code lives
|
||||
|
||||
`src/intent-store.ts` (store + pure helpers, singleton), the `transcript:user_prompt` event in `src/transcript-watcher.ts`, capture wiring in `src/web/server.ts` (`captureIntentPrompt`), context assembly in `src/readmymind-context.ts` (pure) + `src/readmymind-collectors.ts` (transcript tail + git IO), the predictor in `src/readmymind-predictor.ts`, routes in `src/web/routes/readmymind-routes.ts`, schemas in `src/web/schemas.ts`, frontend in `src/web/public/readmymind-ui.js`. Tests: `test/intent-store.test.ts`, `test/readmymind-context.test.ts`, `test/readmymind-collectors.test.ts`, `test/readmymind-predictor.test.ts`, `test/routes/readmymind-routes.test.ts`, and the capture cases in `test/transcript-watcher.test.ts`.
|
||||
@@ -1,7 +1,7 @@
|
||||
# Remote Sessions (SSH)
|
||||
|
||||
Codeman can run a session's agent on a **remote host over SSH** instead of the
|
||||
local machine. The agent (Claude, OpenCode, Codex, Antigravity, Gemini, or a plain shell)
|
||||
local machine. The agent (Claude, OpenCode, Codex, Antigravity, Gemini, Pi, or a plain shell)
|
||||
runs inside a `tmux` server **on the remote host**, so it survives the SSH
|
||||
connection dropping; Codeman attaches to it the same way it attaches to a local
|
||||
managed session.
|
||||
@@ -30,7 +30,7 @@ Types live in `src/types/session.ts`; persistence in `src/remote-hosts.ts`.
|
||||
| `RemoteHost` (extends `RemoteSshOptions`) | A saved host: `id`, `label`, `host`, `username`, `port?`, `commands?` (per-mode launch command override). |
|
||||
| `RemoteCase` | A working directory on a host: `name`, `type: 'remote'`, `hostId`, `remotePath`. |
|
||||
| `SessionRemote` (extends `RemoteSshOptions`) | The resolved bundle stamped onto a live session: host coordinates + `remotePath` + `commands`, plus **`owned?`** and **`remoteSessionName?`** (COD-105 — see [Ownership](#ownership-launched-vs-discovered-and-attached-cod-105)). Built by `toSessionRemote(host, case)` (sets `owned: true`) for the launch path, or `toAttachedSessionRemote(host, name, path)` (sets `owned: false`) for the attach path. Both copy the advanced SSH options through so every connection is identical. |
|
||||
| `RemoteCommandMode` | `Extract<SessionMode, 'shell' \| 'claude' \| 'opencode' \| 'codex' \| 'gemini' \| 'antigravity'>` — the modes that can run remotely. |
|
||||
| `RemoteCommandMode` | `Extract<SessionMode, 'shell' \| 'claude' \| 'opencode' \| 'codex' \| 'gemini' \| 'antigravity' \| 'pi'>` — the modes that can run remotely. |
|
||||
| `RemoteSessionInfo` (COD-105) | One discovered remote tmux session: `name` (always `codeman-*`), `attached` (a client is connected), `created` (epoch s), `windows`. Returned by `listRemoteCodemanSessions()`. |
|
||||
|
||||
Persistence is two flat JSON arrays in the instance data dir:
|
||||
|
||||
@@ -163,6 +163,57 @@ both self-reporting, so the retest ask is now "open the console and paste the `[
|
||||
- iPhone: Claude or shell session, and whether a full tab kill changes anything.
|
||||
- Browser console: `app.terminalUi?.terminal?.modes?.mouseTrackingMode` (false-path 4).
|
||||
|
||||
## ROUND 3 (2026-08-09): Codex wheel dead — CONFIRMED AND FIXED
|
||||
|
||||
DodgyBadger (Codex latest, Chrome, Windows 11): mouse wheel does nothing in a CODEX session
|
||||
while working fine in shell and web tabs; DRAGGING THE SCROLLBAR WORKS, so xterm's local
|
||||
buffer demonstrably has content for their codex pane. Analysis against the shipped code:
|
||||
|
||||
- `_shouldForwardWheelToApp` returns true UNCONDITIONALLY for `codex` (no version gate, unlike
|
||||
claude's `>= 2.1.187`), so every plain wheel tick is sent as SGR reports to Codex.
|
||||
- The "verified to scroll its transcript on SGR wheel reports" claim for codex predates
|
||||
current Codex builds; if Codex latest ignores SGR wheel, forwarding eats the gesture while
|
||||
the healthy local scrollback (proven by the working scrollbar) sits unused.
|
||||
- The #227 PageUp fallback cannot rescue this: it is gated to `claude` mode AND `baseY === 0`,
|
||||
and codex here has real local scrollback. The `[scroll]` diagnostic will still say
|
||||
`forward-sgr (mode=codex, ...)`, confirming the branch, worth asking the reporter to paste.
|
||||
|
||||
**CONFIRMED by the reporter's `[scroll]` line (2026-08-09, PR #227 comment)**:
|
||||
`forward-sgr (mode=codex, cliVersion=unknown, localScrollbackOptOut=false, mouseTracking=none,
|
||||
localScrollbackRows=967)`. Forwarding branch active, 967 rows of healthy local scrollback
|
||||
unused, Codex ignoring the SGR reports. Environment: Codex latest, Chrome, Windows 11.
|
||||
|
||||
**Measured against codex-cli 0.147.0** (isolated `tmux -L codexwheel`, fake `CODEX_HOME/auth.json`,
|
||||
history built with 401ing prompts), which settles it without needing a version gate at all:
|
||||
|
||||
| Probe | Result |
|
||||
| ---------------------------------------------- | ----------------------------------------------- |
|
||||
| `#{mouse_any_flag}` once the TUI is up | `0`: codex never enables mouse tracking |
|
||||
| `#{alternate_on}` | `0`: inline viewport, not an alt-screen pager |
|
||||
| `#{history_size}` while prompting | grows 3 → 32: the transcript goes to scrollback |
|
||||
| 6 × `\x1b[<64;10;10M` written to the pane | pane capture byte-identical, nothing happens |
|
||||
| control: literal `zz` | pane changes, so the probe can see changes |
|
||||
| `\x1b[<0;12;5M` + release (the click-tap path) | no change either: taps are no-ops, not garbage |
|
||||
|
||||
Codex has no in-app pager to drive: its history lives in the terminal's own scrollback, which is
|
||||
exactly what forwarding was stealing the gesture from. A version gate would be the wrong fix (and
|
||||
`cliVersion=unknown` means there is no codex probe to gate on anyway).
|
||||
|
||||
**Fix (shipped):** `_shouldForwardWheelToApp` now returns true for `claude >= 2.1.187` and nothing
|
||||
else. Codex falls to the normal local-scrollback path like shell/gemini/opencode, so wheel and touch
|
||||
scroll the same history the scrollbar drag was already scrolling. The claude-only PageUp fallback is
|
||||
untouched: codex never needs it, its local buffer is real. Taps stay hand-encoded for codex
|
||||
(`_sessionUsesServerMouseStrip`), measured harmless, so click-to-position is merely unavailable
|
||||
there rather than damaging. Lesson for the next mode added to the forward list: "it is a strip mode"
|
||||
proves nothing, write a real SGR report into a live pane and diff the capture first.
|
||||
|
||||
Verified end-to-end in Chromium against a live codex session on an isolated instance
|
||||
(`CODEMAN_INSTANCE=codexwheel`, port 5055, `envOverrides.CODEX_HOME` pointing at the fake auth
|
||||
dir): trusted `page.mouse.wheel` up now logs
|
||||
`[scroll] … → local-scrollback (mode=codex, …, localScrollbackRows=43)`, moves the viewport
|
||||
39 → 4 (back to the Codex banner), and sends ZERO bytes to the PTY. Unit coverage:
|
||||
`test/terminal-touch-tap.test.ts` ("only claude forwards — codex and gemini keep the local wheel").
|
||||
|
||||
Original plan follows.
|
||||
|
||||
## Reports
|
||||
|
||||
@@ -312,7 +312,7 @@ TOCTOU window.
|
||||
| Route | Cap | Notes |
|
||||
|-------|-----|-------|
|
||||
| `file-content` | 10 MB | text preview |
|
||||
| `file-raw` | 50 MB | inline MIME map; **`X-Content-Type-Options: nosniff` on all responses** |
|
||||
| `file-raw` | 50 MB | inline MIME map; **`X-Content-Type-Options: nosniff` on all responses**; streamed, `Range`-aware (206 slices come from the same validated path, and the cap is checked before the range) |
|
||||
| `POST /api/download` | 50 MB | forced `attachment`; sensitive‑path blocklist |
|
||||
|
||||
### SVG / content‑type XSS
|
||||
@@ -489,7 +489,7 @@ production layout (`~/.codeman`, `-L codeman`, port 3000).
|
||||
Docker cases (1.4.0) run a session inside a per‑case container instead of on the host. The security posture:
|
||||
|
||||
- **Hardened create flags, always** — `--cap-drop ALL`, `--security-opt no-new-privileges`, `--pids-limit` (fork‑bomb guard), `--memory` == `--memory-swap` (a real OOM cap), `--init`, and non‑root: `--user <hostUid>:0` on Linux (host uid → workspace files stay host‑owned; GID 0 keeps `$HOME` writable), `--userns=keep-id` on rootless Podman. **Never** `--privileged`, and **never** the docker socket — the pure builder in `docker-hosts.ts` cannot emit them and the schema cannot represent them.
|
||||
- **Credentials never enter an image** — the convenient default bind‑mounts host cred dirs (`~/.claude`, `~/.codex`, `~/.gemini` — which also carries Antigravity's `antigravity-cli/` state — and `~/.config/{gcloud,opencode}`) read‑write. Bind mounts are physically excluded from `docker commit`, so exported images are secret‑free. API‑key CLIs get their key as an exec‑time NAME‑ONLY `--env OPENAI_API_KEY` (no `=value`, no `ps` leak, never committed); a create‑time `-e` for a secret is never used. The **sealed** profile (`mountCredentials:false` + `network:none`) drops the host mounts; full‑image export is then refused (an in‑container login would ride the committed layer) unless a pre‑commit scrub is opted into.
|
||||
- **Credentials never enter an image** — the convenient default bind‑mounts host cred dirs (`~/.claude`, `~/.codex`, `~/.gemini` — which also carries Antigravity's `antigravity-cli/` state — `~/.config/{gcloud,opencode}`, and five seeded files from `~/.pi/agent`) read‑write. Bind mounts are physically excluded from `docker commit`, so exported images are secret‑free. API‑key CLIs get their key as an exec‑time NAME‑ONLY `--env OPENAI_API_KEY` (no `=value`, no `ps` leak, never committed); a create‑time `-e` for a secret is never used. The **sealed** profile (`mountCredentials:false` + `network:none`) drops the host mounts; full‑image export is then refused (an in‑container login would ride the committed layer) unless a pre‑commit scrub is opted into.
|
||||
- **Blast radius — accept it explicitly** — the convenient profile mounts an arbitrary host workspace RW plus the host credential dirs RW into a network‑enabled container, so container‑run agent code can read/modify those host trees and reach the network at once. Still a net improvement over today's on‑host `--dangerously-skip-permissions` execution; use the sealed profile for genuinely untrusted work.
|
||||
- **Import is untrusted‑bundle‑safe** — `/api/docker-cases/import` validates the manifest + per‑member SHA‑256 before extraction, rejects absolute / `..` tar members (traversal guard), and re‑tags the loaded image into a quarantined namespace so it can never overwrite `codeman/agent:base` or a pre‑existing tag.
|
||||
- **Host guard & the bridge‑hooks listener** — in‑container hook callbacks carry `Host: host.docker.internal` / `host.containers.internal`; both are on the always‑on host‑header allowlist (`DOCKER_HOST_GATEWAY_ALIASES`) and resolve to the host only from inside a container netns, so they are not a browser DNS‑rebinding surface. On a loopback‑only server, in‑container hooks are opt‑in via `CODEMAN_DOCKER_BRIDGE_HOOKS=1`, which binds a SECOND listener on the docker bridge gateway serving **only** the hook endpoints (every other path → `403`) into the same hook‑secret‑gated pipeline. The bridge is host‑internal (containers + host), not the LAN, so it does not widen network exposure; the hook secret is bind‑mounted read‑only and referenced by path.
|
||||
|
||||
@@ -0,0 +1,246 @@
|
||||
# Session lineage lines (spawn lines between tabs)
|
||||
|
||||
**Goal:** when a session spawns another session (the `codeman` agent skill starting a
|
||||
worker, or anything else that says who it is), draw the same kind of glowing connection
|
||||
line the subagent windows already use, but **tab → tab**, so a glance at the strip shows
|
||||
which tab spawned which.
|
||||
|
||||
Status: PLAN. Nothing implemented yet.
|
||||
|
||||
---
|
||||
|
||||
## 1. The blocking fact: no parent relationship exists today
|
||||
|
||||
There is no spawn-parent link between sessions anywhere in the codebase:
|
||||
|
||||
- `SessionState` (`src/types/session.ts:388`) has no `parentSessionId` / `spawnedBy` /
|
||||
`createdBy`.
|
||||
- `POST /api/quick-start` and `POST /api/sessions` record only `owner = ownerFor(req)`,
|
||||
which is the multi-user **human**, not the calling session.
|
||||
- The only parent links that do exist are `TeamConfig.leadSessionId` (agent teams) and
|
||||
`subagent-parents.json` (a frontend **window-layout** store for subagent windows).
|
||||
Neither says "session A spawned session B".
|
||||
- Nothing in the HTTP request identifies the caller: an agent's spawn call is plain
|
||||
`curl` from inside a tmux pane, so there is no socket-level identity to recover
|
||||
(`SO_PEERCRED` needs a unix socket; the API is TCP).
|
||||
|
||||
So the caller has to **tell** us. It already knows its own id: every managed pane gets
|
||||
`CODEMAN_SESSION_ID` exported by `session-cli-builder.ts` (and the skill's §0 preamble
|
||||
already binds it to `$SELF`).
|
||||
|
||||
## 2. Wire format
|
||||
|
||||
Two ways in, because they serve different callers. Body wins when both are present.
|
||||
|
||||
| Where | Shape | Who uses it |
|
||||
| --- | --- | --- |
|
||||
| body field | `"parentSessionId": "<uuid>"` | anything hand-writing one create call |
|
||||
| request header | `X-Codeman-Parent-Session: <uuid>` | the skill: added **once** to the `CURL` array in the §0 preamble, so every present and future create call carries it with no per-recipe edit |
|
||||
|
||||
Rules, all of them deliberate:
|
||||
|
||||
- **Advisory decoration only.** It never grants access, never scopes anything, never
|
||||
affects lifecycle. A child is not killed when its parent dies; the line just stops
|
||||
being drawn once the parent tab is gone.
|
||||
- **Never fails a spawn.** An unknown / stale / foreign parent id is silently dropped
|
||||
(field ends up `undefined`), not a `400`. A cosmetic field must not be able to break
|
||||
worker creation.
|
||||
- **Resolved, not trusted.** The id must match a live session the caller can already
|
||||
see (`canAccessOwned`), and the resolved parent's `owner` must equal the new
|
||||
session's `owner`. Otherwise a user could staple their session under another user's
|
||||
tab in multi-user mode.
|
||||
- Exact id match first; a `>= 8`-char **unique** prefix match as a fallback (ids appear
|
||||
truncated in mux names and UI surfaces; ambiguous prefixes resolve to nothing).
|
||||
|
||||
## 3. Server changes
|
||||
|
||||
| File | Change |
|
||||
| --- | --- |
|
||||
| `src/types/session.ts` | `SessionState.parentSessionId?: string` with a doc comment saying it is UI decoration and never a permission signal |
|
||||
| `src/session.ts` | constructor option `parentSessionId` → `_parentSessionId`, public getter, emitted from `toState()` (~line 1170) |
|
||||
| `src/web/schemas.ts` | `parentSessionId: z.string().max(100).optional()` on `CreateSessionSchema` (272) and `QuickStartSchema` (680). Neither is `.strict()`, so this is additive |
|
||||
| `src/web/route-helpers.ts` | new `resolveParentSessionId(ctx, req, bodyValue, owner)` implementing §2's rules; returns `string \| undefined`, never throws |
|
||||
| `src/web/routes/session-routes.ts` | pass it into the three `new Session({...})` sites: `POST /api/sessions` (846), `POST /api/run` (2522), `POST /api/quick-start` (2896) |
|
||||
| `src/web/server.ts` | recovery path (~2617): `parentSessionId: savedState?.parentSessionId` so the link survives a restart |
|
||||
|
||||
**No new SSE event.** `session_created` / `session_updated` broadcast
|
||||
`getSessionStateWithRespawn(session)`, which is `toState()`-derived, so the field rides
|
||||
along to the browser for free — and the frontend already does
|
||||
`this.sessions.set(data.id, data)`, so `session.parentSessionId` is simply there.
|
||||
|
||||
Optional follow-up: surface it on `/api/sessions/unified` rows so the Session Manager
|
||||
and the home rails can show "spawned by w3-claudeman".
|
||||
|
||||
## 4. Frontend rendering
|
||||
|
||||
### 4.1 Where the code goes
|
||||
|
||||
`_updateConnectionLinesImmediate()` (`subagent-windows.js:242`) is a strict
|
||||
**batched read → batched write** pass, and it already has an extension point:
|
||||
ultracode appends its own layer via `_appendUltracodeConnectionLines(svg, rects)` at
|
||||
the end, sharing the `rects` cache so no layer forces a second reflow.
|
||||
|
||||
Lineage lines follow that exactly: a new module `src/web/public/session-lineage.js`
|
||||
(load order 15.6, after `ultracode-windows.js`) exporting
|
||||
`_appendLineageConnectionLines(svg, rects)` onto `CodemanApp.prototype`, called from the
|
||||
same tail. **The core function keeps ownership of the read/write split**; the new layer
|
||||
only reads through the shared `rects` map and only appends paths.
|
||||
|
||||
The path math itself lives in `constants.js` as a pure
|
||||
`computeLineagePath(parentRect, childRect, stripRect, depth)` — same treatment as
|
||||
`computeTabScrollLeft`, so the geometry is unit-testable without a browser.
|
||||
|
||||
### 4.2 Geometry
|
||||
|
||||
Both endpoints are tabs in one horizontal strip, so the subagent shape (tab-bottom →
|
||||
window-top) does not apply. **One case**, a **U-bridge hanging below the strip** that
|
||||
touches both tabs on their bottom edge:
|
||||
|
||||
```
|
||||
y0 = max(parent.bottom, child.bottom)
|
||||
d = clamp(14 + |x2 - x1| * 0.085, 22, 104) + depth * 8 + |child.bottom - parent.bottom|
|
||||
path: M x1 parent.bottom C x1 y0+d, x2 y0+d, x2 child.bottom
|
||||
```
|
||||
|
||||
`depth` is the child's index among its siblings, so several children of one parent
|
||||
**nest** instead of overprinting.
|
||||
|
||||
> **Superseded (2026-08-14): the two shapes this section used to specify.** The dip was
|
||||
> `clamp(14 + span * 0.06, 16, 44) + depth * 6`, and a wrapped strip
|
||||
> (`tabs-two-rows` / `tabs-auto-wrap`) got its own parent-bottom → child-**top** bezier.
|
||||
> Both were tuned against two tabs side by side and failed at the distances the feature
|
||||
> is used at:
|
||||
>
|
||||
> - a skill worker is appended to the **end** of the strip, so the real span is
|
||||
> 800-1500px, where a 44px cap is a 33px sag, i.e. a line that reads as straight and
|
||||
> crosses the terminal instead of bracketing under the strip;
|
||||
> - and when the strip wraps, parent-bottom (34) to child-top (48) leaves **14px** to
|
||||
> bend in, so the arc was a flat line hidden in the row gap, with siblings drawn on
|
||||
> top of each other. Reported as *"they connect already, but the lines are straight
|
||||
> and not easy visible"*.
|
||||
>
|
||||
> Anchoring both ends at the tab bottoms and hanging the control points below the
|
||||
> **lower** row gives the wrapped case the same bracket as the flat one, and removes the
|
||||
> branch. Pinned by `test/session-lineage-lines.test.ts`.
|
||||
|
||||
A small `<circle r="3.5">` at the child end marks direction (it breathes to 4.5 while that worker is busy) (an SVG `marker` would need a
|
||||
`<defs>` block and fights `stroke-dasharray`).
|
||||
|
||||
Each path gets `class="connection-line lineage-line"`, `data-parent-tab`,
|
||||
`data-child-tab`, and `data-agent-id="lineage:<childId>"` — that last one is what makes
|
||||
the existing entrance machinery (`markConnectionLineEntering` / `_applyLineEntrances`,
|
||||
keyed on `data-agent-id`) work on these lines with **zero** new animation code,
|
||||
including the negative-`animation-delay` resume across the `svg.innerHTML = ''` rebuild.
|
||||
|
||||
### 4.3 Clipping
|
||||
|
||||
`.session-tabs` is `overflow-x: auto`, so a tab scrolled out of the strip still has a
|
||||
rect — one that lies outside the strip box and would draw an arc across the logo or the
|
||||
header buttons. **Skip any edge whose parent or child center falls outside
|
||||
`stripRect` (4px tolerance).** Skipping is honest; clamping would draw a line to a tab
|
||||
that is not there.
|
||||
|
||||
### 4.4 Redraw triggers
|
||||
|
||||
`updateConnectionLines()` already coalesces through `scheduleBackground`, so extra
|
||||
callers are cheap. Needed:
|
||||
|
||||
- `_fullRenderSessionTabs()` — already calls it (app.js:3912). Free.
|
||||
- `_renderSessionTabsImmediate()` — does **not**. A badge appearing widens a tab and
|
||||
moves every tab after it, which slides the arcs off their anchors. Add the call,
|
||||
guarded on `this._lineageEdgeCount > 0` so nobody pays for it without the feature.
|
||||
- **strip `scroll`** (passive listener on `#sessionTabs`) — the arcs must track the
|
||||
scroller. This is new; no existing line layer needed it.
|
||||
- window `resize` — piggyback the throttled handler in `terminal-ui.js:930`.
|
||||
- `_onSessionCreated` — `markConnectionLineEntering('lineage:' + data.id)` so a new
|
||||
child draws in **if** the user has a line-entrance theme on (all entrance styles are
|
||||
`legacy`/off by default, so this is a no-op for an untouched install).
|
||||
|
||||
### 4.5 Styling
|
||||
|
||||
`.connection-line.lineage-line`: blue stroke from the per-skin `--session-blue` token
|
||||
(violet until 2026-08-14, changed because it lost contrast against the terminal's own
|
||||
dim foreground the moment the arc crossed text),
|
||||
`stroke-width: 2.5`, `dasharray 5 5`, `opacity: .72` (`.95` while the child works),
|
||||
softer than the subagent lines so the two layers still read as different things now that
|
||||
hue no longer separates them (shape does most of that work: a lineage arc hangs under the
|
||||
strip and never reaches a window), but the contrast against the terminal comes from a
|
||||
**second, wider glow** rather than more weight, because the first
|
||||
cut (2px / `4 4` / `.55` / one 5px glow) disappeared into terminal text on a real 1080p
|
||||
desktop. `lineage-flow` marches by two dash cycles, so it moves with the dash array
|
||||
(`5 5` → `-20`). Trap to respect: the skin block nests under
|
||||
`html:not([data-skin="og"])`, so a bare `.lineage-line` rule inside it would outrank the
|
||||
base rule at higher specificity. **Define the color as a token per skin, keep exactly
|
||||
one `.lineage-line` rule.** Light skins get a darker stroke.
|
||||
|
||||
Optional signal worth having: `.lineage-line--working` (a slow `stroke-dashoffset`
|
||||
march) only while the **child** session is working, wrapped in
|
||||
`prefers-reduced-motion: no-preference`. Static otherwise — a permanently marching line
|
||||
per tab pair is noise and battery.
|
||||
|
||||
### 4.6 Desktop only, and why
|
||||
|
||||
The SVG overlay is `z-index: 999`. On desktop the header is `z-index: 100`, so arcs
|
||||
paint **over** the header and can touch tab bottoms. Under 1024px `mobile.css` makes the
|
||||
header `position: fixed; z-index: 1200`, which would **bury** the arcs — and the phone
|
||||
strip is a scroller where both endpoints are rarely on screen together anyway. So the
|
||||
layer returns early unless `MobileDetection.getDeviceType() === 'desktop'`.
|
||||
|
||||
Raising the SVG to ~1250 (above the fixed header, below modals at 1300) is a possible
|
||||
phase 2, but it needs a real check against the mobile overview and the drawer.
|
||||
|
||||
### 4.7 Setting
|
||||
|
||||
`sessionLineageLines`, **per-device** — so it goes in the `displayKeys` set in
|
||||
`settings-ui.js` and must **not** be added to `SettingsUpdateSchema` (`.strict()`;
|
||||
sending an undeclared key fails the whole PUT). Rendered as a switch in
|
||||
App Settings → Appearance, beside the entrance-animation pickers.
|
||||
|
||||
**Default: ON for desktop** (phones never render it). This is the one deliberate
|
||||
departure from the "new visual surfaces ship OFF" convention — the feature is the
|
||||
request, and a user with 12 unrelated tabs has a one-click off switch. Flag for the
|
||||
owner if the convention should win instead.
|
||||
|
||||
## 5. Optional extras (call them separately, none are required)
|
||||
|
||||
1. **Order children after their parent** in `sessionOrder` on create, so arcs stay short
|
||||
and the strip reads as a tree. Real cost: it renumbers the Alt+N badges and moves
|
||||
tabs under the user's cursor, so it should be its own toggle, default OFF.
|
||||
2. **Lineage hover focus**: hovering a tab dims unrelated arcs and brightens its own
|
||||
subtree.
|
||||
3. **"Spawned by" in the Session Manager / home rails**, once `parentSessionId` is on
|
||||
the unified rows.
|
||||
4. **Inherited tab tint**: children pick up a faded version of the parent's tab color.
|
||||
|
||||
## 6. Tests
|
||||
|
||||
- `test/session-lineage.test.ts` (route-level, `app.inject`): round-trips through
|
||||
`POST /api/sessions` + `POST /api/quick-start`, header path, body-wins-over-header,
|
||||
unknown id dropped without failing the spawn, cross-owner parent dropped in
|
||||
multi-user, field present in `GET /api/sessions` and persisted state.
|
||||
- `test/session-lineage-lines.test.ts` (jsdom, pure): `computeLineagePath` — same-row U,
|
||||
wrapped-row bezier, sibling nesting depth, off-strip skip, degenerate zero-width rects.
|
||||
- Browser check (not in `test:ci`): spawn two workers with a parent, assert two
|
||||
`path.lineage-line` elements anchored to the right tabs, then scroll the strip and
|
||||
assert they moved.
|
||||
- Existing guards that must stay green: `test/mobile-header-buttons-policy.test.ts`
|
||||
(nothing new on phones), `test/app-settings-structure.test.ts` (the new switch pairs
|
||||
with its rail section).
|
||||
|
||||
## 7. Skill side (owned by the release session, not this plan)
|
||||
|
||||
One line in the `codeman` skill's §0 preamble covers every spawn recipe:
|
||||
|
||||
```bash
|
||||
CURL=(curl -sk "${AUTH[@]}" -H "X-Codeman-Parent-Session: $SELF")
|
||||
```
|
||||
|
||||
plus a `CODEMAN_PREAMBLE` version bump so stale cached preambles fail loudly instead of
|
||||
silently spawning unparented workers. Recipes that build a create payload by hand can
|
||||
alternatively send `"parentSessionId":"'"$SELF"'"`.
|
||||
|
||||
## 8. Docs to update when it lands
|
||||
|
||||
`CLAUDE.md` (a Key Patterns bullet), `docs/architecture-invariants.md` (new anchor: the
|
||||
resolve-don't-trust rule, the desktop-only z-index reason, the shared `rects` pass),
|
||||
`docs/api-reference.md` (the new field + header on the create endpoints).
|
||||
@@ -43,38 +43,22 @@ const syncData = DEC_SYNC_START + data + DEC_SYNC_END;
|
||||
this.broadcast('session:terminal', { id: sessionId, data: syncData });
|
||||
```
|
||||
|
||||
## Client-Side Implementation (`app.js`)
|
||||
## Client-Side Implementation (`terminal-ui.js`)
|
||||
|
||||
### `batchTerminalWrite(data)`
|
||||
|
||||
1. Checks if flicker filter is enabled (optional, per-session)
|
||||
2. If flicker filter active: buffers screen-clear patterns (`ESC[2J`, `ESC[H ESC[J`, `ESC[nA`)
|
||||
3. Accumulates data in `pendingWrites`
|
||||
4. Schedules `requestAnimationFrame` if not already scheduled
|
||||
5. On rAF callback: checks for incomplete sync blocks (start without end)
|
||||
6. If incomplete: waits up to 50ms via `syncWaitTimeout`
|
||||
7. Calls `flushPendingWrites()` when complete
|
||||
|
||||
### `extractSyncSegments(data)`
|
||||
|
||||
- Parses DEC 2026 markers, returns array of content segments
|
||||
- Content before sync blocks returned as-is
|
||||
- Content inside sync blocks returned without markers
|
||||
- Incomplete blocks (start without end) returned with marker for next chunk
|
||||
4. Calls `_scheduleTerminalWriteFlush()` if no flush is pending
|
||||
5. The yielded callback clears its scheduled flag before calling `flushPendingWrites()`
|
||||
6. Large batches schedule their own next chunk until the queue is empty
|
||||
|
||||
### `flushPendingWrites()`
|
||||
|
||||
```javascript
|
||||
const segments = extractSyncSegments(this.pendingWrites);
|
||||
this.pendingWrites = ''; // Clear before writing
|
||||
for (const segment of segments) {
|
||||
if (segment && !segment.startsWith(DEC_SYNC_START)) {
|
||||
terminal.write(segment); // Skip incomplete blocks (start with marker)
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Note: Segments starting with `DEC_SYNC_START` are incomplete blocks awaiting more data. These are skipped (discarded if timeout forces flush).
|
||||
- Joins the queued terminal data and passes DEC 2026 markers through to xterm.js 6, which handles synchronized output natively.
|
||||
- Writes at most 32KB per yield for Codex and 64KB for other modes.
|
||||
- Requeues the remainder and immediately schedules another safe yield. A final large response therefore drains without waiting for another SSE event.
|
||||
|
||||
### `chunkedTerminalWrite(buffer, chunkSize=128KB)`
|
||||
|
||||
@@ -116,17 +100,15 @@ When detected, buffers 50ms of subsequent output before flushing atomically.
|
||||
|
||||
## Edge Cases
|
||||
|
||||
- **Incomplete sync blocks**: 50ms timeout forces flush (content discarded to prevent freeze)
|
||||
- **Incomplete sync blocks**: xterm.js retains synchronized output until its closing marker
|
||||
- **Large buffers**: Chunked writing prevents UI freeze
|
||||
- **Server shutdown**: Skips batching via `_isStopping` flag
|
||||
- **Session switch**: Clears flicker filter state, pending writes, and sync timeout (prevents cross-session data bleed)
|
||||
- **SSE reconnect**: `handleInit()` clears all pending write state
|
||||
|
||||
**Trade-off:** If a sync block is split across SSE packets and the end marker doesn't arrive within 50ms, the incomplete content is discarded. This prioritizes responsiveness over completeness. In practice this is rare since the server always sends complete `SYNC_START...SYNC_END` pairs and SSE typically delivers them atomically.
|
||||
|
||||
## DEC Mode 2026 Compatibility
|
||||
|
||||
Terminals that natively support DEC 2026 will buffer and render atomically. Terminals that don't support it ignore the escape sequences harmlessly. xterm.js doesn't support DEC 2026 natively, so the client implements its own buffering by parsing the markers.
|
||||
Terminals that natively support DEC 2026 buffer and render atomically. Codeman uses xterm.js 6, so the client passes the markers through instead of parsing or discarding partial blocks.
|
||||
|
||||
**Supporting terminals:** WezTerm, Kitty, Ghostty, iTerm2 3.5+, Windows Terminal, VSCode terminal
|
||||
|
||||
@@ -135,4 +117,4 @@ Terminals that natively support DEC 2026 will buffer and render atomically. Term
|
||||
| File | Key Functions |
|
||||
|------|---------------|
|
||||
| `src/web/server.ts` | `batchTerminalData()`, `flushTerminalBatches()`, `broadcast()` |
|
||||
| `src/web/public/app.js` | `batchTerminalWrite()`, `extractSyncSegments()`, `flushPendingWrites()`, `flushFlickerBuffer()`, `chunkedTerminalWrite()` |
|
||||
| `src/web/public/terminal-ui.js` | `batchTerminalWrite()`, `_scheduleTerminalWriteFlush()`, `flushPendingWrites()`, `flushFlickerBuffer()`, `chunkedTerminalWrite()` |
|
||||
|
||||
+25
-1
@@ -46,7 +46,11 @@ Codeman, including a phone that is not on the tailnet.
|
||||
|
||||
`direct` mode (a plain cross-origin iframe) still exists and is cheaper, but it only
|
||||
works for an HTTPS dashboard that permits framing. The **Test** button probes from
|
||||
the server and tells you which mode applies.
|
||||
the server and tells you which mode applies. Note what Test actually verifies:
|
||||
**server-to-upstream reachability, nothing else**. It does not exercise the browser
|
||||
sandbox, cookies, CORS, CSP, or any reverse proxy sitting in front of Codeman, so a
|
||||
passing Test does not guarantee the embedded page will render (see the
|
||||
cookie-authenticated reverse proxy caveat below).
|
||||
|
||||
## The sandbox, and when to turn it off
|
||||
|
||||
@@ -66,6 +70,17 @@ Even in trusted mode, Codeman never forwards its own credentials upstream: the
|
||||
`Authorization` header and the `codeman_session` cookie are stripped on the way out,
|
||||
so `CODEMAN_PASSWORD` cannot leak into a dashboard.
|
||||
|
||||
⚠️ **Sandboxed tabs may not work when Codeman itself is behind a
|
||||
cookie-authenticated reverse proxy** (Cloudflare Access, Authelia, oauth2-proxy and
|
||||
similar). The sandboxed frame is opaque-origin, so its stylesheet, script, and API
|
||||
requests do not carry the proxy's authentication cookie; the proxy redirects them to
|
||||
the login provider, where CORS/CSP kills them, and the embedded app renders
|
||||
unstyled or broken while the Codeman page around it works fine. Trusted mode
|
||||
(**Open sandboxed** off) keeps a real origin and the cookie, so it works. The
|
||||
**Test** button cannot catch this: it checks that the Codeman *server* can reach the
|
||||
upstream, not that a sandboxed *browser* frame can load assets through the public
|
||||
authentication layer.
|
||||
|
||||
## How the proxy authenticates
|
||||
|
||||
A sandboxed iframe is opaque-origin, so every request it makes is cross-site: the
|
||||
@@ -137,6 +152,15 @@ then every API call fails, which looks like the dashboard being broken.
|
||||
- **Login-protected dashboards need trusted mode**, since a sandboxed frame has no
|
||||
cookie jar. A server-side per-dashboard cookie jar would lift this and is the
|
||||
natural next step if it becomes annoying.
|
||||
- **Cookie-authenticated reverse proxies in front of Codeman break sandboxed tabs**
|
||||
(#238). The sandboxed frame's requests carry no auth cookie, so the proxy bounces
|
||||
them to its login provider and the app loads broken while Test reports reachable.
|
||||
Use trusted mode behind Cloudflare Access and friends; see the warning above.
|
||||
- **Slow endpoints and the upstream timeout** (#237). The proxy waits
|
||||
`CODEMAN_WEBVIEW_TIMEOUT_MS` (default 300s) for the upstream's response *headers*,
|
||||
then streams the body without any time bound; a header timeout is logged
|
||||
server-side and answered as a 502 that names the limit. WebSocket handshakes use
|
||||
the separate `CODEMAN_WEBVIEW_WS_HANDSHAKE_TIMEOUT_MS` (default 30s).
|
||||
- **Not a security boundary.** The proxy reaches whatever the Codeman server can
|
||||
reach. That is not an escalation for someone who already commands
|
||||
`--dangerously-skip-permissions` agents, but in multi-user mode it does mean a
|
||||
|
||||
@@ -0,0 +1,207 @@
|
||||
# Agent CLIs
|
||||
|
||||
Codeman drives seven run modes: six agent CLIs plus a plain shell. This page covers picking
|
||||
one, setting it up, and the differences that actually change how you work.
|
||||
|
||||
## The seven modes
|
||||
|
||||
| Mode | CLI | Get it |
|
||||
| -------------------- | ---------------------------- | ---------------------------------------------------------------------- |
|
||||
| **Claude Code** | `claude` | [docs.anthropic.com](https://docs.anthropic.com/en/docs/claude-code) |
|
||||
| **OpenCode** | `opencode` | [opencode.ai](https://opencode.ai) |
|
||||
| **Codex** | `codex` | [developers.openai.com/codex/cli](https://developers.openai.com/codex/cli) |
|
||||
| **Gemini** | `gemini` | [github.com/google-gemini/gemini-cli](https://github.com/google-gemini/gemini-cli) |
|
||||
| **Antigravity** | `agy` | [antigravity.google](https://antigravity.google) |
|
||||
| **Pi** | `pi` | [pi.dev](https://pi.dev) |
|
||||
| **Terminal / Shell** | your `$SHELL` | Already installed. |
|
||||
|
||||
Any combination works, including all of them. The run mode is chosen per session from the
|
||||
arrow beside the **Run** button, so one case can have a Claude session and a Codex session
|
||||
open side by side.
|
||||
|
||||
## Codeman does not manage your logins
|
||||
|
||||
Install each CLI yourself and log it in once by hand. Codeman never collects, stores, or
|
||||
refreshes your CLI credentials. It launches the binary and attaches to the result.
|
||||
|
||||
The one place credentials are touched is [Docker Cases](Docker-Cases), where host
|
||||
credentials are copied into a container read-only at launch so you do not have to log in
|
||||
again inside it. Even there, the container keeps its own copies and never writes back to
|
||||
your host credential stores.
|
||||
|
||||
## Making a CLI visible to Codeman
|
||||
|
||||
Codeman resolves each binary from the environment the **server** runs in, which is not
|
||||
necessarily the shell you tested in.
|
||||
|
||||
```bash
|
||||
codeman doctor # what Codeman can actually see
|
||||
codeman doctor --json
|
||||
```
|
||||
|
||||
If a CLI is installed but a Run button for it never appears:
|
||||
|
||||
1. Check `which <cli>` in a plain login shell, not just your interactive one.
|
||||
2. If Codeman runs as a service, remember that launchd hands a job
|
||||
`/usr/bin:/bin:/usr/sbin:/sbin`. `codeman service install` bakes your PATH into the unit
|
||||
precisely to avoid this; a hand-written plist or unit will not.
|
||||
3. Restart the server after installing a new CLI.
|
||||
|
||||
`pi` is additionally version-probed rather than trusted by name, because `pi` is a generic
|
||||
enough command that something else on your PATH may answer to it.
|
||||
|
||||
## Claude is the reference mode
|
||||
|
||||
A number of Codeman features exist only for Claude sessions. This is structural, not a
|
||||
backlog: they depend on Claude Code's hook system, or on parsing Claude's specific terminal
|
||||
output. The other CLIs expose no equivalent.
|
||||
|
||||
| Feature | Claude | Other CLIs |
|
||||
| ------------------------------------------------ | ------ | --------------------------------------------------- |
|
||||
| Sessions, tabs, scrollback, exactly-once input | Yes | Yes |
|
||||
| Respawn cycling and unattended runs | Yes | Yes |
|
||||
| Cron jobs | Yes | Yes |
|
||||
| Docker cases, remote SSH cases | Yes | Yes |
|
||||
| Precise idle detection (hook-driven) | Yes | Output-stabilization fallback, coarser |
|
||||
| Auto-resume when a usage limit resets | Yes | No |
|
||||
| Plan usage chip | Yes | No |
|
||||
| Approvals Inbox | Yes | No |
|
||||
| Read My Mind | Yes | No |
|
||||
| Ralph loop and its task tracker | Yes | No |
|
||||
| Subagent and team windows | Yes | No |
|
||||
| Model, effort, and ultracode controls | Yes | No |
|
||||
| `stop` and `blocked` wait signals | Yes | 400 if you ask for them explicitly |
|
||||
| The bundled agent skill | Yes | No |
|
||||
|
||||
Everything that makes a session a session works everywhere. What is Claude-only is mostly
|
||||
the machinery that needs to know *what* the agent is doing rather than *that* it is doing
|
||||
something.
|
||||
|
||||
## Per-CLI notes
|
||||
|
||||
### Claude Code
|
||||
|
||||
The defaults you will care about, all under **App Settings**:
|
||||
|
||||
- **Model** (Models section). Written into the case's `.claude/settings.local.json` as a
|
||||
soft default, so `/model` still works mid-session. The 1M-context Opus variant is a
|
||||
switch on the model card rather than a separate model.
|
||||
- **Effort** (`low` through `max`) or **ultracode** for dynamic multi-agent workflows. Also
|
||||
a soft default: `/effort` overrides it any time. Effort is deliberately not passed as an
|
||||
environment variable, because that would hard-lock it and block in-session switching.
|
||||
- **Startup permission mode** (Agents & CLIs section). The default is
|
||||
`--dangerously-skip-permissions`, which is why the security model matters. You can switch
|
||||
new sessions to Anthropic's classifier-guarded `auto` mode, normal prompting, or an
|
||||
explicit allowed-tools list.
|
||||
|
||||
**Separate Claude accounts per session.** Set `CLAUDE_CONFIG_DIR` in a session's environment
|
||||
overrides to point it at a different Claude config directory, which is how you run one
|
||||
session on a client's subscription and another on your own. One caveat: a relocated config
|
||||
directory writes transcripts outside `~/.claude/projects`, which blinds the response viewer,
|
||||
subagent windows, ultracode panel, and Read My Mind for that session. Symlink `projects`
|
||||
back into the shared tree to keep them working:
|
||||
|
||||
```bash
|
||||
ln -s ~/.claude/projects <configDir>/projects
|
||||
```
|
||||
|
||||
### OpenCode
|
||||
|
||||
Renders its own TUI, so Codeman treats readiness as output stabilization rather than
|
||||
watching for a prompt marker. Requires tmux, with no direct-PTY fallback, because its
|
||||
environment is injected through socket-scoped `tmux setenv` rather than the command line.
|
||||
|
||||
Integration detail: [`docs/opencode-integration.md`](https://github.com/Ark0N/Codeman/blob/master/docs/opencode-integration.md).
|
||||
|
||||
### Codex
|
||||
|
||||
Two behaviours that are deliberate and worth knowing:
|
||||
|
||||
- **Predictive echo instead of buffered echo.** Codex's composer reacts to every keystroke,
|
||||
a `/` opens a live-filtering picker, arrows edit server-side state. Buffering keystrokes
|
||||
until Enter starved it, so Codex paints each keystroke at the predicted cell while the
|
||||
bytes on the wire stay byte-identical to what you typed.
|
||||
- **The wheel is not forwarded** into its transcript. Codex ignores the mouse reports
|
||||
Codeman would send, so forwarding produced a dead wheel. Scrolling in a Codex session is
|
||||
local scrollback.
|
||||
|
||||
### Gemini
|
||||
|
||||
Enterprise only, since Google's June 2026 consumer cutover. Its environment allowlist
|
||||
includes the broad `GOOGLE_*` namespace, deliberately, because Vertex AI authentication
|
||||
needs `GOOGLE_CLOUD_PROJECT`, `GOOGLE_APPLICATION_CREDENTIALS`, and
|
||||
`GOOGLE_GENAI_USE_VERTEXAI`. That is the loosest allowlist entry in Codeman and it affects
|
||||
only the CLI you spawned yourself.
|
||||
|
||||
### Antigravity
|
||||
|
||||
Google's successor to the consumer Gemini CLI, invoked as `agy`. It keeps all of its state
|
||||
in `~/.gemini/antigravity-cli/`, so the credential handling that applies to Gemini applies
|
||||
to it as well.
|
||||
|
||||
### Pi
|
||||
|
||||
Pi needs the opposite instincts from every other CLI here.
|
||||
|
||||
- **It has no permission prompts and no sandbox.** There is no bypass flag to send, and
|
||||
Codeman does not invent one.
|
||||
- **Its privileged setting is project trust**, a three-way `--approve` / `--no-approve` /
|
||||
unset. Approving trust makes Pi **execute repo-local `.pi/extensions` TypeScript**, so
|
||||
point it at a repository you trust. In multi-user mode, a user without an explicit grant
|
||||
gets `--no-approve` even when no configuration exists.
|
||||
- **Authentication is `/login` inside the session**, or the server process's own
|
||||
environment. Pi's roughly 34 provider keys (`ANTHROPIC_API_KEY`, `OPENAI_API_KEY`,
|
||||
`HF_TOKEN`, and so on) share no common prefix, and the environment allowlist is global
|
||||
rather than per mode, so admitting them for Pi would widen the allowlist for every mode at
|
||||
once. They stay out.
|
||||
|
||||
Guide: [`docs/pi-integration.md`](https://github.com/Ark0N/Codeman/blob/master/docs/pi-integration.md).
|
||||
|
||||
### Terminal / Shell
|
||||
|
||||
A plain shell in a tmux session. No agent, no hooks, no idle detection.
|
||||
|
||||
On phones a shell session automatically swaps the keyboard accessory bar for terminal
|
||||
controls: Ctrl, Esc, Tab, arrows, paste. **Ctrl is a one-shot modifier**: tap it, then tap a
|
||||
letter, and the control byte is sent. It disarms on use, on a second tap, on any other
|
||||
accessory key, on a session switch, and when the keyboard closes. Details in
|
||||
[Mobile Guide](Mobile-Guide).
|
||||
|
||||
## Environment overrides
|
||||
|
||||
Per-session environment variables are set when creating a session and persist across
|
||||
respawns. Which variables are accepted depends on the mode:
|
||||
|
||||
| Mode | Allowed prefixes |
|
||||
| ----------- | --------------------------------- |
|
||||
| Claude | `CLAUDE_CODE_*`, plus the exact key `CLAUDE_CONFIG_DIR` |
|
||||
| OpenCode | `OPENCODE_*` |
|
||||
| Codex | `CODEX_*` |
|
||||
| Gemini | `GEMINI_*`, `GOOGLE_*` |
|
||||
| Antigravity | `ANTIGRAVITY_*` |
|
||||
| Pi | `PI_*` |
|
||||
|
||||
Anything outside the allowlist is rejected at the schema. This is intentional: the allowlist
|
||||
is one global list, so widening it for one CLI widens it for all of them.
|
||||
|
||||
Two things that deliberately do **not** travel as environment variables: **effort**, because
|
||||
an environment variable hard-locks it and blocks `/effort`, and **model**, which is written
|
||||
into the case's `.claude/settings.local.json` so that `/model` keeps working.
|
||||
|
||||
## Choosing a mode
|
||||
|
||||
- **Claude Code** if you want every Codeman feature. Unattended overnight runs, usage-limit
|
||||
auto-resume, the Approvals Inbox, and subagent visualization all assume it.
|
||||
- **Codex, OpenCode, Gemini, Antigravity** when you prefer that agent or that model. You get
|
||||
the session layer, respawn, cron, Docker, and remote SSH; you do not get the hook-driven
|
||||
features.
|
||||
- **Pi** if you want a fast, unsandboxed agent and you understand what project trust does.
|
||||
- **Shell** for the times you want a terminal on your phone with no agent at all. It is a
|
||||
genuinely useful mode, not a fallback.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Core Concepts](Core-Concepts) - run modes versus location overlays.
|
||||
- [Settings Reference](Settings-Reference) - model, effort, and permission-mode settings.
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - what idle detection does per mode.
|
||||
- [Security](Security) - what skipping permission prompts actually means.
|
||||
@@ -0,0 +1,97 @@
|
||||
# Autonomous Loops
|
||||
|
||||
Two features that go further than "keep the session going": the **Ralph loop**, which works
|
||||
a task list to completion in one session, and the **Orchestrator**, which turns a goal into
|
||||
a phased plan and drives it across agents.
|
||||
|
||||
Both are Claude-only, both are off by default, and neither is where to start. If what you
|
||||
want is an agent that keeps working overnight, that is
|
||||
[Keeping Agents Running](Keeping-Agents-Running), and it is simpler, better understood, and
|
||||
what most people actually use.
|
||||
|
||||
## Which one, if either
|
||||
|
||||
| You have | Use |
|
||||
| ------------------------------------------------- | ------------------------------------------------------------ |
|
||||
| A session that stops too early | [Respawn](Keeping-Agents-Running) |
|
||||
| A written task list to grind through | Ralph loop |
|
||||
| One large goal that needs planning and checkpoints | Orchestrator |
|
||||
| Work that should start at a certain time | [Cron Jobs](Cron-Jobs) |
|
||||
| Several workers to fan out and supervise | [Driving Codeman From An Agent](Driving-Codeman-From-An-Agent) |
|
||||
|
||||
## The Ralph loop
|
||||
|
||||
Named after the Ralph Wiggum pattern: keep feeding the agent its own task list until the
|
||||
list is empty.
|
||||
|
||||
The shape of it:
|
||||
|
||||
- The task list lives in a plan file in the case, conventionally `fix_plan.md`.
|
||||
- Each cycle the agent reads the plan, works the next incomplete task, and marks progress.
|
||||
- Codeman watches the file, tracks todos, and detects stalls.
|
||||
- The loop ends when the agent signals completion, when the iteration cap is reached, or
|
||||
when you stop it.
|
||||
|
||||
Start it from **Session Options → Ralph / Todo**, or from the wizard on the welcome screen.
|
||||
|
||||
| Setting | What it does |
|
||||
| ---------------------- | ------------------------------------------------------------------------ |
|
||||
| Max iterations | Hard ceiling on cycles. |
|
||||
| Max todos | Cap on tracked tasks, default 500, oldest evicted first. |
|
||||
| Todo expiration | Auto-expiry for stale todos, default 60 minutes. |
|
||||
| Plan file | Which file holds the task list. |
|
||||
|
||||
A **circuit breaker** sits behind it to stop respawn thrashing: it moves from closed to
|
||||
half-open to open, and is reset explicitly from the session's Ralph controls.
|
||||
|
||||
Honest assessment: Ralph is functional but is not where development attention goes. It
|
||||
predates the respawn presets, which cover most of what people originally used it for with
|
||||
less ceremony. Treat it as a specialised tool rather than the headline feature.
|
||||
|
||||
Full background, including the upstream pattern it is based on:
|
||||
[`docs/ralph-wiggum-guide.md`](https://github.com/Ark0N/Codeman/blob/master/docs/ralph-wiggum-guide.md).
|
||||
|
||||
## The Orchestrator
|
||||
|
||||
A state machine that turns one goal into a phased plan and drives it to completion:
|
||||
|
||||
```
|
||||
idle → planning → approval → executing → verifying → (replanning) → completed / failed
|
||||
```
|
||||
|
||||
- **Planning** turns your goal into phases.
|
||||
- **Approval** is yours. You see the plan before anything runs.
|
||||
- **Executing** runs each phase, using team agents and the task queue.
|
||||
- **Verifying** gates each phase before the next one starts. A failed gate can send it back
|
||||
to replanning rather than forward.
|
||||
|
||||
Open it from the Orchestrator panel in the toolbar. State persists in `state.json`, so a
|
||||
server restart does not lose an in-flight plan.
|
||||
|
||||
Where it differs from Ralph: Ralph is one session grinding a list, the Orchestrator
|
||||
coordinates phases and agents with verification between them. It suits work that has a
|
||||
natural shape ("migrate this, then update callers, then update the tests") rather than a
|
||||
flat backlog.
|
||||
|
||||
Architecture: [`docs/orchestrator-loop-architecture.md`](https://github.com/Ark0N/Codeman/blob/master/docs/orchestrator-loop-architecture.md).
|
||||
|
||||
## Running any of this safely
|
||||
|
||||
Autonomous loops are the features most able to spend money and change code while you are not
|
||||
looking. Some habits that pay off:
|
||||
|
||||
- **Run them in a case that is a git repository**, on a branch you are willing to throw
|
||||
away. Being able to read the diff afterwards is the whole safety net.
|
||||
- **Consider a container.** [Docker Cases](Docker-Cases) gives the agent its own filesystem
|
||||
and network, and one checkbox is all it costs.
|
||||
- **Set the iteration cap deliberately.** It is the ceiling on the spend.
|
||||
- **Turn on notifications** so a blocked loop reaches you: see
|
||||
[Notifications And Approvals](Notifications-And-Approvals).
|
||||
- **Read the run summary and lifecycle log afterwards**, not just the final diff. They show
|
||||
where it went sideways and recovered.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - the simpler feature that usually fits better.
|
||||
- [Watching Agents Work](Watching-Agents-Work) - seeing what a loop is doing while it runs.
|
||||
- [Docker Cases](Docker-Cases) - a sandbox for unattended work.
|
||||
@@ -0,0 +1,118 @@
|
||||
# Contributing
|
||||
|
||||
The full guide lives in
|
||||
[CONTRIBUTING.md](https://github.com/Ark0N/Codeman/blob/master/.github/CONTRIBUTING.md).
|
||||
This page is the short orientation, plus how to fix a page in this wiki.
|
||||
|
||||
## Where things go
|
||||
|
||||
| You have | Send it to |
|
||||
| --------------------------- | ---------------------------------------------------------------------------------------------- |
|
||||
| A bug | An [issue](https://github.com/Ark0N/Codeman/issues), with OS, install method, browser, and which CLI the session was running. |
|
||||
| A question or setup problem | [Discussions](https://github.com/Ark0N/Codeman/discussions). |
|
||||
| An idea | [Ideas](https://github.com/Ark0N/Codeman/discussions/categories/ideas), where it gets voted on. |
|
||||
| A small fix | Straight to a PR. |
|
||||
| A bigger feature | An issue or Discussion first, then build once the design has a nod. |
|
||||
| A security problem | Never a public issue. See [SECURITY.md](https://github.com/Ark0N/Codeman/blob/master/.github/SECURITY.md). |
|
||||
|
||||
Issues usually get a response within a day, and every release credits its contributors and
|
||||
bug reporters by name.
|
||||
|
||||
## Dev setup
|
||||
|
||||
```bash
|
||||
git clone https://github.com/Ark0N/Codeman.git
|
||||
cd Codeman
|
||||
npm install # postinstall builds the vendored xterm addon bundles
|
||||
npm run dev # http://localhost:3000
|
||||
```
|
||||
|
||||
Requirements: Node 22+, tmux, and at least one agent CLI on your PATH.
|
||||
|
||||
The frontend is plain JavaScript with no bundler in dev: edit a `.js` or `.css` file and
|
||||
reload. The exception is `index.html`, which is read once at server start, so markup changes
|
||||
need a restart.
|
||||
|
||||
## Before you push
|
||||
|
||||
CI runs all of these, so running them locally saves a round trip:
|
||||
|
||||
```bash
|
||||
npm run typecheck
|
||||
npm run lint
|
||||
npm run format:check
|
||||
npm run check:frontend-syntax
|
||||
npm test -- test/<file>.test.ts # one file, the normal way
|
||||
npm run test:ci # the full CI sweep
|
||||
```
|
||||
|
||||
**Never run bare `npm test`.** The default configuration includes browser-driven Playwright
|
||||
suites that need a live server, Chromium, and environment-specific baselines; they hang or
|
||||
fail on a normal machine. `test:ci` is the honest "run everything".
|
||||
|
||||
Tests are tmux-safe by design: under vitest the tmux layer becomes an in-memory mock, so
|
||||
tests cannot touch real sessions. If you add a test that binds a port, pick a unique one at
|
||||
3150 or above, and never 3000.
|
||||
|
||||
## Finding your way around
|
||||
|
||||
- Every source file opens with a `@fileoverview` block. Read it before the file; it is the
|
||||
map.
|
||||
- [`CLAUDE.md`](https://github.com/Ark0N/Codeman/blob/master/CLAUDE.md) at the repo root is
|
||||
the densest architecture primer there is. It is written for AI coding agents, but its
|
||||
invariants apply identically to humans, and most review feedback traces back to something
|
||||
already written there.
|
||||
- [`docs/architecture-invariants.md`](https://github.com/Ark0N/Codeman/blob/master/docs/architecture-invariants.md)
|
||||
holds the deep mechanisms and the history behind each rule.
|
||||
|
||||
## Good first contributions
|
||||
|
||||
- **A theme skin.** A skin is four things kept in sync, and a static test checks the sync, so
|
||||
if the test passes your skin works.
|
||||
- **A language.** The i18n module is dependency-free, English is canonical, and Simplified
|
||||
Chinese is a complete example to copy.
|
||||
- **Docs.** If you got stuck and then figured it out, the sentence that would have unstuck
|
||||
you is a pull request.
|
||||
- Anything labelled
|
||||
[good first issue](https://github.com/Ark0N/Codeman/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22).
|
||||
|
||||
Worth discussing first: new CLI backends, and real-device testing reports, especially
|
||||
mobile, which always find things emulation cannot.
|
||||
|
||||
## PR expectations
|
||||
|
||||
- One change per PR. Small and focused reviews fast; a grab bag stalls.
|
||||
- Target `master`.
|
||||
- **Keep your branch mergeable.** A PR with conflicts silently gets no CI runs at all, which
|
||||
is a GitHub quirk rather than a Codeman one. Rebase when conflicts appear.
|
||||
- Include or update tests when you change behaviour.
|
||||
- Do not bump versions or edit the changelog; releases are handled after merge.
|
||||
- AI-assisted contributions are welcome, with one condition: understand what you are
|
||||
submitting, and actually run it. "The model said it works" is not a test.
|
||||
|
||||
## Fixing this wiki
|
||||
|
||||
These pages are generated from
|
||||
[`docs/wiki/`](https://github.com/Ark0N/Codeman/tree/master/docs/wiki) in the main
|
||||
repository, and pushed here automatically when master changes.
|
||||
|
||||
**Editing a page in the browser will be overwritten by the next sync.** Send a pull request
|
||||
against `docs/wiki/` instead. It is plain markdown, and a documentation PR is a genuinely
|
||||
useful contribution.
|
||||
|
||||
Conventions for wiki pages:
|
||||
|
||||
- Links between pages use the wiki form: `[Remote Access](Remote-Access)`, no `.md`.
|
||||
- Links into the repository are absolute `https://github.com/Ark0N/Codeman/blob/master/...`
|
||||
URLs.
|
||||
- Images are referenced from the main repository over raw URLs rather than being copied into
|
||||
the wiki.
|
||||
- Say what the default is, especially when it is off. Most of Codeman is opt-in.
|
||||
- Label Claude-only behaviour every time it appears. Six of the seven run modes are not
|
||||
Claude.
|
||||
|
||||
## Conduct
|
||||
|
||||
Be kind, be direct, assume good faith. Report unacceptable behaviour privately via the
|
||||
contact in
|
||||
[SECURITY.md](https://github.com/Ark0N/Codeman/blob/master/.github/SECURITY.md).
|
||||
@@ -0,0 +1,183 @@
|
||||
# Core Concepts
|
||||
|
||||
The five ideas the rest of the manual assumes: cases, sessions, run modes, location
|
||||
overlays, and tmux. Plus what actually persists, and where it lives on disk.
|
||||
|
||||
## Case
|
||||
|
||||
A **case** is a named working directory that Codeman remembers. It is the unit you pick in
|
||||
the toolbar before hitting Run, and every session belongs to exactly one.
|
||||
|
||||
A case is not a container or a sandbox. It is a folder plus a name plus a little
|
||||
Codeman-side configuration:
|
||||
|
||||
- Which CLI the Run button should default to.
|
||||
- Per-case toggles (Agent Teams, 1M Opus context).
|
||||
- Where it runs, if it is not the local filesystem: see [Location overlays](#location-overlays).
|
||||
|
||||
Three ways to get one, all under **+** next to the case picker:
|
||||
|
||||
| How | Result |
|
||||
| ----------------- | ------------------------------------------------------------------------------------------------------ |
|
||||
| **Create New** | A fresh `~/codeman-cases/<name>` with a scaffolded `CLAUDE.md`. |
|
||||
| **Clone Repo** | A public repo cloned into `~/codeman-cases/<name>` and registered as a case. |
|
||||
| **Link Existing** | An existing folder anywhere on disk, registered in place. Nothing is copied or moved. |
|
||||
|
||||
Linked cases keep living where they are. Deleting a case in Codeman removes the
|
||||
registration, and for a linked case that is all it removes.
|
||||
|
||||
**Cases created from scratch are the only copy of that code.** Uninstalling Codeman does not
|
||||
delete `~/codeman-cases/`, but treat that directory as real work, not scratch space.
|
||||
|
||||
## Session
|
||||
|
||||
A **session** is one CLI process running in one tmux session, streamed to your browser.
|
||||
|
||||
Sessions are named `w<n>-<case>`, so `w1-myproject` is the first worker in the `myproject`
|
||||
case. Each has a stable id, and that id is what the API, the wait primitives, and every
|
||||
event use.
|
||||
|
||||
Several sessions can share one case. That is the normal way to parallelize: three workers
|
||||
in the same repo, three tabs, one case.
|
||||
|
||||
A session carries state the case does not:
|
||||
|
||||
- Its run mode, model, effort level, and environment overrides.
|
||||
- Its respawn configuration and Ralph loop state.
|
||||
- Its terminal scrollback.
|
||||
- Its owner, in [Multi-User Mode](Multi-User-Mode).
|
||||
|
||||
## Run mode
|
||||
|
||||
The **run mode** is which CLI the session runs: `claude`, `opencode`, `codex`, `gemini`,
|
||||
`antigravity`, `pi`, or `shell`. It is chosen at start and does not change afterwards; to
|
||||
switch, start another session.
|
||||
|
||||
Claude is the reference mode. Six of the seven are not Claude, and a number of Codeman
|
||||
features are Claude-only for structural reasons rather than missing effort: they depend on
|
||||
Claude Code's hook system or on parsing its terminal output. Every such feature is labelled
|
||||
Claude-only where it appears, and [Agent CLIs](Agent-CLIs) lists them in one place.
|
||||
|
||||
## Location overlays
|
||||
|
||||
Where a case runs is **separate from** which CLI it runs. There are three locations:
|
||||
|
||||
| Location | What happens |
|
||||
| -------------- | ------------------------------------------------------------------------------------------------------------ |
|
||||
| **Local** | The default. tmux and the CLI run on the Codeman host. |
|
||||
| **Docker** | One long-lived container per case; sessions `docker exec` into it. See [Docker Cases](Docker-Cases). |
|
||||
| **Remote SSH** | A durable tmux server on the remote host, fronted by a local pane running `ssh`. See [Remote SSH Sessions](Remote-SSH-Sessions). |
|
||||
|
||||
This matters because it is a common source of confusion: Docker is **not** an eighth run
|
||||
mode. All seven run modes work in all three locations. A case is docker-backed or
|
||||
ssh-backed; a session is claude or codex or shell.
|
||||
|
||||
**Web tabs** are the other thing that is not a session. A saved dashboard URL renders as a
|
||||
tab beside your agents, but there is no PTY, no tmux, and no respawn behind it. See
|
||||
[Web Tabs](Web-Tabs).
|
||||
|
||||
## Why tmux
|
||||
|
||||
tmux is a hard requirement, and it is the reason Codeman behaves the way it does.
|
||||
|
||||
The agent runs inside a tmux session. Codeman attaches to it, the same way your terminal
|
||||
would. That indirection buys:
|
||||
|
||||
- **Survival.** The agent outlives your browser tab, your network, your laptop lid, and a
|
||||
restart of the Codeman server itself.
|
||||
- **Real scrollback.** History is held by tmux, so reconnecting replays what happened while
|
||||
you were gone instead of starting from blank.
|
||||
- **Attach from anywhere else.** The same session is reachable from a terminal over SSH
|
||||
with the `sc` chooser, or plain `tmux -L codeman attach`.
|
||||
- **Secrets off the command line.** Environment overrides are injected with socket-scoped
|
||||
`tmux setenv` rather than being visible in the spawn command.
|
||||
|
||||
The socket is `tmux -L codeman`, separate from your personal tmux server, so Codeman
|
||||
sessions never appear in a bare `tmux ls`.
|
||||
|
||||
## What persists
|
||||
|
||||
| Survives | Does not survive |
|
||||
| -------------------------------------------- | --------------------------------------------------- |
|
||||
| Closing the browser | `tmux -L codeman kill-server` |
|
||||
| Losing the network | A machine reboot (tmux dies with it) |
|
||||
| Restarting the Codeman server | Killing the session from the UI |
|
||||
| `codeman web --stop` | |
|
||||
| A dropped SSH link, for remote cases | |
|
||||
| A container restart, for docker cases | |
|
||||
|
||||
Conversation history is a separate question: Claude transcripts live in `~/.claude/`, so a
|
||||
conversation can be resumed even after the tmux session is gone. That is what the welcome
|
||||
screen's **Resume Conversation** list offers.
|
||||
|
||||
## State on disk
|
||||
|
||||
Everything Codeman knows lives under `~/.codeman/`:
|
||||
|
||||
| File | Holds |
|
||||
| ---------------------------------------- | -------------------------------------------------------------------- |
|
||||
| `state.json` | Sessions, settings, respawn config, orchestrator state, cron jobs. |
|
||||
| `settings.json` | User preferences that sync across your devices. |
|
||||
| `mux-sessions.json` | tmux recovery data. |
|
||||
| `session-lifecycle.jsonl` | Append-only audit log of session starts, exits, and kills. |
|
||||
| `linked-cases.json` | Registered cases. |
|
||||
| `remote-hosts.json`, `docker-hosts.json` | Location overlay configuration. |
|
||||
| `webviews.json` | Saved dashboard URLs. |
|
||||
| `users.json` | Multi-user accounts, mode 0600. |
|
||||
| `push-*.json` | Web push keys and subscriptions. |
|
||||
| `certs/` | Self-signed TLS for `--https`. |
|
||||
|
||||
None of it needs root, none of it leaves the machine, and deleting `~/.codeman/` resets
|
||||
Codeman to a fresh install without touching your code.
|
||||
|
||||
## Instances
|
||||
|
||||
The data directory and the tmux socket are both **process wide**. Two Codeman servers
|
||||
started on one machine share them, which means the second one discovers the first one's
|
||||
live sessions and attaches to them, resizing and mutating sessions you did not expect it to
|
||||
touch.
|
||||
|
||||
To run two on purpose, give each its own instance name:
|
||||
|
||||
```bash
|
||||
CODEMAN_INSTANCE=beta CODEMAN_PORT=5000 codeman web
|
||||
```
|
||||
|
||||
That scopes the data directory and the tmux socket together, which is the only safe way to
|
||||
do it. `CODEMAN_DATA_DIR` and `CODEMAN_TMUX_SOCKET` can be set individually if you need
|
||||
them apart, but setting only one of the two reproduces exactly the problem you were trying
|
||||
to avoid.
|
||||
|
||||
## Hooks
|
||||
|
||||
For Claude sessions, Codeman writes a hooks configuration into the case so Claude Code can
|
||||
report events back: a permission prompt appeared, the turn finished, the agent went idle, a
|
||||
task completed. Those events drive tab alerts, the Approvals Inbox, notifications, and the
|
||||
wait primitives.
|
||||
|
||||
This is why some features are Claude-only. The other CLIs have no equivalent hook system,
|
||||
so for them Codeman falls back to watching terminal output, which is coarser: it can see
|
||||
that something happened, not what it was.
|
||||
|
||||
See [Hooks And Integrations](Hooks-And-Integrations).
|
||||
|
||||
## Vocabulary
|
||||
|
||||
| Term | Means |
|
||||
| --------------- | ---------------------------------------------------------------------------- |
|
||||
| **Case** | Named working directory. |
|
||||
| **Session** | One CLI in one tmux session. |
|
||||
| **Run mode** | Which CLI: claude, opencode, codex, gemini, antigravity, pi, shell. |
|
||||
| **Respawn** | Restarting the CLI on idle to keep an unattended run going. |
|
||||
| **Ralph loop** | An autonomous single-session task loop. |
|
||||
| **Orchestrator**| A phased plan driven across multiple agents. |
|
||||
| **Subagent** | An agent the CLI spawned itself, shown live in its own window. |
|
||||
| **Web tab** | A saved dashboard URL rendered as a tab. Not a session. |
|
||||
| **Instance** | One Codeman server with its own data directory and tmux socket. |
|
||||
|
||||
## Read next
|
||||
|
||||
- [The Dashboard](The-Dashboard) - what the UI is showing you.
|
||||
- [Agent CLIs](Agent-CLIs) - the seven run modes in detail.
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - respawn, idle detection, usage limits.
|
||||
- [`docs/architecture-invariants.md`](https://github.com/Ark0N/Codeman/blob/master/docs/architecture-invariants.md) - the mechanisms behind all of this, for contributors.
|
||||
@@ -0,0 +1,161 @@
|
||||
# Cron Jobs
|
||||
|
||||
Saved, named jobs that start a session and send it a prompt on a schedule. Cron for agent
|
||||
sessions: *every weekday at 03:00, open a Claude session in `~/proj` and tell it to update
|
||||
dependencies and open a PR.*
|
||||
|
||||
The ⏰ **Cron** header button is opt-in. Turn it on in
|
||||
**App Settings → Header & Panels**.
|
||||
|
||||
## Creating a job
|
||||
|
||||
1. Click **⏰ Cron**, then **+ New Job**.
|
||||
2. Give it a name, pick the agent type and working directory.
|
||||
3. Write the prompt, or point at a file containing it.
|
||||
4. Choose a schedule and leave **Enabled** on.
|
||||
5. **Save**. The job appears with its computed next run.
|
||||
|
||||
**Run Now** fires it immediately without touching the schedule, which is the fastest way to
|
||||
find out whether the prompt does what you meant.
|
||||
|
||||
## The fields
|
||||
|
||||
| Field | Notes |
|
||||
| ------------------------ | ------------------------------------------------------------------------------------------- |
|
||||
| **Name** | Also used as the created session's name. |
|
||||
| **Agent type** | Any run mode, including `shell`. |
|
||||
| **Working directory** | Validated when you save **and** again when the job fires. Blocked system trees are refused. |
|
||||
| **Launch command** | Shell jobs only. Sent as the first line once the shell is up, before the prompt. |
|
||||
| **Prompt** | Inline text, or a path to a file read at fire time. |
|
||||
| **Input mode** | `typed` behaves like a human typing. `paste` writes directly. |
|
||||
| **Schedule** | `once`, `interval`, `daily`, or `weekly`. |
|
||||
| **Enabled** | Disabled jobs never fire on their own. **Run Now** still works. |
|
||||
| **Concurrency policy** | What to do if sessions of the same type are already running. |
|
||||
| **Auto-close previous** | Recurring jobs only. Closes the session the previous run created. Default on. |
|
||||
| **Notes** | Free text for you. |
|
||||
|
||||
## Schedules
|
||||
|
||||
All wall-clock times are in the **server's local timezone**, not your browser's. A job set
|
||||
for 03:00 fires at 03:00 where the server is.
|
||||
|
||||
| Type | Behaviour |
|
||||
| ---------- | -------------------------------------------------------------------------------------------------- |
|
||||
| `once` | Fires at an absolute time, then disables itself. A job missed because the server was down still fires once on the next tick. |
|
||||
| `interval` | Every N minutes, from 1 minute to a year. |
|
||||
| `daily` | At `HH:MM` every day. If today's time has passed, the next run is tomorrow. |
|
||||
| `weekly` | At `HH:MM` on the weekdays you pick. |
|
||||
|
||||
Interval jobs re-anchor to when they actually fired, not to an ideal cadence, so a slow tick
|
||||
or a server restart shifts later runs slightly. That drift is accepted rather than corrected.
|
||||
|
||||
## Prompts are single line
|
||||
|
||||
This is the rule people trip over. Programmatic input into an agent session is single line
|
||||
everywhere in Codeman, because the terminal UIs these CLIs use treat a newline as submit. A
|
||||
multi-line prompt would be silently mangled, so it is **rejected** instead: the form refuses
|
||||
it, and a prompt file whose contents are multi-line fails the run with a clear message.
|
||||
|
||||
For anything longer than a sentence, put the instructions in a file and make the prompt tell
|
||||
the agent to read it:
|
||||
|
||||
```
|
||||
read TASKS.md and work through it
|
||||
```
|
||||
|
||||
That is also easier to edit than a job field.
|
||||
|
||||
### Prompt files
|
||||
|
||||
Reading the prompt from a file at fire time is useful when the instructions change more
|
||||
often than the schedule. The path is confined to the job's working directory, symlinks are
|
||||
resolved before the check, sensitive trees are refused, and the file has to be a regular
|
||||
file under 1 MiB.
|
||||
|
||||
If any of that fails, the run is recorded as failed and **no session is created**.
|
||||
|
||||
## Concurrency
|
||||
|
||||
Applies to scheduled runs only, never to **Run Now**:
|
||||
|
||||
| Policy | Behaviour |
|
||||
| ------------------------------- | -------------------------------------------------------------------------------------- |
|
||||
| `warn_only` | Always launch. The count of live same-type sessions is shown but does not block. |
|
||||
| `skip_if_same_agent_running` | Skip this fire if another live session of that mode exists. |
|
||||
|
||||
The skip policy has the details you would want it to have:
|
||||
|
||||
- Only **live** sessions block. A tab whose CLI already exited does not count.
|
||||
- Sessions the job created on its own previous runs never block it, otherwise a recurring
|
||||
job would deadlock on itself after the first fire.
|
||||
- A skipped `once` job is not consumed. It stays armed and fires when the blocker goes away.
|
||||
- Consecutive skips are collapsed into one record per streak, so a perpetually skipped job
|
||||
cannot bloat your state file.
|
||||
|
||||
## Run history
|
||||
|
||||
Every fire is recorded per job, with a status:
|
||||
|
||||
| Status | Meaning |
|
||||
| --------- | -------------------------------------------------------------------- |
|
||||
| `created` | The run started and a session was created. |
|
||||
| `skipped` | The concurrency policy blocked it. Not counted as a run. |
|
||||
| `failed` | The prompt could not be resolved, or the working directory was gone. |
|
||||
|
||||
The schedule is advanced **before** the session launches, so a slow start cannot cause the
|
||||
same job to re-trigger.
|
||||
|
||||
## Cron versus the other autonomy features
|
||||
|
||||
| Want | Use |
|
||||
| --------------------------------------------------- | ------------------------------------------------------- |
|
||||
| Start work at a specific time | Cron |
|
||||
| Keep an existing session working | [Keeping Agents Running](Keeping-Agents-Running) |
|
||||
| Drive one goal to completion across phases | [Autonomous Loops](Autonomous-Loops) |
|
||||
|
||||
There is also an older, deliberately separate `ScheduledRun` concept behind
|
||||
`/api/scheduled`: a run-now, duration-bounded loop with no recurrence and no saved jobs. The
|
||||
two systems never interact, and Cron is the one you want.
|
||||
|
||||
## From the API
|
||||
|
||||
```bash
|
||||
API=http://localhost:3000
|
||||
|
||||
curl -s -X POST "$API/api/cron/jobs" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"name": "nightly-deps",
|
||||
"agentType": "claude",
|
||||
"workingDir": "/home/me/proj",
|
||||
"promptMode": "inline_text",
|
||||
"promptText": "Update dependencies and open a PR",
|
||||
"inputMode": "typed",
|
||||
"scheduleType": "daily",
|
||||
"dailyTime": "03:00",
|
||||
"enabled": true,
|
||||
"concurrencyPolicy": "warn_only"
|
||||
}' | jq
|
||||
|
||||
curl -s "$API/api/cron/jobs" | jq
|
||||
curl -s -X POST "$API/api/cron/jobs/<jobId>/run" | jq
|
||||
curl -s "$API/api/cron/jobs/<jobId>/runs" | jq
|
||||
```
|
||||
|
||||
Add `-u admin:"$CODEMAN_PASSWORD"` when a password is set, and `-k` with the `https://` URL
|
||||
on an HTTPS install.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **Times are the server's, not yours.** Obvious until you are travelling.
|
||||
- **A `pi` job starts slowly.** The readiness poll looks for markers pi does not print, so it
|
||||
burns its poll budget before sending the prompt. The job still works.
|
||||
- **A deleted working directory fails the run**, by design, rather than creating a session
|
||||
somewhere unexpected.
|
||||
- **Auto-close only touches sessions this job created.** Your own tabs are never closed.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - continuing work rather than starting it.
|
||||
- [Notifications And Approvals](Notifications-And-Approvals) - hearing about a job that got stuck.
|
||||
- [`docs/cron-guide.md`](https://github.com/Ark0N/Codeman/blob/master/docs/cron-guide.md) - the complete reference, including the API and SSE events.
|
||||
@@ -0,0 +1,177 @@
|
||||
# Docker Cases
|
||||
|
||||
Run a case inside its own container instead of directly on your host: for isolation, for a
|
||||
reproducible toolchain, and for the ability to pick the whole environment up and move it to
|
||||
another machine.
|
||||
|
||||
A docker case is a **location overlay**, not a run mode. All seven run modes work inside a
|
||||
container. See [Core Concepts](Core-Concepts).
|
||||
|
||||
## One-time setup: the base image
|
||||
|
||||
The container needs an image carrying the agent toolchain (node, the CLIs, git, tmux). It
|
||||
builds itself on first use with progress streamed to the UI, or you can build it ahead of
|
||||
time:
|
||||
|
||||
```bash
|
||||
node scripts/build-agent-image.mjs --no-cache
|
||||
```
|
||||
|
||||
**Always pass `--no-cache`.** The CLIs are installed in a single `npm install -g` layer, so
|
||||
a plain rebuild reuses that layer from the cache and the CLIs stay frozen at whatever
|
||||
versions the image was *first* built with. This has shipped a broken CLI while reporting a
|
||||
successful build.
|
||||
|
||||
A zero exit code proves the layers ran, not that the toolchain works. Verify:
|
||||
|
||||
```bash
|
||||
docker run --rm codeman/agent:base bash -lc \
|
||||
'for c in claude codex gemini opencode agy pi; do printf "%-9s " $c; $c --version 2>&1 | head -1; done'
|
||||
```
|
||||
|
||||
The image is secret-free. Credentials are delivered at runtime, never baked in, so exports
|
||||
never leak them. A full image lands around 1.6GB.
|
||||
|
||||
Prerequisite: Docker or Podman with a reachable daemon.
|
||||
|
||||
## The quick way
|
||||
|
||||
On **Add Case → Create New**, tick **🐳 Run in an isolated Docker container**. That alone is
|
||||
enough: Codeman creates the case folder, spins up a hardened container with sensible
|
||||
defaults, and starts the session inside it.
|
||||
|
||||
Expanding **Container settings** offers a template:
|
||||
|
||||
| Template | Memory | CPUs | GPUs |
|
||||
| ----------------- | ------ | ---- | ------------------------------------- |
|
||||
| Small | 2 GB | 1 | none |
|
||||
| Medium (default) | 4 GB | 2 | none |
|
||||
| Large | 8 GB | 4 | none |
|
||||
| GPU | 8 GB | 4 | all (needs the NVIDIA container toolkit) |
|
||||
|
||||
Disk is elastic: storage grows as data arrives, bounded only by host disk. Changing any
|
||||
setting creates a dedicated host profile for that case, so it never mutates the shared
|
||||
default.
|
||||
|
||||
## The full way
|
||||
|
||||
**Add Case → Docker** exposes everything:
|
||||
|
||||
| Field | Meaning |
|
||||
| -------------------- | ----------------------------------------------------------------------------------------------- |
|
||||
| **Case name** | As usual. |
|
||||
| **Workspace path** | A real host directory, bind-mounted into the container at the **same absolute path**. |
|
||||
| **Host ID** | A reusable profile (image, network, resources). Share one across cases to share settings. |
|
||||
| **Network** | `bridge` (internet on, default), `none` (fully isolated), or a custom bridge. |
|
||||
| **Advanced** | Memory and CPU caps, host credential seeding, and whether to resume the last conversation on relaunch. |
|
||||
|
||||
The same-absolute-path bind mount is what keeps the File Viewer, attachments, and watchers
|
||||
operating on real host bytes rather than a copy.
|
||||
|
||||
## One container per case
|
||||
|
||||
Exactly one long-lived container per case, shared by every session in it.
|
||||
|
||||
- Killing one session kills only that session's in-container tmux. Siblings keep running and
|
||||
the container stays up.
|
||||
- Reconnecting after a Codeman restart lands back in the same live agent.
|
||||
- A container stop or a host reboot restarts the container and **resumes the last
|
||||
conversation** from the bind-mounted transcript.
|
||||
- Deleting the case removes the container. The workspace on the host survives.
|
||||
|
||||
## Credentials
|
||||
|
||||
Your existing host logins work inside the container without logging in again. Credentials
|
||||
are **seeded**: mounted read-only and copied in once at launch, so in-container CLIs never
|
||||
write refreshed tokens back to your host credential stores. Onboarding and trust prompts are
|
||||
pre-answered so no wizard appears.
|
||||
|
||||
Turn seeding **off** for a sealed sandbox: no host credentials, and with `network: none`, no
|
||||
outbound access either. That is the profile for genuinely untrusted work; you log in inside
|
||||
the container instead.
|
||||
|
||||
Bind mounts are excluded from image capture, so exports stay secret-free.
|
||||
|
||||
One consequence worth knowing: Pi's credentials are seeded per file rather than as a whole
|
||||
directory, because that directory also holds sessions, extensions, and installed packages,
|
||||
which can be gigabytes. So in-container Pi sessions are invisible from the host, and `pi -c`
|
||||
inside a docker case sees only that container's history.
|
||||
|
||||
## Isolation
|
||||
|
||||
Every container runs hardened by default:
|
||||
|
||||
- `--cap-drop ALL`
|
||||
- `--security-opt no-new-privileges`
|
||||
- Non-root, running as your host uid so workspace files stay host-owned
|
||||
- PID limit, memory cap with swap pinned to it, `--init`
|
||||
- **Never** `--privileged`, and **never** the docker socket
|
||||
|
||||
Rootless engines without cgroup-v2 systemd delegation cannot enforce resource caps; linking
|
||||
such a host warns that the caps are advisory.
|
||||
|
||||
## Configuration drift is refused, not ignored
|
||||
|
||||
Editing a docker host's configuration (image, memory, network) after a container exists is
|
||||
detected on the next launch by comparing a configuration hash against the container's label.
|
||||
A mismatch **refuses the launch** and offers to recreate rather than silently running with
|
||||
stale configuration.
|
||||
|
||||
Recreating is refused while sessions of that case are live. The workspace and the
|
||||
conversation both survive it.
|
||||
|
||||
## Moving a case to another machine
|
||||
|
||||
**Export**, from the Docker tab:
|
||||
|
||||
| Option | Contents |
|
||||
| -------------------------- | ------------------------------------------------------------------------- |
|
||||
| **Full image + workspace** | The whole toolchain, installed packages, and files, in one `.tgz`. |
|
||||
| **Workspace only** | Just the project files. Fast and small. |
|
||||
|
||||
The container is paused across the capture so image and workspace are consistent, free space
|
||||
is checked first, and the intermediate image is cleaned up. Exports run in the background
|
||||
and notify you when the bundle is ready.
|
||||
|
||||
**Import** on the other machine: copy the `.tgz` into `~/.codeman/docker-exports/` and
|
||||
import it into a new case. The manifest and per-member checksums are verified, the workspace
|
||||
tar is extracted with a traversal guard, and the image is loaded under a **quarantined tag**
|
||||
so it can never overwrite a local image. The destination supplies its own credentials, so
|
||||
nothing secret crosses machines.
|
||||
|
||||
## Hooks need to reach the server
|
||||
|
||||
In-container hooks (permission events, idle and stop notifications) call back to Codeman
|
||||
over the docker bridge gateway. If Codeman binds **loopback only**, which is the default and
|
||||
the production configuration, the container cannot reach it and **in-container hooks do not
|
||||
fire**.
|
||||
|
||||
The session still works fully: idle detection falls back to output-based detection through
|
||||
the exec PTY, and with permission prompts skipped there is nothing to forward anyway.
|
||||
|
||||
To enable them:
|
||||
|
||||
```bash
|
||||
CODEMAN_DOCKER_BRIDGE_HOOKS=1
|
||||
```
|
||||
|
||||
Codeman then starts a second listener bound to the docker bridge gateway that serves **only**
|
||||
the hook endpoints and rejects everything else with a 403. The bridge is host-internal, so
|
||||
this does not widen your network exposure. Add it to the service unit and restart.
|
||||
|
||||
## Limits
|
||||
|
||||
- Per-session environment overrides, effort, and per-CLI configuration are **rejected** for
|
||||
docker cases, because they do not cross into the container. Configure the container through
|
||||
the docker host's per-mode command override instead.
|
||||
- tmux must exist in the base image. It is a hard prerequisite and is probed when linking a
|
||||
host.
|
||||
- On macOS, Docker Desktop takes a dedicated uid path, and memory caps are subject to the
|
||||
VM's own ceiling.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Core Concepts](Core-Concepts) - why this is an overlay rather than a run mode.
|
||||
- [Security](Security) - where containers fit in the model.
|
||||
- [Remote SSH Sessions](Remote-SSH-Sessions) - the other overlay.
|
||||
- [`docs/docker-cases.md`](https://github.com/Ark0N/Codeman/blob/master/docs/docker-cases.md) - the full reference.
|
||||
@@ -0,0 +1,188 @@
|
||||
# Driving Codeman From An Agent
|
||||
|
||||
Everything the dashboard does is HTTP, so an agent can do it too. This page is for the case
|
||||
that makes Codeman interesting: **Claude Code running inside a Codeman session, spawning and
|
||||
supervising other sessions.**
|
||||
|
||||
Two routes. Start with the skill.
|
||||
|
||||
## The agent skill
|
||||
|
||||
A Claude Code skill that teaches the agent the whole API, so you ask in plain English
|
||||
instead of pasting endpoint documentation into prompts.
|
||||
|
||||
### Install it
|
||||
|
||||
| How | Command | Scope |
|
||||
| ------------ | ----------------------------------------------------------- | ----------------------------------------------------------- |
|
||||
| Skills CLI | `npx skills add Ark0N/Codeman --skill codeman -g` | Global, any skills-aware agent. |
|
||||
| Bundled CLI | `codeman skill install` | Global, at `~/.claude/skills/codeman`. |
|
||||
| Bundled CLI | `codeman skill install --case <name>` | One case. |
|
||||
| Web UI | **App Settings → Agents & CLIs → Claude → Agent Skill** | Injects into each case when a Claude session is created. Off by default. |
|
||||
|
||||
`codeman skill uninstall [--case <name>]` reverses the CLI installs, and never touches a
|
||||
`skills/codeman` you wrote yourself.
|
||||
|
||||
### Then just ask
|
||||
|
||||
| You say | What happens |
|
||||
| --------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------ |
|
||||
| "What sessions are running right now?" | Lists them with name, mode, and status. Read-only. |
|
||||
| "Start a shell worker on the `myapp` case, run the test suite, tell me if it passes." | Spawns, waits on a completion marker, reads the exit code, cleans up. |
|
||||
| "Spin up 3 workers for lint, typecheck and tests, run them in parallel, report failures." | One session per task, all started first, then gathered as each finishes. |
|
||||
| "Have a claude worker summarize `src/session.ts`, then close it." | Spawns, runs the readiness ladder, sends and waits, reads the answer, deletes the session. |
|
||||
| "Watch session w4 and tell me if it gets stuck on a permission prompt." | Blocks on the `blocked` signal and surfaces the question to **you**. |
|
||||
|
||||
Sessions the agent creates get deleted when it is done. You can watch the tabs appear and
|
||||
disappear in the dashboard while it works.
|
||||
|
||||
### What it will and will not do
|
||||
|
||||
- **It self-gates.** Outside a Codeman session it refuses to act and does not guess an API
|
||||
URL, so a global install costs an unrelated Claude Code session nothing.
|
||||
- **Unprompted, it may only** spawn sessions, prompt them, and delete ones **it created in
|
||||
that conversation, by exact id**, behind a guard that refuses to delete the agent's own
|
||||
session.
|
||||
- **It will not** answer another session's permission prompt on your behalf. It surfaces the
|
||||
question instead.
|
||||
- **Deleting a case** (which erases a real directory of your code), bulk kills, respawn,
|
||||
Ralph, cron, orchestrator, and settings writes all require you to ask, naming the target.
|
||||
|
||||
Turning the setting back off **does not remove already-injected copies**, because a
|
||||
create-time sweep would yank the skill out from under other live sessions sharing that
|
||||
directory. Remove them per case with `codeman skill uninstall --case <name>`.
|
||||
|
||||
The skill ships with the verb index always loaded, plus on-demand references for the verbs,
|
||||
worked multi-worker recipes, endpoint tables, and cross-session messaging.
|
||||
|
||||
## The manual path
|
||||
|
||||
The same operations as raw HTTP, for a CI bot, a shell script, or an agent without skill
|
||||
support.
|
||||
|
||||
### Detect that you are inside Codeman
|
||||
|
||||
These are set in every managed session. Read them rather than hardcoding anything:
|
||||
|
||||
| Variable | Meaning |
|
||||
| -------------------------- | ------------------------------------------------------------------------ |
|
||||
| `CODEMAN_MUX=1` | You are in a managed tmux session. Never `tmux kill-session`, `pkill claude`, or `pkill tmux`: you will kill yourself or a sibling. |
|
||||
| `CODEMAN_API_URL` | Base URL, with the correct scheme. |
|
||||
| `CODEMAN_SESSION_ID` | Your own session id. Use it to avoid acting on yourself. |
|
||||
| `CODEMAN_HOOK_SECRET_FILE` | Path to the hook secret. |
|
||||
|
||||
### Rules of the road
|
||||
|
||||
Read these before writing any code. Each one has cost somebody an afternoon.
|
||||
|
||||
1. **Input is single line and must end with `\r`.** Enter fires only when the payload
|
||||
contains a carriage return. Without it the text sits unsubmitted on the prompt, the
|
||||
request still succeeds, and a combined wait burns its full timeout on a turn that never
|
||||
started. Embedded newlines are stripped rather than rejected, so `"echo A\necho B\r"` runs
|
||||
the joined `echo Aecho B`. One line per call.
|
||||
2. **Make input idempotent.** Send a stable `clientId` and a monotonic per-session `seq`. The
|
||||
server deduplicates, so a retry after a dropped connection cannot double-deliver.
|
||||
3. **Auth.** With `CODEMAN_PASSWORD` set, use HTTP Basic or the session cookie. A missing
|
||||
`Origin` is allowed, so plain curl works. A `401` replies with the bare string
|
||||
`Unauthorized`, **not** the JSON envelope, so piping it into `jq` throws a parse error
|
||||
instead of showing the failure. Check the status before parsing.
|
||||
4. **Envelope.** Most endpoints return `{ "success": true, "data": ... }`. A few legacy GETs
|
||||
return bare bodies, so handle both: `body.data ?? body`.
|
||||
5. **Wait instead of polling, and a timeout is not an error.** The wait endpoints answer
|
||||
`200` with `wait.timedOut: true`. Loop over short waits rather than one long call, because
|
||||
tunnels cut idle connections.
|
||||
6. **Only `claude` sessions emit `stop` and `blocked`.** They come from Claude Code hooks.
|
||||
Shell and the external CLIs accept only `idle`, `working`, and `exit`; asking for `stop`
|
||||
explicitly there is a `400`, while omitting `until` is always safe. On a shell session
|
||||
`idle` fires **once at startup and never again**, so synchronize hook-less sessions with an
|
||||
output marker instead.
|
||||
7. **Nothing reports "ready", so wait for it explicitly.** A new session answers
|
||||
`{"signal":"exit","immediate":true}` until its PID exists, and that means *not started*,
|
||||
not *crashed*. A Claude worker in a fresh case then sits on the CLI's trust dialog; prompt
|
||||
it there and the wait resolves on idle in about two seconds looking exactly like a finished
|
||||
turn, while your text sits stuck in the dialog.
|
||||
|
||||
### Recipes
|
||||
|
||||
```bash
|
||||
API="${CODEMAN_API_URL:-http://localhost:3000}"
|
||||
# Add -u admin:"$CODEMAN_PASSWORD" if a password is set, and -k on an HTTPS install.
|
||||
|
||||
# What is running
|
||||
curl -s "$API/api/sessions" | jq '.data[] | {id, name, mode, status}'
|
||||
|
||||
# Spawn a worker in a case
|
||||
curl -s -X POST "$API/api/quick-start" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"caseName":"myapp","mode":"shell"}' | jq
|
||||
|
||||
# Send a prompt (note the \r)
|
||||
curl -s -X POST "$API/api/sessions/$ID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"run the tests\r","clientId":"my-agent","seq":1}' | jq
|
||||
|
||||
# Send and block until the turn finishes (registers the wait BEFORE writing)
|
||||
curl -s -X POST "$API/api/sessions/$ID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"summarize src/session.ts\r","wait":["stop"],"waitTimeout":120000}' | jq
|
||||
|
||||
# Or wait for a marker in the output, which works on shell sessions too
|
||||
curl -s "$API/api/sessions/$ID/wait-output?contains=DONE_17909&from=buffer" | jq
|
||||
|
||||
# Read the terminal back
|
||||
curl -s "$API/api/sessions/$ID/terminal?tail=4000" | jq -r '.data.output'
|
||||
|
||||
# Clean up, by exact id
|
||||
curl -s -X DELETE "$API/api/sessions/$ID" | jq
|
||||
```
|
||||
|
||||
Use `POST /api/quick-start` rather than `POST /api/sessions` when a case might be remote:
|
||||
the plain create endpoint validates the working directory locally and has no case concept.
|
||||
|
||||
### The split-marker trick
|
||||
|
||||
For hook-less sessions, synchronize on a marker in the output. The catch: your own
|
||||
keystrokes echo into the output stream, so an unsplit marker matches **before the command
|
||||
has run**.
|
||||
|
||||
Split it so the typed line never contains the string you are waiting for:
|
||||
|
||||
```bash
|
||||
M=DONE; R=17909
|
||||
# typed: echo ${M}_${R} → output contains DONE_17909, the typed line does not
|
||||
```
|
||||
|
||||
Make it unique per call, because tmux repaints replay old screen text.
|
||||
|
||||
### Reading output
|
||||
|
||||
Use `terminal?tail=`, not `/output`. The latter's text field is empty for every tmux-backed
|
||||
session, which is every interactive session. `tail` counts **bytes**, and what comes back is
|
||||
terminal data with ANSI sequences included.
|
||||
|
||||
## Fan-out, and why it needs care
|
||||
|
||||
Wait signals are **edge triggered with no history**. A signal that fires with no waiter
|
||||
registered is unobservable afterwards.
|
||||
|
||||
So a fan-out must register its waits before or as it dispatches: use send-and-wait per
|
||||
worker, or latched output markers. Dispatching all the workers and then waiting on them one
|
||||
at a time loses the signals of everyone who finished early.
|
||||
|
||||
Send-and-wait registers the waiter **before** the write for the same reason. A separate POST
|
||||
followed by a wait races, and reports the previous turn's state.
|
||||
|
||||
## Lineage
|
||||
|
||||
A create request can name the session that spawned it, through a body field or a header, and
|
||||
the dashboard then draws a lineage arc from parent to child. The skill sets it automatically.
|
||||
|
||||
It is resolved rather than trusted: an unresolvable parent is dropped silently rather than
|
||||
failing the spawn, because a cosmetic field must never break a worker.
|
||||
|
||||
## Read next
|
||||
|
||||
- [HTTP API](HTTP-API) - the endpoint map and the envelope.
|
||||
- [Hooks And Integrations](Hooks-And-Integrations) - events flowing the other way.
|
||||
- [Watching Agents Work](Watching-Agents-Work) - seeing the fan-out in the UI.
|
||||
- [`skills/codeman/SKILL.md`](https://github.com/Ark0N/Codeman/blob/master/skills/codeman/SKILL.md) - the skill itself.
|
||||
@@ -0,0 +1,250 @@
|
||||
# FAQ
|
||||
|
||||
The questions that keep arriving in
|
||||
[Discussions](https://github.com/Ark0N/Codeman/discussions) and issues. For "why is it
|
||||
doing that", go to [Troubleshooting](Troubleshooting) instead.
|
||||
|
||||
## The basics
|
||||
|
||||
### What is Codeman, in one sentence?
|
||||
|
||||
A self-hosted dashboard that runs AI coding agents in persistent tmux sessions on your own
|
||||
machine and lets you drive them from any browser, including a phone.
|
||||
|
||||
### Is it free? What is the licence?
|
||||
|
||||
MIT, free, and open source. There is no paid tier and no account.
|
||||
|
||||
### Do I need an API key?
|
||||
|
||||
No. Codeman drives agent CLIs you have already installed and logged in yourself. Whatever
|
||||
subscription or key that CLI uses is what pays for the tokens. Codeman never collects,
|
||||
stores, or refreshes your credentials.
|
||||
|
||||
### Does Codeman send my code or prompts anywhere?
|
||||
|
||||
No. There is no telemetry, no analytics, and no phone-home. The only network traffic
|
||||
Codeman itself makes is between your browser and your server.
|
||||
|
||||
Your agent CLI is a separate matter: Claude Code talks to Anthropic, Codex talks to OpenAI,
|
||||
and so on. That traffic is the CLI's, on your own account, exactly as it would be in a
|
||||
terminal.
|
||||
|
||||
Two features do send data outward, both off by default and both stated where they appear:
|
||||
voice dictation through your own Claude login, and the Read My Mind prediction call.
|
||||
|
||||
### Does it work on Windows?
|
||||
|
||||
Through WSL2. Codeman requires tmux. Install it inside WSL, run your agent CLI inside WSL,
|
||||
and `http://localhost:3000` works from your Windows browser. Work in the Linux filesystem
|
||||
rather than `/mnt/c/...`, which is dramatically slower for file watching and git.
|
||||
|
||||
### Is there a mobile app?
|
||||
|
||||
The web UI is built for phones and installs as a PWA. There is no App Store or Play Store
|
||||
app.
|
||||
|
||||
## Sessions and persistence
|
||||
|
||||
### Do my agents keep running when I close the browser?
|
||||
|
||||
Yes. Agents run in tmux on the server, not in your browser. Close the tab, close the laptop,
|
||||
lose the network. When you come back, the session is still there with its scrollback.
|
||||
|
||||
The same holds when the Codeman server itself restarts. What does end a session is killing
|
||||
the tmux server or rebooting the machine.
|
||||
|
||||
### What happens after a reboot?
|
||||
|
||||
tmux dies with the machine, so the sessions are gone. Conversations are not: Claude
|
||||
transcripts persist on disk, and the welcome screen's **Resume Conversation** list picks
|
||||
them back up. Install Codeman as a service and the server itself comes back on boot.
|
||||
|
||||
### How many sessions can I run at once?
|
||||
|
||||
The design target is 20 sessions and 50 agent windows at 60fps. The hard cap is higher, and
|
||||
what you will actually hit first is the CPU and memory of the machine running the agents.
|
||||
|
||||
### Can I run Claude Code and Codex side by side?
|
||||
|
||||
Yes, that is a normal setup. The run mode is per session, so one case can have a Claude tab,
|
||||
a Codex tab, and a shell tab open at the same time, each with its own colour. Some Codeman
|
||||
features are Claude-only; [Agent CLIs](Agent-CLIs) lists exactly which.
|
||||
|
||||
### Can I attach to a session from a terminal instead of the browser?
|
||||
|
||||
Yes. `sc` is an interactive chooser (`sc 2` attaches directly, `sc -l` lists), or use tmux
|
||||
directly on the `codeman` socket. Detach with `Ctrl+A D`.
|
||||
|
||||
## Running unattended
|
||||
|
||||
### I hit my Claude usage limit overnight. Can Codeman resume automatically?
|
||||
|
||||
Yes, and it is the reason the feature exists. Turn on auto-resume at the top of the Respawn
|
||||
tab for that session. When Claude halts on a subscription limit, Codeman parses the reset
|
||||
time from the message, waits until two minutes past it, and continues the conversation.
|
||||
|
||||
Respawn cycles are blocked while a session is limit-paused, which is what stops a `/clear`
|
||||
from wiping the conversation you are waiting to resume. Claude-only.
|
||||
|
||||
### Will it keep prompting my agent forever?
|
||||
|
||||
Only if you configure it to. Respawn cycling is per session and off unless you turn it on,
|
||||
and it has presets ranging from a 60 minute solo session to an 8 hour overnight run. There
|
||||
are circuit breakers to stop a thrashing session from spinning indefinitely. See
|
||||
[Keeping Agents Running](Keeping-Agents-Running).
|
||||
|
||||
### Does an idle session cost tokens?
|
||||
|
||||
No. An idle agent is a process waiting for input. Tokens are spent when a turn runs, so what
|
||||
costs money is the re-prompting you configured, not the session sitting there.
|
||||
|
||||
### Can I schedule work for a specific time?
|
||||
|
||||
Yes. [Cron Jobs](Cron-Jobs) saves named jobs on a `once`, `interval`, `daily`, or `weekly`
|
||||
schedule; each spins up a session and sends a prompt when due, with per-job run history.
|
||||
|
||||
## Access
|
||||
|
||||
### How do I reach Codeman from my phone when I am away from home?
|
||||
|
||||
Tailscale is the recommended answer: your devices join a private network, Codeman keeps its
|
||||
loopback bind, and you get real HTTPS. The installer sets it up, and `install.sh tailscale`
|
||||
retrofits it onto an existing install.
|
||||
|
||||
A Cloudflare tunnel gives a public URL faster, and requires `CODEMAN_PASSWORD`. Full
|
||||
comparison in [Remote Access](Remote-Access).
|
||||
|
||||
### Why can't other devices reach Codeman?
|
||||
|
||||
Because the default bind is `127.0.0.1`, on purpose. Codeman starts agents with permission
|
||||
prompts skipped, so whoever reaches the dashboard can run code on your machine. Exposing it
|
||||
is a deliberate step, and [Remote Access](Remote-Access) covers the safe ways.
|
||||
|
||||
### My reverse proxy domain is rejected with `403 host not allowed`
|
||||
|
||||
The always-on Host-header allowlist blocks DNS rebinding, and it does not know your domain.
|
||||
Add it:
|
||||
|
||||
```bash
|
||||
CODEMAN_ALLOWED_HOSTS='codeman.example.com,.internal.example.com'
|
||||
```
|
||||
|
||||
A leading dot matches subdomains. Also make sure the proxy forwards WebSocket upgrades.
|
||||
|
||||
### Do I have to type a password on my phone?
|
||||
|
||||
No. Scan the QR code shown on the desktop dashboard. Tokens are single use and rotate every
|
||||
60 seconds. The password remains the fallback.
|
||||
|
||||
## Multiple people, multiple instances
|
||||
|
||||
### Can several people share one Codeman?
|
||||
|
||||
Yes, with `codeman web --multiuser`. Each person gets a login and their own case space, and
|
||||
sessions, cases, search, and events are scoped to their owner.
|
||||
|
||||
Be clear about what that is: it separates **workspaces**, not operating system accounts.
|
||||
Every session still runs as the same OS user, so a determined user's agent can reach another
|
||||
user's files. For real isolation, pair users with Docker cases or run separate instances
|
||||
under separate OS accounts. See [Multi-User Mode](Multi-User-Mode).
|
||||
|
||||
### How do I run a second instance, a beta beside my main one?
|
||||
|
||||
Give it its own instance name, which scopes the data directory and the tmux socket together:
|
||||
|
||||
```bash
|
||||
CODEMAN_INSTANCE=beta CODEMAN_PORT=5000 codeman web
|
||||
```
|
||||
|
||||
Do not skip this. The data directory and tmux socket are process wide, so a second server on
|
||||
the defaults discovers and attaches your live sessions.
|
||||
|
||||
## Updating and maintenance
|
||||
|
||||
### What is the right way to update Codeman?
|
||||
|
||||
| Install route | Update with |
|
||||
| ------------- | --------------------------------------------------------------------------- |
|
||||
| Installer | Re-run the install one-liner, or **App Settings → System → Updates**. |
|
||||
| npm | `npm update -g aicodeman` |
|
||||
| git clone | `git pull && npm install && npm run build`, then restart the service. |
|
||||
|
||||
The in-app updater covers git-clone installs supervised by systemd or launchd. It stashes a
|
||||
dirty tree rather than discarding it, and streams progress across the restart. npm installs
|
||||
report as non-updatable.
|
||||
|
||||
### Will updating kill my running sessions?
|
||||
|
||||
No. Sessions live in tmux, so restarting the server reattaches to them.
|
||||
|
||||
### Where is my data?
|
||||
|
||||
Everything under `~/.codeman/`, with cases created from scratch in `~/codeman-cases/`.
|
||||
Nothing needs root and nothing leaves the machine. Uninstalling does not delete either
|
||||
directory.
|
||||
|
||||
## Features
|
||||
|
||||
### What is the difference between respawn, Ralph, and the orchestrator?
|
||||
|
||||
- **Respawn** restarts a session's CLI when it goes idle, to keep a long run going. It is the
|
||||
one most people want.
|
||||
- **Ralph loop** is an autonomous single-session task loop with its own tracker.
|
||||
- **Orchestrator** turns one goal into a phased plan and drives it across agents.
|
||||
|
||||
[Keeping Agents Running](Keeping-Agents-Running) and [Autonomous Loops](Autonomous-Loops)
|
||||
cover them properly.
|
||||
|
||||
### Can agents start and supervise other agents?
|
||||
|
||||
Yes. Codeman ships an agent skill that lets an agent inside a session drive the HTTP API:
|
||||
list sessions, spawn workers, send prompts, and block until a worker's turn finishes. It is
|
||||
off by default and enabled per case.
|
||||
|
||||
See [Driving Codeman From An Agent](Driving-Codeman-From-An-Agent).
|
||||
|
||||
### Can I run a case in a container?
|
||||
|
||||
Yes. One container per case, shared by all its sessions, non-root and capability-dropped by
|
||||
default, with your host CLI logins seeded in so nothing asks you to log in again. You can
|
||||
export a container plus its workspace and move it to another machine. See
|
||||
[Docker Cases](Docker-Cases).
|
||||
|
||||
### Can the agent run on a different machine?
|
||||
|
||||
Yes. Point a case at a remote host over SSH and the agent runs there, inside a durable
|
||||
remote tmux, so a dropped connection does not kill the run. See
|
||||
[Remote SSH Sessions](Remote-SSH-Sessions).
|
||||
|
||||
### Can I put my Grafana or other dashboards in here?
|
||||
|
||||
Yes. Saved URLs render as tabs beside your sessions, proxied through Codeman's own origin so
|
||||
that mixed content and frame-blocking headers do not break them. See [Web Tabs](Web-Tabs).
|
||||
|
||||
### Why is a feature I read about not on screen?
|
||||
|
||||
Most of Codeman's UI is opt-in and defaults to off, so a stock install stays small. Check
|
||||
**App Settings → Header & Panels**. [Settings Reference](Settings-Reference) lists the
|
||||
defaults.
|
||||
|
||||
## Contributing
|
||||
|
||||
### How do I request a feature?
|
||||
|
||||
Open an [Idea](https://github.com/Ark0N/Codeman/discussions/categories/ideas) and it gets
|
||||
voted on. Roadmap decisions happen there.
|
||||
|
||||
### How do I contribute code?
|
||||
|
||||
[CONTRIBUTING.md](https://github.com/Ark0N/Codeman/blob/master/.github/CONTRIBUTING.md) has
|
||||
the full map. Small fixes can go straight to a PR; anything larger starts as an issue or
|
||||
Discussion so the design gets a nod first. Skins, translations, and docs are good first
|
||||
contributions.
|
||||
|
||||
### How do I fix a mistake in this wiki?
|
||||
|
||||
These pages are generated from
|
||||
[`docs/wiki/`](https://github.com/Ark0N/Codeman/tree/master/docs/wiki) in the main
|
||||
repository. Editing a page in the browser gets overwritten on the next sync, so send a PR
|
||||
against that directory instead.
|
||||
@@ -0,0 +1,164 @@
|
||||
# HTTP API
|
||||
|
||||
Codeman's HTTP and SSE API is a **stable contract**. Everything the dashboard does goes
|
||||
through it, so anything the dashboard can do, a script can do.
|
||||
|
||||
This page is the orientation. The complete specification, including every wait semantic and
|
||||
the SSE catalogue, is
|
||||
[`docs/api-reference.md`](https://github.com/Ark0N/Codeman/blob/master/docs/api-reference.md).
|
||||
|
||||
## What is stable
|
||||
|
||||
Covered by semantic versioning: endpoint paths under `/api/v1`, the response envelope,
|
||||
`errorCode` values, and SSE event names.
|
||||
|
||||
Not covered, and free to change in a patch release: on-disk state files, internal modules,
|
||||
and anything marked experimental. The full statement is in
|
||||
[Versioning](Versioning).
|
||||
|
||||
`/api/v1/*` is a versioned alias of `/api/*`. Prefer the versioned form in anything you
|
||||
intend to keep.
|
||||
|
||||
## The envelope
|
||||
|
||||
```json
|
||||
{ "success": true, "data": { } }
|
||||
```
|
||||
|
||||
```json
|
||||
{ "success": false, "error": "human readable", "errorCode": "NOT_FOUND" }
|
||||
```
|
||||
|
||||
A few legacy GET handlers return bare bodies rather than the envelope, so a robust client
|
||||
reads `body.data ?? body`.
|
||||
|
||||
Branch on `errorCode`, which is stable. The HTTP status is reliable too:
|
||||
|
||||
| `errorCode` | HTTP | Meaning |
|
||||
| ------------------ | ---- | ------------------------------------------------ |
|
||||
| `INVALID_INPUT` | 400 | Malformed request or failed validation. |
|
||||
| `UNAUTHORIZED` | 401 | Authentication required or failed. |
|
||||
| `NOT_FOUND` | 404 | No such resource. |
|
||||
| `SESSION_BUSY` | 409 | The session is busy. |
|
||||
| `CONFLICT` | 409 | Conflicts with current state. |
|
||||
| `ALREADY_EXISTS` | 409 | Resource already exists. |
|
||||
| `OPERATION_FAILED` | 422 | Well formed, could not be completed. |
|
||||
| `RATE_LIMITED` | 429 | Too many requests. |
|
||||
| `INTERNAL_ERROR` | 500 | Unexpected server error. |
|
||||
|
||||
New error codes are non-breaking. Removing or renaming one is a major change.
|
||||
|
||||
**A `401` is the bare string `Unauthorized`, not the envelope.** Piping it into `jq` throws
|
||||
a parse error rather than showing the failure, so check the status first.
|
||||
|
||||
## Authentication
|
||||
|
||||
With no password set, and the default loopback bind, there is none. With `CODEMAN_PASSWORD`
|
||||
set, use HTTP Basic or the session cookie:
|
||||
|
||||
```bash
|
||||
curl -s -u admin:"$CODEMAN_PASSWORD" "$API/api/sessions"
|
||||
```
|
||||
|
||||
A **missing** `Origin` header is allowed, so curl and CLI tools work unchanged. A
|
||||
present-but-foreign origin is rejected by the CSRF guard. On an HTTPS install with the
|
||||
self-signed certificate, add `-k`.
|
||||
|
||||
## Endpoint map
|
||||
|
||||
Roughly 200 handlers across 24 route modules. By domain:
|
||||
|
||||
| Domain | Handlers | Covers |
|
||||
| ------------------- | -------- | --------------------------------------------------- |
|
||||
| System | 45 | Status, settings, search, digest, updates. |
|
||||
| Sessions | 34 | Create, input, terminal, wait, kill. |
|
||||
| Cases | 29 | Create, link, clone, remote and docker cases. |
|
||||
| Files | 16 | Preview, edit, raw, attachments, path picker. |
|
||||
| Orchestrator | 10 | Plans and phases. |
|
||||
| Ralph | 9 | Loop control and configuration. |
|
||||
| Cron | 9 | Jobs and run history. |
|
||||
| Admin | 8 | Multi-user administration. |
|
||||
| Plan | 8 | Plan orchestration. |
|
||||
| Respawn | 7 | Respawn configuration and presets. |
|
||||
| Webviews | 6 | Saved dashboards, plus the proxy. |
|
||||
| Mux | 5 | tmux operations. |
|
||||
| Push | 4 | Web push subscriptions. |
|
||||
| Read My Mind | 4 | Intent profiles and prediction. |
|
||||
| Scheduled | 4 | The legacy scheduled-run concept. |
|
||||
| Approvals | 3 | The inbox and answering. |
|
||||
| Teams, me, search, hooks, clipboard, telemetry, voice, ws | 1-2 each | |
|
||||
|
||||
Each route module documents its own endpoints in its file header.
|
||||
|
||||
## Long-polling instead of polling
|
||||
|
||||
Three calls block until something happens, so an agent driving Codeman from a shell can wait
|
||||
rather than spin:
|
||||
|
||||
| Call | Blocks until |
|
||||
| ----------------------------------------- | -------------------------------------------------------- |
|
||||
| `GET /api/v1/sessions/:id/wait` | One of a set of lifecycle signals fires. |
|
||||
| `GET /api/v1/sessions/:id/wait-output` | A literal string appears in the session's output. |
|
||||
| `POST /api/v1/sessions/:id/input` + `wait`| The input is delivered **and then** a signal fires. |
|
||||
|
||||
Three semantics that break callers who assume otherwise:
|
||||
|
||||
1. **A timeout is `200`, not an error.** It answers with `wait.timedOut: true`. Loop over
|
||||
short waits; a single long call gets cut by tunnels and proxies.
|
||||
2. **Send-and-wait is not a POST followed by a wait.** It registers the waiter *before*
|
||||
writing, which closes the window where a separate wait sees the session still idle from
|
||||
the previous turn and answers instantly about the wrong turn.
|
||||
3. **Signals are edge triggered with no history.** One that fires with no waiter registered
|
||||
is unobservable afterwards. Fan-outs must register their waits as they dispatch.
|
||||
|
||||
`wait-output` matches a **literal substring, never a regex.** That is deliberate: no regex
|
||||
means no catastrophic backtracking on attacker-influenced output.
|
||||
|
||||
Only `claude` sessions emit `stop` and `blocked`, because those come from Claude Code hooks.
|
||||
Shell and external CLI sessions accept `idle`, `working`, and `exit`.
|
||||
|
||||
## SSE
|
||||
|
||||
`GET /api/events` is the live event stream. 155 event names, kept in sync between server and
|
||||
client with a test that fails on drift.
|
||||
|
||||
The heartbeat is a **named** `sse:heartbeat` event rather than an SSE comment, because
|
||||
comments are invisible to `EventSource` by specification and a client could not observe
|
||||
them. That is what lets the browser detect a stream that has silently stopped delivering.
|
||||
|
||||
```js
|
||||
const es = new EventSource('/api/events');
|
||||
es.addEventListener('session:created', (e) => console.log(JSON.parse(e.data)));
|
||||
```
|
||||
|
||||
## Quick examples
|
||||
|
||||
```bash
|
||||
API="${CODEMAN_API_URL:-http://localhost:3000}"
|
||||
|
||||
curl -s "$API/api/status" | jq # whole-system snapshot
|
||||
curl -s "$API/api/sessions" | jq '.data[].name' # live sessions
|
||||
curl -s "$API/api/sessions/unified" | jq # live + historical, deduped
|
||||
curl -s "$API/api/subagents" | jq # background agents
|
||||
curl -s "$API/api/search?q=deploy" | jq # cross-session search
|
||||
```
|
||||
|
||||
## Limits
|
||||
|
||||
| Limit | Default |
|
||||
| --------------------- | ------------------------------------------ |
|
||||
| Max sessions | 50 |
|
||||
| Max agent windows | 500 |
|
||||
| Max SSE clients | 100 |
|
||||
| Terminal buffer | 32 MB per session |
|
||||
| Text payload | 1 MB |
|
||||
| Wait timeout ceiling | 600 s, and the response tells you what was applied |
|
||||
|
||||
Most are environment-overridable. See `src/config/`.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Driving Codeman From An Agent](Driving-Codeman-From-An-Agent) - the practical version, with recipes.
|
||||
- [Hooks And Integrations](Hooks-And-Integrations) - events flowing back into Codeman.
|
||||
- [Versioning](Versioning) - what the version number promises.
|
||||
- [`docs/api-reference.md`](https://github.com/Ark0N/Codeman/blob/master/docs/api-reference.md) - the full specification.
|
||||
@@ -0,0 +1,136 @@
|
||||
<p align="center">
|
||||
<img src="https://raw.githubusercontent.com/Ark0N/Codeman/master/docs/images/codeman-title.svg" alt="Codeman" height="56">
|
||||
</p>
|
||||
|
||||
<h3 align="center">Mission control for AI coding agents</h3>
|
||||
|
||||
Codeman runs your coding agents on your own machine and puts them behind one dashboard you
|
||||
can open from any device. It spawns Claude Code, OpenCode, Codex, Antigravity, Gemini, or
|
||||
Pi inside persistent tmux sessions, streams the real terminal to the browser, and keeps
|
||||
working while you are away from the keyboard: it re-prompts idle agents, resumes when a
|
||||
subscription limit resets, runs jobs on a schedule, and shows every background subagent
|
||||
live.
|
||||
|
||||
This wiki is the manual. The [README](https://github.com/Ark0N/Codeman) is the overview,
|
||||
and the deep internals live in
|
||||
[`docs/`](https://github.com/Ark0N/Codeman/tree/master/docs).
|
||||
|
||||
```bash
|
||||
curl -fsSL https://getcodeman.com/install | bash
|
||||
codeman web # then open http://localhost:3000
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Start here
|
||||
|
||||
**New to Codeman**
|
||||
|
||||
1. [Installation](Installation) - requirements, the installer, npm and git clone routes, updating.
|
||||
2. [Quick Start](Quick-Start) - from a running server to a working agent in five minutes.
|
||||
3. [Core Concepts](Core-Concepts) - cases, sessions, run modes, and what survives a restart.
|
||||
4. [The Dashboard](The-Dashboard) - reading the tab strip, the status dots, and the alerts.
|
||||
|
||||
**Already running it**
|
||||
|
||||
- [Agent CLIs](Agent-CLIs) - the seven run modes, their setup, and which features are Claude-only.
|
||||
- [Mobile Guide](Mobile-Guide) - phone and tablet use, QR login, the touch keyboard bar.
|
||||
- [Remote Access](Remote-Access) - Tailscale, Cloudflare tunnel, LAN plus password, QR login.
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - idle detection, respawn cycling, auto-resume on usage limits.
|
||||
- [Troubleshooting](Troubleshooting) - symptom-first index of things that actually break.
|
||||
|
||||
**Driving it from code**
|
||||
|
||||
- [Driving Codeman From An Agent](Driving-Codeman-From-An-Agent) - the bundled skill, worker sessions, wait primitives.
|
||||
- [HTTP API](HTTP-API) - the envelope, auth, the endpoint map, SSE events.
|
||||
- [Hooks And Integrations](Hooks-And-Integrations) - events flowing back into Codeman.
|
||||
|
||||
---
|
||||
|
||||
## Everything in the manual
|
||||
|
||||
### Getting started
|
||||
|
||||
| Page | What it answers |
|
||||
| ------------------------------- | --------------------------------------------------- |
|
||||
| [Installation](Installation) | How do I install it, update it, and remove it? |
|
||||
| [Quick Start](Quick-Start) | How do I get one agent working right now? |
|
||||
| [Core Concepts](Core-Concepts) | What is a case, a session, a run mode? |
|
||||
|
||||
### Using it
|
||||
|
||||
| Page | What it answers |
|
||||
| ------------------------------------------ | ---------------------------------------------------------- |
|
||||
| [The Dashboard](The-Dashboard) | What is the UI telling me? |
|
||||
| [Agent CLIs](Agent-CLIs) | Which agent should this session run, and how do I set it up? |
|
||||
| [Working With Files](Working-With-Files) | How do I read, edit, and attach files? |
|
||||
| [Input And Voice](Input-And-Voice) | How do I talk to an agent, including by voice? |
|
||||
| [Mobile Guide](Mobile-Guide) | How well does this work on a phone? |
|
||||
| [Keyboard Shortcuts](Keyboard-Shortcuts) | What can I drive from the keyboard? |
|
||||
| [Settings Reference](Settings-Reference) | What does this setting do, and why did it not follow me to my phone? |
|
||||
|
||||
### Keeping agents running
|
||||
|
||||
| Page | What it answers |
|
||||
| ------------------------------------------------------------- | --------------------------------------------------- |
|
||||
| [Keeping Agents Running](Keeping-Agents-Running) | How does it run unattended overnight? |
|
||||
| [Notifications And Approvals](Notifications-And-Approvals) | How do I know an agent needs me, and answer from my phone? |
|
||||
| [Cron Jobs](Cron-Jobs) | How do I run an agent on a schedule? |
|
||||
| [Autonomous Loops](Autonomous-Loops) | What are the Ralph and Orchestrator loops for? |
|
||||
| [Watching Agents Work](Watching-Agents-Work) | How do I see what the subagents are doing? |
|
||||
|
||||
### Where it runs
|
||||
|
||||
| Page | What it answers |
|
||||
| --------------------------------------------- | -------------------------------------------- |
|
||||
| [Docker Cases](Docker-Cases) | How do I sandbox a project in a container? |
|
||||
| [Remote SSH Sessions](Remote-SSH-Sessions) | How do I run the agent on another machine? |
|
||||
| [Web Tabs](Web-Tabs) | Can my Grafana live in here too? |
|
||||
| [Multi-User Mode](Multi-User-Mode) | Can several people share one Codeman? |
|
||||
|
||||
### Access and security
|
||||
|
||||
| Page | What it answers |
|
||||
| ------------------------------- | ---------------------------------------------------------- |
|
||||
| [Remote Access](Remote-Access) | How do I reach it from outside this machine, safely? |
|
||||
| [Security](Security) | What is exposed, what protects it, what do I have to do? |
|
||||
|
||||
### Automation and integration
|
||||
|
||||
| Page | What it answers |
|
||||
| ----------------------------------------------------------------- | -------------------------------------------- |
|
||||
| [Driving Codeman From An Agent](Driving-Codeman-From-An-Agent) | How does an agent spawn and drive workers? |
|
||||
| [HTTP API](HTTP-API) | What can I call, and what comes back? |
|
||||
| [Hooks And Integrations](Hooks-And-Integrations) | How do I wire Codeman into something else? |
|
||||
|
||||
### Operating it
|
||||
|
||||
| Page | What it answers |
|
||||
| --------------------------------------------- | -------------------------------------------------- |
|
||||
| [Running As A Service](Running-As-A-Service) | How do I keep it up across reboots, and update it? |
|
||||
| [Troubleshooting](Troubleshooting) | Why is it doing that? |
|
||||
| [FAQ](FAQ) | The questions that keep coming up. |
|
||||
| [Contributing](Contributing) | How do I send a fix? |
|
||||
| [Versioning](Versioning) | What does the version number promise? |
|
||||
|
||||
---
|
||||
|
||||
## Requirements at a glance
|
||||
|
||||
| Thing | Needed |
|
||||
| ------------ | --------------------------------------------------------------------------- |
|
||||
| OS | macOS or Linux. Windows works through WSL2. |
|
||||
| Node.js | 22 or newer. |
|
||||
| tmux | Required. Sessions live in tmux, which is what makes them survive restarts. |
|
||||
| An agent CLI | At least one of Claude Code, OpenCode, Codex, Gemini, Antigravity, Pi. Plain shell sessions need none. |
|
||||
| Network | Binds to `127.0.0.1` by default. Reaching it from another device is a deliberate step: see [Remote Access](Remote-Access). |
|
||||
|
||||
Codeman is MIT licensed, self-hosted, and sends no telemetry. Everything runs on your
|
||||
machine.
|
||||
|
||||
## Getting help
|
||||
|
||||
- **Questions and setup help**: [Discussions](https://github.com/Ark0N/Codeman/discussions), especially [Q&A](https://github.com/Ark0N/Codeman/discussions/categories/q-a).
|
||||
- **Bugs**: [Issues](https://github.com/Ark0N/Codeman/issues). Include your OS, install method, browser, and which CLI the session was running.
|
||||
- **Ideas and roadmap**: [Ideas](https://github.com/Ark0N/Codeman/discussions/categories/ideas).
|
||||
- **Security**: never a public issue. See [SECURITY.md](https://github.com/Ark0N/Codeman/blob/master/.github/SECURITY.md).
|
||||
@@ -0,0 +1,97 @@
|
||||
# Hooks and Integrations
|
||||
|
||||
Events flowing **back** into Codeman, and the four seams a third party can build against.
|
||||
|
||||
## Hooks
|
||||
|
||||
Claude Code can run a command when something happens in a session. Codeman writes a hooks
|
||||
configuration into each Claude case so those events post back to it, which is what turns a
|
||||
terminal into something that can notify you.
|
||||
|
||||
| Event | Fires when | Drives |
|
||||
| ---------------------- | ----------------------------------------------- | --------------------------------------------- |
|
||||
| `permission_prompt` | The agent asks for permission. | Red tab alert, Approvals Inbox, push. |
|
||||
| `idle_prompt` | The agent is waiting for input. | Yellow tab alert, the `idle` wait signal. |
|
||||
| `stop` | A turn ends. | The `stop` wait signal, idle detection. |
|
||||
| `elicitation_dialog` | A dialog opens. | Approvals Inbox. |
|
||||
| `elicitation_complete` | The dialog closes. | Clearing the alert. |
|
||||
| `elicitation_response` | The dialog is answered. | Clearing the alert. |
|
||||
| `teammate_idle` | An agent-team member goes idle. | Team surfaces. |
|
||||
| `task_completed` | A task finishes. | Task tracking, run summary. |
|
||||
|
||||
This is why several Codeman features are Claude-only. The other CLIs have no hook system, so
|
||||
for them Codeman watches terminal output, which reveals that something happened but not what
|
||||
it was.
|
||||
|
||||
### How hooks get installed
|
||||
|
||||
Codeman writes them into the case when a Claude session is created. Hook blocks are
|
||||
**marker-owned**: Codeman only ever updates a block it wrote, and never touches
|
||||
configuration you added yourself.
|
||||
|
||||
If tab alerts and approvals never fire in a particular case, that case is missing its hook
|
||||
block. Recreating the case rewrites it.
|
||||
|
||||
### The hook secret
|
||||
|
||||
`/api/hook-event` and `/api/status-telemetry` skip HTTP Basic authentication, because they
|
||||
are called from localhost by the CLI itself. When authentication is on, that bypass
|
||||
additionally requires a per-instance hook secret, because Codeman cannot tell a genuine
|
||||
loopback call from a request arriving through your own loopback reverse proxy.
|
||||
|
||||
The secret lives in the data directory, and its path is exported into every managed session.
|
||||
|
||||
### Two things that break hooks
|
||||
|
||||
- **HTTPS.** Hook callbacks must accept the self-signed certificate. Recent versions
|
||||
self-heal existing cases; older cases need recreating.
|
||||
- **Docker cases on a loopback bind.** A container cannot reach `127.0.0.1` on the host, so
|
||||
in-container hooks silently do not fire. Set `CODEMAN_DOCKER_BRIDGE_HOOKS=1` to open a
|
||||
hooks-only listener on the bridge gateway. See [Docker Cases](Docker-Cases).
|
||||
|
||||
## Integration seams
|
||||
|
||||
Codeman has **no plugin runtime**, and that is a decision rather than a gap. A plugin runtime
|
||||
means running third-party code inside a process that spawns agents with your credentials, on
|
||||
a server people routinely expose over a tunnel. Codeman's security posture is one of its
|
||||
reasons to exist, so it does not trade that away for an extension mechanism.
|
||||
|
||||
What exists instead is four documented seams.
|
||||
|
||||
### 1. Web tabs
|
||||
|
||||
Anything with a web UI can live inside Codeman as a tab, proxied through Codeman's own
|
||||
origin. The lowest-effort integration by a wide margin: if your tool has a dashboard, it can
|
||||
sit beside the agents with no code at all. See [Web Tabs](Web-Tabs).
|
||||
|
||||
### 2. SSE events
|
||||
|
||||
`GET /api/events` streams everything Codeman knows: session lifecycle, output, agent
|
||||
activity, approvals, cron runs. 155 named events, stable under semantic versioning.
|
||||
|
||||
This is the seam for anything that reacts. A bot that pings your chat channel when an agent
|
||||
needs a human is a short script over this stream.
|
||||
|
||||
### 3. HTTP API and CLI
|
||||
|
||||
Everything the dashboard does. Create sessions, send input, block on wait primitives, read
|
||||
terminals, manage cron. See [HTTP API](HTTP-API) and
|
||||
[Driving Codeman From An Agent](Driving-Codeman-From-An-Agent).
|
||||
|
||||
### 4. Hooks
|
||||
|
||||
The seam above, in the other direction: your own hook commands can run alongside Codeman's
|
||||
in a case, as long as you leave Codeman's marker-owned block alone.
|
||||
|
||||
## Publishing an integration
|
||||
|
||||
There is no registry to submit to. Share it in
|
||||
[Show and tell](https://github.com/Ark0N/Codeman/discussions/300), and if it needs a change
|
||||
in Codeman to work properly, open an issue or a Discussion first.
|
||||
|
||||
## Read next
|
||||
|
||||
- [HTTP API](HTTP-API) - the endpoint map and envelope.
|
||||
- [Driving Codeman From An Agent](Driving-Codeman-From-An-Agent) - the agent-facing path.
|
||||
- [`docs/extending-codeman.md`](https://github.com/Ark0N/Codeman/blob/master/docs/extending-codeman.md) - the seams in full, with examples.
|
||||
- [`docs/claude-code-hooks-reference.md`](https://github.com/Ark0N/Codeman/blob/master/docs/claude-code-hooks-reference.md) - upstream hook semantics.
|
||||
@@ -0,0 +1,135 @@
|
||||
# Input and Voice
|
||||
|
||||
Getting words into an agent: typing, dictating, and letting Codeman guess. Plus the input
|
||||
machinery that only shows up when it goes wrong.
|
||||
|
||||
## Typing
|
||||
|
||||
Click into the terminal and type. It is a real terminal, so everything the CLI supports
|
||||
works, slash commands included.
|
||||
|
||||
| Key | Effect |
|
||||
| ---------------------------- | --------------------------------------------- |
|
||||
| `Enter` | Send. |
|
||||
| `Shift+Enter` / `Ctrl+Enter` | Newline without sending. |
|
||||
| `Ctrl+C` | Copy if text is selected, otherwise interrupt. |
|
||||
| `Ctrl+Shift+C` | Copy, never interrupts. |
|
||||
| `Ctrl+L` | Clear the terminal. |
|
||||
|
||||
### Exactly-once delivery
|
||||
|
||||
Browser input goes through a durable layer rather than a plain socket write. Each prompt
|
||||
carries a stable client id and a per-session sequence number, held in local storage until
|
||||
the server acknowledges it.
|
||||
|
||||
The result is the property you want on a phone: a connection that drops mid-prompt never
|
||||
loses the prompt and never delivers it twice. Two browser tabs on the same session coexist,
|
||||
and only a reconnect from the *same* tab supersedes the old connection.
|
||||
|
||||
## Zero-lag local echo
|
||||
|
||||
On touch devices, keystrokes are painted in the terminal immediately and sent when you press
|
||||
Enter, instead of waiting for each character to round-trip to the server and back. Over a
|
||||
mobile connection that is the difference between usable and not.
|
||||
|
||||

|
||||
|
||||
The consequence to remember: **text on screen has not necessarily reached the agent yet.**
|
||||
It is flushed on Enter. If a prompt appears to have been ignored, press Enter, or the phone
|
||||
toolbar's **Enter** button.
|
||||
|
||||
Default on for touch devices, off for desktop, and switchable in
|
||||
**App Settings → Terminal & Input**.
|
||||
|
||||
### Codex is different on purpose
|
||||
|
||||
Codex's composer reacts to every keystroke: `/` opens a live-filtering picker, arrows edit
|
||||
state on its side, the composer grows as text wraps. Buffering until Enter starved it, so
|
||||
Codex sessions use **predictive echo** instead: each keystroke is painted at its predicted
|
||||
position while the bytes actually sent stay identical to what you typed. Predictions
|
||||
reconcile against the real buffer and only apply while the cursor is on the composer row.
|
||||
|
||||
## CJK input
|
||||
|
||||
Chinese, Japanese, and Korean input needs an IME, and an IME needs a real text field.
|
||||
Turning on CJK input in **App Settings → Terminal & Input** puts an always-visible textarea
|
||||
below the terminal that owns composition, then delivers the composed text to the session.
|
||||
|
||||
## Voice dictation
|
||||
|
||||
`Ctrl+Shift+V`, or the microphone button. There are three providers and the default is
|
||||
`auto`, which prefers them in this order:
|
||||
|
||||
| Provider | Needs | Notes |
|
||||
| ------------------ | ---------------------------------------------- | ------------------------------------------------------------ |
|
||||
| **Claude** | Claude Code logged in on the server. Opt-in. | Uses this machine's existing Claude login. No extra key. |
|
||||
| **Deepgram** | A Deepgram API key. | Nova-3, with automatic silence detection. |
|
||||
| **Web Speech** | Nothing. | Browser-provided, quality varies. |
|
||||
|
||||
### Dictating through your Claude login
|
||||
|
||||
Off by default; enable it in **App Settings → Voice**.
|
||||
|
||||
Claude Code has its own voice mode, but it opens the **host's** microphone, and in Codeman
|
||||
the CLI runs headless in a tmux pane while you are in a browser somewhere else entirely. So
|
||||
Codeman captures audio in your browser and borrows only the backend: audio goes browser to
|
||||
Codeman to Anthropic, and the page never sees the OAuth token.
|
||||
|
||||
Two deliberate limits:
|
||||
|
||||
- **Credentials are read only.** Codeman never refreshes your Claude token, because a
|
||||
refresh rotates the refresh token and could sign you out of your own CLI. An expired token
|
||||
is reported as expired rather than silently renewed.
|
||||
- **Capture is raw PCM** at 16 kHz mono, which requires an AudioWorklet rather than the
|
||||
usual browser recorder.
|
||||
|
||||
## Read My Mind
|
||||
|
||||
**Claude only, off by default.** Turn it on in **App Settings**, and a 🧠 button appears in
|
||||
the header (on phones, in the keyboard bar instead).
|
||||
|
||||
It keeps a per-case **intent profile**: goals you or your agent write down, plus the prompts
|
||||
you actually submitted in that case. Pressing 🧠 feeds that profile plus live session signals
|
||||
to a single model call and shows a predicted next prompt.
|
||||
|
||||
What you can do with the result:
|
||||
|
||||
- **Send** it, **Insert** it into the composer, or edit it first.
|
||||
- Pick one of the alternate suggestions, which swaps into the editable field without losing
|
||||
your edits.
|
||||
- **Rethink**, optionally with a steer note, to reject the whole set and try again.
|
||||
|
||||
**Nothing is ever sent automatically.** Every path requires a click.
|
||||
|
||||
Where the data lives: the profile is keyed by owner and the resolved working directory, so
|
||||
it survives `/clear` and respawns. Prompts can contain secrets, so the store is written
|
||||
0600 and is deliberately excluded from cross-session search.
|
||||
|
||||
Guide: [`docs/readmymind.md`](https://github.com/Ark0N/Codeman/blob/master/docs/readmymind.md).
|
||||
|
||||
## Programmatic input
|
||||
|
||||
Sending prompts over the API has one rule that catches everyone: **the payload must end with
|
||||
`\r`** or Enter is never sent. The request still succeeds, the text sits unsubmitted in the
|
||||
composer, and any wait burns its whole timeout on a turn that never started.
|
||||
|
||||
Input is also **single line**. Embedded newlines are stripped rather than rejected, so
|
||||
`"echo A\necho B\r"` runs the joined `echo Aecho B`. Put multi-line content in a file and
|
||||
tell the agent to read it.
|
||||
|
||||
See [Driving Codeman From An Agent](Driving-Codeman-From-An-Agent).
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **Typed text sitting on screen has not been sent.** Press Enter.
|
||||
- **`Ctrl+C` with a selection copies.** Clear the selection to interrupt.
|
||||
- **Voice needs HTTPS.** Microphone access requires a secure context, same as push
|
||||
notifications.
|
||||
- **Read My Mind goes blind for sessions using a relocated Claude config directory**, along
|
||||
with the other transcript-backed features. See [Agent CLIs](Agent-CLIs).
|
||||
|
||||
## Read next
|
||||
|
||||
- [Mobile Guide](Mobile-Guide) - the keyboard bar and touch input.
|
||||
- [Keyboard Shortcuts](Keyboard-Shortcuts) - the full list.
|
||||
- [Working With Files](Working-With-Files) - images and attachments as input.
|
||||
@@ -0,0 +1,234 @@
|
||||
# Installation
|
||||
|
||||
Getting Codeman onto a machine, verifying it works, updating it, and removing it.
|
||||
|
||||
## Requirements
|
||||
|
||||
| Requirement | Notes |
|
||||
| ---------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| **macOS or Linux** | Windows works through WSL2. See [Windows](#windows-wsl) below. |
|
||||
| **Node.js 22+** | The installer offers to install it if missing. |
|
||||
| **tmux** | Not optional. Sessions live inside tmux, which is what makes them survive a server restart, a dropped connection, or a closed laptop. |
|
||||
| **An agent CLI** | At least one of [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), [Pi](https://pi.dev). Plain shell sessions need none. See [Agent CLIs](Agent-CLIs). |
|
||||
|
||||
Codeman itself sends no telemetry and phones no home. The only network traffic is your
|
||||
browser to your server, and whatever the agent CLI you chose does on its own.
|
||||
|
||||
## Route A: the installer (recommended)
|
||||
|
||||
```bash
|
||||
curl -fsSL https://getcodeman.com/install | bash
|
||||
```
|
||||
|
||||
This installs Node.js and tmux if they are missing, clones Codeman into `~/.codeman/app`,
|
||||
and builds it.
|
||||
|
||||
What it asks you:
|
||||
|
||||
1. **Permission for every system change.** Package installs and agent CLI downloads are
|
||||
prompted individually. Nothing is installed silently.
|
||||
2. **How the dashboard should be reachable.** Three choices:
|
||||
- **Tailscale** (recommended for phone access): keeps the loopback bind and walks you
|
||||
through `tailscale serve`, including the tailnet HTTPS toggle, then verifies the result
|
||||
end to end.
|
||||
- **Your local network** (`0.0.0.0`): prompts for a password. Skipping the password takes
|
||||
an explicit confirmation and ends on a loud warning.
|
||||
- **This machine only** (`127.0.0.1`): the safest option, and the default for a bare
|
||||
`codeman web` regardless of what you pick here.
|
||||
|
||||
Which one is preselected depends on what the installer finds. A fresh install defaults to
|
||||
the local network, unless Tailscale is already connected, in which case it defaults to
|
||||
Tailscale. An existing loopback install defaults to keeping loopback, or to Tailscale when
|
||||
a serve mapping for Codeman is already there. A bare Enter never pulls in new software,
|
||||
and a non-interactive run always keeps the safe loopback default.
|
||||
3. **What to do when it finishes.** Run in this terminal, install as a background service
|
||||
that starts on boot, or do nothing yet.
|
||||
|
||||
Re-running the same one-liner **updates an existing install in place**. Local changes in
|
||||
`~/.codeman/app` are stashed rather than discarded, a running service is restarted and
|
||||
verified, and your existing network binding is preserved. An interrupted first install
|
||||
resumes instead of restarting.
|
||||
|
||||
Two other entry points exist:
|
||||
|
||||
```bash
|
||||
install.sh update # update only
|
||||
install.sh uninstall # remove
|
||||
install.sh tailscale # retrofit Tailscale access onto an existing install
|
||||
```
|
||||
|
||||
**Automation and CI**: with no terminal attached, any step that would change the system
|
||||
aborts with instructions instead of running silently. Set `CODEMAN_NONINTERACTIVE=1` to
|
||||
approve those steps. `CODEMAN_TAILSCALE=1` preselects the Tailscale answer, and never
|
||||
installs Tailscale itself non-interactively.
|
||||
|
||||
## Route B: npm
|
||||
|
||||
```bash
|
||||
npm install -g aicodeman
|
||||
codeman web
|
||||
```
|
||||
|
||||
The npm package is named `aicodeman`; the product is Codeman. Both `codeman` and
|
||||
`aicodeman` are installed as commands.
|
||||
|
||||
The trade-off against Route A: no guided network setup, and the in-app self-updater does
|
||||
not apply. npm installs report as non-updatable in **App Settings → System → Updates**, and
|
||||
you update with `npm update -g aicodeman`.
|
||||
|
||||
## Route C: git clone
|
||||
|
||||
For contributing, or for running unreleased code.
|
||||
|
||||
```bash
|
||||
git clone https://github.com/Ark0N/Codeman.git
|
||||
cd Codeman
|
||||
npm install # postinstall builds the vendored xterm addon bundles
|
||||
npm run dev # dev server on http://localhost:3000
|
||||
```
|
||||
|
||||
For a production run from a clone:
|
||||
|
||||
```bash
|
||||
npm run build
|
||||
npm run start
|
||||
```
|
||||
|
||||
`npm run dev` runs TypeScript directly through `tsx` with no build step. The frontend is
|
||||
plain JavaScript served from `src/web/public/` with no bundler, so editing a `.js` or `.css`
|
||||
file and reloading the page is enough. The one exception is `index.html`, which is read once
|
||||
at server start, so markup changes need a restart.
|
||||
|
||||
See [Contributing](Contributing) for the rest of the development loop.
|
||||
|
||||
## Installing an agent CLI
|
||||
|
||||
Codeman drives CLIs, it does not bundle them. Install at least one:
|
||||
|
||||
| CLI | Install | Notes |
|
||||
| --------------- | ------------------------------------------------------------------ | -------------------------------------------------------------------------- |
|
||||
| **Claude Code** | `npm i -g @anthropic-ai/claude-code` | The primary target. Some Codeman features are Claude-only: see [Agent CLIs](Agent-CLIs). |
|
||||
| **OpenCode** | See [opencode.ai](https://opencode.ai) | |
|
||||
| **Codex** | See [developers.openai.com/codex/cli](https://developers.openai.com/codex/cli) | |
|
||||
| **Antigravity** | See [antigravity.google](https://antigravity.google) | Google's successor to the consumer Gemini CLI. |
|
||||
| **Gemini CLI** | See [github.com/google-gemini/gemini-cli](https://github.com/google-gemini/gemini-cli) | Enterprise only since Google's June 2026 consumer cutover. |
|
||||
| **Pi** | See [pi.dev](https://pi.dev) | No permission prompts and no sandbox by design. Read [Agent CLIs](Agent-CLIs) before using it on a repo you care about. |
|
||||
|
||||
Log each CLI in once, by hand, before pointing Codeman at it. Codeman never collects or
|
||||
stores your CLI credentials.
|
||||
|
||||
## Verify the install
|
||||
|
||||
```bash
|
||||
codeman doctor # checks Node, tmux, the agent CLIs, document converters
|
||||
codeman --version
|
||||
codeman web # then open http://localhost:3000
|
||||
```
|
||||
|
||||
`codeman doctor --json` gives machine-readable output, and `--category core` narrows it to
|
||||
the things a session cannot start without.
|
||||
|
||||
If the dashboard loads and **+ New Session** opens, you are done. Continue to
|
||||
[Quick Start](Quick-Start).
|
||||
|
||||
## Where things live
|
||||
|
||||
| Path | What |
|
||||
| ----------------------- | ------------------------------------------------------------------------------------------------ |
|
||||
| `~/.codeman/app` | The installed code (installer route only). |
|
||||
| `~/.codeman/` | All state: `state.json`, settings, session history, push keys, TLS certs. See [Core Concepts](Core-Concepts). |
|
||||
| `~/codeman-cases/` | Cases created from scratch. Linked cases stay wherever they already are. |
|
||||
| `~/.codeman/web.log` | Log for a detached (`-d`) server. |
|
||||
|
||||
Everything is under your home directory, and nothing needs root.
|
||||
|
||||
## Keeping it running
|
||||
|
||||
A bare `codeman web` dies with the shell that started it. Two ways to outlive that:
|
||||
|
||||
```bash
|
||||
codeman web -d # detached; --status and --stop manage it
|
||||
codeman service install # systemd user unit or macOS LaunchAgent; survives reboots
|
||||
```
|
||||
|
||||
Full detail, including logs and the self-updater, is in
|
||||
[Running As A Service](Running-As-A-Service).
|
||||
|
||||
## Updating
|
||||
|
||||
| Install route | How to update |
|
||||
| ------------- | ----------------------------------------------------------------- |
|
||||
| Installer | Re-run the one-liner, or **App Settings → System → Updates** in the UI. |
|
||||
| npm | `npm update -g aicodeman` |
|
||||
| git clone | `git pull && npm install && npm run build`, then restart. |
|
||||
|
||||
The in-app updater covers git-clone installs supervised by systemd or launchd. It restarts
|
||||
the process that is running it, so the actual work happens in a detached script and the
|
||||
browser polls across the restart. Progress appears in the UI.
|
||||
|
||||
## Uninstalling
|
||||
|
||||
```bash
|
||||
install.sh uninstall # installer route
|
||||
npm uninstall -g aicodeman # npm route
|
||||
```
|
||||
|
||||
Neither removes `~/.codeman/` or `~/codeman-cases/`. Delete those by hand if you want the
|
||||
state and your case folders gone as well, and check `~/codeman-cases/` first: linked cases
|
||||
point at directories you already had, but cases created from scratch have their only copy
|
||||
there.
|
||||
|
||||
Running tmux sessions are not killed by an uninstall. `tmux -L codeman kill-server` ends
|
||||
them.
|
||||
|
||||
## Windows (WSL)
|
||||
|
||||
```powershell
|
||||
wsl bash -c "curl -fsSL https://getcodeman.com/install | bash"
|
||||
```
|
||||
|
||||
Codeman requires tmux, so Windows runs it inside
|
||||
[WSL2](https://learn.microsoft.com/en-us/windows/wsl/install). If you do not have WSL yet:
|
||||
run `wsl --install` in an admin PowerShell, reboot, open Ubuntu, and install your agent CLI
|
||||
*inside* WSL. `http://localhost:3000` then works from your Windows browser.
|
||||
|
||||
Work inside the Linux filesystem (`~/project`), not `/mnt/c/...`. Filesystem watching and
|
||||
git are both dramatically slower across the Windows mount, and agents notice.
|
||||
|
||||
## macOS notes
|
||||
|
||||
**`Error: posix_spawnp failed.` on every session start.** node-pty publishes its macOS
|
||||
`spawn-helper` without the executable bit, and macOS launches every PTY through it. Codeman
|
||||
detects this and repairs it automatically on the first failure. If you hit it on a clone
|
||||
install and want to fix it by hand:
|
||||
|
||||
```bash
|
||||
npm run fix:node-pty
|
||||
```
|
||||
|
||||
This is a `chmod`, not a rebuild. Look in `prebuilds/darwin-<arch>/`, not
|
||||
`build/Release/`, which does not exist on macOS. Linux cannot reproduce this.
|
||||
|
||||
**launchd and PATH.** A LaunchAgent gets `/usr/bin:/bin:/usr/sbin:/sbin`, which finds
|
||||
neither a Homebrew or nvm `node` nor `tmux` or `claude`. `codeman service install` bakes
|
||||
your current PATH into the unit for exactly this reason, so prefer it over a hand-written
|
||||
plist.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **`tmux: command not found` after a successful install.** The installer asks before
|
||||
installing packages, and a declined prompt is a valid answer it remembers. Install tmux
|
||||
and re-run.
|
||||
- **Port 3000 in use.** `codeman web --port 8080`, or set `CODEMAN_PORT`.
|
||||
- **Two Codemans on one machine.** The data directory and the tmux socket are both process
|
||||
wide, so a second instance discovers and attaches the first one's live sessions. Give each
|
||||
a distinct `CODEMAN_INSTANCE` before starting a second. See [Core Concepts](Core-Concepts).
|
||||
- **The dashboard is not reachable from your phone.** That is the default, not a fault. The
|
||||
server binds `127.0.0.1`. See [Remote Access](Remote-Access).
|
||||
|
||||
## Read next
|
||||
|
||||
- [Quick Start](Quick-Start) - your first working session.
|
||||
- [Agent CLIs](Agent-CLIs) - picking and setting up a run mode.
|
||||
- [Remote Access](Remote-Access) - reaching it from another device.
|
||||
- [Troubleshooting](Troubleshooting) - when the above did not go as written.
|
||||
@@ -0,0 +1,159 @@
|
||||
# Keeping Agents Running
|
||||
|
||||
Codeman exists for the hours you are not at the keyboard. This page covers how it notices an
|
||||
agent has stopped, what it does about it, and how to run a session overnight without
|
||||
babysitting it.
|
||||
|
||||
Everything here is **per session and off by default**. A session you never configure just
|
||||
sits there when it finishes, which is usually what you want.
|
||||
|
||||
## How Codeman knows an agent is idle
|
||||
|
||||
Harder than it sounds, and worth understanding, because it is what every other feature here
|
||||
is built on.
|
||||
|
||||
**For Claude sessions**, the naive signal does not work. Claude redraws its prompt marker
|
||||
roughly once a second all the way through a turn, so "saw a prompt, waited two seconds,
|
||||
called it idle" flipped working sessions to idle a couple of seconds into every turn. Its
|
||||
real working indicator is an animated line whose glyph and wording both change, and terminal
|
||||
repaints arrive in partial fragments, so matching it in the output stream does not work
|
||||
either.
|
||||
|
||||
So Codeman waits for the pane to go quiet, then **asks the screen** what is on it before
|
||||
believing the session is idle. Turn-start detection works the same way in reverse: a
|
||||
sustained run of repaints marks a turn as started, with the same screen check vetoing mere
|
||||
keystroke echo. Idle now lands a few seconds after a turn genuinely ends.
|
||||
|
||||
There are several layers stacked on that: a completion message from the CLI, an AI check,
|
||||
output silence, and token stability.
|
||||
|
||||
**For every other CLI**, there are no hooks to lean on, so detection is output
|
||||
stabilization: the session is idle when output stops changing. Coarser, and it is why the
|
||||
features further down this page are Claude-only.
|
||||
|
||||
## The Respawn Controller
|
||||
|
||||
Respawn keeps a session working past the point where the agent would otherwise stop. When
|
||||
the session goes idle, Codeman runs a cycle and starts it again.
|
||||
|
||||
A cycle is up to four steps, each optional:
|
||||
|
||||
1. **Update prompt.** Ask the agent to write down where it got to, so the next round can pick
|
||||
it up.
|
||||
2. **`/clear`.** Reset the context window.
|
||||
3. **`/init`.** Re-read the project's `CLAUDE.md`.
|
||||
4. **Kickstart prompt.** Tell it to continue.
|
||||
|
||||
Steps 2 and 3 are what make long runs possible: without a context reset, a multi-hour
|
||||
session eventually spends its whole window on its own history.
|
||||
|
||||
Configure it in **Session Options → Respawn**, then press **Enable**. It repeats until the
|
||||
duration you set runs out.
|
||||
|
||||
| Setting | What it controls |
|
||||
| ---------------------- | ----------------------------------------------------------------------- |
|
||||
| **Idle timeout** | How long the session must be quiet before a cycle starts. |
|
||||
| **Duration** | How long the whole arrangement stays armed. |
|
||||
| **Inter-step delay** | Pause between the steps above, so a step is not sent into a busy pane. |
|
||||
| **`/clear` + `/init`** | Whether the context reset happens at all. |
|
||||
| **Update prompt** | What the agent is asked to record before the reset. |
|
||||
| **Kickstart prompt** | What starts the next round. |
|
||||
| **Auto-accept prompts**| Answer routine confirmation dialogs automatically. |
|
||||
|
||||
### Presets
|
||||
|
||||
Five built-ins, and the numbers matter more than the names. The idle timeout is the main
|
||||
difference: a lead session coordinating subagents is legitimately silent for a minute at a
|
||||
time, and a three second timeout would interrupt it constantly.
|
||||
|
||||
| Preset | Idle timeout | Duration | Built for |
|
||||
| -------------- | ------------ | -------- | --------------------------------------------------------------- |
|
||||
| **Solo** | 3s | 60 min | One agent working alone, fast cycles with a context reset. |
|
||||
| **Subagents** | 45s | 240 min | A lead session running Task subagents; tolerates their silences. |
|
||||
| **Team** | 90s | 480 min | Leading an agent team; tolerates long silences. |
|
||||
| **Ralph/Todo** | 8s | 480 min | Working through a task list with progress tracking. |
|
||||
| **Overnight** | 10s | 480 min | Unattended overnight runs with a full reset between cycles. |
|
||||
|
||||
Start from the preset that matches your shape of work and adjust the idle timeout first.
|
||||
Presets you build yourself can be saved alongside these.
|
||||
|
||||
### What it costs
|
||||
|
||||
Every cycle is real tokens: the update prompt, the reset, and the kickstart, plus whatever
|
||||
work follows. An overnight run is a deliberate spend, not a background nicety. The duration
|
||||
setting is the ceiling, and it is worth setting honestly.
|
||||
|
||||
## Auto-resume when a usage limit resets
|
||||
|
||||
**Claude only.** At the top of the Respawn tab.
|
||||
|
||||
When Claude halts on a subscription limit, the message names the time the limit resets.
|
||||
Codeman parses it, arms a timer for two minutes after that, then sends Escape followed by
|
||||
`continue`.
|
||||
|
||||
The important part is what it does **not** do: respawn cycles are blocked while a session is
|
||||
limit-paused. Without that, the next cycle would fire `/clear` and wipe the conversation you
|
||||
are waiting to resume. This is the single most useful setting for overnight runs on a
|
||||
subscription plan.
|
||||
|
||||
## The plan usage chip
|
||||
|
||||
**Claude only.** A header chip showing live subscription usage, on by default on desktop and
|
||||
off on phones.
|
||||
|
||||
It works by installing a status line exporter into Claude Code, which posts Claude's own
|
||||
rate limit data back to Codeman. The exporter is marker-identified, so it only ever touches
|
||||
a status line Codeman installed, never one you wrote yourself, and it prints your footer
|
||||
through so the in-terminal status line still works.
|
||||
|
||||
The chip and the exporter are the same setting. Turning the chip on without the exporter
|
||||
would leave it showing a dash forever, so resolve it in one place: **App Settings**.
|
||||
|
||||
## Circuit breakers
|
||||
|
||||
Two, and they are unrelated:
|
||||
|
||||
- **The Ralph breaker** stops respawn thrashing. It moves from closed to half-open to open,
|
||||
and is reset from the session's Ralph controls.
|
||||
- **The PTY-exit breaker** trips when a session's process exits repeatedly and quickly, and
|
||||
blocks automatic restarts so a broken configuration cannot spin forever.
|
||||
|
||||
The PTY-exit breaker resets **only** on an explicit clear. Reattaching to the session does
|
||||
not clear it, deliberately, so a UI reconnect cannot paper over a session that is genuinely
|
||||
failing to start.
|
||||
|
||||
## A working overnight setup
|
||||
|
||||
1. Start a Claude session in the case you want worked on.
|
||||
2. Give it a clear goal and let it start. Respawn continues work, it does not invent it.
|
||||
3. **Session Options → Respawn → Overnight preset.**
|
||||
4. Turn on **auto-resume on usage limit**.
|
||||
5. Set the duration to how long you actually want it running.
|
||||
6. Press **Enable**.
|
||||
7. Optionally turn on push notifications so a blocking question reaches your phone: see
|
||||
[Notifications And Approvals](Notifications-And-Approvals).
|
||||
|
||||
In the morning, the **Away Digest** summarizes what happened while you were gone, and the
|
||||
run summary and lifecycle log carry the detail.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **Respawn without a context reset stalls eventually.** The window fills with history and
|
||||
the agent gets less useful every cycle.
|
||||
- **An idle timeout that is too short interrupts real work.** If the agent runs long tool
|
||||
calls or coordinates subagents, raise it. That is what the Subagents and Team presets are.
|
||||
- **The update prompt is what makes a reset survivable.** After `/clear`, everything the
|
||||
agent knows comes from that summary and the project files. A vague update prompt produces
|
||||
a vague next cycle.
|
||||
- **Non-Claude sessions can respawn**, but with output-based idle detection and no
|
||||
usage-limit auto-resume.
|
||||
- **Do not run respawn on a session you are actively typing in.** It will send prompts
|
||||
underneath you.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Autonomous Loops](Autonomous-Loops) - Ralph and the orchestrator, for structured
|
||||
autonomous work rather than "keep going".
|
||||
- [Cron Jobs](Cron-Jobs) - starting work on a schedule instead of continuing it.
|
||||
- [Notifications And Approvals](Notifications-And-Approvals) - being told when it needs you.
|
||||
- [`docs/respawn-state-machine.md`](https://github.com/Ark0N/Codeman/blob/master/docs/respawn-state-machine.md) - the state machine itself.
|
||||
@@ -0,0 +1,72 @@
|
||||
# Keyboard Shortcuts
|
||||
|
||||
Every binding, and how to change them. `Ctrl` also accepts `Cmd` on macOS.
|
||||
|
||||
Press `Ctrl+?` in the app for the same list in a floating overlay.
|
||||
|
||||
## Sessions and tabs
|
||||
|
||||
| Shortcut | Action |
|
||||
| ------------------------------- | --------------------------------------------------------------- |
|
||||
| `Ctrl+K` (also `Cmd+K`, `Alt+K`)| Find an open session or start a new one. |
|
||||
| `Ctrl+W` | Kill the active session. |
|
||||
| `Ctrl+Tab` | Next session. |
|
||||
| `Alt+[` / `Alt+]` | Previous / next tab. |
|
||||
| `Alt+1` to `Alt+9` | Switch to tab N. Physical keys, so macOS Option layouts work. |
|
||||
| `Ctrl+Shift+{` / `Ctrl+Shift+}` | Move the active tab left / right. |
|
||||
| `Alt+B` | Collapse / expand the session sidebar, when that layout is on. |
|
||||
|
||||
## Terminal
|
||||
|
||||
| Shortcut | Action |
|
||||
| ----------------------- | --------------------------------------------------------------- |
|
||||
| `Enter` | Send. |
|
||||
| `Shift+Enter` | Insert a newline without sending. |
|
||||
| `Ctrl+Enter` | Same. |
|
||||
| `Ctrl+C` | Copy the selection, or interrupt when nothing is selected. |
|
||||
| `Ctrl+Shift+C` | Copy the selection. Never interrupts. |
|
||||
| `Ctrl+L` | Clear the terminal. |
|
||||
| `Ctrl+Shift+R` | Restore terminal size. |
|
||||
| `Ctrl` `+` / `Ctrl` `-` | Font size. |
|
||||
| `Shift+Wheel` | Scroll the local buffer, even where the wheel is forwarded to the CLI. |
|
||||
|
||||
## Everything else
|
||||
|
||||
| Shortcut | Action |
|
||||
| -------------- | ------------------------------- |
|
||||
| `Ctrl+Shift+V` | Toggle voice input. |
|
||||
| `Ctrl+?` | Shortcut reference overlay. |
|
||||
| `Escape` | Close panels and modals. |
|
||||
|
||||
## Rebinding
|
||||
|
||||
**App Settings → Shortcuts.** Bindings live in a registry with per-user overrides, so a
|
||||
rebind is stored as an override on top of the default rather than replacing the table.
|
||||
|
||||
Two things are deliberately not rebindable:
|
||||
|
||||
- **`Ctrl+C` smart copy.** The generic dispatch loop calls `preventDefault()` on every
|
||||
shortcut it handles, and doing that to `Ctrl+C` would swallow the interrupt when nothing
|
||||
is selected. It is handled separately for that reason.
|
||||
- **`Escape`**, which closes whatever is open.
|
||||
|
||||
## Why some chords behave oddly
|
||||
|
||||
The terminal sees keystrokes before the app does. Any chord the app claims has to also be
|
||||
swallowed at the terminal layer, or xterm writes the control byte into the session as well
|
||||
as triggering the action. If you rebind something to a chord the terminal cares about
|
||||
(`Ctrl+D`, say), expect the CLI to see it too.
|
||||
|
||||
`Alt+1` through `Alt+9` are matched on **physical key position** rather than the character
|
||||
produced, so macOS Option layouts that produce `¡™£` still switch tabs.
|
||||
|
||||
## On phones
|
||||
|
||||
There is no physical keyboard, so the equivalents live in the keyboard accessory bar: `Esc`,
|
||||
`Ctrl` as a one-shot modifier, `Tab`, arrows, and quick actions. See
|
||||
[Mobile Guide](Mobile-Guide).
|
||||
|
||||
## Read next
|
||||
|
||||
- [The Dashboard](The-Dashboard) - what the shortcuts are navigating.
|
||||
- [Settings Reference](Settings-Reference) - where the overrides are stored.
|
||||
@@ -0,0 +1,145 @@
|
||||
# Mobile Guide
|
||||
|
||||
Codeman on a phone is not a shrunken desktop UI. It is the surface most of its design
|
||||
attention has gone into, because checking on an agent from a bus is the thing this software
|
||||
is for.
|
||||
|
||||
<p align="center">
|
||||
<img src="https://raw.githubusercontent.com/Ark0N/Codeman/master/docs/screenshots/mobile-session-keyboard-20260727.png" alt="Answering an agent prompt on a phone" width="300">
|
||||
</p>
|
||||
|
||||
## Getting there
|
||||
|
||||
1. **Set up access.** Tailscale is the recommended route and gives you real HTTPS. See
|
||||
[Remote Access](Remote-Access).
|
||||
2. **Log in by QR.** Open the dashboard on your desktop and scan the code. No password
|
||||
typing. Tokens are single use and rotate every 60 seconds.
|
||||
3. **Install it to your home screen.** On iOS this is mandatory for push notifications;
|
||||
Safari does not deliver push to tabs. On Android it makes the app full screen.
|
||||
|
||||
HTTPS matters for more than security here: microphone access and push notifications both
|
||||
require a secure context.
|
||||
|
||||
## The layout
|
||||
|
||||
| Element | Where |
|
||||
| -------------------- | --------------------------------------------------------------------- |
|
||||
| Header | Fixed at the top, deliberately minimal. Desktop-only controls never appear. |
|
||||
| Tab strip | Scrolls horizontally. The active tab is always scrolled into view. |
|
||||
| Terminal | The rest of the screen. |
|
||||
| Toolbar | Bottom: Run, Stop, **Enter**, case picker, voice, settings. |
|
||||
| Keyboard bar | Above the on-screen keyboard when it is open. |
|
||||
|
||||
Layout respects notch and home-indicator safe areas, touch targets are 44px, and the case
|
||||
picker is a bottom sheet rather than a dropdown.
|
||||
|
||||
**Swipe left and right** on the terminal to switch sessions.
|
||||
|
||||
## The home screen
|
||||
|
||||
Tapping the "C" logo gives a session overview rather than a welcome page:
|
||||
|
||||
1. **NEEDS YOU** first: sessions blocked on a question, with answer strips so you can
|
||||
resolve them without opening the session.
|
||||
2. **CURRENT SESSIONS** with live status.
|
||||
3. **PAST SESSIONS**, resumable.
|
||||
|
||||
Row status uses the same language as the tabs: green when fine, pulsing while working,
|
||||
yellow when waiting for input, red when a question is pending.
|
||||
|
||||
The split Run button carries the same per-backend colours as the desktop toolbar, and its
|
||||
picker mirrors the desktop run-mode menu.
|
||||
|
||||
On by default; it can be turned off in settings.
|
||||
|
||||
## The keyboard accessory bar
|
||||
|
||||
A row of keys above the virtual keyboard, and what it contains depends on the session.
|
||||
|
||||
**Agent sessions** get quick actions: `/init`, `/clear`, `/compact`, a clipboard key, `Esc`,
|
||||
a path picker, an image key, and 🧠 when Read My Mind is on. Destructive commands need a
|
||||
double press, so you cannot fire `/clear` with a stray thumb.
|
||||
|
||||
**Shell sessions** automatically swap it for terminal controls: `Ctrl`, `Esc`, `Tab`, four
|
||||
arrows, paste, and dismiss. Your normal preference is remembered and restored when you
|
||||
switch back to an agent session, so a settings change during a shell session cannot strip
|
||||
the bar away permanently.
|
||||
|
||||
### One-shot Ctrl
|
||||
|
||||
`Ctrl` on the shell bar is a **one-shot modifier**: tap `Ctrl`, then tap `c`, and the
|
||||
control byte is sent. It disarms on use, on a second tap, on any other accessory key, on a
|
||||
session switch, and when the keyboard closes.
|
||||
|
||||
That list matters. A modifier left armed turns your next innocent keystroke into a control
|
||||
byte, so it is deliberately eager to disarm. Keys with no control equivalent pass through
|
||||
unchanged, exactly like a hardware keyboard.
|
||||
|
||||
## The Enter button
|
||||
|
||||
The toolbar's dedicated **Enter** button exists because of local echo. On a phone, the
|
||||
characters you type are painted locally and have not reached the agent yet; Enter flushes
|
||||
them and then submits.
|
||||
|
||||
It replays the keypress through the terminal rather than sending a bare carriage return.
|
||||
Sending a bare `\r` would submit an empty line and strand your typed text on screen, which
|
||||
looks exactly like a dead button.
|
||||
|
||||
On phones this button replaces the desktop's **Run Shell** control; starting a shell moved
|
||||
into the Run dropdown.
|
||||
|
||||
## Scrolling and the keyboard
|
||||
|
||||
- The terminal and toolbar shift up when the keyboard opens, tracked through the browser's
|
||||
visual viewport rather than guessed.
|
||||
- **Two ways to dismiss the keyboard**: tap outside the terminal on inert space, or tap twice
|
||||
on inert terminal content. Tapping a control never dismisses it, and tapping the prompt row
|
||||
keeps focus so you can place the caret.
|
||||
- A scroll is never mistaken for a tap: travel is measured from the start of the gesture, and
|
||||
multi-touch never counts.
|
||||
|
||||
## Voice
|
||||
|
||||
The microphone button, or the keyboard bar. Providers and setup are covered in
|
||||
[Input And Voice](Input-And-Voice). Dictating is often faster than typing a prompt on a
|
||||
phone, and it is the main reason the feature exists.
|
||||
|
||||
## Notifications
|
||||
|
||||
Push notifications reach you with no tab open, and with the Approvals Inbox on they carry
|
||||
**Approve** and **Deny** buttons handled by the service worker, so you can unblock an agent
|
||||
from the lock screen.
|
||||
|
||||
Setup in [Notifications And Approvals](Notifications-And-Approvals).
|
||||
|
||||
## Reading long answers
|
||||
|
||||
The terminal viewport is small. **Last Response** (opt-in header button) renders the agent's
|
||||
last answer as scrollable text instead, with a **More** button for additional context.
|
||||
|
||||
The [File Viewer](Working-With-Files) works on phones too, including edit mode, which is
|
||||
enough to fix a typo an agent introduced while you are away from your desk.
|
||||
|
||||
## What is deliberately not on phones
|
||||
|
||||
- Extra header buttons. New header controls are kept off phones by policy, with a test that
|
||||
enforces it.
|
||||
- The Approvals bell. Phones get the NEEDS YOU strips on the home screen instead.
|
||||
- The desktop home tab rail, which needs a wide window.
|
||||
- Lineage arcs, which are a desktop overlay.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **Typed text sitting on screen has not been sent.** Press Enter.
|
||||
- **iOS needs the home screen install for push**, not just a bookmark.
|
||||
- **iOS Safari can serve stale JavaScript after an update** until the tab is fully closed.
|
||||
Close it and reopen.
|
||||
- **Plain HTTP over a LAN address disables voice and push.** Use HTTPS.
|
||||
- **An armed `Ctrl` is visibly highlighted.** If it looks the same as a resting key, you are
|
||||
on an old version, on a light skin.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Remote Access](Remote-Access) - getting the phone connected in the first place.
|
||||
- [Notifications And Approvals](Notifications-And-Approvals) - being told when you are needed.
|
||||
- [Input And Voice](Input-And-Voice) - local echo, dictation, and the input rules.
|
||||
@@ -0,0 +1,97 @@
|
||||
# Multi-User Mode
|
||||
|
||||
Share one Codeman with a small trusted team. Each person gets their own login and workspace,
|
||||
and sessions, cases, search, and live events are scoped to their owner.
|
||||
|
||||
**Off by default.** Without the flag, behaviour is identical to single-user Codeman, because
|
||||
every scoping check short-circuits.
|
||||
|
||||
## Read this before enabling it
|
||||
|
||||
**Multi-user mode separates workspaces. It does not sandbox users from each other.**
|
||||
|
||||
Every session still runs as the **same operating system account**. A determined user's agent
|
||||
can reach another user's files, because at the OS level they are the same user. This is a
|
||||
convenience and organization feature, not a security boundary.
|
||||
|
||||
If you need real isolation:
|
||||
|
||||
- Pair each user with [Docker Cases](Docker-Cases), which gives their work its own
|
||||
filesystem and network.
|
||||
- Or run separate Codeman instances under separate OS accounts, each with its own
|
||||
`CODEMAN_INSTANCE`.
|
||||
|
||||
"Small trusted team" is the honest description of who this is for.
|
||||
|
||||
## Enabling it
|
||||
|
||||
```bash
|
||||
codeman users add alice --admin # create the first admin, prompts for a password
|
||||
codeman web --multiuser # or CODEMAN_MULTIUSER=1
|
||||
```
|
||||
|
||||
Then manage users from the CLI or the **Users** entry in App Settings:
|
||||
|
||||
```bash
|
||||
codeman users add bob # a regular user
|
||||
codeman users list
|
||||
codeman users passwd bob # reset to a one-time password
|
||||
codeman users rm bob
|
||||
```
|
||||
|
||||
`--password-stdin` reads the password from standard input, for scripts.
|
||||
|
||||
Accounts live in `~/.codeman/users.json` with scrypt-hashed passwords, mode 0600.
|
||||
Administrative actions are audited to `~/.codeman/admin-audit.jsonl`.
|
||||
|
||||
## What each user gets
|
||||
|
||||
| Thing | Scope |
|
||||
| ------------------- | ---------------------------------------------------------------------------- |
|
||||
| **Case space** | `~/codeman-users/<name>/cases`, their own. |
|
||||
| **Sessions** | Only theirs are listed, reachable, or controllable. |
|
||||
| **Events** | Live event routing is per owner, and fails closed. |
|
||||
| **Search** | Scoped on read, including historical results. |
|
||||
| **File previews** | Scoped to sessions they own. |
|
||||
| **Path picker** | Only their own user space as a root, not the whole home directory. |
|
||||
|
||||
Admins see everything.
|
||||
|
||||
Ownership threads through every list endpoint, the session lookup helper, the WebSocket
|
||||
layer, and file previews. A user cannot address another user's session even by id.
|
||||
|
||||
## Safer defaults for regular users
|
||||
|
||||
Non-admins get tighter defaults, and lifting them is an explicit per-user grant:
|
||||
|
||||
| Default | Meaning |
|
||||
| ---------------------------------- | ------------------------------------------------------------------------ |
|
||||
| Claude runs in `auto` permission mode | Anthropic's classifier-guarded mode instead of skip-prompts. |
|
||||
| Raw shell sessions require a grant | A plain shell is unmediated machine access. |
|
||||
| Skip-permissions requires a grant | Same reasoning. |
|
||||
| Cron `launchCommand` requires a grant | It is an arbitrary command on a schedule. |
|
||||
| Pi project trust defaults to off | Trust makes Pi execute repo-local TypeScript. |
|
||||
|
||||
These exist because the OS boundary is shared. They narrow what a normal account can do
|
||||
casually; they do not make the account a sandbox.
|
||||
|
||||
## Accounts and sessions
|
||||
|
||||
Each user authenticates with their own name and password rather than the shared
|
||||
`CODEMAN_PASSWORD`. Logins are individually revocable: disable, reset, or delete an account
|
||||
at any time, and existing browser sessions can be revoked.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **Enabling it does not migrate existing cases** into a user space. They stay where they
|
||||
are, owned by whoever the ownership rules resolve them to.
|
||||
- **Admins see everything**, including other users' sessions. Choose admins accordingly.
|
||||
- **The audit log is append-only and local.** Ship it somewhere if you care about it.
|
||||
- **It is not a substitute for OS accounts.** Restating this because it is the one thing
|
||||
people get wrong.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Security](Security) - where this fits in the model, and what it does not cover.
|
||||
- [Docker Cases](Docker-Cases) - the isolation story that actually isolates.
|
||||
- [`docs/multi-user-plan.md`](https://github.com/Ark0N/Codeman/blob/master/docs/multi-user-plan.md) - the design.
|
||||
@@ -0,0 +1,147 @@
|
||||
# Notifications and Approvals
|
||||
|
||||
An agent that stops to ask a question, with nobody watching, is a run that quietly wasted an
|
||||
hour. This page covers every way Codeman tells you it needs you, and how to answer without
|
||||
opening the session.
|
||||
|
||||
## The signals, cheapest first
|
||||
|
||||
| Surface | Reaches you | Default |
|
||||
| ---------------------- | ------------------------------------------------- | ------- |
|
||||
| Tab alert | While the dashboard is open | On |
|
||||
| Browser title flash | Another tab in the same browser | On |
|
||||
| Desktop notification | Another window on the same machine | Opt-in |
|
||||
| Push notification | Anywhere, even with no tab open | Opt-in |
|
||||
| Approvals Inbox | One queue across every session | Opt-in |
|
||||
| Phone overview | Phone home screen, NEEDS YOU section | On |
|
||||
| Away Digest | Afterwards, as a summary | Opt-in |
|
||||
|
||||
## Tab alerts
|
||||
|
||||
The tab itself changes state:
|
||||
|
||||
| State | Meaning |
|
||||
| -------------------- | ---------------------------------------------------------- |
|
||||
| Yellow, blinking | The agent is waiting for input from you. |
|
||||
| Red, blinking | A question or permission prompt is blocking the session. |
|
||||
|
||||
These are a steady colour with a pulse layered on top, not a blink to transparent, so a tab
|
||||
needing attention looks that way at every point in the cycle.
|
||||
|
||||
They survive a reload. The alert state is re-seeded from the server on page load, so
|
||||
reloading the dashboard while a permission dialog is blocking a session does not leave you
|
||||
with a normal-looking tab.
|
||||
|
||||
For Claude sessions, these come from Claude Code's hooks and are precise about *why* the
|
||||
session stopped. For other CLIs there are no hooks, so you get the coarser output-based
|
||||
signal.
|
||||
|
||||
## Window title and OS notifications
|
||||
|
||||
The browser tab title is prefixed `codeman:<host>`, so several Codeman instances across
|
||||
several machines stay distinguishable at a glance. Override the hostname with
|
||||
`codeman web --title-hostname <name>`.
|
||||
|
||||
Desktop notifications use the same prefix. Enable them in **App Settings → Notifications**.
|
||||
|
||||
## Push notifications
|
||||
|
||||
Push reaches your phone with **no Codeman tab open at all**, which is the only option that
|
||||
works while you are actually away.
|
||||
|
||||
Setup:
|
||||
|
||||
1. Open Codeman over **HTTPS**. Web push requires a secure context. Tailscale gives you real
|
||||
HTTPS; `--https` gives you a self-signed certificate; plain HTTP over a LAN address will
|
||||
not work.
|
||||
2. **App Settings → Notifications → Subscribe**, and accept the browser prompt.
|
||||
3. On **iOS**, add Codeman to your home screen first. Safari only delivers web push to
|
||||
installed web apps, not to tabs.
|
||||
|
||||
Once subscribed, a blocking prompt reaches your phone even from a locked screen.
|
||||
|
||||
## The Approvals Inbox
|
||||
|
||||
**Opt-in, off by default. Claude sessions only.**
|
||||
|
||||
One queue of every prompt currently waiting on a human, across all your sessions, answerable
|
||||
in place. When you have eight workers running, this is the difference between checking eight
|
||||
tabs and checking one list.
|
||||
|
||||
Turn it on in **App Settings**. Surfaces:
|
||||
|
||||
- **A header bell** with a count, hidden entirely while the count is zero. Never shown on
|
||||
phones.
|
||||
- **A drawer** listing each waiting card.
|
||||
- **NEEDS YOU strips** at the top of the phone overview home screen.
|
||||
|
||||
Each card shows the session, the case, and the captured prompt with its options. Answering
|
||||
sends the keystroke into the session for you: a digit for a menu choice, Escape to decline,
|
||||
or free text for an idle prompt.
|
||||
|
||||
Behaviour worth knowing:
|
||||
|
||||
- **One item per session.** A newer prompt supersedes the older one, because the older one
|
||||
is no longer on screen.
|
||||
- **Menu answers are validated against the live screen.** Codeman re-captures the pane before
|
||||
sending, and refuses with a conflict if the dialog is no longer there. Otherwise your
|
||||
keystroke would land in the composer as stray text.
|
||||
- **Permission and question items clear only on definitive signals**: the turn ending, the
|
||||
dialog completing, an answer, a supersede, the session exiting, or a 12 hour timeout. They
|
||||
do not clear on a heuristic "looks busy again" signal, because that signal is wrong often
|
||||
enough to lose a real prompt.
|
||||
- **In memory only.** Restarting the server clears the queue; the prompts themselves are
|
||||
still sitting in the sessions.
|
||||
|
||||
### Approve and Deny from the notification
|
||||
|
||||
With the inbox enabled, push notifications carry **Approve** and **Deny** buttons. Those are
|
||||
handled by the service worker directly, so they work with no tab open: tap Approve on a
|
||||
locked phone and the agent continues.
|
||||
|
||||
With the inbox off, the buttons are stripped from the notification payload entirely rather
|
||||
than being shown and failing.
|
||||
|
||||
## The phone overview
|
||||
|
||||
On phones, tapping the "C" logo gives a session overview with **NEEDS YOU** first, then
|
||||
current sessions, then past ones. Rows use the same language as the tab strip: a green dot
|
||||
when fine, pulsing while working, yellow when waiting for input, red when a question is
|
||||
pending.
|
||||
|
||||
Answer strips let you resolve a prompt straight from the home screen without opening the
|
||||
session.
|
||||
|
||||
## The Away Digest
|
||||
|
||||
Retrospective rather than live: what happened while you were gone, aggregated from the
|
||||
lifecycle log, run summaries, live sessions, token statistics, and recent subagents.
|
||||
|
||||
It is the morning-after view for an overnight run. Enable its header button in
|
||||
**App Settings → Header & Panels**.
|
||||
|
||||
## Recommended setup for unattended runs
|
||||
|
||||
1. HTTPS access, ideally Tailscale. See [Remote Access](Remote-Access).
|
||||
2. Push notifications subscribed, with Codeman installed to the home screen on iOS.
|
||||
3. Approvals Inbox on.
|
||||
4. Auto-resume on usage limit on, for each session you leave running. See
|
||||
[Keeping Agents Running](Keeping-Agents-Running).
|
||||
|
||||
That combination means a blocking question wakes your phone and can be answered in two taps
|
||||
from the lock screen.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **No push over plain HTTP.** It is a browser requirement, not a Codeman one.
|
||||
- **iOS needs the home screen install.** A Safari tab will never receive push.
|
||||
- **The bell is invisible at zero.** That is deliberate, not a broken setting.
|
||||
- **Approvals are Claude-only.** They are built on hook events the other CLIs do not emit.
|
||||
- **A stale menu answer is refused, not sent.** If you answer a card for a dialog that has
|
||||
since gone away, Codeman declines rather than typing a digit into the composer.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - what to configure before walking away.
|
||||
- [Mobile Guide](Mobile-Guide) - the phone surfaces in full.
|
||||
- [Settings Reference](Settings-Reference) - where each of these toggles lives.
|
||||
@@ -0,0 +1,153 @@
|
||||
# Quick Start
|
||||
|
||||
From an installed Codeman to a working agent, in about five minutes. If you have not
|
||||
installed yet, start at [Installation](Installation).
|
||||
|
||||
## 1. Start the server
|
||||
|
||||
```bash
|
||||
codeman web
|
||||
```
|
||||
|
||||
It prints a URL, `http://localhost:3000` by default. Open it.
|
||||
|
||||
The server binds `127.0.0.1` only, so this URL works from the machine running it and
|
||||
nowhere else. That is deliberate: Codeman starts agents with permission prompts skipped by
|
||||
default, so anyone who can reach the dashboard can run code on this machine. Reaching it
|
||||
from your phone is a separate, deliberate step covered in [Remote Access](Remote-Access).
|
||||
|
||||
To keep it alive after you close the terminal, use `codeman web -d` instead, or install it
|
||||
as a service. See [Running As A Service](Running-As-A-Service).
|
||||
|
||||
## 2. Meet the welcome screen
|
||||
|
||||
With no sessions running you get the welcome screen:
|
||||
|
||||
- **Run buttons** for each agent CLI Codeman found on your PATH. If you expected one and it
|
||||
is missing, its binary is not visible to the server; see [Agent CLIs](Agent-CLIs).
|
||||
- **A QR code**, if a password is set. Scanning it logs a phone in without typing anything.
|
||||
- **Resume Conversation**, a list of past sessions, including Claude conversations started
|
||||
outside Codeman. Empty on a fresh install.
|
||||
- **Search**, across sessions, events, and files.
|
||||
|
||||
You can click a Run button right now and get a working agent in your current case. The rest
|
||||
of this page is the deliberate version.
|
||||
|
||||
## 3. Pick or create a case
|
||||
|
||||
A **case** is a named working directory that Codeman remembers. Every session runs inside
|
||||
one. The case picker is in the bottom toolbar.
|
||||
|
||||
To make a new one, click **+** next to the picker. The Add Case dialog has three tabs:
|
||||
|
||||
| Tab | Use it when |
|
||||
| ----------------- | ------------------------------------------------------------------------------------------------------------------ |
|
||||
| **Create New** | Starting a fresh project. Creates `~/codeman-cases/<name>` and scaffolds a `CLAUDE.md` into it. |
|
||||
| **Clone Repo** | Working on an existing public repo. Paste the URL; Codeman preflights it as you type, offers the repo's real branches and tags, and fills in the case name. |
|
||||
| **Link Existing** | The code is already on disk. Point at the folder, with **Browse** if you would rather click than type. |
|
||||
|
||||
The gear next to the picker holds two per-case toggles: **Agent Teams** and
|
||||
**1M Opus Context**. Both are off by default and both are safe to ignore for now.
|
||||
|
||||
**Create New** also has a checkbox for running the case inside a Docker container, and a
|
||||
**Remote** panel for running it over SSH on another machine. Those are
|
||||
[Docker Cases](Docker-Cases) and [Remote SSH Sessions](Remote-SSH-Sessions); skip them for
|
||||
your first session.
|
||||
|
||||
## 4. Pick a run mode and hit Run
|
||||
|
||||
The **Run** button starts an agent in the selected case. The arrow next to it picks which
|
||||
one:
|
||||
|
||||
| Mode | What starts |
|
||||
| -------------------- | -------------------------------------------------------------- |
|
||||
| **Claude Code** | The default, and the mode every Codeman feature supports. |
|
||||
| **OpenCode** | |
|
||||
| **Codex** | OpenAI's CLI. |
|
||||
| **Gemini** | Enterprise only since Google's consumer cutover. |
|
||||
| **Antigravity** | Google's successor to the consumer Gemini CLI. |
|
||||
| **Pi** | No permission prompts and no sandbox by design. |
|
||||
| **Terminal / Shell** | A plain shell, no agent. Also the **Run Shell** button. |
|
||||
|
||||
The dropdown also lists any saved dashboard URLs ([Web Tabs](Web-Tabs)) and your recent
|
||||
sessions. Those do not change the run mode: Run always means "start an agent".
|
||||
|
||||
Click **Run**. A tab appears, and Codeman spawns the CLI on a real PTY inside a tmux
|
||||
session and streams it to your browser.
|
||||
|
||||
The number spinner beside the button starts several sessions at once, up to 20. Useful for
|
||||
fanning the same case out across parallel workers; unnecessary for a first run.
|
||||
|
||||
## 5. Talk to the agent
|
||||
|
||||
Click into the terminal and type. It is a real terminal (xterm.js over a real PTY), so full
|
||||
TUIs render properly and everything the CLI supports works, slash commands included.
|
||||
|
||||
| Key | Effect |
|
||||
| ---------------------------- | --------------------------------------------- |
|
||||
| `Enter` | Send. |
|
||||
| `Shift+Enter` / `Ctrl+Enter` | Newline without sending. |
|
||||
| `Ctrl+C` | Copy if text is selected, otherwise interrupt. |
|
||||
| `Ctrl+Shift+V` | Voice input. |
|
||||
|
||||
You can also paste or drag an image straight into the session, and register external files
|
||||
as attachments. See [Working With Files](Working-With-Files) and
|
||||
[Input And Voice](Input-And-Voice).
|
||||
|
||||
Input is delivered **exactly once**, even if your connection drops mid-prompt. A dropped
|
||||
link never loses a prompt and never sends it twice.
|
||||
|
||||
## 6. Read the tab
|
||||
|
||||
The tab tells you what the session is doing without opening it:
|
||||
|
||||
| Signal | Meaning |
|
||||
| --------------------- | ---------------------------------------------------------- |
|
||||
| Green dot | Alive and idle. |
|
||||
| Pulsing green dot | Working on a turn. |
|
||||
| Yellow, blinking | Waiting for you to type something. |
|
||||
| Red, blinking | A question or permission prompt is blocking the agent. |
|
||||
|
||||
Full tour in [The Dashboard](The-Dashboard). If you want a phone notification when an agent
|
||||
needs you, that is [Notifications And Approvals](Notifications-And-Approvals).
|
||||
|
||||
## 7. Leave, and come back
|
||||
|
||||
Close the browser tab. Close the laptop. The agent keeps running, because it lives in tmux
|
||||
and not in your browser.
|
||||
|
||||
Reopen the dashboard and the session is still there with its scrollback intact. First load
|
||||
of a session pulls the full tmux scrollback, so you get the history, not just what arrived
|
||||
after you reconnected.
|
||||
|
||||
This also survives restarting the Codeman server itself. What does not survive is killing
|
||||
the tmux server or rebooting the machine.
|
||||
|
||||
## 8. Stop things
|
||||
|
||||
| To do this | Do that |
|
||||
| ------------------------- | ------------------------------------------------------------------- |
|
||||
| Interrupt the current turn | `Ctrl+C` with nothing selected, or the **Stop** button. |
|
||||
| Close one session | `Ctrl+W`, or the tab's close control. |
|
||||
| Stop the server, keep agents | `codeman web --stop`. The tmux sessions stay alive. |
|
||||
| Stop everything | `tmux -L codeman kill-server`. |
|
||||
|
||||
If you are working *inside* a Codeman-managed session (`echo $CODEMAN_MUX` prints `1`),
|
||||
never run `tmux kill-session` or `pkill claude` by hand. You will kill the session you are
|
||||
sitting in, along with its siblings.
|
||||
|
||||
## Where to go next
|
||||
|
||||
**Make it run without you.** [Keeping Agents Running](Keeping-Agents-Running) covers idle
|
||||
detection, respawn cycling, and auto-resume when a subscription limit resets. That is the
|
||||
feature Codeman exists for.
|
||||
|
||||
**Get it on your phone.** [Remote Access](Remote-Access), then
|
||||
[Mobile Guide](Mobile-Guide).
|
||||
|
||||
**Understand what you just used.** [Core Concepts](Core-Concepts) explains cases, sessions,
|
||||
run modes, and what state lives where.
|
||||
|
||||
**Automate it.** [Cron Jobs](Cron-Jobs) for scheduled work,
|
||||
[Driving Codeman From An Agent](Driving-Codeman-From-An-Agent) for agents that spawn and
|
||||
supervise other agents.
|
||||
@@ -0,0 +1,209 @@
|
||||
# Remote Access
|
||||
|
||||
Reaching your Codeman from a phone, a laptop on the other side of the house, or a hotel
|
||||
network. This is the page to read carefully, because Codeman's dashboard is a
|
||||
remote-code-execution surface by design: it starts agents with permission prompts skipped,
|
||||
so whoever can reach it can run code on your machine.
|
||||
|
||||
## Start from the default
|
||||
|
||||
`codeman web` binds `127.0.0.1`. It is reachable from the machine running it and nothing
|
||||
else, which is why the no-password default is safe out of the box. Every option below is a
|
||||
deliberate step away from that.
|
||||
|
||||
Two rules that make the rest of this page simple:
|
||||
|
||||
1. **Never expose Codeman on a network without `CODEMAN_PASSWORD`.** Binding a non-loopback
|
||||
host without one starts, but prints a loud warning with the fixes.
|
||||
2. **Prefer keeping the loopback bind** and putting an authenticated tunnel in front of it,
|
||||
over binding wide and relying on a password alone.
|
||||
|
||||
## Pick an approach
|
||||
|
||||
| Approach | Good for | Cost |
|
||||
| --------------------- | ----------------------------------------------------- | --------------------------------------------------------- |
|
||||
| **Tailscale** | Phone access, permanently. The recommended setup. | Install Tailscale on both devices. |
|
||||
| **Cloudflare tunnel** | A public URL, quickly, from anywhere. | Public URL, so a password is mandatory. |
|
||||
| **LAN + password** | Home network only, no extra software. | Every device on your LAN can reach the login page. |
|
||||
| **SSH port forward** | You already SSH to the box. | Manual, per session, terminal-bound. |
|
||||
|
||||
## Tailscale (recommended)
|
||||
|
||||
Your devices join a private network, and Codeman stays bound to loopback. Nothing is
|
||||
published to the internet, and you get real HTTPS with a real certificate.
|
||||
|
||||
The installer sets this up for you, including installing Tailscale, logging in, enabling
|
||||
tailnet HTTPS, and verifying the result end to end. To retrofit it onto an existing
|
||||
install:
|
||||
|
||||
```bash
|
||||
install.sh tailscale
|
||||
```
|
||||
|
||||
By hand:
|
||||
|
||||
```bash
|
||||
tailscale serve --bg 3000
|
||||
tailscale serve status
|
||||
```
|
||||
|
||||
Then open `https://<machine>.<tailnet>.ts.net` from any device on your tailnet.
|
||||
|
||||
Notes:
|
||||
|
||||
- Keep the loopback bind. `tailscale serve` connects to `127.0.0.1:3000` locally, so
|
||||
binding wider adds exposure and buys nothing.
|
||||
- Your tailnet is the authentication boundary. Setting `CODEMAN_PASSWORD` as well is
|
||||
reasonable defence in depth, especially if other people have devices on your tailnet.
|
||||
- Codeman's Host-header allowlist already accepts `.ts.net`, so no extra configuration is
|
||||
needed.
|
||||
- The installer never resets or rewrites `serve` mappings other than the one pointing at
|
||||
Codeman's port, so unrelated serve configuration is left alone.
|
||||
|
||||
## Cloudflare tunnel
|
||||
|
||||
A free [quick tunnel](https://developers.cloudflare.com/cloudflare-one/connections/connect-networks/do-more-with-tunnels/trycloudflare/)
|
||||
gives you a public HTTPS URL with no port forwarding, no DNS, and no static IP:
|
||||
|
||||
```
|
||||
Browser → Cloudflare edge (HTTPS) → cloudflared → localhost:3000
|
||||
```
|
||||
|
||||
Prerequisites: [`cloudflared`](https://developers.cloudflare.com/cloudflare-one/connections/connect-networks/downloads/)
|
||||
installed, and `CODEMAN_PASSWORD` set.
|
||||
|
||||
```bash
|
||||
./scripts/tunnel.sh start # starts the tunnel, prints the public URL
|
||||
./scripts/tunnel.sh url
|
||||
./scripts/tunnel.sh status
|
||||
./scripts/tunnel.sh stop
|
||||
```
|
||||
|
||||
The quick-tunnel URL is a random `*.trycloudflare.com` address that changes every time the
|
||||
tunnel restarts. For a stable hostname, `./scripts/tunnel.sh named setup` walks through a
|
||||
named tunnel.
|
||||
|
||||
To survive reboots:
|
||||
|
||||
```bash
|
||||
systemctl --user enable codeman-tunnel
|
||||
loginctl enable-linger $USER
|
||||
```
|
||||
|
||||
There is also a toggle in **App Settings → System → Remote access**.
|
||||
|
||||
**The tunnel refuses to start without a password.** That is on purpose: a public URL with no
|
||||
authentication is a terminal on your machine handed to the internet. Acknowledging the risk
|
||||
explicitly is possible from the UI toggle, and only from there; the API will not do it for
|
||||
you.
|
||||
|
||||
## LAN plus password
|
||||
|
||||
```bash
|
||||
export CODEMAN_PASSWORD='something long'
|
||||
codeman web -H 0.0.0.0 --https
|
||||
```
|
||||
|
||||
Every device on your local network can now reach the login page. `--https` generates a
|
||||
self-signed certificate into `~/.codeman/certs/`, which your browser will warn about once.
|
||||
|
||||
`CODEMAN_USERNAME` defaults to `admin`.
|
||||
|
||||
The installer offers this path and prompts for the password. On re-runs it preserves
|
||||
whichever binding you already chose.
|
||||
|
||||
## SSH port forward
|
||||
|
||||
No configuration at all, if you already have SSH access:
|
||||
|
||||
```bash
|
||||
ssh -L 3000:localhost:3000 you@your-box
|
||||
```
|
||||
|
||||
Then open `http://localhost:3000` on the local machine. Codeman keeps its loopback bind and
|
||||
sees a local connection. Good for occasional access, awkward as a permanent arrangement
|
||||
because it dies with the SSH session.
|
||||
|
||||
## Logging in from a phone
|
||||
|
||||
Typing a long password on a phone keyboard is miserable, so Codeman issues **single-use QR
|
||||
tokens**. The desktop dashboard shows a QR code; scan it and the phone is authenticated.
|
||||
|
||||
How it behaves:
|
||||
|
||||
- The code rotates every 60 seconds, with a 90 second grace window so scanning during a
|
||||
rotation still works.
|
||||
- Each token is **single use**. The moment a phone consumes it, a new one is generated.
|
||||
- The URL contains a 6-character lookup code, not the secret, so it does not leak through
|
||||
browser history, `Referer` headers, or the tunnel provider's logs.
|
||||
- The desktop shows a toast naming the device and browser that just authenticated, with a
|
||||
one-click revoke.
|
||||
- QR attempts are rate limited separately from password attempts, so a mistyped password
|
||||
cannot lock out your QR login and vice versa.
|
||||
|
||||
Someone holding only the tunnel URL still meets the normal password prompt. The QR is the
|
||||
fast path, not a bypass.
|
||||
|
||||
Design detail and the threat analysis it is built against:
|
||||
[`docs/qr-auth-plan.md`](https://github.com/Ark0N/Codeman/blob/master/docs/qr-auth-plan.md).
|
||||
|
||||
## Behind a reverse proxy
|
||||
|
||||
Codeman enforces a Host-header allowlist on every request to block DNS rebinding, and the
|
||||
same allowlist gates the cross-site Origin check. It accepts `localhost`, IP literals, the
|
||||
bind host, `.ts.net`, `.trycloudflare.com`, `.cfargotunnel.com`, and the active managed
|
||||
tunnel.
|
||||
|
||||
**Your own domain is not on that list.** Add it:
|
||||
|
||||
```bash
|
||||
CODEMAN_ALLOWED_HOSTS='codeman.example.com,.internal.example.com'
|
||||
```
|
||||
|
||||
A bare entry matches that exact host; a leading dot matches subdomains. Without this, a
|
||||
correctly configured proxy still gets `403 host not allowed`, which reads like a proxy bug
|
||||
and is not one.
|
||||
|
||||
Also make sure the proxy forwards WebSocket upgrades. The terminal is a WebSocket, and the
|
||||
upgrade runs the same Host and Origin checks, closing with code `4003` on failure.
|
||||
|
||||
## Session cookies and rate limits
|
||||
|
||||
The first request prompts for HTTP Basic credentials. On success the server issues an opaque
|
||||
`codeman_session` cookie (24 hour lifetime, extended on activity, validated server-side so
|
||||
it cannot be forged offline). Ten failed attempts from one IP produce a `429` with a 15
|
||||
minute decay.
|
||||
|
||||
A valid cookie or a correct password recovers immediately even while an attacker is hammering
|
||||
the same IP, which matters because all tunnel traffic arrives from one loopback address.
|
||||
|
||||
## Terminal alternatives
|
||||
|
||||
You do not have to use a browser. `sc` is a thumb-friendly session chooser for SSH clients
|
||||
like Termius or Blink:
|
||||
|
||||
```bash
|
||||
sc # interactive chooser
|
||||
sc 2 # attach to session 2
|
||||
sc -l # list
|
||||
```
|
||||
|
||||
Detach with `Ctrl+A D`. The sessions are the same ones the dashboard shows.
|
||||
|
||||
## Common problems
|
||||
|
||||
| Symptom | Cause and fix |
|
||||
| ----------------------------------------------------------- | ------------------------------------------------------------------------------------------------------- |
|
||||
| `403 host not allowed` | Your domain is not in the allowlist. Set `CODEMAN_ALLOWED_HOSTS`. |
|
||||
| Phone shows the login page but the terminal never connects | The proxy is not forwarding WebSocket upgrades. |
|
||||
| Browser warns about the certificate | Expected with `--https` and its self-signed certificate. Tailscale gives you a real one instead. |
|
||||
| LAN IP does not respond, but a tunnel to the same box works | The server is bound to loopback. That is the default. A tunnel reaches it; a LAN browser cannot. |
|
||||
| Hooks stopped working after switching to HTTPS | Hook callbacks need `-k` for the self-signed certificate. Recent versions self-heal existing cases; if yours predates that, recreate the case's hooks. |
|
||||
| Everything is slow over the tunnel | Quick tunnels route through Cloudflare's edge. Tailscale is usually a direct connection and much faster. |
|
||||
|
||||
## Read next
|
||||
|
||||
- [Security](Security) - the whole model, and the hardening checklist.
|
||||
- [Mobile Guide](Mobile-Guide) - once you can reach it from the phone.
|
||||
- [Running As A Service](Running-As-A-Service) - keeping server and tunnel up across reboots.
|
||||
- [`docs/security-architecture.md`](https://github.com/Ark0N/Codeman/blob/master/docs/security-architecture.md) - the full model.
|
||||
@@ -0,0 +1,101 @@
|
||||
# Remote SSH Sessions
|
||||
|
||||
Point a case at another machine and the agent runs **there**, with the same dashboard,
|
||||
mobile UI, and autonomy features. Your laptop becomes a window onto a session living on the
|
||||
remote host.
|
||||
|
||||
Like Docker, this is a **location overlay** on a case, not a run mode. All seven run modes
|
||||
work remotely. See [Core Concepts](Core-Concepts).
|
||||
|
||||
## Why bother
|
||||
|
||||
The agent runs where the work is: a build server, a NAS, a GPU box, a machine reachable only
|
||||
through a jump host. Your laptop can sleep, change networks, or close, and the run continues.
|
||||
|
||||
## Setting it up
|
||||
|
||||
**Add Case → Remote**:
|
||||
|
||||
| Field | Notes |
|
||||
| --------------------- | -------------------------------------------------------------------- |
|
||||
| **Host** | Hostname or IP. |
|
||||
| **Username** | The SSH user. |
|
||||
| **Port** | Defaults to 22. |
|
||||
| **Identity file** | `~` and `$HOME` are expanded for you. |
|
||||
| **Jump host** | The `-J` equivalent, `[user@]host[:port]`. |
|
||||
| **SOCKS proxy** | For hosts reachable only through a proxy. |
|
||||
| **Extra SSH options** | Any `KEY=VALUE` options your normal connection needs. |
|
||||
| **Remote path** | The working directory on that machine. |
|
||||
|
||||
Hosts are saved and reusable, so a second case on the same machine is just a path. Host
|
||||
profiles can also carry per-run-mode launch command overrides, for when the binary lives
|
||||
somewhere unusual on that host.
|
||||
|
||||
The remote host needs **tmux**. Codeman probes for it when you link the host rather than
|
||||
failing later at launch.
|
||||
|
||||
## What actually runs
|
||||
|
||||
The agent lives inside a dedicated tmux server on the **remote** host, and Codeman fronts it
|
||||
with a local tmux pane running `ssh`.
|
||||
|
||||
That two-layer arrangement is what makes it durable: a dropped SSH connection, a network
|
||||
change, or a closed laptop kills the local pane, not the remote session. Reconnecting lands
|
||||
back in the same live conversation.
|
||||
|
||||
The remote session name is deliberately chosen so that a Codeman **running on the target
|
||||
host** will not adopt it as one of its own. Two Codemans, one host, no interference.
|
||||
|
||||
## Auto-reconnect
|
||||
|
||||
A watcher with bounded backoff notices a dead SSH pane and quietly reattaches to the still
|
||||
running remote session. On by default; the kill switch is in
|
||||
**App Settings → Agents & CLIs → Remote auto-reconnect**.
|
||||
|
||||
Intentional kills are never revived. Closing a session means closing it.
|
||||
|
||||
## Discover and attach
|
||||
|
||||
Codeman can list the `codeman-*` sessions already running on a host, whether that machine's
|
||||
own Codeman started them or another operator did, and attach to one.
|
||||
|
||||
The distinction that matters:
|
||||
|
||||
| Session | On tab close |
|
||||
| ------------ | ------------------------------------------------ |
|
||||
| **Launched** | Killed, like any local session. |
|
||||
| **Attached** | **Detached, never killed.** |
|
||||
|
||||
Attaching to someone else's session and closing your tab must not end their run, so it does
|
||||
not. Several clients can attach the same remote session at different window sizes without
|
||||
clamping each other, and discovery shows a shared badge with the client count.
|
||||
|
||||
## Security
|
||||
|
||||
Every SSH command line in Codeman flows through one builder that shell-escapes every
|
||||
user-supplied field: identity paths, jump hosts, proxy commands, and extra options. That is
|
||||
the entire injection surface, and it is deliberately a single function rather than string
|
||||
concatenation spread across the codebase.
|
||||
|
||||
Host, path, and identity fields are schema-validated on top of that.
|
||||
|
||||
Codeman does not store SSH passwords. Use keys, as you would for any other automation.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **The remote host needs tmux.** Probed at link time, so you find out immediately.
|
||||
- **The local working directory is meaningless** for a remote session, and is not used.
|
||||
- **Run flows must go through the quick-start path** for remote cases. This matters if you
|
||||
are driving Codeman over the API: the plain session-create endpoint validates the working
|
||||
directory locally and has no case concept, so it will reject or misroute a remote case.
|
||||
- **Latency is SSH latency.** Local echo helps the typing feel, but a slow link is a slow
|
||||
link.
|
||||
- **Transcript-backed features follow the transcript.** Subagent windows and similar surfaces
|
||||
read files on the machine where the agent runs.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Core Concepts](Core-Concepts) - overlays versus run modes.
|
||||
- [Docker Cases](Docker-Cases) - the other overlay.
|
||||
- [Security](Security) - the wider model.
|
||||
- [`docs/remote-sessions.md`](https://github.com/Ark0N/Codeman/blob/master/docs/remote-sessions.md) - the full design.
|
||||
@@ -0,0 +1,197 @@
|
||||
# Running As A Service
|
||||
|
||||
Keeping Codeman up: past the shell you started it in, past a logout, past a reboot. Plus
|
||||
logs, updates, and running more than one instance.
|
||||
|
||||
## Three levels
|
||||
|
||||
| Level | Survives | Command |
|
||||
| -------------------- | ----------------------------------------- | ------------------------- |
|
||||
| Foreground | Nothing. Dies with the terminal. | `codeman web` |
|
||||
| Detached | Closing the shell and logging out. | `codeman web -d` |
|
||||
| Service | Reboots. | `codeman service install` |
|
||||
|
||||
Agents themselves survive all three, because they live in tmux. Stopping the server never
|
||||
stops the agents.
|
||||
|
||||
## Detached mode
|
||||
|
||||
```bash
|
||||
codeman web -d # start detached; logs to ~/.codeman/web.log
|
||||
codeman web --status # is it up, and on which pid
|
||||
codeman web --stop # graceful stop; agents keep running
|
||||
```
|
||||
|
||||
`-d` waits until the server actually answers before reporting success, so a port clash never
|
||||
reads as a successful start.
|
||||
|
||||
Two implementation details that explain the behaviour:
|
||||
|
||||
- It relaunches the same entry script detached, so there is no controlling terminal and no
|
||||
shell job entry. `nohup` is **not** what makes this work: Node re-arms the hangup signal to
|
||||
its default even when it inherits "ignore", and Codeman handles that signal with a graceful
|
||||
shutdown, so a delivered hangup would still stop the server.
|
||||
- `--stop` verifies the process still looks like a Codeman server before signalling it,
|
||||
because process ids get recycled.
|
||||
|
||||
**It refuses to start a second server on the same data directory.** Two servers sharing a
|
||||
tmux socket attach to each other's live sessions.
|
||||
|
||||
## Installing as a service
|
||||
|
||||
```bash
|
||||
codeman service install # systemd user unit on Linux, LaunchAgent on macOS
|
||||
codeman service status
|
||||
codeman service uninstall
|
||||
```
|
||||
|
||||
The installer's final menu offers this too.
|
||||
|
||||
Notable behaviours:
|
||||
|
||||
- **Your PATH is baked into the unit.** launchd hands a job
|
||||
`/usr/bin:/bin:/usr/sbin:/sbin`, which finds neither a Homebrew or nvm `node` nor `tmux`
|
||||
or `claude`. This is the single most common cause of a hand-written unit that starts and
|
||||
immediately dies.
|
||||
- **`CODEMAN_PASSWORD` is never written into the unit file.** Add it yourself if the service
|
||||
needs authentication.
|
||||
- **It refuses when a server is already running** on that data directory, for the same reason
|
||||
detached mode does.
|
||||
- **It verifies rather than assumes.** `launchctl load` and a clean spawn are both silent
|
||||
about a server that starts and immediately exits, so the parent polls until the child
|
||||
answers or dies.
|
||||
|
||||
On Linux, if you want the service running while you are not logged in:
|
||||
|
||||
```bash
|
||||
loginctl enable-linger $USER
|
||||
```
|
||||
|
||||
### Writing the unit by hand
|
||||
|
||||
**Linux (systemd user unit):**
|
||||
|
||||
```bash
|
||||
mkdir -p ~/.config/systemd/user
|
||||
cat > ~/.config/systemd/user/codeman-web.service << EOF
|
||||
[Unit]
|
||||
Description=Codeman Web Server
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
ExecStart=$(which node) $HOME/.codeman/app/dist/index.js web
|
||||
Restart=always
|
||||
RestartSec=10
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
EOF
|
||||
systemctl --user daemon-reload
|
||||
systemctl --user enable --now codeman-web
|
||||
loginctl enable-linger $USER
|
||||
```
|
||||
|
||||
**macOS (LaunchAgent):**
|
||||
|
||||
```bash
|
||||
mkdir -p ~/Library/LaunchAgents
|
||||
cat > ~/Library/LaunchAgents/com.codeman.web.plist << EOF
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN"
|
||||
"http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>com.codeman.web</string>
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>$(which node)</string>
|
||||
<string>$HOME/.codeman/app/dist/index.js</string>
|
||||
<string>web</string>
|
||||
</array>
|
||||
<key>RunAtLoad</key><true/>
|
||||
<key>KeepAlive</key><true/>
|
||||
<key>StandardOutPath</key>
|
||||
<string>/tmp/codeman.log</string>
|
||||
<key>StandardErrorPath</key>
|
||||
<string>/tmp/codeman.log</string>
|
||||
</dict>
|
||||
</plist>
|
||||
EOF
|
||||
launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/com.codeman.web.plist
|
||||
```
|
||||
|
||||
Prefer `codeman service install` where you can. It handles the PATH problem for you.
|
||||
|
||||
## Logs
|
||||
|
||||
```bash
|
||||
journalctl --user -u codeman-web -f # systemd
|
||||
tail -f ~/.codeman/web.log # detached mode
|
||||
log stream --predicate 'process == "node"' # macOS, noisy
|
||||
```
|
||||
|
||||
## Updating
|
||||
|
||||
| Install route | Update with |
|
||||
| ------------- | ------------------------------------------------------------------------ |
|
||||
| Installer | Re-run the one-liner, or **App Settings → System → Updates**. |
|
||||
| npm | `npm update -g aicodeman` |
|
||||
| git clone | `git pull && npm install && npm run build`, then restart. |
|
||||
|
||||
### The in-app updater
|
||||
|
||||
**App Settings → System → Updates**, for git-clone installs supervised by systemd or
|
||||
launchd. npm installs report as non-updatable, and an unsupervised install is told to
|
||||
restart manually.
|
||||
|
||||
The interesting part is that the update restarts the very process running it. So the real
|
||||
work runs in a **detached script that outlives the restart** and writes progress to a status
|
||||
file, which the browser polls across the connection drop. A dirty tree is stashed rather
|
||||
than discarded.
|
||||
|
||||
### After updating
|
||||
|
||||
Sessions are unaffected: they live in tmux and the server reattaches. If the UI looks stale,
|
||||
reload; on iOS Safari, close the tab completely and reopen.
|
||||
|
||||
## Running two instances
|
||||
|
||||
The data directory and the tmux socket are process wide, so a second server on the defaults
|
||||
will discover and attach the first one's sessions. Scope both together:
|
||||
|
||||
```bash
|
||||
CODEMAN_INSTANCE=beta CODEMAN_PORT=5000 codeman web
|
||||
```
|
||||
|
||||
Service unit names are instance-scoped too, so a beta instance can be installed as its own
|
||||
service without colliding with the main one. `CODEMAN_DATA_DIR` and `CODEMAN_TMUX_SOCKET`
|
||||
exist for the rare case where they need to differ, but setting only one of them recreates
|
||||
exactly the problem you were avoiding.
|
||||
|
||||
## The tunnel as a service
|
||||
|
||||
```bash
|
||||
systemctl --user enable codeman-tunnel
|
||||
loginctl enable-linger $USER
|
||||
```
|
||||
|
||||
Or the toggle in **App Settings → System → Remote access**. See
|
||||
[Remote Access](Remote-Access).
|
||||
|
||||
## Health checks
|
||||
|
||||
```bash
|
||||
curl -s localhost:3000/api/status | jq '.version, .uptime'
|
||||
codeman web --status
|
||||
codeman doctor
|
||||
```
|
||||
|
||||
Add `-k` and the `https://` URL on an HTTPS install.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Installation](Installation) - the routes and what each supports.
|
||||
- [Remote Access](Remote-Access) - exposing it once it stays up.
|
||||
- [Troubleshooting](Troubleshooting) - when it does not.
|
||||
@@ -0,0 +1,120 @@
|
||||
# Security
|
||||
|
||||
The honest version first: **Codeman's dashboard is a remote code execution surface, by
|
||||
design.** It starts agents with permission prompts skipped by default, so anyone who can
|
||||
reach it can run arbitrary code as your user, on your machine. Every protection in Codeman
|
||||
exists to control who that is.
|
||||
|
||||
That is not a flaw to be fixed. It is what "run my coding agent for me" means. The job is to
|
||||
make sure the set of people who can reach it is exactly the set you intended.
|
||||
|
||||
## The default is safe
|
||||
|
||||
A bare `codeman web` binds `127.0.0.1`. Only processes on that machine can reach it, which
|
||||
is why shipping with no password by default is defensible. Everything risky starts when you
|
||||
expose it.
|
||||
|
||||
## Hardening checklist
|
||||
|
||||
In order of how much they matter:
|
||||
|
||||
1. **Do not expose it without `CODEMAN_PASSWORD`.** Binding a non-loopback host without one
|
||||
starts, but warns loudly. A tunnel refuses outright unless you acknowledge the exposure
|
||||
in the UI.
|
||||
2. **Prefer Tailscale over a public tunnel.** Keeping the loopback bind and putting a
|
||||
private network in front of it removes the public attack surface entirely, and gives you
|
||||
real HTTPS. See [Remote Access](Remote-Access).
|
||||
3. **Use a long password.** It is the only thing between a public URL and your shell.
|
||||
4. **Consider the permission mode.** **App Settings → Agents & CLIs → Claude → Startup
|
||||
Mode** can switch new sessions from skip-prompts to Anthropic's classifier-guarded `auto`
|
||||
mode, to normal prompting, or to an explicit allowed-tools list.
|
||||
5. **Use Docker cases for untrusted work.** If you are pointing an autonomous loop at a repo
|
||||
you did not write, [Docker Cases](Docker-Cases) gives it its own filesystem and network
|
||||
for the cost of one checkbox.
|
||||
6. **Keep it updated.** Browser-driven attack paths were closed in 0.9.x and hardening is
|
||||
ongoing.
|
||||
|
||||
## What protects what
|
||||
|
||||
These run on **every** request, including on a default no-password loopback install:
|
||||
|
||||
| Layer | What it stops |
|
||||
| ---------------------------- | ----------------------------------------------------------------------------------------------- |
|
||||
| **Host-header allowlist** | DNS rebinding. A domain rebound to `127.0.0.1` is rejected before any handler runs. Add your own domains with `CODEMAN_ALLOWED_HOSTS`. |
|
||||
| **Cross-site Origin guard** | CSRF on state-changing requests. A *missing* Origin is allowed so curl, the CLI, and hooks keep working; a foreign or opaque one is rejected. |
|
||||
| **Raw `text/plain` bodies** | The CORS simple-request CSRF vector, where a cross-site form could smuggle JSON into a write route with no preflight. |
|
||||
| **WebSocket origin check** | Cross-site WebSocket hijacking. The terminal upgrade closes with code `4003` on failure. |
|
||||
| **Output escaping** | Stored XSS from agent-derived strings: tool names, command arguments, subagent descriptions. |
|
||||
| **Security headers** | A strict content security policy, `nosniff`, frame options, and HSTS over HTTPS. CORS is reflected only for loopback origins. |
|
||||
|
||||
When authentication is enabled:
|
||||
|
||||
| Layer | Behaviour |
|
||||
| ------------------- | ------------------------------------------------------------------------------------------------ |
|
||||
| **HTTP Basic** | `CODEMAN_USERNAME` (default `admin`) and `CODEMAN_PASSWORD`. |
|
||||
| **Session cookie** | A 256-bit opaque token validated server side, so it cannot be forged offline. 24 hours, extended on activity, with a device-context audit trail. |
|
||||
| **Rate limiting** | Ten failed attempts per IP produce a `429` with a 15 minute decay. A correct password or valid cookie recovers immediately even under attack, which matters because all tunnel traffic shares one loopback address. |
|
||||
| **QR auth** | Single-use 60-second tokens with their own separate rate limiter, so a mistyped password cannot lock out QR login. |
|
||||
| **Hook endpoints** | The hook and telemetry endpoints skip Basic auth because they are called from localhost by the CLI, but when auth is on, that bypass additionally requires a per-instance hook secret. |
|
||||
|
||||
## File access
|
||||
|
||||
Three separate file surfaces, each confined differently, because a single shared rule would
|
||||
be wrong for at least one of them:
|
||||
|
||||
| Surface | Rules |
|
||||
| -------------------- | -------------------------------------------------------------------------------------------- |
|
||||
| **File Viewer** | Real path resolution before boundary checks, so symlinks cannot escape. Sensitive trees blocked. Edit mode adds an extension allowlist, a size cap, `.git` denial, and optimistic concurrency. It never creates files. |
|
||||
| **Attachments** | An id-based registry, so browser requests never carry absolute paths. The magic-link scanner is prompt-injectable by nature and is therefore force-confined to the session's workspace. Extension allowlist, not a blocklist. |
|
||||
| **Path picker** | Its own root allowlist rather than the workspace confinement. In multi-user mode a non-admin gets only their own user space, because per-user spaces live inside the home directory. |
|
||||
|
||||
Downloads block sensitive paths outright (`.env`, credentials files, `~/.ssh`, AWS
|
||||
credentials), and SVG and HTML are served as downloads with `nosniff` so they cannot execute
|
||||
in the page.
|
||||
|
||||
## Supply chain and isolation
|
||||
|
||||
- Security-sensitive transitive dependencies are pinned to patched versions, and lockfile
|
||||
integrity is checked on every push and pull request: every entry must resolve to the public
|
||||
registry with a hash.
|
||||
- Public assets are scanned for NUL bytes and syntax-checked in CI.
|
||||
- `CODEMAN_INSTANCE` scopes the tmux socket and the data directory together, so two
|
||||
instances never attach each other's live sessions.
|
||||
|
||||
## What Codeman does not protect against
|
||||
|
||||
Stated plainly, because a security page that only lists strengths is not useful:
|
||||
|
||||
- **Multi-user mode is not a sandbox.** It separates workspaces. Every session still runs as
|
||||
the same OS account, so a determined user's agent can reach another user's files. For real
|
||||
isolation, pair users with Docker cases or run separate instances under separate OS
|
||||
accounts.
|
||||
- **An agent you gave shell access can do anything you can.** Permission modes narrow this;
|
||||
they do not remove it.
|
||||
- **A tunnel makes your machine reachable from the internet.** The password is the whole
|
||||
boundary. Treat it accordingly.
|
||||
- **Codeman cannot detect your own loopback reverse proxy**, which is why the hook-endpoint
|
||||
bypass requires a secret unconditionally when auth is on.
|
||||
- **The agent CLIs have their own trust models.** Pi's project trust executes repo-local
|
||||
TypeScript, for instance. See [Agent CLIs](Agent-CLIs).
|
||||
|
||||
## Privacy
|
||||
|
||||
No telemetry, no analytics, no phone-home. Codeman's only network traffic is between your
|
||||
browser and your server. Your agent CLI's traffic is its own, on your account.
|
||||
|
||||
Two features send data outward, both off by default and both stated where they appear: voice
|
||||
dictation through your Claude login, and the Read My Mind prediction call.
|
||||
|
||||
## Reporting a vulnerability
|
||||
|
||||
**Never in a public issue.**
|
||||
[SECURITY.md](https://github.com/Ark0N/Codeman/blob/master/.github/SECURITY.md) has the
|
||||
private disclosure process and the current list of known limitations.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Remote Access](Remote-Access) - the safe ways to expose it.
|
||||
- [Multi-User Mode](Multi-User-Mode) - what it does and does not separate.
|
||||
- [Docker Cases](Docker-Cases) - real isolation for untrusted work.
|
||||
- [`docs/security-architecture.md`](https://github.com/Ark0N/Codeman/blob/master/docs/security-architecture.md) - the complete model.
|
||||
@@ -0,0 +1,172 @@
|
||||
# Settings Reference
|
||||
|
||||
Two settings surfaces, and the rule that explains why a setting you changed on your laptop
|
||||
did not follow you to your phone.
|
||||
|
||||
| Surface | Scope | Opened from |
|
||||
| ------------------- | ------------------------------ | ---------------------------- |
|
||||
| **App Settings** | Global, this Codeman install. | The header gear. |
|
||||
| **Session Options** | One session. | The session's tab. |
|
||||
|
||||
App Settings is a single scrolling document with a rail acting as a table of contents;
|
||||
clicking a rail entry scrolls rather than switching. Session Options genuinely switches
|
||||
panels.
|
||||
|
||||
## Per-device versus synced
|
||||
|
||||
Some settings live on the server and follow you to every device. Others are stored in the
|
||||
browser and stay put. This is deliberate, not an oversight: your phone wants a different
|
||||
font size, a different keyboard bar, and a different set of header buttons than your
|
||||
desktop.
|
||||
|
||||
| Category | Examples |
|
||||
| ----------------------- | ------------------------------------------------------------------------------- |
|
||||
| **Per-device, local** | Skin, WebGL renderer, local echo, CJK input, extended keyboard bar, File Viewer and Cron header buttons. Never sent to the server at all. |
|
||||
| **Per-device policy** | Most `show*` toggles, plan usage chip, language. Stored server-side, but a device only takes the server value when it has no local one of its own. |
|
||||
| **Synced** | Models, effort, CLI options, notification preferences, voice settings, display name, the agent skill and approvals toggles. |
|
||||
|
||||
The practical rule: **appearance and input are per device, behaviour is shared.** If a change
|
||||
did not follow you, it is in one of the first two rows, and you change it again on that
|
||||
device.
|
||||
|
||||
## App Settings
|
||||
|
||||
### Updates
|
||||
|
||||
Current version, a manual check, and the in-app updater. Covers git-clone installs
|
||||
supervised by systemd or launchd; npm installs report as non-updatable. See
|
||||
[Running As A Service](Running-As-A-Service).
|
||||
|
||||
### Terminal & Input
|
||||
|
||||
| Setting | Default | Notes |
|
||||
| ----------------------------- | -------------------- | --------------------------------------------------------------------- |
|
||||
| Local Echo | On for touch devices | Paints keystrokes locally and flushes on Enter. See [Input And Voice](Input-And-Voice). |
|
||||
| CJK Input | Off | IME composition through a dedicated text field. |
|
||||
| Extended Keyboard Bar | Per device | Which accessory bar phones get. Shell sessions override it while they are active. |
|
||||
| Wheel Scrolls Local History | Off | Keeps the wheel on the local buffer instead of forwarding it to the CLI. |
|
||||
| WebGL Renderer | On | With a GPU-stall watchdog that falls back to DOM rendering. |
|
||||
| Gesture Control | Off | Camera hand tracking. Also needs `CODEMAN_GESTURE=1` on the server. |
|
||||
|
||||
### Header & Panels
|
||||
|
||||
Chips for every optional header control, with a live preview of the resulting header:
|
||||
|
||||
Run, Font Size, System Stats, Redraw Terminal, Response Viewer, Away Digest, Session
|
||||
Manager, Attachments, File Viewer, Multi-monitor, Plan Usage, Lifecycle Log, Monitor,
|
||||
Project Insights, File Browser, Subagents, Approvals Inbox, Read My Mind, Ultracode Agents,
|
||||
Ultracode Windows, Cron.
|
||||
|
||||
Most default to off. The stock desktop header is system stats, File Viewer, and the gear.
|
||||
New header controls never appear on phones.
|
||||
|
||||
This section also holds background-agent tracking, including whether to track agents for
|
||||
every session or only the active tab.
|
||||
|
||||
### Appearance
|
||||
|
||||
| Setting | Notes |
|
||||
| ---------------------- | ----------------------------------------------------------------------------------------- |
|
||||
| Skin | Theme palettes, light ones included. Applied before first paint, so no flash of the wrong theme. |
|
||||
| Entrance Animations | Per-surface animation styles for tabs, terminals, windows, and lineage lines. All default to the legacy no-animation behaviour. |
|
||||
| Display Name | Your name in the UI. Cosmetic only; it never renames the package, CLI, API, or storage. |
|
||||
| Interface Language | English or Simplified Chinese. Per device. |
|
||||
| Session List Layout | Header tab strip (default) or a collapsible left sidebar. See [The Dashboard](The-Dashboard#session-list-layout). |
|
||||
| Tall Tabs | Taller tab strip. |
|
||||
| Pop-out Button on Tabs | Adds the detach control to tabs, with a per-tab override. |
|
||||
| Spawn Lineage Lines | Arcs from a parent tab to sessions it spawned. Desktop only, on by default. |
|
||||
| Overview Home Screen | The phone home screen. On by default. |
|
||||
|
||||
### Models
|
||||
|
||||
Claude model cards, the 1M context window switch, and the thinking effort segment. The cards
|
||||
and the switch compose into one model choice, so there is no separate "which one wins"
|
||||
question.
|
||||
|
||||
Model and effort are both **soft defaults**: the model is written into the case's
|
||||
`.claude/settings.local.json` and effort is passed at start, so `/model` and `/effort`
|
||||
inside a session override them at any time.
|
||||
|
||||
### Agents & CLIs
|
||||
|
||||
| Setting | Notes |
|
||||
| -------------------------------- | -------------------------------------------------------------------------------------------- |
|
||||
| Startup Mode | Claude's permission mode for new sessions. Default skips prompts; `auto` uses Anthropic's classifier-guarded mode; `normal` prompts; or give an explicit allowed-tools list. |
|
||||
| Allowed Tools | The list used by the explicit mode. |
|
||||
| Ralph / Todo Tracker | Enables the Ralph loop surfaces. |
|
||||
| Agent Teams | Experimental teams. Also needs the CLI's own environment flag. |
|
||||
| Codeman Agent Skill | Injects the agent skill into new Claude sessions per case. Off by default. See [Driving Codeman From An Agent](Driving-Codeman-From-An-Agent). |
|
||||
| Remote auto-reconnect | Reattaches dropped remote SSH sessions. On by default. |
|
||||
| Nice priority / value | Runs agent processes at a lower CPU priority. |
|
||||
| Bypass approvals and sandbox | Pi's project trust. Read [Agent CLIs](Agent-CLIs) before enabling. |
|
||||
| Animated status effects | Cosmetic. |
|
||||
|
||||
### Notifications
|
||||
|
||||
Master toggle, browser notifications, push subscription, audio alerts, and the idle
|
||||
threshold that decides when a quiet session counts as needing you. See
|
||||
[Notifications And Approvals](Notifications-And-Approvals).
|
||||
|
||||
### Voice
|
||||
|
||||
Active provider and the engine behind it, insert mode, language, domain keywords to bias
|
||||
recognition, the Deepgram API key, and the opt-in switch for transcribing through this
|
||||
server's Claude login, with its live credential status. See
|
||||
[Input And Voice](Input-And-Voice).
|
||||
|
||||
### Shortcuts
|
||||
|
||||
Rebinding for the shortcut registry. See [Keyboard Shortcuts](Keyboard-Shortcuts).
|
||||
|
||||
### System
|
||||
|
||||
`CLAUDE.md` template for new cases, default working directory, the image watcher, and
|
||||
Cloudflare tunnel controls including the tunnel and upload URLs. In multi-user mode, the
|
||||
**Users** administration entry is injected here.
|
||||
|
||||
## Session Options
|
||||
|
||||
Per session, from the tab.
|
||||
|
||||
| Panel | Contains |
|
||||
| ---------------- | ------------------------------------------------------------------------------------------- |
|
||||
| **Respawn** | Auto-resume on usage limit, the respawn cycle configuration, presets, duration. See [Keeping Agents Running](Keeping-Agents-Running). |
|
||||
| **Session** | Name, working directory, environment overrides, per-tab pop-out override. |
|
||||
| **Ralph / Todo** | Loop configuration, iteration and todo caps, circuit breaker reset. See [Autonomous Loops](Autonomous-Loops). |
|
||||
| **Summary** | What this session has done: tokens, activity, run summary. |
|
||||
|
||||
Panels that only make sense for Claude are hidden for other run modes rather than shown and
|
||||
failing.
|
||||
|
||||
## Environment variables
|
||||
|
||||
Some things are configured before the server starts, not in the UI:
|
||||
|
||||
| Variable | Effect |
|
||||
| ----------------------------------- | ---------------------------------------------------------------------- |
|
||||
| `CODEMAN_PORT` | Listen port. |
|
||||
| `CODEMAN_HOST` | Bind address. Loopback by default. |
|
||||
| `CODEMAN_PASSWORD` / `CODEMAN_USERNAME` | HTTP Basic credentials. Username defaults to `admin`. |
|
||||
| `CODEMAN_ALLOWED_HOSTS` | Extra Host and Origin allowlist entries for a reverse proxy. |
|
||||
| `CODEMAN_INSTANCE` | Scopes the data directory and tmux socket together. Required for a second instance. |
|
||||
| `CODEMAN_MULTIUSER` | Enables multi-user mode. |
|
||||
| `CODEMAN_GESTURE` | Makes gesture control available to be enabled. |
|
||||
| `CODEMAN_DOCKER_BRIDGE_HOOKS` | Lets in-container hooks reach the host on a loopback bind. |
|
||||
| `CODEMAN_FILE_PICKER_ROOTS` | Extra roots for the path picker. |
|
||||
| `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK` | Acknowledges exposing the server with no password. |
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **A setting that did not sync is per device.** Change it again on that device.
|
||||
- **The plan usage chip and its telemetry exporter are one setting.** Enabling the chip
|
||||
without the exporter would leave it blank forever, so it is deliberately not separable.
|
||||
- **Toggling a header button does nothing on a phone.** Phones deliberately ignore most of
|
||||
the header chips.
|
||||
- **Enabling a feature does not retroactively configure existing sessions.** The agent skill
|
||||
injection, for instance, applies at session creation.
|
||||
|
||||
## Read next
|
||||
|
||||
- [The Dashboard](The-Dashboard) - what each control does once visible.
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - the Respawn panel in depth.
|
||||
- [Agent CLIs](Agent-CLIs) - model, effort, and permission modes.
|
||||
@@ -0,0 +1,204 @@
|
||||
# The Dashboard
|
||||
|
||||
What the interface is telling you, and which parts of it are hidden until you turn them on.
|
||||
|
||||
Most of Codeman's UI is **opt-in**. A stock install shows a deliberately small header, and a
|
||||
feature you read about here may simply not be on screen yet. Where that is the case, this
|
||||
page says so and names the setting.
|
||||
|
||||

|
||||
|
||||
## Layout
|
||||
|
||||
| Region | What lives there |
|
||||
| ------------------ | -------------------------------------------------------------------------------------- |
|
||||
| **Header, left** | The "C" logo (goes home) and the session list, unless you moved it to the sidebar. |
|
||||
| **Header, right** | Status chips and panel buttons, most of them off by default. |
|
||||
| **Center** | The terminal for the active session, or the home screen when nothing is selected. |
|
||||
| **Bottom toolbar** | Run, Stop, Run Shell, the case picker, and the instance counters. |
|
||||
| **Overlays** | Panels and modals: Respawn, Cron, Subagents, File Viewer, Settings. |
|
||||
|
||||
## Session list layout
|
||||
|
||||
The session list lives in the header as a horizontal strip by default. With a lot of
|
||||
sessions open that strip stops being scannable, so **App Settings → Appearance → Tabs →
|
||||
Session List Layout** can move it into a vertical sidebar on the left instead.
|
||||
|
||||
| Layout | Behaviour |
|
||||
| -------------------- | --------------------------------------------------------------------------------- |
|
||||
| **Header tab strip** | The default. Wraps to a second row on desktop, scrolls sideways on a phone. |
|
||||
| **Left sidebar** | A vertical list with a filter box and a live session count. `Alt+B` collapses it to a narrow rail that keeps the status dots and task badges visible. On a phone it is an off-canvas drawer rather than a docked rail. |
|
||||
|
||||
It is the same list either way, just re-hosted: tab order, drag-to-reorder, the `Alt+1`
|
||||
to `Alt+9` numbers and every status colour below behave identically in both. The setting is
|
||||
per device, so a sidebar on your desktop does not force one onto your phone.
|
||||
|
||||
## Session tabs
|
||||
|
||||
One tab per session, in your order, and that order syncs across your devices.
|
||||
|
||||
**Status is carried by the dot and the tab's own styling:**
|
||||
|
||||
| Look | Meaning |
|
||||
| ----------------------------- | ----------------------------------------------------------------------- |
|
||||
| Green dot | Alive, not currently working. |
|
||||
| Pulsing green dot with a ring | Working on a turn. |
|
||||
| Yellow tab, blinking | The agent is waiting for input from you. |
|
||||
| Red tab, blinking | A question or permission prompt is blocking the session. |
|
||||
| No dot | The session is not running. |
|
||||
|
||||

|
||||
|
||||
The alert states are steady colour with a pulse layered on top, not a blink between the
|
||||
alert colour and nothing, so a tab that needs you looks like it needs you at every point in
|
||||
the cycle. They survive a page reload: the state is re-seeded from the server on load, so
|
||||
reloading while a permission prompt is blocking does not lose the red tab.
|
||||
|
||||
**Navigation:**
|
||||
|
||||
| Action | Keys |
|
||||
| ------------------------------- | ------------------------------------------------------- |
|
||||
| Jump to tab N | `Alt+1` to `Alt+9` (the number on the tab) |
|
||||
| Next / previous | `Ctrl+Tab`, `Alt+[`, `Alt+]` |
|
||||
| Move the active tab | `Ctrl+Shift+{`, `Ctrl+Shift+}` |
|
||||
| Close | `Ctrl+W` |
|
||||
| Find any session, open or past | `Ctrl+K` (also `Cmd+K` and `Alt+K`) |
|
||||
|
||||
Tabs can also be dragged to reorder.
|
||||
|
||||
On phones the strip scrolls horizontally instead of wrapping, and the active tab is always
|
||||
scrolled into view. It is not reordered to the front, so the `Alt+N` numbering stays stable.
|
||||
|
||||
### Lineage arcs
|
||||
|
||||
When one session spawns another (an agent starting a worker through the API), Codeman draws
|
||||
a coloured arc under the strip connecting parent to child, with one colour per child. It is
|
||||
how a fan-out of eight workers stays readable.
|
||||
|
||||
Desktop only, and on by default. Turn it off in **App Settings → Appearance**. Arcs are
|
||||
skipped for tabs scrolled out of the strip.
|
||||
|
||||
## Header controls
|
||||
|
||||
The right side of the header. Almost all of these are off until you enable them in
|
||||
**App Settings → Header & Panels**.
|
||||
|
||||
| Control | Default | What it does |
|
||||
| ---------------------- | ------------------ | ------------------------------------------------------------------------------- |
|
||||
| Connection dot | Always on | SSE connection health. Green is connected. |
|
||||
| Font size `-` / `+` | Always on | `Ctrl +` / `Ctrl -` do the same. |
|
||||
| CPU / MEM bars | On | Server resource use. |
|
||||
| File Viewer | On | Toggles the file browser panel. |
|
||||
| Settings gear | Always on | App Settings. |
|
||||
| Plan usage chip | On, desktop only | Live Claude subscription usage. Claude-only, and needs its telemetry exporter, which the same setting installs. |
|
||||
| Session Manager | Off | The full session list, live and historical. |
|
||||
| Approvals bell | Off | Cross-session queue of prompts waiting on a human. Appears only when the count is above zero. Never shown on phones. |
|
||||
| Read My Mind 🧠 | Off | Predicts your next prompt for this case. Claude-only. |
|
||||
| Attachments | Off | Registered external files. |
|
||||
| Away Digest | Off | What happened while you were gone. |
|
||||
| Last Response | Off | Readable view of the agent's last answer, useful on phones. |
|
||||
| Ultracode / Workflow | Off | Live workflow-run agents. |
|
||||
| Notifications | Off | Notification history and settings. |
|
||||
| Lifecycle Log | Off | Session start, exit, and kill audit trail. |
|
||||
| Cron ⏰ | Off | Scheduled jobs. |
|
||||
| Multi-monitor | Off, macOS | Opens a window spanning every display. |
|
||||
| Tunnel indicator | When a tunnel runs | Cloudflare tunnel status. |
|
||||
| Admin panel | Multi-user only | User administration. |
|
||||
|
||||
New header controls never appear on phones. Phone layout is deliberately minimal and is
|
||||
covered in [Mobile Guide](Mobile-Guide).
|
||||
|
||||
## Connection state
|
||||
|
||||
The dot in the header is the quick read. Two louder surfaces exist because a cached page
|
||||
with no server behind it used to look identical to a page with no sessions:
|
||||
|
||||
- **A full-screen overlay** when the page has never loaded server state. There is nothing
|
||||
behind it worth preserving.
|
||||
- **A banner** when the connection drops after state had loaded, so your scrollback stays
|
||||
readable.
|
||||
|
||||
Both wait about 2.5 seconds before appearing, so a deploy that restarts the server does not
|
||||
flash a warning at you every time. If the browser reports itself offline, the grace period
|
||||
is skipped.
|
||||
|
||||
There is also a watchdog for the case where the connection stops delivering without
|
||||
erroring. If the server's heartbeat stops arriving, Codeman reconnects on its own rather
|
||||
than sitting on a green dot showing frozen data.
|
||||
|
||||
## The terminal
|
||||
|
||||
A real terminal: xterm.js in the browser, a real PTY on the server, tmux in between. Full
|
||||
TUIs render correctly.
|
||||
|
||||
Worth knowing:
|
||||
|
||||
- **Scrollback.** The first time you open a session, Codeman pulls the entire tmux
|
||||
scrollback, not just the recent tail. Scrolling to the very top pulls again on demand.
|
||||
- **Wheel and touch scrolling** are forwarded into Claude's own transcript on recent Claude
|
||||
versions, so the wheel scrolls the conversation rather than the terminal. `Shift+Wheel` is
|
||||
always local scrollback. Other CLIs scroll locally.
|
||||
- **Selection copy.** `Ctrl+C` copies when text is selected and interrupts when it is not.
|
||||
`Ctrl+Shift+C` always copies.
|
||||
- **Zero-lag input.** On touch devices, keystrokes paint locally before the round trip. See
|
||||
[Input And Voice](Input-And-Voice).
|
||||
- **Renderer.** WebGL by default, with a watchdog that falls back to DOM rendering if the
|
||||
GPU stalls. `?nowebgl` forces DOM rendering for one page load.
|
||||
|
||||
## The home screen
|
||||
|
||||
With no session selected you get the welcome screen: run buttons for the CLIs Codeman
|
||||
found, a QR code when a password is set, cross-session search, and **Resume Conversation**,
|
||||
which lists past sessions including Claude conversations started outside Codeman entirely.
|
||||
|
||||
Two extras depending on the device:
|
||||
|
||||
- **Desktop, wide windows**: your open tabs appear as a rail docked to the left edge, in tab
|
||||
order, with created and last-active stamps. It needs at least 1180px of width; below that
|
||||
it is hidden so it cannot overlap the search panel.
|
||||
- **Phones**: tapping the "C" logo gives a session overview instead: NEEDS YOU first, then
|
||||
current sessions, then past ones. On by default.
|
||||
|
||||
## Panels
|
||||
|
||||
| Panel | Opened from | Covered in |
|
||||
| ---------------- | --------------------------------- | ---------------------------------------------------------------- |
|
||||
| Respawn | Session Options | [Keeping Agents Running](Keeping-Agents-Running) |
|
||||
| Ralph | Session Options | [Autonomous Loops](Autonomous-Loops) |
|
||||
| Orchestrator | Toolbar | [Autonomous Loops](Autonomous-Loops) |
|
||||
| Cron | Header ⏰ (opt-in) | [Cron Jobs](Cron-Jobs) |
|
||||
| Subagents | Automatic while agents run | [Watching Agents Work](Watching-Agents-Work) |
|
||||
| Ultracode | Header (opt-in) | [Watching Agents Work](Watching-Agents-Work) |
|
||||
| File Viewer | Header | [Working With Files](Working-With-Files) |
|
||||
| Attachments | Header (opt-in) | [Working With Files](Working-With-Files) |
|
||||
| Approvals | Header bell (opt-in) | [Notifications And Approvals](Notifications-And-Approvals) |
|
||||
| App Settings | Header gear | [Settings Reference](Settings-Reference) |
|
||||
|
||||
Session-specific configuration lives in **Session Options**, reachable from the tab. App
|
||||
Settings is global; Session Options is per session.
|
||||
|
||||
## Search and the session palette
|
||||
|
||||
`Ctrl+K` opens the session palette: every session, live or historical, filtered as you
|
||||
type. Picking a past one resumes its conversation.
|
||||
|
||||
The search box on the home screen is wider in scope. It federates over session metadata,
|
||||
run-summary events, and attachment history, filtered by type, case, status, and date. It
|
||||
does substring matching over data already in memory, with no regex and no filesystem reads,
|
||||
so it is fast and cannot be turned into a traversal.
|
||||
|
||||
## Appearance
|
||||
|
||||
**App Settings → Appearance** carries the theme skins, including light ones. The choice is
|
||||
applied before the first paint, so there is no flash of the wrong theme on load.
|
||||
|
||||
The same section has the entrance animations for tabs, terminals, agent windows, and
|
||||
lineage lines. All of them default to the legacy no-animation behaviour, so an untouched
|
||||
install animates nothing.
|
||||
|
||||
## Read next
|
||||
|
||||
- [Keyboard Shortcuts](Keyboard-Shortcuts) - the full list, and how to rebind.
|
||||
- [Settings Reference](Settings-Reference) - every setting, and why some follow you across devices and others do not.
|
||||
- [Mobile Guide](Mobile-Guide) - what changes on a phone.
|
||||
- [Watching Agents Work](Watching-Agents-Work) - subagent windows and workflow runs.
|
||||
@@ -0,0 +1,290 @@
|
||||
# Troubleshooting
|
||||
|
||||
Symptom first. Find the line that matches what you are seeing.
|
||||
|
||||
Before anything else, check what version you are on and whether the problem is already
|
||||
fixed:
|
||||
|
||||
```bash
|
||||
codeman --version
|
||||
codeman doctor
|
||||
```
|
||||
|
||||
## Installing and starting
|
||||
|
||||
### `Failed to start claude: error: posix_spawnp failed` on macOS
|
||||
|
||||
node-pty ships its macOS `spawn-helper` without the executable bit, and macOS launches
|
||||
every PTY through it. Codeman detects this and repairs it on the first failure, so updating
|
||||
usually fixes it outright. To repair by hand on a clone install:
|
||||
|
||||
```bash
|
||||
npm run fix:node-pty
|
||||
```
|
||||
|
||||
It is a `chmod`, not a rebuild, so it does not need Xcode command line tools. The helper
|
||||
lives in `prebuilds/darwin-<arch>/`, not `build/Release/`, which does not exist on macOS.
|
||||
Linux never sees this.
|
||||
|
||||
### `tmux: command not found`
|
||||
|
||||
The installer asks before installing packages and remembers a declined answer. Install tmux
|
||||
and start again. There is no tmux-free mode: sessions live in tmux.
|
||||
|
||||
### The port is already in use
|
||||
|
||||
```bash
|
||||
codeman web --port 8080 # or set CODEMAN_PORT
|
||||
```
|
||||
|
||||
If you believe nothing is on 3000, check for a Codeman you already started:
|
||||
|
||||
```bash
|
||||
codeman web --status
|
||||
```
|
||||
|
||||
### The terminal area is blank, and the console mentions a missing vendor file
|
||||
|
||||
Clone installs build the vendored xterm addon bundles in `postinstall`. If `npm install`
|
||||
was interrupted or run with `--ignore-scripts`, those bundles are missing:
|
||||
|
||||
```bash
|
||||
npm install
|
||||
```
|
||||
|
||||
They are intentionally not committed to the repository.
|
||||
|
||||
### `Case path not found` when clicking Run
|
||||
|
||||
The case points at a directory that no longer exists, usually because it was deleted or
|
||||
moved outside Codeman. Re-link the case, or create it again.
|
||||
|
||||
### The server starts but nothing is reachable
|
||||
|
||||
That is the default behaviour, not a failure. Codeman binds `127.0.0.1`. See
|
||||
[Remote Access](Remote-Access).
|
||||
|
||||
## Reaching the interface
|
||||
|
||||
### The dashboard will not load from another device
|
||||
|
||||
Check, in order: the bind (loopback by default), a firewall, and then
|
||||
[Remote Access](Remote-Access) for a supported way to expose it.
|
||||
|
||||
### `403 host not allowed`
|
||||
|
||||
The Host header is not in the allowlist, which is the DNS-rebinding guard doing its job. Add
|
||||
your domain:
|
||||
|
||||
```bash
|
||||
CODEMAN_ALLOWED_HOSTS='codeman.example.com,.internal.example.com'
|
||||
```
|
||||
|
||||
A leading dot matches subdomains.
|
||||
|
||||
### The page loads but the terminal never connects
|
||||
|
||||
The terminal is a WebSocket. Behind a reverse proxy, the upgrade must be forwarded. The
|
||||
upgrade also runs the Host and Origin checks and closes with code `4003` when they fail.
|
||||
|
||||
### The UI looks stale after updating
|
||||
|
||||
The app shell is cached by a service worker, and static assets are served with a long cache
|
||||
lifetime. `index.html` is not cached, and every asset reference is version-stamped, so a
|
||||
normal reload picks up a new build.
|
||||
|
||||
Two exceptions worth knowing:
|
||||
|
||||
- **iOS Safari** can keep serving old JavaScript until the tab is fully closed, not just
|
||||
reloaded. Close the tab and reopen it.
|
||||
- If you edit files in dev, changes to `index.html` need a server restart. Changes to `.js`
|
||||
and `.css` do not.
|
||||
|
||||
### A full-screen "cannot reach the server" overlay appears
|
||||
|
||||
The server is genuinely unreachable, or the connection dropped. Codeman waits about 2.5
|
||||
seconds before showing it, so a quick restart does not flash it. Retry re-arms both the
|
||||
event stream and the terminal socket.
|
||||
|
||||
## Sessions
|
||||
|
||||
### A session shows idle while it is clearly working
|
||||
|
||||
Update. Claude redraws its prompt roughly once a second throughout a turn, and older idle
|
||||
detection treated that as the end of the turn, flipping working sessions to idle a couple of
|
||||
seconds in. Current versions confirm against the actual screen before believing it.
|
||||
|
||||
### A session is stuck showing busy
|
||||
|
||||
For non-Claude CLIs, idle detection is output-based and coarser by necessity: those CLIs
|
||||
expose no hooks. A session that has genuinely gone quiet will settle. If it never does,
|
||||
interrupt it (`Ctrl+C` with nothing selected).
|
||||
|
||||
### The agent asks about bypass permissions every time
|
||||
|
||||
That prompt comes from Claude Code, not Codeman. Codeman's default is to start with
|
||||
permission prompts skipped, which is what the security model is built around. If you would
|
||||
rather it prompted, change **App Settings → Agents & CLIs → Claude → Startup Mode**.
|
||||
|
||||
### Sessions vanished after a reboot
|
||||
|
||||
Expected. tmux does not survive a reboot, so the sessions are gone. Conversations are not:
|
||||
Claude transcripts persist, so the welcome screen's **Resume Conversation** list can pick
|
||||
them back up.
|
||||
|
||||
### A session restarts, then refuses to restart again
|
||||
|
||||
That is the PTY-exit circuit breaker. Repeated rapid PTY exits trip it, and it blocks
|
||||
automatic restarts so a broken configuration does not spin forever. Reset it explicitly from
|
||||
the session's controls. Reattaching does not clear it, deliberately.
|
||||
|
||||
### Sessions I did not create appeared, or my session resized itself
|
||||
|
||||
Two Codeman servers are running against the same data directory and tmux socket. The second
|
||||
one discovers and attaches the first one's sessions. Give each instance its own scope:
|
||||
|
||||
```bash
|
||||
CODEMAN_INSTANCE=beta CODEMAN_PORT=5000 codeman web
|
||||
```
|
||||
|
||||
`codeman web -d` and `codeman service install` both refuse to start a second server on one
|
||||
data directory for exactly this reason.
|
||||
|
||||
## The terminal
|
||||
|
||||
### I cannot scroll back through history
|
||||
|
||||
Scrollback behaviour depends on the CLI, and Codeman adjusts what it strips per mode.
|
||||
Things to try:
|
||||
|
||||
- `Shift+Wheel` always scrolls the local buffer, whatever else is going on.
|
||||
- On Claude sessions with a recent CLI, the wheel is forwarded into Claude's own transcript,
|
||||
so it scrolls the conversation rather than the terminal buffer. That is intended.
|
||||
- Scrolling to the very top pulls the full tmux scrollback again on demand.
|
||||
|
||||
### The wheel does nothing in a Codex session
|
||||
|
||||
Codex ignores the mouse reports that forwarding would send, so Codeman does not forward
|
||||
there. Scrolling is local, and `Shift+Wheel` behaves the same way.
|
||||
|
||||
### `Ctrl+C` copies when I wanted to interrupt
|
||||
|
||||
With a selection, `Ctrl+C` copies. With no selection, it interrupts. Clear the selection
|
||||
first, or use the **Stop** button. `Ctrl+Shift+C` always copies and never interrupts.
|
||||
|
||||
### I typed a prompt but nothing was sent
|
||||
|
||||
On touch devices, keystrokes are painted locally and flushed when you press Enter, so text
|
||||
on screen has not necessarily reached the agent yet. Press Enter, or the phone toolbar's
|
||||
**Enter** button.
|
||||
|
||||
If you are sending input over the API instead, your payload must end with `\r` or no Enter
|
||||
is ever sent. The request still succeeds and the text sits unsubmitted in the composer. See
|
||||
[Driving Codeman From An Agent](Driving-Codeman-From-An-Agent).
|
||||
|
||||
## Mobile
|
||||
|
||||
### The keyboard covers the terminal, or scroll position jumps
|
||||
|
||||
Update first; several rounds of fixes have gone into keyboard resize and scroll restoration.
|
||||
|
||||
### I cannot reach the rightmost tabs
|
||||
|
||||
The strip scrolls horizontally on phones and the active tab is scrolled into view
|
||||
automatically. Swipe the strip itself. If a background render snaps you back, update.
|
||||
|
||||
### The space key does nothing on Android
|
||||
|
||||
A long-standing Android keyboard bug, fixed some time ago. Update.
|
||||
|
||||
### The keyboard will not close
|
||||
|
||||
Tap outside the terminal, or tap twice on inert terminal content. Tapping a control does not
|
||||
dismiss it, by design.
|
||||
|
||||
## Agents and CLIs
|
||||
|
||||
### A CLI is installed but Codeman does not offer it
|
||||
|
||||
Codeman resolves binaries from the environment the **server** runs in.
|
||||
|
||||
```bash
|
||||
codeman doctor
|
||||
```
|
||||
|
||||
If it runs as a service, launchd gives the job a minimal PATH. `codeman service install`
|
||||
bakes your PATH into the unit; a hand-written plist does not. Restart the server after
|
||||
installing a new CLI.
|
||||
|
||||
### Hooks stopped working after switching to HTTPS
|
||||
|
||||
Hook callbacks have to accept the self-signed certificate. Recent versions self-heal
|
||||
existing cases; if yours predates that, recreate the case so its hooks are rewritten.
|
||||
|
||||
### The model or effort I chose is not being used
|
||||
|
||||
Both are **soft defaults**, on purpose. The model is written into the case's
|
||||
`.claude/settings.local.json` and effort is passed on the command line at start, so `/model`
|
||||
and `/effort` inside the session override them at any time. Effort is deliberately never
|
||||
passed as an environment variable, because that hard-locks it.
|
||||
|
||||
### Tab alerts and approvals never fire in one of my repos
|
||||
|
||||
That case is missing its hooks block. Recreating the case rewrites it.
|
||||
|
||||
## Docker and remote
|
||||
|
||||
### Docker sessions do not detect idle
|
||||
|
||||
On a loopback-only bind, a container cannot reach `127.0.0.1` on the host, so in-container
|
||||
hooks have nothing to call. Set `CODEMAN_DOCKER_BRIDGE_HOOKS=1` to open a hooks-only
|
||||
listener on the docker bridge gateway. Without it, idle detection falls back to output
|
||||
watching.
|
||||
|
||||
### A rebuilt agent image still has old CLI versions
|
||||
|
||||
Always rebuild with `--no-cache`:
|
||||
|
||||
```bash
|
||||
node scripts/build-agent-image.mjs --no-cache
|
||||
```
|
||||
|
||||
A plain rebuild reuses the cached `npm install -g` layer and keeps the CLIs frozen at their
|
||||
original versions while reporting success.
|
||||
|
||||
### A remote SSH session dropped and did not come back
|
||||
|
||||
A bounded-backoff watcher reattaches dropped sessions, and it is on by default. Intentional
|
||||
kills are never revived. Check the host is reachable and that the remote tmux server is
|
||||
still running.
|
||||
|
||||
## Gathering diagnostics
|
||||
|
||||
```bash
|
||||
codeman doctor # dependency check
|
||||
curl -s localhost:3000/api/status | jq # full app state
|
||||
tmux -L codeman list-sessions # what tmux thinks is alive
|
||||
journalctl --user -u codeman-web -f # service logs (Linux)
|
||||
tail -f ~/.codeman/web.log # detached mode logs
|
||||
```
|
||||
|
||||
On an HTTPS install, add `-k` to the curl commands and use the `https://` URL.
|
||||
|
||||
## Filing a good bug report
|
||||
|
||||
Open an [issue](https://github.com/Ark0N/Codeman/issues) with:
|
||||
|
||||
- OS and version.
|
||||
- Install method: installer, npm, or git clone.
|
||||
- `codeman --version`.
|
||||
- Browser and version, if the problem is in the UI.
|
||||
- Which CLI the session was running, and its version.
|
||||
- What you did, what happened, what you expected.
|
||||
|
||||
Reports usually get a response within a day, and every release credits its reporters by
|
||||
name.
|
||||
|
||||
Questions and setup help fit better in
|
||||
[Discussions](https://github.com/Ark0N/Codeman/discussions). Security problems never go in a
|
||||
public issue; see
|
||||
[SECURITY.md](https://github.com/Ark0N/Codeman/blob/master/.github/SECURITY.md).
|
||||
@@ -0,0 +1,72 @@
|
||||
# Versioning
|
||||
|
||||
Codeman follows [semantic versioning](https://semver.org/). This page says what the version
|
||||
number actually promises, which matters if you are building anything against Codeman.
|
||||
|
||||
## Covered by the version number
|
||||
|
||||
Breaking any of these after 1.0 requires a **major** bump:
|
||||
|
||||
1. **The CLI.** Command names, documented flags, and their behaviour. The npm package is
|
||||
`aicodeman` and installs both the `aicodeman` and `codeman` commands; renaming either is
|
||||
breaking.
|
||||
2. **The HTTP API and SSE channel**, served under `/api/v1` with the uniform envelope and
|
||||
conventional status codes. Endpoint paths, the envelope, `errorCode` values, and SSE event
|
||||
names are all stable.
|
||||
3. **Documented deployment environment variables**: `CODEMAN_PASSWORD`, `CODEMAN_USERNAME`,
|
||||
`CODEMAN_HOST`, `CODEMAN_PORT`, `CODEMAN_INSTANCE`, `CODEMAN_ALLOWED_HOSTS`,
|
||||
`CODEMAN_DATA_DIR`, `CODEMAN_TMUX_SOCKET`, plus the `--host`, `--port`, and `--https`
|
||||
flags.
|
||||
4. **The published `xterm-zerolag-input` library**, on its own independent version line.
|
||||
Codeman reaching 1.0 says nothing about that package's version.
|
||||
|
||||
Additive changes are **not** breaking: new endpoints, new optional fields, new error codes,
|
||||
new SSE events. Genuinely breaking API changes would ship under a new prefix rather than
|
||||
changing `/api/v1`.
|
||||
|
||||
## Not covered
|
||||
|
||||
These can change in a minor or even patch release:
|
||||
|
||||
1. **The `~/.codeman/` state file formats.** Migrations are made on a best-effort basis and
|
||||
have been done across renames, but the on-disk shape is not a contract. Do not write
|
||||
tooling against it.
|
||||
2. **Internal TypeScript modules.** The npm package is CLI-only. There is no stable library
|
||||
entry point, and importing it programmatically is unsupported.
|
||||
3. **Experimental and opt-in features**, whatever the app's version: gesture control, agent
|
||||
teams, and anything labelled experimental in the UI or docs.
|
||||
|
||||
## Deprecation
|
||||
|
||||
- Additive changes are preferred over breaking ones.
|
||||
- A covered surface slated for removal is deprecated first: it keeps working for at least one
|
||||
minor release, with a runtime warning and a changelog note pointing at the replacement,
|
||||
then is removed in the next major.
|
||||
- Backwards-compatibility shims are kept until a major boundary.
|
||||
|
||||
## Releases
|
||||
|
||||
Releases are managed with changesets. Every release:
|
||||
|
||||
- Bumps the version and updates
|
||||
[`CHANGELOG.md`](https://github.com/Ark0N/Codeman/blob/master/CHANGELOG.md).
|
||||
- Publishes to npm as `aicodeman`.
|
||||
- Cuts a GitHub release, tagged `codeman@X.Y.Z`.
|
||||
- **Credits its contributors and bug reporters by name** in the release notes.
|
||||
|
||||
There is no fixed cadence. Patches ship when fixes are ready, which in practice is often.
|
||||
|
||||
## Which version am I on?
|
||||
|
||||
```bash
|
||||
codeman --version
|
||||
```
|
||||
|
||||
Or **App Settings → Updates**, which also checks for a newer one and can install it. See
|
||||
[Running As A Service](Running-As-A-Service).
|
||||
|
||||
## Read next
|
||||
|
||||
- [HTTP API](HTTP-API) - the stable API surface itself.
|
||||
- [Contributing](Contributing) - how changes get made.
|
||||
- [`docs/versioning-policy.md`](https://github.com/Ark0N/Codeman/blob/master/docs/versioning-policy.md) - the authoritative statement.
|
||||
@@ -0,0 +1,112 @@
|
||||
# Watching Agents Work
|
||||
|
||||
Modern agents fan out. A single Claude session can be running six subagents, and the parent
|
||||
terminal shows you almost none of it. Codeman surfaces that hidden work as live windows,
|
||||
panels, and after-the-fact summaries.
|
||||
|
||||
Everything on this page is Claude-only. It reads Claude Code's transcripts and team state;
|
||||
the other CLIs expose no equivalent.
|
||||
|
||||

|
||||
|
||||
## Subagent windows
|
||||
|
||||
When a Claude session spawns subagents, each one gets its own floating window with a live
|
||||
transcript: what it was asked to do, what it is doing, and what it returned.
|
||||
|
||||
- Windows are draggable and resizable, and their positions persist across reloads.
|
||||
- A connection line links each window to the session tab that spawned it, so with four
|
||||
sessions running you can still tell whose worker is whose.
|
||||
- Closing a window does not stop the subagent. It only stops you watching it.
|
||||
|
||||
This is the feature that makes a fan-out legible. Without it, a lead session that spawned
|
||||
eight workers looks like a stalled terminal for several minutes.
|
||||
|
||||
## Session lineage arcs
|
||||
|
||||
The tab strip draws a coloured arc from a parent tab to any tab it spawned, one colour per
|
||||
child. That covers the other direction of fan-out: not subagents inside one session, but
|
||||
whole sessions started by an agent through the API.
|
||||
|
||||
Desktop only, on by default, and toggled in **App Settings → Appearance**. Arcs are skipped
|
||||
for tabs scrolled out of view.
|
||||
|
||||
See [Driving Codeman From An Agent](Driving-Codeman-From-An-Agent) for the spawning side.
|
||||
|
||||
## Agent teams
|
||||
|
||||
Claude Code's experimental agent teams appear as teammates alongside subagents. Enable them
|
||||
in the CLI's own environment:
|
||||
|
||||
```bash
|
||||
CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1
|
||||
```
|
||||
|
||||
and turn the per-case **Agent Teams** toggle on in the case settings gear.
|
||||
|
||||
Codeman watches the team directory and matches teammates to the session leading them.
|
||||
Teammates are in-process threads rather than separate CLI processes, so they show up as
|
||||
windows, not tabs.
|
||||
|
||||
Notes and the experiment log:
|
||||
[`docs/agent-teams/`](https://github.com/Ark0N/Codeman/tree/master/docs/agent-teams).
|
||||
|
||||
## Ultracode and workflow runs
|
||||
|
||||
When Claude runs a Workflow, dozens of agents can be in flight at once. The completion
|
||||
artifact for a run is only written at the **end**, so a live run would otherwise be
|
||||
invisible until it finished. Codeman synthesizes the in-flight view from the transcripts and
|
||||
lets the real artifact supersede it when it lands.
|
||||
|
||||
Two independent toggles, both off by default:
|
||||
|
||||
| Setting | Shows |
|
||||
| ---------------------- | ----------------------------------------- |
|
||||
| Ultracode panel | A docked panel listing the run's agents. |
|
||||
| Ultracode windows | Floating windows, like subagents. |
|
||||
|
||||
Turning on either starts the watcher.
|
||||
|
||||
## Reading the answer, not the terminal
|
||||
|
||||
**Last Response** (header button, opt-in) renders the agent's last answer as scrollable text
|
||||
rather than terminal output. It exists mostly for phones, where reading a long answer in a
|
||||
terminal viewport is painful. **More** loads additional context.
|
||||
|
||||
## After the fact
|
||||
|
||||
| Surface | Answers |
|
||||
| ------------------ | -------------------------------------------------------------- |
|
||||
| **Away Digest** | What happened while I was gone? |
|
||||
| **Run summary** | What did this run actually do? |
|
||||
| **Lifecycle log** | When did sessions start, exit, or get killed, and why? |
|
||||
| **Token stats** | What did it cost? |
|
||||
|
||||
The Away Digest aggregates the lifecycle log, run summary events, live sessions, token
|
||||
statistics, and recent subagents into one view. It is the right first thing to open in the
|
||||
morning after an overnight run.
|
||||
|
||||
All of these header buttons are opt-in: **App Settings → Header & Panels**.
|
||||
|
||||
## Performance
|
||||
|
||||
The design target is 20 sessions and 50 agent windows at 60fps. If you routinely run more
|
||||
than that, expect the browser rather than the server to be the limit, and close windows you
|
||||
are not reading.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **A session pointed at a relocated Claude config directory goes blind here.** Transcripts
|
||||
written outside `~/.claude/projects` are invisible to the watchers, so subagent windows,
|
||||
the ultracode panel, the response viewer, and Read My Mind all stop working for that
|
||||
session. Symlink `projects` back into the shared tree to fix it. See
|
||||
[Agent CLIs](Agent-CLIs).
|
||||
- **Closing a window does not cancel the agent.** Nothing on this page controls agents; it
|
||||
observes them.
|
||||
- **Windows are opt-in for ultracode, automatic for subagents.**
|
||||
|
||||
## Read next
|
||||
|
||||
- [The Dashboard](The-Dashboard) - where these surfaces live.
|
||||
- [Driving Codeman From An Agent](Driving-Codeman-From-An-Agent) - the other kind of fan-out.
|
||||
- [Autonomous Loops](Autonomous-Loops) - the loops that generate this much activity.
|
||||
@@ -0,0 +1,101 @@
|
||||
# Web Tabs
|
||||
|
||||
Open any dashboard you run, Grafana, Uptime Kuma, Portainer, a status page on port 4000, as
|
||||
a tab beside your agent sessions. Codeman becomes one mission control instead of Codeman
|
||||
plus a pile of browser tabs.
|
||||
|
||||
A web tab is **not a session**. There is no PTY, no tmux, and no respawn behind it, the same
|
||||
way a docker case is not a run mode.
|
||||
|
||||
## Adding one
|
||||
|
||||
1. Click the chevron next to **Run**.
|
||||
2. Under **Web / URL**, pick **Add URL**.
|
||||
3. Name it, paste the URL, optionally hit **Test**, and **Save**.
|
||||
|
||||
It opens immediately and appears in the dropdown from then on. Web tabs share the tab strip
|
||||
with sessions, continue the same `Alt+1` to `Alt+9` numbering, and carry a globe icon so
|
||||
they never read as a running agent.
|
||||
|
||||
**Closing a tab is not deleting it.** The tab's `x` closes; the `x` on its **dropdown row**
|
||||
deletes the saved dashboard. Each dropdown row also has a gear for editing the URL.
|
||||
|
||||
Switching tabs does not reload a dashboard. Frames stay alive in the background, so one that
|
||||
took a while to authenticate is still there when you come back. Past six live frames, the
|
||||
least recently viewed is dropped to bound memory.
|
||||
|
||||
## Why dashboards are proxied
|
||||
|
||||
A plain cross-origin iframe fails three ways at once in the setup Codeman actually ships in:
|
||||
|
||||
| Blocker | What happens |
|
||||
| ------------------- | ----------------------------------------------------------------------------------------------- |
|
||||
| **Mixed content** | Production is HTTPS, and browsers hard-block `http://` iframes on an HTTPS page. No override, and none at all on iOS Safari. |
|
||||
| **Framing refusal** | Grafana, Portainer, Home Assistant and many others send `X-Frame-Options: DENY`. |
|
||||
| **Codeman's CSP** | `default-src 'self'` blocks a cross-origin frame before it starts. |
|
||||
|
||||
So by default the dashboard is served **through Codeman's own origin**: the browser loads a
|
||||
path on Codeman, and Codeman relays to the dashboard, stripping the framing refusal,
|
||||
rewriting redirects, cookies and root-absolute URLs, and relaying WebSockets so live panels
|
||||
still update.
|
||||
|
||||
A useful side effect: the dashboard is fetched **by the Codeman server**, so a tailnet-only
|
||||
or localhost-only dashboard works from any device that can reach Codeman, including a phone
|
||||
that is not on your tailnet.
|
||||
|
||||
There is also a `direct` mode, a plain cross-origin iframe, which is cheaper but only works
|
||||
for an HTTPS dashboard that permits framing.
|
||||
|
||||
## The Test button, and what it does not test
|
||||
|
||||
**Test** probes from the server and tells you which mode applies. It verifies
|
||||
**server-to-upstream reachability and nothing else**. It does not exercise the browser
|
||||
sandbox, cookies, CORS, CSP, or any reverse proxy in front of Codeman.
|
||||
|
||||
A passing Test does not guarantee the embedded page renders.
|
||||
|
||||
## The sandbox, and when to turn it off
|
||||
|
||||
Because a proxied dashboard is served from Codeman's own address, the browser considers it
|
||||
same-origin with Codeman. Unchecked, its JavaScript could read the Codeman page and call the
|
||||
API that spawns agents.
|
||||
|
||||
So the frame is sandboxed **without** same-origin access by default. The page runs in an
|
||||
opaque origin: it cannot touch Codeman, and it gets no cookies or local storage of its own.
|
||||
|
||||
Unchecking **Open sandboxed** grants a real origin. Do that only for a dashboard you fully
|
||||
trust, and only when you need it, which in practice means one with its own login that stores
|
||||
a session in a cookie.
|
||||
|
||||
Either way, Codeman never forwards its own credentials upstream. The `Authorization` header
|
||||
and the `codeman_session` cookie are stripped on the way out, so `CODEMAN_PASSWORD` cannot
|
||||
leak into a dashboard.
|
||||
|
||||
## Known incompatibility: cookie-authenticated reverse proxies
|
||||
|
||||
If Codeman itself sits behind Cloudflare Access, Authelia, oauth2-proxy, or similar, a
|
||||
**sandboxed** tab may render unstyled or broken while the Codeman page around it works fine.
|
||||
|
||||
The reason: an opaque-origin frame's stylesheet, script, and API requests do not carry the
|
||||
proxy's authentication cookie. The proxy redirects them to the login provider, and CORS or
|
||||
CSP kills them there.
|
||||
|
||||
Trusted mode keeps a real origin and the cookie, so it works. Test cannot catch this, because
|
||||
it checks the server's reach, not the browser's.
|
||||
|
||||
## Security notes
|
||||
|
||||
The proxy authenticates on an in-memory capability embedded in the path, which is why it is
|
||||
exempt from the cookie and Origin checks that every API route enforces. That exemption is
|
||||
fenced to safe methods and non-API paths, and there is a test pinning it in place.
|
||||
|
||||
Two failure modes that only appear inside a sandboxed frame, and that curl can never
|
||||
reproduce, are handled: runtime-built root-absolute URLs escaping the injected base, and
|
||||
same-host requests being CORS-checked with a null origin. Both present as the dashboard's own
|
||||
"Failed to fetch" while the page itself renders fine.
|
||||
|
||||
## Read next
|
||||
|
||||
- [The Dashboard](The-Dashboard) - the tab strip these share.
|
||||
- [Security](Security) - why the sandbox default is what it is.
|
||||
- [`docs/web-tabs.md`](https://github.com/Ark0N/Codeman/blob/master/docs/web-tabs.md) - the full reference.
|
||||
@@ -0,0 +1,159 @@
|
||||
# Working With Files
|
||||
|
||||
Reading, editing, attaching, and previewing files without leaving the dashboard. Useful on
|
||||
a desktop; on a phone it is the difference between reviewing an agent's work and waiting
|
||||
until you get home.
|
||||
|
||||
## The File Viewer
|
||||
|
||||
A panel that browses the active session's working directory. Its header button is on by
|
||||
default; if it is missing, re-enable it in **App Settings → Header & Panels**.
|
||||
|
||||
It renders what it can:
|
||||
|
||||
| Kind | Behaviour |
|
||||
| ------------------------ | ------------------------------------------------------------------------- |
|
||||
| Text and code | Syntax-aware preview. Long files are truncated in plain preview. |
|
||||
| Images | Inline. |
|
||||
| Audio and video | Inline with a working scrub bar, because range requests are supported. |
|
||||
| PDF and Office documents | Converted for preview when a converter is available. |
|
||||
| Anything else | Download. |
|
||||
|
||||
Caps: 10 MB for text preview, 50 MB for raw and download. Sensitive paths (`.env`, anything
|
||||
matching credentials, `~/.ssh`, AWS credentials) are blocked from download, and SVG and HTML
|
||||
are served as downloads rather than rendered, so they cannot execute in the page.
|
||||
|
||||
Closing the preview pauses and unloads any playing media. A video that keeps playing after
|
||||
you close the panel means you are on an old version.
|
||||
|
||||
## Editing in place
|
||||
|
||||
Text files can be edited and saved directly in the viewer. Click the pencil in the preview
|
||||
header, edit, **Save**.
|
||||
|
||||
The guardrails are worth knowing, because they are what makes editing safe rather than
|
||||
convenient:
|
||||
|
||||
- **Extension allowlist**, not a blocklist. Code, docs, config, and markup are editable.
|
||||
Anything not on the list is not.
|
||||
- **512 KB cap** on both read and write.
|
||||
- **Edit mode never truncates.** The plain preview does truncate long files, and saving a
|
||||
truncated buffer would silently delete the rest, so the editor loads the whole file or
|
||||
refuses.
|
||||
- **Optimistic concurrency.** The save carries a hash of what you started from. If the file
|
||||
changed underneath you (likely, when an agent is working in the same repo), the save is
|
||||
rejected rather than clobbering their work.
|
||||
- **No file creation.** Writes go to a temporary file and are renamed over the original, and
|
||||
the open never creates. Editing in place is structural, not a rule.
|
||||
- **Line endings are preserved** server-side, so editing two lines of a CRLF file does not
|
||||
produce a whole-file diff.
|
||||
- **`.git/` is denied outright.** Hooks are executable code, and a corrupted index looks
|
||||
unrecoverable to someone who wanted to fix a typo.
|
||||
- **Non-UTF-8 content is refused**, verified by a round-trip comparison.
|
||||
|
||||
## Attachments
|
||||
|
||||
Attachments are live references to files **outside** the session's workspace: a spec on your
|
||||
desktop, a PDF in Downloads, a design document elsewhere on the machine.
|
||||
|
||||
Register one from the CLI:
|
||||
|
||||
```bash
|
||||
codeman attach /path/to/spec.pdf
|
||||
```
|
||||
|
||||
An attachment card appears in the session, and the file can be previewed inline. The
|
||||
attachment gets a stable id, and browser requests use that id rather than carrying absolute
|
||||
paths around.
|
||||
|
||||
Agents can register attachments too, by emitting a `codeman://attach?...` link in their
|
||||
output. That path is **prompt-injectable by nature**, so it is force-confined to the
|
||||
session's workspace: a hostile prompt cannot use it to pull arbitrary host files into the
|
||||
event stream. The gate is an extension allowlist rather than a blocklist.
|
||||
|
||||
Document conversion for previews is globally rate limited. Without that, ten large documents
|
||||
detected at once would fork ten multi-minute converter processes.
|
||||
|
||||
## Clicking a path
|
||||
|
||||
File paths in a session are links. That works in two places:
|
||||
|
||||
- **In the terminal**, on any absolute path an agent prints.
|
||||
- **In the response viewer**, where paths are usually written as prose or in backticks. They
|
||||
render as underlined monospace links.
|
||||
|
||||
Clicking one opens it in the preview: images and PDFs render, video and audio play with a
|
||||
working scrub bar, documents convert, text and Markdown show inline. Log-shaped files open in
|
||||
the tail viewer instead, which follows a file that is still being written.
|
||||
|
||||
Paths **outside** the session's workspace work too, which matters because that is where most
|
||||
of an agent's output lands: a screenshot in `/tmp`, a capture in its own scratchpad, a file in
|
||||
another checkout. Those are served through the attachment routes rather than the workspace
|
||||
ones, so the same rules apply as to any other attachment: secret trees are blocked, the
|
||||
extension allowlist decides what can be opened, and symlinks are resolved before either check.
|
||||
|
||||
Outside the workspace the allowlist is images, video, audio, PDF, Office documents, and text
|
||||
files, where "text" is the same list the viewer will let you edit: code, config, logs, csv,
|
||||
markdown. The reasoning is that a session can already `cat` any of those, so the file suffix
|
||||
was never what kept anything secret; the path guard is. Types outside the list (`.svg`,
|
||||
`.bmp`) say so rather than failing silently, and `.html` previews as source rather than being
|
||||
rendered, so nothing served this way can execute in the page.
|
||||
|
||||
Text previews are capped at the first 500 lines, fetched as a partial read, so clicking a
|
||||
one-gigabyte log does not try to paint one.
|
||||
|
||||
Log-shaped files inside the workspace still open in the tail viewer, which follows a file as
|
||||
it is written. Outside the workspace they open in the preview instead: the tail viewer runs
|
||||
`tail -f`, and that is deliberately restricted to the workspace, `/var/log` and `~/logs`.
|
||||
|
||||
Nothing is registered until you click. Opening a file this way does not add an attachment card.
|
||||
|
||||
## The path picker
|
||||
|
||||
For choosing a path rather than typing one. It appears in two places:
|
||||
|
||||
- **Browse** in **Add Case → Link Existing**.
|
||||
- The **📁 Path** key on the mobile keyboard bar.
|
||||
|
||||
It browses one directory at a time and can show hidden entries on request. The picker
|
||||
inserts the path into your prompt **without** pressing Enter, so nothing is submitted by
|
||||
accident. Its sibling **⌫ All** key clears the unsent prompt, and never sends the agent's
|
||||
`/clear` command.
|
||||
|
||||
This is a separate file-serving surface from the viewer, with its own rules: it allowlists
|
||||
your home directory, the cases directory, and anything in `CODEMAN_FILE_PICKER_ROOTS`, and
|
||||
blocks sensitive trees. In multi-user mode a non-admin gets only their own user space as a
|
||||
root, because per-user spaces live inside the home directory and a home-directory root would
|
||||
expose everyone.
|
||||
|
||||
## Images into a session
|
||||
|
||||
Paste from the clipboard or drag and drop straight onto the terminal. The image is written
|
||||
where the agent can read it and the reference is inserted into your prompt. On a phone, the
|
||||
image key in the keyboard bar opens the camera or photo library.
|
||||
|
||||
HEIC images from an iPhone are converted to JPEG on the way in.
|
||||
|
||||
## Generated artifacts
|
||||
|
||||
When an agent produces a file the UI can show (a chart, a diagram, a document), it can
|
||||
surface as an artifact attachment rather than a path you have to go and find.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **The viewer follows the active session's workspace.** Switching tabs changes what you are
|
||||
browsing.
|
||||
- **A save can be rejected, and that is the feature.** It means the agent edited the file
|
||||
while you were typing. Re-open, re-apply, save again.
|
||||
- **Attachments live outside the workspace on purpose.** For files inside it, just use the
|
||||
viewer.
|
||||
- **`.env` files are readable in the viewer if the extension policy allows the preview, but
|
||||
never downloadable.** Do not treat the viewer as a secrets boundary; treat the machine as
|
||||
the boundary.
|
||||
|
||||
## Read next
|
||||
|
||||
- [The Dashboard](The-Dashboard) - where the panels live.
|
||||
- [Input And Voice](Input-And-Voice) - other ways to get content into a session.
|
||||
- [Security](Security) - how the file surfaces are confined.
|
||||
- [`docs/file-viewer-edit-plan.md`](https://github.com/Ark0N/Codeman/blob/master/docs/file-viewer-edit-plan.md) - the edit-mode design.
|
||||
@@ -0,0 +1,9 @@
|
||||
Documents Codeman **{{VERSION}}**. Something wrong or missing on this page? These pages are
|
||||
generated from [`docs/wiki/`](https://github.com/Ark0N/Codeman/tree/master/docs/wiki) in
|
||||
the main repository, so browser edits here are overwritten on the next sync. Send a pull
|
||||
request against that directory instead, or open a
|
||||
[Discussion](https://github.com/Ark0N/Codeman/discussions).
|
||||
|
||||
<!-- {{VERSION}} is replaced with the current major.minor series by
|
||||
.github/workflows/wiki-sync.yml at publish time. Do not hardcode a
|
||||
version here: it went stale every release when it was hand-written. -->
|
||||
@@ -0,0 +1,53 @@
|
||||
### [Codeman Wiki](Home)
|
||||
|
||||
[README](https://github.com/Ark0N/Codeman)
|
||||
|
||||
**Getting started**
|
||||
|
||||
- [Installation](Installation)
|
||||
- [Quick Start](Quick-Start)
|
||||
- [Core Concepts](Core-Concepts)
|
||||
|
||||
**Using it**
|
||||
|
||||
- [The Dashboard](The-Dashboard)
|
||||
- [Agent CLIs](Agent-CLIs)
|
||||
- [Working With Files](Working-With-Files)
|
||||
- [Input And Voice](Input-And-Voice)
|
||||
- [Mobile Guide](Mobile-Guide)
|
||||
- [Keyboard Shortcuts](Keyboard-Shortcuts)
|
||||
- [Settings Reference](Settings-Reference)
|
||||
|
||||
**Keeping agents running**
|
||||
|
||||
- [Unattended Runs](Keeping-Agents-Running)
|
||||
- [Notifications & Approvals](Notifications-And-Approvals)
|
||||
- [Cron Jobs](Cron-Jobs)
|
||||
- [Autonomous Loops](Autonomous-Loops)
|
||||
- [Watching Agents Work](Watching-Agents-Work)
|
||||
|
||||
**Where it runs**
|
||||
|
||||
- [Docker Cases](Docker-Cases)
|
||||
- [Remote SSH Sessions](Remote-SSH-Sessions)
|
||||
- [Web Tabs](Web-Tabs)
|
||||
- [Multi-User Mode](Multi-User-Mode)
|
||||
|
||||
**Access & security**
|
||||
|
||||
- [Remote Access](Remote-Access)
|
||||
- [Security](Security)
|
||||
|
||||
**Automation**
|
||||
|
||||
- [Driving It From An Agent](Driving-Codeman-From-An-Agent)
|
||||
- [HTTP API](HTTP-API)
|
||||
- [Hooks & Integrations](Hooks-And-Integrations)
|
||||
|
||||
**Operating it**
|
||||
|
||||
- [Running As A Service](Running-As-A-Service)
|
||||
- [Troubleshooting](Troubleshooting)
|
||||
- [FAQ](FAQ)
|
||||
- [Contributing](Contributing)
|
||||
- [Versioning](Versioning)
|
||||
@@ -0,0 +1,121 @@
|
||||
# Warm worker pool: sub-second claude worker spawns
|
||||
|
||||
Design sketch. Status: **proposed**, not started. Opt-in (`workerPoolSize`, default 0 = off); a user who touches nothing sees no change at all.
|
||||
|
||||
---
|
||||
|
||||
## 1. Problem and numbers
|
||||
|
||||
Measured against prod 1.18.3 on 2026-08-15, AFTER the SKILL.md fast-path hardening
|
||||
(no recon turns), on the identical "spawn two codeman workers" prompt:
|
||||
|
||||
- **Cold orchestrator** (fresh session, skill loaded from disk): **20.2 s** prompt to
|
||||
final report. Breakdown: 3.9 s Skill-load turn, 6.4 s generating the one fused Bash
|
||||
call, **4.4 s spawn call**, 5.5 s summary. Tabs appeared at 10.5 s.
|
||||
- **Warm orchestrator** (skill already in context, no Skill turn): **12.8 s**, spawn
|
||||
call 6.0 s.
|
||||
- Inside the spawn call, session + tmux + case creation is cheap: the workers (and
|
||||
their tabs) appeared 0.2-1.7 s in, both siblings within ~350 ms of each other. The
|
||||
remaining **~4-5 s is claude CLI boot plus the composer-readiness wait**, paid again
|
||||
on every cold spawn. That slice is the pool's entire target.
|
||||
|
||||
The honest framing after the hardening: model turns dominate the skill flow (~16 of
|
||||
20 cold seconds) and no server feature can shrink those. The pool attacks the
|
||||
tool-side floor, and it has two distinct beneficiaries:
|
||||
|
||||
- **Skill/API orchestration**: the spawn call drops from ~4.4-6 s to ~1 s. Cold runs
|
||||
land ~16-17 s, warm ~8 s. Tab appearance barely moves for this consumer (it is
|
||||
model-turn-bound at ~10 s cold / ~4 s warm).
|
||||
- **The UI Run button and direct quick-start callers**: a click today waits the full
|
||||
boot + readiness before the worker can take a prompt; a pooled claim makes the tab
|
||||
appear and the worker READY sub-second. This is the most visible win, and it
|
||||
involves no skill at all.
|
||||
|
||||
Target: hand out an already-ready worker in **under 1 s**.
|
||||
|
||||
## 2. Shape
|
||||
|
||||
A new `src/worker-pool.ts` singleton service, following the `CronService` pattern: it **reuses the existing session layer** (`SessionManager` create + the normal spawn path) and never rebuilds tmux logic.
|
||||
|
||||
A pool member is a real claude `Session`, pre-spawned in a reserved scratch case (`~/codeman-cases/.pool-<n>`, created with the standard scaffold + hooks), already past readiness: composer drawn, hooks installed, preamble file seeded. It sits idle at the composer costing no tokens.
|
||||
|
||||
The claim happens **transparently inside `POST /api/quick-start`**: when a request is pool-eligible (§3) and a healthy member is available, quick-start returns that member instead of cold-spawning. The agent skill, the UI Run button, and every existing caller change **nothing**. Ineligible or pool-empty requests cold-spawn exactly as today, so the pool is only ever a fast path, never a behavior change.
|
||||
|
||||
## 3. Eligibility gate
|
||||
|
||||
Claim only when ALL of these hold; otherwise fall through to a cold spawn:
|
||||
|
||||
- `mode === 'claude'` (external CLIs have different readiness semantics and inject secrets via `tmux setenv` at spawn; out of scope).
|
||||
- No `envOverrides`, no `CLAUDE_CONFIG_DIR`, and `modelOverride`/`effort` unset or equal to what the pool member was spawned with. Env vars flow at spawn time and cannot be applied to a running CLI.
|
||||
- The requested case is **fresh** (does not exist yet). A linked case, an existing directory, a remote-SSH case, or a Docker case means the caller wants a specific workspace; pool members cannot provide one.
|
||||
- Single-user mode, or the requester owns the pool (v1 ships single-user only; §11).
|
||||
|
||||
## 4. What a claim does (~300 ms)
|
||||
|
||||
1. Pop a ready member (in-memory check-and-remove; Node's single thread makes this atomic, so two concurrent quick-starts cannot claim the same member).
|
||||
2. Health-probe it: `isPaneDead` (the existing ~750 ms-cached mux probe) plus one `capturePaneText` asserting a clean composer. A dead, limit-paused, or dirty member is recycled, and the claim tries the next member or falls through to cold spawn.
|
||||
3. Rename the session to the normal `w<n>-<case>` name, set `parentSessionId` via the existing `resolveParentSessionId()`, clear the pool flag, persist state.
|
||||
4. Emit `session_created` **now** (it was suppressed at warm-spawn time, §5). The tab appears here, sub-second after the request.
|
||||
5. Return the **pool case** as `casePath`/`workingDir` and do NOT create a directory under the requested name: an empty dir the worker's CLI does not run in is a trap (files written there are invisible to the worker at cwd), and the agent skill greps the RETURNED `casePath` for Codeman hooks before trusting the worker, so the response must point at the directory that really carries them.
|
||||
6. Kick a background refill (§6).
|
||||
|
||||
**The identity wrinkle, stated honestly:** the session id, `CODEMAN_SESSION_ID` inside the pane, the seeded preamble file, and the CLI's cwd are all fixed at warm-spawn and survive the claim unchanged. So a claimed worker's `workingDir` is the pool dir, not `~/codeman-cases/<requested-name>`; the requested name is a **label**. The API must report the truthful `workingDir`. Transcript projHash, response viewer, subagent windows, and Read My Mind all key off the real path and keep working precisely because we do not lie about it. This is acceptable for the dominant use (ephemeral skill workers that are deleted after answering) and is documented in the skill; a caller that needs the real case as cwd is by definition not pool-eligible.
|
||||
|
||||
**Verified skill compatibility (zero preamble changes).** Checked against the shipped 1.18.3 preamble: `spawn_worker`'s readiness probe (`_composer_up`) is a `wait-output` call with `from=buffer`, which scans output that already scrolled past before blocking, so a pooled member's long-since-drawn composer matches instantly instead of stranding a fresh-stream wait. The trust-dialog fallback never fires (members passed the dialog at warm time), and the hooks grep passes because the pool case carries the standard scaffold. Pooled and cold spawns are indistinguishable to the skill except in speed and the additive `pooled: true`.
|
||||
|
||||
## 5. Hiding pre-claim members
|
||||
|
||||
Pool members must be invisible until claimed or they read as ghost tabs. `Session.isPoolWorker` gates, at minimum:
|
||||
|
||||
- `GET /api/sessions` and `GET /api/sessions/unified` (and therefore the Cmd+K palette and the session-history-index snapshot that feeds `/api/search`).
|
||||
- `session_created` SSE at warm-spawn (deferred to claim time). All other per-session SSE for a hidden member is suppressed at the broadcast call sites it would reach.
|
||||
- Push notifications and the Approvals Inbox (a warm member showing a trust dialog must recycle, not notify).
|
||||
- The phone overview / home rail (both render from the session list, so the list filter covers them).
|
||||
- The lifecycle log records `pool_warm` / `pool_claim` events rather than user-visible session history.
|
||||
|
||||
`maxSessions` (50) **counts** pool members, and the pool refuses to warm within `poolSize + 2` of the cap so it can never starve real session creation.
|
||||
|
||||
## 6. Refill, TTL, drain
|
||||
|
||||
- **Refill** after each claim, debounced, at most one warm spawn in flight (a claim burst falls back to cold spawns rather than forking N CLIs at once; same reasoning as the document-conversion limiter).
|
||||
- **TTL ~30 min**: recycle members older than that so they cannot drift from settings, hooks config, or a self-updated CLI on disk.
|
||||
- **Drain and respawn** on: `claudeModel` change, hooks-config regeneration, self-update, and `workerPoolSize` changes. On server shutdown, kill pool sessions (they are stateless and ours). On boot, kill any leftover `.pool-*` tmux sessions found via `mux-sessions.json` rather than adopting them; adoption buys nothing for stateless members.
|
||||
|
||||
## 7. Failure modes
|
||||
|
||||
| Failure | Handling |
|
||||
| --- | --- |
|
||||
| Member died idle (PTY exit, crash) | Health probe at claim catches it; recycle + try next; PTY-exit breaker applies unchanged |
|
||||
| Member hit a usage limit while idle | `isLimitPaused` members are never handed out; recycle |
|
||||
| Composer dirty (stray keystrokes, dialog) | `capturePaneText` probe refuses it; recycle |
|
||||
| Claim race | Impossible by construction (synchronous in-memory pop) |
|
||||
| Warm spawn itself fails | Log, back off, retry on next refill tick; pool empty just means cold spawns |
|
||||
|
||||
## 8. Cost
|
||||
|
||||
Each warm member is one tmux session + one idle claude process (order 150-300 MB RSS; **measure before defaulting the size above 0**, including whether an idle CLI makes any background requests via its statusline refresh). Zero token cost while idle. Suggested starting size for users who opt in: 2.
|
||||
|
||||
## 9. Settings and API surface
|
||||
|
||||
- `workerPoolSize` (int, 0-4, default 0): **synced** setting in `SettingsUpdateSchema`. The watcher that resizes the pool on `PUT /api/settings` must resolve from `merged`, never the raw body (the partial-PUT gotcha in CLAUDE.md).
|
||||
- One internal status endpoint, `GET /api/worker-pool` (size, members' ages, claims served, fall-through count), for debugging. No new SSE events: the claim emits the existing `session_created`.
|
||||
- No new public API semantics: `/api/quick-start`'s contract is unchanged apart from a `pooled: true` field in the response data, which is additive.
|
||||
|
||||
## 10. Considered and rejected
|
||||
|
||||
- **Renaming the pool case dir to the requested name at claim.** Linux keeps the process cwd working across the rename (inode-based), but claude computed its transcript projHash from the old path string at boot, so transcripts, subagent windows, and the response viewer go blind, the exact failure mode the `CLAUDE_CONFIG_DIR` docs warn about. Truthful label semantics (§4) beat a clever rename.
|
||||
- **A new explicit claim endpoint.** Transparency inside quick-start means the skill, the UI, and every existing script get the speedup with zero changes; a new endpoint means new docs, new drift, and callers that must know the pool exists.
|
||||
- **Pooling external CLI modes.** Readiness there is output stabilization, secrets ride `tmux setenv` at spawn, and codex/pi composer semantics differ per CLI. Claude-only until someone measures a need.
|
||||
- **Returning quick-start at creation instead of readiness (no pool).** Would move tabs earlier on cold spawns too, but `sendwait` immediately after would then race the composer; readiness is what makes immediate tasking safe, and the pool makes the whole question moot for eligible spawns.
|
||||
|
||||
## 11. Phasing
|
||||
|
||||
1. **v1**: single-user, claude-only, fixed-size pool, transparent claim, status endpoint. Everything above.
|
||||
2. **v2**: per-owner pools for multi-user mode (pool members must carry an owner because ownership scoping is structural); possibly model-matched pools (one warm set per configured `claudeModel`).
|
||||
3. **Explicitly out**: warming linked/repo cases (spawning where the work is has no hooks and is the skill's documented costliest mistake; a warm pool must not make it faster to reach).
|
||||
|
||||
## 12. Testing
|
||||
|
||||
- Unit: pool manager logic pure and mock-driven (eligibility gate, TTL, refill debounce, drain triggers), `MockSession` from `test/mocks/`.
|
||||
- Route: `app.inject` on quick-start asserting claim vs cold-spawn per eligibility row in §3, plus the double-claim race (two concurrent injects, one pool member: exactly one `pooled: true`).
|
||||
- Live: re-run the pinned baselines against a warmed beta instance. Before (2026-08-15, prod 1.18.3, post-hardening): cold orchestrator **20.2 s** / warm **12.8 s** end to end, spawn call 4.4-6.0 s. Acceptance: spawn call under 1 s, cold ~16-17 s, warm ~8-9 s, and a UI Run click to a READY worker in under 1 s.
|
||||
+52
-5
@@ -116,6 +116,15 @@ GEMINI_SEARCH_PATHS=(
|
||||
"$HOME/bin/gemini"
|
||||
)
|
||||
|
||||
# Pi CLI search paths (from src/utils/pi-cli-resolver.ts)
|
||||
PI_SEARCH_PATHS=(
|
||||
"$HOME/.local/bin/pi"
|
||||
"/usr/local/bin/pi"
|
||||
"$HOME/.bun/bin/pi"
|
||||
"$HOME/.npm-global/bin/pi"
|
||||
"$HOME/bin/pi"
|
||||
)
|
||||
|
||||
# Antigravity CLI search paths (from src/utils/antigravity-cli-resolver.ts)
|
||||
ANTIGRAVITY_SEARCH_PATHS=(
|
||||
"$HOME/.local/bin/agy"
|
||||
@@ -529,6 +538,37 @@ get_antigravity_path() {
|
||||
done
|
||||
}
|
||||
|
||||
# `pi` is a short, generic name (Raspberry Pi tooling, personal scripts), so the
|
||||
# server-side resolver additionally probes `pi --version`. Detection here only feeds
|
||||
# the "you have no AI CLI" hint, so a plain executable test is enough.
|
||||
check_pi() {
|
||||
if command -v pi &>/dev/null; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
for path in "${PI_SEARCH_PATHS[@]}"; do
|
||||
if [[ -x "$path" ]]; then
|
||||
return 0
|
||||
fi
|
||||
done
|
||||
|
||||
return 1
|
||||
}
|
||||
|
||||
get_pi_path() {
|
||||
if command -v pi &>/dev/null; then
|
||||
command -v pi
|
||||
return
|
||||
fi
|
||||
|
||||
for path in "${PI_SEARCH_PATHS[@]}"; do
|
||||
if [[ -x "$path" ]]; then
|
||||
echo "$path"
|
||||
return
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
check_cloudflared() {
|
||||
# Check ~/.local/bin first (matches tunnel-manager.ts resolution order)
|
||||
if [[ -x "$HOME/.local/bin/cloudflared" ]]; then
|
||||
@@ -2029,12 +2069,13 @@ main() {
|
||||
fi
|
||||
fi
|
||||
|
||||
# AI CLI (Codeman drives one of: Claude Code, OpenCode, Codex, Gemini, Antigravity)
|
||||
# AI CLI (Codeman drives one of: Claude Code, OpenCode, Codex, Gemini, Antigravity, Pi)
|
||||
local has_claude=false
|
||||
local has_opencode=false
|
||||
local has_codex=false
|
||||
local has_gemini=false
|
||||
local has_antigravity=false
|
||||
local has_pi=false
|
||||
|
||||
info "Checking AI CLI tools..."
|
||||
if check_claude; then
|
||||
@@ -2057,17 +2098,21 @@ main() {
|
||||
has_antigravity=true
|
||||
success "Antigravity CLI found at $(get_antigravity_path)"
|
||||
fi
|
||||
if check_pi; then
|
||||
has_pi=true
|
||||
success "Pi CLI found at $(get_pi_path)"
|
||||
fi
|
||||
|
||||
if [[ "$has_claude" == "false" && "$has_opencode" == "false" && "$has_codex" == "false" && "$has_gemini" == "false" && "$has_antigravity" == "false" ]]; then
|
||||
if [[ "$has_claude" == "false" && "$has_opencode" == "false" && "$has_codex" == "false" && "$has_gemini" == "false" && "$has_antigravity" == "false" && "$has_pi" == "false" ]]; then
|
||||
echo ""
|
||||
warn "No AI CLI found. Codeman needs at least one: Claude Code, OpenCode, Codex, Antigravity, or Gemini."
|
||||
warn "No AI CLI found. Codeman needs at least one: Claude Code, OpenCode, Codex, Antigravity, Gemini, or Pi."
|
||||
headless_guard "install an AI CLI (curl | bash from its vendor)"
|
||||
echo ""
|
||||
echo -e " ${BOLD}Which AI CLI would you like to install?${NC}"
|
||||
echo -e " ${CYAN}1)${NC} Claude Code (Anthropic)"
|
||||
echo -e " ${CYAN}2)${NC} OpenCode (open-source)"
|
||||
echo -e " ${CYAN}3)${NC} Both"
|
||||
echo -e " ${CYAN}4)${NC} Skip (I'll install one myself, e.g. Codex or Antigravity)"
|
||||
echo -e " ${CYAN}4)${NC} Skip (I'll install one myself, e.g. Codex, Antigravity or Pi)"
|
||||
echo ""
|
||||
|
||||
local cli_choice=""
|
||||
@@ -2114,6 +2159,7 @@ main() {
|
||||
warn "Skipping AI CLI install. Codeman will run, but sessions need a CLI to drive."
|
||||
info "Install one later, e.g.: npm install -g @openai/codex (Codex)"
|
||||
info " or: curl -fsSL https://antigravity.google/cli/install.sh | bash (Antigravity)"
|
||||
info " or: npm install -g --ignore-scripts @earendil-works/pi-coding-agent (Pi)"
|
||||
elif [[ "$has_claude" == "false" ]] && [[ "$has_opencode" == "false" ]]; then
|
||||
die "The selected AI CLI failed to install. Install one manually and re-run the installer."
|
||||
fi
|
||||
@@ -2413,12 +2459,13 @@ main() {
|
||||
echo -e " https://github.com/Ark0N/Codeman"
|
||||
echo ""
|
||||
|
||||
if ! check_claude && ! check_opencode && ! check_codex && ! check_gemini && ! check_antigravity; then
|
||||
if ! check_claude && ! check_opencode && ! check_codex && ! check_gemini && ! check_antigravity && ! check_pi; then
|
||||
echo -e " ${YELLOW}${BOLD}Reminder:${NC} Install at least one AI CLI to start using Codeman:"
|
||||
echo -e " ${CYAN}curl -fsSL https://claude.ai/install.sh | bash${NC} # Claude Code"
|
||||
echo -e " ${CYAN}curl -fsSL https://opencode.ai/install | bash${NC} # OpenCode"
|
||||
echo -e " ${CYAN}npm install -g @openai/codex${NC} # Codex"
|
||||
echo -e " ${CYAN}curl -fsSL https://antigravity.google/cli/install.sh | bash${NC} # Antigravity"
|
||||
echo -e " ${CYAN}npm install -g --ignore-scripts @earendil-works/pi-coding-agent${NC} # Pi"
|
||||
echo ""
|
||||
fi
|
||||
|
||||
|
||||
Generated
+15
-3
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "aicodeman",
|
||||
"version": "1.12.2",
|
||||
"version": "1.19.6",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "aicodeman",
|
||||
"version": "1.12.2",
|
||||
"version": "1.19.6",
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"workspaces": [
|
||||
@@ -60,6 +60,7 @@
|
||||
"pixelmatch": "^6.0.0",
|
||||
"playwright": "^1.58.0",
|
||||
"pngjs": "^7.0.0",
|
||||
"postcss": "^8.5.15",
|
||||
"prettier": "^3.4.0",
|
||||
"puppeteer": "^24.36.0",
|
||||
"remotion": "4.0.473",
|
||||
@@ -4547,6 +4548,16 @@
|
||||
"integrity": "sha512-b3fMOsyLVuCeNJWxolACEUED0vm7qC0cy4wRvf3oURSzDTYVQiGPhTnhWZwIHdvC48Y+oLhvYXnY4XDXPoJo6A==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@xterm/headless": {
|
||||
"version": "6.0.0",
|
||||
"resolved": "https://registry.npmjs.org/@xterm/headless/-/headless-6.0.0.tgz",
|
||||
"integrity": "sha512-5Yj1QINYCyzrZtf8OFIHi47iQtI+0qYFPHmouEfG8dHNxbZ9Tb9YGSuLcsEwj9Z+OL75GJqPyJbyoFer80a2Hw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"workspaces": [
|
||||
"addons/*"
|
||||
]
|
||||
},
|
||||
"node_modules/@xterm/xterm": {
|
||||
"version": "6.0.0",
|
||||
"resolved": "https://registry.npmjs.org/@xterm/xterm/-/xterm-6.0.0.tgz",
|
||||
@@ -12333,9 +12344,10 @@
|
||||
}
|
||||
},
|
||||
"packages/xterm-zerolag-input": {
|
||||
"version": "0.1.8",
|
||||
"version": "0.3.0",
|
||||
"license": "MIT",
|
||||
"devDependencies": {
|
||||
"@xterm/headless": "^6.0.0",
|
||||
"jsdom": "^24.1.3",
|
||||
"tsup": "^8.5.1",
|
||||
"typescript": "^5.5.0",
|
||||
|
||||
+12
-4
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "aicodeman",
|
||||
"version": "1.12.2",
|
||||
"version": "1.19.6",
|
||||
"description": "Mission control for AI coding agents - run 20 autonomous agents with real-time monitoring and session persistence",
|
||||
"type": "module",
|
||||
"main": "dist/index.js",
|
||||
@@ -17,10 +17,15 @@
|
||||
"dev": "tsx src/index.ts web",
|
||||
"web": "node dist/index.js web",
|
||||
"clean": "rm -rf dist",
|
||||
"test": "vitest run --config config/vitest.config.ts",
|
||||
"test:watch": "vitest --config config/vitest.config.ts",
|
||||
"test:coverage": "vitest run --config config/vitest.config.ts --coverage",
|
||||
"test": "vitest run --config config/vitest.ci.config.ts",
|
||||
"test:watch": "vitest --config config/vitest.ci.config.ts",
|
||||
"test:coverage": "vitest run --config config/vitest.ci.config.ts --coverage",
|
||||
"test:ci": "vitest run --config config/vitest.ci.config.ts",
|
||||
"test:browser": "vitest run --config config/vitest.browser.config.ts",
|
||||
"test:perf": "vitest run --config config/vitest.perf.config.ts",
|
||||
"test:all": "vitest run --config config/vitest.config.ts",
|
||||
"pretest:mobile": "node scripts/prepare-test-vendor.mjs",
|
||||
"test:mobile": "vitest run --config test/mobile/vitest.config.ts",
|
||||
"check:frontend-syntax": "node scripts/check-frontend-syntax.mjs",
|
||||
"fix:node-pty": "node scripts/fix-node-pty.mjs",
|
||||
"typecheck": "tsc --noEmit",
|
||||
@@ -56,6 +61,7 @@
|
||||
"opencode",
|
||||
"codex",
|
||||
"antigravity",
|
||||
"pi",
|
||||
"gemini-cli",
|
||||
"ai-agents",
|
||||
"agent",
|
||||
@@ -118,6 +124,7 @@
|
||||
"pixelmatch": "^6.0.0",
|
||||
"playwright": "^1.58.0",
|
||||
"pngjs": "^7.0.0",
|
||||
"postcss": "^8.5.15",
|
||||
"prettier": "^3.4.0",
|
||||
"puppeteer": "^24.36.0",
|
||||
"remotion": "4.0.473",
|
||||
@@ -159,6 +166,7 @@
|
||||
"dist",
|
||||
"scripts/postinstall.js",
|
||||
"scripts/fix-node-pty.mjs",
|
||||
"skills",
|
||||
"LICENSE",
|
||||
"README.md"
|
||||
]
|
||||
|
||||
@@ -1,5 +1,30 @@
|
||||
# xterm-zerolag-input
|
||||
|
||||
## 0.3.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- 55bff4a: Zero-lag predictive echo for Codex sessions (mosh-style write-through prediction).
|
||||
|
||||
Codex's per-keystroke composer forced 1.12.2 to disable the local-echo overlay (issues #218/#219/#220/#222), leaving Codex typing at full round-trip latency on remote links. This release adds a second echo mode instead of re-enabling the first: every keystroke still goes to the PTY exactly as before (byte-identical wire behavior, pinned by vm-level and end-to-end trace-equality tests), while the new `PredictiveEchoAddon` in `xterm-zerolag-input` 0.2.0 paints the predicted glyph at the predicted cell. When the real echo lands, the prediction is confirmed and its span removed (an invisible swap); mispredictions self-heal via a two-pass mismatch cascade and a TTL.
|
||||
- Reconciliation reads the parsed terminal buffer, never the raw stream: full-line redraws, ECH gap painting and tmux's in-place deltas all converge to the same cells. Confirmation requires the cell match PLUS a cursor advance, so placeholder glyphs and identical repaints never false-confirm; blank cells are neutral (codex clears its placeholder on the first echo).
|
||||
- Predictions paint only while the cursor sits on the measured Codex composer row (`/^› /`, codex-cli 0.147): trust/approval modals and wrapped continuation rows get no ghosts, deliberately falling back to real echo.
|
||||
- Ships as a SEPARATE `vendor/xterm-predictive-echo.js` bundle: the existing zerolag bundle is byte-identical (sha256-verified), and a missing or broken bundle degrades Codex to exact 1.12.2 behavior. The per-device `localEchoEnabled` toggle is the kill switch.
|
||||
- Claude/Gemini/OpenCode/Antigravity keep buffer mode untouched; shell stays off.
|
||||
- A post-build adversarial review added the anchor-hold rule: after an unpredicted wire edit (backspace into echoed text, cleared input, IME text commits) new predictions hold until the next parsed write, so a stale displayed cursor can never mis-anchor a run.
|
||||
- Tests: 55 new package tests including replay suites driven by fixtures recorded from a real codex TUI through the production tmux+strip pipeline (`scripts/dev/record-codex-frames.mjs`) and a 500-iteration seeded fuzz; new vm policy/wire-neutrality suites; a 10-scenario Playwright E2E against real codex covering the #218/#219/#220/#222 retests, byte-identity, and a simulated 300ms-RTT run. The package test suite now runs in CI.
|
||||
|
||||
## 0.2.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- **New addon: `PredictiveEchoAddon`, mosh-style write-through prediction.** The second echo mode for per-keystroke TUIs (OpenAI Codex's composer, live pickers) that buffer-until-Enter starves. Every keystroke is sent by the consumer immediately and unchanged; the addon paints the predicted glyph at the predicted cell and reconciles against the PARSED terminal buffer: confirmation requires the cell match plus a cursor advance past the record, foreign non-blank content on two consecutive passes cascades a drop, blank cells are neutral, a TTL bounds everything, and scroll/resize/sustained cursor moves clear the run. Visual-only by construction; it cannot gate, delay or rewrite input.
|
||||
- Anchor-hold rule: after an unpredicted wire edit (backspace into echoed text, cleared input, an IME text commit) new predictions hold until the next parsed write, so a stale displayed cursor can never mis-anchor a run (worst case: exactly one unpredicted keystroke).
|
||||
- New exports: `PredictiveEchoAddon`, `PredictiveEchoOptions`, `PredictionState`, plus the long-intended `charCellWidth` / `stringCellWidth` helpers.
|
||||
- `XtermTerminal` type gains OPTIONAL members (`buffer.active.cursorX/cursorY`, `getLine().getCell?`, `onWriteParsed?`, `onResize?`). Additive only: existing consumers and mocks are unaffected.
|
||||
- IIFE build exposes `window.PredictiveEchoAddon` and a self-activating `window.PredictiveEchoOverlay`, alongside the unchanged `ZerolagInputAddon` / `LocalEchoOverlay` globals.
|
||||
- Tests: 52 new (30 addon-law specs, renderer geometry, 6 replay suites driven by fixtures recorded from real codex 0.147 through tmux + the production strip, and a 500-iteration seeded fuzz with per-op invariants). `@xterm/headless` as a devDependency; runtime dependencies remain zero.
|
||||
|
||||
## 0.1.8
|
||||
|
||||
### Patch Changes
|
||||
|
||||
@@ -9,7 +9,7 @@
|
||||
<a href="https://opensource.org/licenses/MIT"><img src="https://img.shields.io/badge/License-MIT-1e3a5f?style=flat-square" alt="MIT"></a>
|
||||
<img src="https://img.shields.io/badge/Dependencies-0-22c55e?style=flat-square" alt="Zero dependencies">
|
||||
<img src="https://img.shields.io/badge/Size-6.1%20kB%20gzip-22c55e?style=flat-square" alt="6.1 kB gzipped">
|
||||
<img src="https://img.shields.io/badge/Tests-175-22c55e?style=flat-square" alt="175 tests">
|
||||
<img src="https://img.shields.io/badge/Tests-227-22c55e?style=flat-square" alt="175 tests">
|
||||
<img src="https://img.shields.io/badge/xterm.js-v5%20%7C%20v7+-3b82f6?style=flat-square" alt="xterm.js v5 and v7+">
|
||||
</p>
|
||||
</p>
|
||||
@@ -46,6 +46,15 @@ Same keystroke, same link. The only difference is who you wait for: the server,
|
||||
|
||||
**No backend changes. No protocol. No server support.** It is a client-side addon that never touches the wire.
|
||||
|
||||
Since 0.2.0 the package ships **two addons for two kinds of TUIs**:
|
||||
|
||||
| Addon | Model | Use when |
|
||||
| --------------------- | -------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------- |
|
||||
| `ZerolagInputAddon` | **Buffer**: hold keystrokes locally, flush on Enter | The remote side is a line-oriented prompt (shells, REPLs, Claude Code's composer) that only needs the finished line |
|
||||
| `PredictiveEchoAddon` | **Predictive write-through**: send every keystroke immediately, paint a prediction, confirm against the parsed buffer | The remote side is a per-keystroke TUI (OpenAI Codex's composer, live pickers) that buffering would starve |
|
||||
|
||||
`ZerolagInputAddon` is documented below; jump to [PredictiveEchoAddon](#predictiveechoaddon-write-through-prediction) for the second mode.
|
||||
|
||||
## Why this one
|
||||
|
||||
| | |
|
||||
@@ -268,6 +277,110 @@ Finds text that exists after the prompt but was never typed through the overlay.
|
||||
|
||||
---
|
||||
|
||||
## `PredictiveEchoAddon` (write-through prediction)
|
||||
|
||||
Buffering is the wrong model for TUIs that react to every keystroke: a slash
|
||||
command picker filters live, arrows edit server-side state, the composer
|
||||
rewraps as it grows. For those, `PredictiveEchoAddon` works like
|
||||
[mosh](https://mosh.org/): the keystroke goes to the PTY **immediately and
|
||||
unchanged**, and the addon simultaneously paints the predicted glyph at the
|
||||
predicted cell. When the real echo lands, the prediction is confirmed and its
|
||||
span removed: an invisible swap, identical glyph beneath. Mispredictions
|
||||
self-heal via a mismatch cascade and a TTL. It is visual-only by construction:
|
||||
nothing it does can gate, delay, reorder or rewrite what you send.
|
||||
|
||||
```typescript
|
||||
import { Terminal } from '@xterm/xterm';
|
||||
import { PredictiveEchoAddon } from 'xterm-zerolag-input';
|
||||
|
||||
const terminal = new Terminal();
|
||||
const predictor = new PredictiveEchoAddon({
|
||||
// Optional: only predict when the cursor sits on a composer row
|
||||
predictWhen: (t) => {
|
||||
const buf = t.buffer.active;
|
||||
const line = buf.getLine(buf.baseY + buf.cursorY);
|
||||
return !!line && /^› /.test(line.translateToString(true));
|
||||
},
|
||||
});
|
||||
terminal.loadAddon(predictor);
|
||||
|
||||
terminal.onData((data) => {
|
||||
const cps = Array.from(data);
|
||||
if (cps.length === 1) {
|
||||
const cp = cps[0].codePointAt(0);
|
||||
if (cp === 0x7f) predictor.predictBackspace();
|
||||
else if (cp >= 0x20) predictor.predictChar(data);
|
||||
else predictor.clearPredictions(); // Enter, Ctrl+C, ...
|
||||
} else if (data.charCodeAt(0) === 0x1b) {
|
||||
predictor.clearPredictions(); // nav keys, bracketed paste
|
||||
}
|
||||
pty.write(data); // ALWAYS, unconditionally
|
||||
});
|
||||
```
|
||||
|
||||
### How reconciliation works
|
||||
|
||||
Predictions are reconciled against the **parsed terminal buffer** (cells after
|
||||
xterm's parser ran), never the raw output stream. That distinction is
|
||||
load-bearing: TUIs redraw whole lines, paint gaps with `ECH` + cursor-forward
|
||||
instead of spaces, and multiplexers like tmux rewrite everything into minimal
|
||||
deltas. Stream matching breaks on all of that; buffer cells converge to the
|
||||
same values no matter how the bytes arrived.
|
||||
|
||||
A prediction is **confirmed** only when its cell shows the predicted glyph AND
|
||||
the cursor has advanced past it (so a placeholder that happens to match, or an
|
||||
identical in-place repaint, never false-confirms). A cell showing foreign
|
||||
non-blank content on two consecutive passes drops that prediction and all
|
||||
later ones (one pass tolerates half-parsed frames). Blank cells are neutral:
|
||||
they are what "not yet echoed" looks like. Whatever remains is dropped by TTL.
|
||||
Scrolling up, resizing, or a sustained cursor move clears the run. After a
|
||||
backspace into already-echoed text, a cleared input, or a multi-char commit,
|
||||
the addon **holds** new predictions until the next parsed write: the displayed
|
||||
cursor is stale for one round trip, and anchoring on it would paint ghosts one
|
||||
cell off (worst case: exactly one unpredicted keystroke, whose own echo
|
||||
releases the hold).
|
||||
|
||||
### API
|
||||
|
||||
```typescript
|
||||
predictChar(ch: string): boolean; // false = suppressed (still SEND the key)
|
||||
predictBackspace(): boolean; // pops the newest prediction (still send \x7f)
|
||||
clearPredictions(): void;
|
||||
reconcile(): void; // manual pass (no onWriteParsed available)
|
||||
setPredictWhen(fn | null): void; // swap the gate at runtime
|
||||
refreshFont(): void; // after font/theme changes
|
||||
get hasPredictions(): boolean;
|
||||
get state(): PredictionState; // { outstanding, confirmedTotal, droppedTotal, anchor }
|
||||
```
|
||||
|
||||
### Options
|
||||
|
||||
```typescript
|
||||
{
|
||||
zIndex?: number, // Default: 7
|
||||
underlinePredictions?: boolean, // Default: false (underline unconfirmed glyphs)
|
||||
foregroundColor?: string, // Default: terminal theme / computed .xterm-rows style
|
||||
backgroundColor?: string, // Default: terminal theme background
|
||||
ttlMs?: number, // Default: 1000
|
||||
maxPending?: number, // Default: 32
|
||||
cursorGraceMs?: number, // Default: 150
|
||||
edgeMarginCells?: number, // Default: 4 (suppress near the right edge)
|
||||
predictWhen?: (t) => boolean, // Default: predict everywhere
|
||||
}
|
||||
```
|
||||
|
||||
### Which addon should I use?
|
||||
|
||||
- The remote program shows a **line prompt** and ignores partial input:
|
||||
`ZerolagInputAddon`. You also get backspace-before-send and batching.
|
||||
- The remote program **reacts per keystroke** (pickers, filters, composers
|
||||
that rewrap): `PredictiveEchoAddon`. It never withholds bytes, so the TUI
|
||||
behaves exactly as with no addon at all; you just stop waiting for the RTT.
|
||||
- Both can be loaded on one terminal and toggled per session mode; that is
|
||||
exactly what Codeman does (buffer for Claude Code, predict for Codex).
|
||||
|
||||
---
|
||||
|
||||
## Integration patterns
|
||||
|
||||
### Buffered input (hold until Enter)
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "xterm-zerolag-input",
|
||||
"version": "0.1.8",
|
||||
"version": "0.3.0",
|
||||
"description": "Instant keystroke feedback overlay for xterm.js: Mosh-inspired local echo that removes perceived input latency over SSH, tunnels and other high-RTT connections",
|
||||
"type": "module",
|
||||
"main": "dist/index.cjs",
|
||||
@@ -37,7 +37,9 @@
|
||||
"ssh",
|
||||
"remote-terminal",
|
||||
"overlay",
|
||||
"addon"
|
||||
"addon",
|
||||
"predictive",
|
||||
"write-through"
|
||||
],
|
||||
"license": "MIT",
|
||||
"homepage": "https://github.com/Ark0N/Codeman/tree/master/packages/xterm-zerolag-input#readme",
|
||||
@@ -50,6 +52,7 @@
|
||||
"directory": "packages/xterm-zerolag-input"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@xterm/headless": "^6.0.0",
|
||||
"jsdom": "^24.1.3",
|
||||
"tsup": "^8.5.1",
|
||||
"typescript": "^5.5.0",
|
||||
|
||||
@@ -11,37 +11,36 @@ import type { XtermTerminal, CellDimensions } from './types.js';
|
||||
* unavailable.
|
||||
*/
|
||||
export function getCellDimensions(terminal: XtermTerminal): CellDimensions | null {
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const t = terminal as any;
|
||||
const dpr = typeof devicePixelRatio === 'number' && devicePixelRatio > 0
|
||||
? devicePixelRatio : 1;
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const t = terminal as any;
|
||||
const dpr = typeof devicePixelRatio === 'number' && devicePixelRatio > 0 ? devicePixelRatio : 1;
|
||||
|
||||
// Try v7+ public API first
|
||||
if (t.dimensions?.css?.cell) {
|
||||
const cellH = t.dimensions.css.cell.height;
|
||||
return {
|
||||
width: t.dimensions.css.cell.width,
|
||||
height: cellH,
|
||||
charTop: (t.dimensions?.device?.char?.top ?? 0) / dpr,
|
||||
charHeight: (t.dimensions?.device?.char?.height ?? (cellH * dpr)) / dpr,
|
||||
};
|
||||
// Try v7+ public API first
|
||||
if (t.dimensions?.css?.cell) {
|
||||
const cellH = t.dimensions.css.cell.height;
|
||||
return {
|
||||
width: t.dimensions.css.cell.width,
|
||||
height: cellH,
|
||||
charTop: (t.dimensions?.device?.char?.top ?? 0) / dpr,
|
||||
charHeight: (t.dimensions?.device?.char?.height ?? cellH * dpr) / dpr,
|
||||
};
|
||||
}
|
||||
|
||||
// Fall back to v5 private API
|
||||
try {
|
||||
const dims = t._core?._renderService?.dimensions;
|
||||
if (dims?.css?.cell) {
|
||||
const cellH = dims.css.cell.height;
|
||||
return {
|
||||
width: dims.css.cell.width,
|
||||
height: cellH,
|
||||
charTop: (dims.device?.char?.top ?? 0) / dpr,
|
||||
charHeight: (dims.device?.char?.height ?? cellH * dpr) / dpr,
|
||||
};
|
||||
}
|
||||
} catch {
|
||||
// Private API may throw in some environments
|
||||
}
|
||||
|
||||
// Fall back to v5 private API
|
||||
try {
|
||||
const dims = t._core?._renderService?.dimensions;
|
||||
if (dims?.css?.cell) {
|
||||
const cellH = dims.css.cell.height;
|
||||
return {
|
||||
width: dims.css.cell.width,
|
||||
height: cellH,
|
||||
charTop: (dims.device?.char?.top ?? 0) / dpr,
|
||||
charHeight: (dims.device?.char?.height ?? (cellH * dpr)) / dpr,
|
||||
};
|
||||
}
|
||||
} catch {
|
||||
// Private API may throw in some environments
|
||||
}
|
||||
|
||||
return null;
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -1,10 +1,13 @@
|
||||
export { ZerolagInputAddon } from './zerolag-input-addon.js';
|
||||
export { PredictiveEchoAddon } from './predictive-echo-addon.js';
|
||||
export { charCellWidth, stringCellWidth } from './overlay-renderer.js';
|
||||
export type {
|
||||
XtermTerminal,
|
||||
XtermAddon,
|
||||
ZerolagInputOptions,
|
||||
ZerolagInputState,
|
||||
PromptFinder,
|
||||
PromptPosition,
|
||||
CellDimensions,
|
||||
XtermTerminal,
|
||||
XtermAddon,
|
||||
ZerolagInputOptions,
|
||||
ZerolagInputState,
|
||||
PromptFinder,
|
||||
PromptPosition,
|
||||
CellDimensions,
|
||||
} from './types.js';
|
||||
export type { PredictiveEchoOptions, PredictionState } from './predictive-echo-addon.js';
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
/**
|
||||
* Incremental DOM renderer for PredictiveEchoAddon.
|
||||
*
|
||||
* Unlike overlay-renderer.ts (which paints whole lines with an opaque
|
||||
* background out to totalCols), prediction spans cover ONLY the predicted
|
||||
* glyph's own cells: anything wider would blank real echo arriving around
|
||||
* a prediction. Spans are keyed by prediction seq for O(1) removal.
|
||||
*/
|
||||
import type { CellDimensions, FontStyle } from './types.js';
|
||||
|
||||
export interface PredictionSpanParams {
|
||||
seq: number;
|
||||
/** Viewport-relative row (0-based). */
|
||||
row: number;
|
||||
/** Column (0-based). */
|
||||
col: number;
|
||||
char: string;
|
||||
/** Cell width of the glyph (1 or 2). */
|
||||
width: 1 | 2;
|
||||
dims: CellDimensions;
|
||||
font: FontStyle;
|
||||
underline: boolean;
|
||||
}
|
||||
|
||||
export function addPredictionSpan(
|
||||
container: HTMLElement,
|
||||
map: Map<number, HTMLSpanElement>,
|
||||
p: PredictionSpanParams
|
||||
): void {
|
||||
const span = document.createElement('span');
|
||||
// cellH+1 height: covers the sub-pixel seam between rows (same trick the
|
||||
// buffer overlay renderer ships with). Background covers only this glyph's
|
||||
// cells, never a full row.
|
||||
span.style.cssText =
|
||||
`position:absolute;left:${p.col * p.dims.width}px;top:${p.row * p.dims.height}px;` +
|
||||
`width:${p.width * p.dims.width}px;height:${p.dims.height + 1}px;line-height:${p.dims.height}px;` +
|
||||
`text-align:center;pointer-events:none;` +
|
||||
`font-family:${p.font.fontFamily};font-size:${p.font.fontSize};font-weight:${p.font.fontWeight};` +
|
||||
(p.font.letterSpacing ? `letter-spacing:${p.font.letterSpacing};` : '') +
|
||||
`color:${p.font.color};background-color:${p.font.backgroundColor};` +
|
||||
`font-feature-settings:'liga' 0,'calt' 0;` +
|
||||
(p.underline ? 'text-decoration:underline;' : '');
|
||||
span.textContent = p.char;
|
||||
map.set(p.seq, span);
|
||||
container.appendChild(span);
|
||||
}
|
||||
|
||||
export function removePredictionSpan(map: Map<number, HTMLSpanElement>, seq: number): void {
|
||||
const span = map.get(seq);
|
||||
if (span) {
|
||||
span.remove();
|
||||
map.delete(seq);
|
||||
}
|
||||
}
|
||||
|
||||
export function clearAllSpans(map: Map<number, HTMLSpanElement>): void {
|
||||
for (const span of map.values()) span.remove();
|
||||
map.clear();
|
||||
}
|
||||
@@ -0,0 +1,480 @@
|
||||
/**
|
||||
* PredictiveEchoAddon: mosh-style write-through local echo.
|
||||
*
|
||||
* The consumer sends every keystroke to the PTY unchanged (write-through);
|
||||
* this addon simultaneously paints the predicted glyph at the predicted cell.
|
||||
* When the real echo lands, the prediction is confirmed and its span removed
|
||||
* (an invisible swap: identical glyph beneath). Mispredictions self-heal via
|
||||
* a mismatch cascade and a TTL. Everything here is visual-only: no method
|
||||
* gates, delays, or rewrites what the consumer sends.
|
||||
*
|
||||
* Reconciliation reads the parsed terminal BUFFER (cells after xterm's parser
|
||||
* ran), never the raw output stream. Full-line redraws, ECH-based gap
|
||||
* painting, and tmux's in-place deltas all converge to the same cells; stream
|
||||
* matching cannot survive them (see docs/local-echo-overlay-plan.md's
|
||||
* "What NOT to Do" in the consuming repo).
|
||||
*
|
||||
* Coordinate base: xterm's `cursorY` is relative to `baseY`, so the absolute
|
||||
* buffer line for a viewport row is `baseY + row`. `viewportY` would only
|
||||
* coincide while scrolled to the bottom; this file never relies on that.
|
||||
*/
|
||||
import { getCellDimensions } from './cell-dimensions.js';
|
||||
import { charCellWidth } from './overlay-renderer.js';
|
||||
import { addPredictionSpan, clearAllSpans, removePredictionSpan } from './prediction-renderer.js';
|
||||
import type { FontStyle, XtermAddon, XtermTerminal } from './types.js';
|
||||
|
||||
export interface PredictiveEchoOptions {
|
||||
/** Z-index of the span container. @default 7 (same layer as the buffer overlay) */
|
||||
zIndex?: number;
|
||||
/** Render predicted glyphs underlined (visual hedge on unreliable links). @default false */
|
||||
underlinePredictions?: boolean;
|
||||
/** Predicted glyph color. @default theme foreground / computed .xterm-rows color */
|
||||
foregroundColor?: string;
|
||||
/** Predicted glyph background. @default theme background */
|
||||
backgroundColor?: string;
|
||||
/** Drop predictions older than this. @default 1000 */
|
||||
ttlMs?: number;
|
||||
/** Maximum outstanding predictions per run. @default 32 */
|
||||
maxPending?: number;
|
||||
/** How long the cursor may sit off the anchor row before predictions clear. @default 150 */
|
||||
cursorGraceMs?: number;
|
||||
/** Suppress predictions that would land within this many cells of the right edge. @default 4 */
|
||||
edgeMarginCells?: number;
|
||||
/** Gate: return false to suppress prediction (e.g. cursor not on a composer row). */
|
||||
predictWhen?: (terminal: XtermTerminal) => boolean;
|
||||
}
|
||||
|
||||
export interface PredictionState {
|
||||
outstanding: number;
|
||||
confirmedTotal: number;
|
||||
droppedTotal: number;
|
||||
anchor: { row: number; col: number } | null;
|
||||
}
|
||||
|
||||
interface PredictionRecord {
|
||||
seq: number;
|
||||
char: string;
|
||||
/** Cells this glyph occupies. */
|
||||
width: 1 | 2;
|
||||
/** Cumulative cell offset from the anchor column BEFORE this char. */
|
||||
offsetCells: number;
|
||||
/** Cell content at predict time, '' normalized to ' '. */
|
||||
snapshot: string;
|
||||
sentAt: number;
|
||||
/** Consecutive reconcile passes that saw foreign non-blank content. */
|
||||
mismatches: number;
|
||||
}
|
||||
|
||||
const DEFAULT_OPTIONS = {
|
||||
zIndex: 7,
|
||||
underlinePredictions: false,
|
||||
ttlMs: 1000,
|
||||
maxPending: 32,
|
||||
cursorGraceMs: 150,
|
||||
edgeMarginCells: 4,
|
||||
} as const;
|
||||
|
||||
const DEFAULT_BG = '#000000';
|
||||
const DEFAULT_FG = '#ffffff';
|
||||
|
||||
export class PredictiveEchoAddon implements XtermAddon {
|
||||
private _terminal: XtermTerminal | null = null;
|
||||
private _container: HTMLDivElement | null = null;
|
||||
private _spans = new Map<number, HTMLSpanElement>();
|
||||
private _outstanding: PredictionRecord[] = [];
|
||||
private _anchor: { row: number; col: number } | null = null;
|
||||
private _cursorOffRowSince: number | null = null;
|
||||
private _seq = 0;
|
||||
private _confirmedTotal = 0;
|
||||
private _droppedTotal = 0;
|
||||
private _ttlTimer: ReturnType<typeof setTimeout> | null = null;
|
||||
/** Anchor hold: set after an unpredicted wire edit (backspace into echoed
|
||||
* text, any cleared input, an IME text commit). While held, new
|
||||
* predictions are suppressed: the displayed cursor is stale until the
|
||||
* next parsed write, and anchoring on it paints ghosts one cell off
|
||||
* (found by review: backspace-then-retype within RTT). Cleared by the
|
||||
* onWriteParsed pass and by public reconcile(), never by the inline
|
||||
* predictChar pass (which runs before the display could catch up). */
|
||||
private _anchorHold = false;
|
||||
private _reconcileScheduled = false;
|
||||
private _disposables: Array<{ dispose(): void }> = [];
|
||||
private _predictWhen: ((terminal: XtermTerminal) => boolean) | null;
|
||||
private _options: Required<Omit<PredictiveEchoOptions, 'foregroundColor' | 'backgroundColor' | 'predictWhen'>> &
|
||||
Pick<PredictiveEchoOptions, 'foregroundColor' | 'backgroundColor'>;
|
||||
private _font: FontStyle = {
|
||||
fontFamily: 'monospace',
|
||||
fontSize: '14px',
|
||||
fontWeight: 'normal',
|
||||
color: DEFAULT_FG,
|
||||
backgroundColor: DEFAULT_BG,
|
||||
letterSpacing: '',
|
||||
};
|
||||
|
||||
constructor(options?: PredictiveEchoOptions) {
|
||||
this._options = {
|
||||
zIndex: options?.zIndex ?? DEFAULT_OPTIONS.zIndex,
|
||||
underlinePredictions: options?.underlinePredictions ?? DEFAULT_OPTIONS.underlinePredictions,
|
||||
ttlMs: options?.ttlMs ?? DEFAULT_OPTIONS.ttlMs,
|
||||
maxPending: options?.maxPending ?? DEFAULT_OPTIONS.maxPending,
|
||||
cursorGraceMs: options?.cursorGraceMs ?? DEFAULT_OPTIONS.cursorGraceMs,
|
||||
edgeMarginCells: options?.edgeMarginCells ?? DEFAULT_OPTIONS.edgeMarginCells,
|
||||
foregroundColor: options?.foregroundColor,
|
||||
backgroundColor: options?.backgroundColor,
|
||||
};
|
||||
this._predictWhen = options?.predictWhen ?? null;
|
||||
}
|
||||
|
||||
// ─── Lifecycle ────────────────────────────────────────────────────
|
||||
|
||||
/** Called by `terminal.loadAddon()`. Do not call directly. */
|
||||
activate(terminal: XtermTerminal): void {
|
||||
this._terminal = terminal;
|
||||
|
||||
this._container = document.createElement('div');
|
||||
this._container.setAttribute('data-predictive-echo', '');
|
||||
this._container.style.cssText = `position:absolute;left:0;top:0;z-index:${this._options.zIndex};pointer-events:none`;
|
||||
const screen = terminal.element?.querySelector('.xterm-screen');
|
||||
if (screen) screen.appendChild(this._container);
|
||||
|
||||
this._readFontStyle();
|
||||
|
||||
// Debounced post-parse reconcile: xterm fires onWriteParsed after the
|
||||
// parser finishes a write chunk, so buffer reads see consistent state.
|
||||
// The microtask coalesces multi-chunk bursts into one pass.
|
||||
if (typeof terminal.onWriteParsed === 'function') {
|
||||
try {
|
||||
this._disposables.push(
|
||||
terminal.onWriteParsed(() => {
|
||||
if (this._reconcileScheduled) return;
|
||||
this._reconcileScheduled = true;
|
||||
queueMicrotask(() => {
|
||||
this._reconcileScheduled = false;
|
||||
this._anchorHold = false; // a parse pass ran: the display caught up
|
||||
this._safeReconcile();
|
||||
});
|
||||
})
|
||||
);
|
||||
} catch {
|
||||
/* consumers without a working emitter fall back to manual reconcile() */
|
||||
}
|
||||
}
|
||||
if (typeof terminal.onResize === 'function') {
|
||||
try {
|
||||
this._disposables.push(terminal.onResize(() => this.clearPredictions()));
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
dispose(): void {
|
||||
this.clearPredictions();
|
||||
for (const d of this._disposables) {
|
||||
try {
|
||||
d.dispose();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
this._disposables = [];
|
||||
this._container?.remove();
|
||||
this._container = null;
|
||||
this._terminal = null;
|
||||
}
|
||||
|
||||
// ─── Public API ───────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Predict a single typed character at the current insertion point.
|
||||
* Returns false when suppressed; the consumer sends the keystroke to the
|
||||
* PTY either way (the return value is informational, never a send gate).
|
||||
*/
|
||||
predictChar(ch: string): boolean {
|
||||
try {
|
||||
this._reconcile();
|
||||
if (this._anchorHold) return false; // display has not caught up with a wire edit
|
||||
|
||||
const t = this._terminal;
|
||||
if (!t || !this._container) return false;
|
||||
const dims = getCellDimensions(t);
|
||||
if (!dims) return false;
|
||||
const buf = t.buffer.active;
|
||||
if (typeof buf.cursorX !== 'number' || typeof buf.cursorY !== 'number') return false;
|
||||
if (buf.viewportY !== buf.baseY) return false;
|
||||
if (this._predictWhen && this._predictWhen(t) === false) return false;
|
||||
|
||||
const cps = Array.from(ch);
|
||||
if (cps.length !== 1) return false;
|
||||
const cp = cps[0].codePointAt(0)!;
|
||||
if (cp < 0x20 || cp === 0x7f) return false;
|
||||
const w = charCellWidth(t, cps[0]);
|
||||
if (w !== 1 && w !== 2) return false;
|
||||
if (w === 2 && !this._hasGetCell()) return false; // ASCII fallback misaligns on wide cols
|
||||
if (this._outstanding.length >= this._options.maxPending) return false;
|
||||
|
||||
if (this._outstanding.length === 0) {
|
||||
this._anchor = { row: buf.cursorY, col: buf.cursorX };
|
||||
this._cursorOffRowSince = null;
|
||||
}
|
||||
const anchor = this._anchor!;
|
||||
const last = this._outstanding[this._outstanding.length - 1];
|
||||
const offset = last ? last.offsetCells + last.width : 0;
|
||||
const col = anchor.col + offset;
|
||||
if (col + w > t.cols - this._options.edgeMarginCells) return false;
|
||||
|
||||
const rec: PredictionRecord = {
|
||||
seq: this._seq++,
|
||||
char: cps[0],
|
||||
width: w,
|
||||
offsetCells: offset,
|
||||
snapshot: this._readCell(anchor.row, col),
|
||||
sentAt: performance.now(),
|
||||
mismatches: 0,
|
||||
};
|
||||
this._outstanding.push(rec);
|
||||
addPredictionSpan(this._container, this._spans, {
|
||||
seq: rec.seq,
|
||||
row: anchor.row,
|
||||
col,
|
||||
char: rec.char,
|
||||
width: w,
|
||||
dims,
|
||||
font: this._font,
|
||||
underline: this._options.underlinePredictions,
|
||||
});
|
||||
this._armTtl();
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pop the newest outstanding prediction (visual only). Returns false when
|
||||
* none are outstanding. The consumer forwards \x7f UNCONDITIONALLY either
|
||||
* way; deleting already-echoed text renders at RTT.
|
||||
*/
|
||||
predictBackspace(): boolean {
|
||||
try {
|
||||
const rec = this._outstanding.pop();
|
||||
if (!rec) {
|
||||
// \x7f goes to the wire and will delete ECHOED text: the cursor is
|
||||
// about to move in a way we cannot see yet
|
||||
this._anchorHold = true;
|
||||
return false;
|
||||
}
|
||||
removePredictionSpan(this._spans, rec.seq);
|
||||
if (this._outstanding.length === 0) this._resetRun();
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Drop every outstanding prediction and its spans. Also arms the anchor
|
||||
* hold: consumers clear on inputs (Enter, Esc, arrows, pastes) whose
|
||||
* cursor effect is unknown until the next parsed write. */
|
||||
clearPredictions(): void {
|
||||
try {
|
||||
this._anchorHold = true;
|
||||
this._droppedTotal += this._outstanding.length;
|
||||
this._outstanding = [];
|
||||
clearAllSpans(this._spans);
|
||||
this._resetRun();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
|
||||
/** Manual reconcile pass, for consumers without onWriteParsed. By contract
|
||||
* it is called after writes parsed, so it also releases the anchor hold. */
|
||||
reconcile(): void {
|
||||
this._anchorHold = false;
|
||||
this._safeReconcile();
|
||||
}
|
||||
|
||||
/** Swap the prediction gate at runtime (mirrors the buffer addon's setPrompt). */
|
||||
setPredictWhen(fn: ((terminal: XtermTerminal) => boolean) | null): void {
|
||||
this._predictWhen = fn;
|
||||
}
|
||||
|
||||
/** Re-read font/theme (call after skin or font-size changes). */
|
||||
refreshFont(): void {
|
||||
this._readFontStyle();
|
||||
}
|
||||
|
||||
get hasPredictions(): boolean {
|
||||
return this._outstanding.length > 0;
|
||||
}
|
||||
|
||||
get state(): PredictionState {
|
||||
return {
|
||||
outstanding: this._outstanding.length,
|
||||
confirmedTotal: this._confirmedTotal,
|
||||
droppedTotal: this._droppedTotal,
|
||||
anchor: this._anchor ? { ...this._anchor } : null,
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Reconciliation ───────────────────────────────────────────────
|
||||
|
||||
private _safeReconcile(): void {
|
||||
try {
|
||||
this._reconcile();
|
||||
} catch {
|
||||
/* predictions may degrade, never break input */
|
||||
}
|
||||
}
|
||||
|
||||
private _reconcile(): void {
|
||||
const t = this._terminal;
|
||||
if (!t) return;
|
||||
if (this._outstanding.length === 0) return; // streaming cost: one boolean
|
||||
const buf = t.buffer.active;
|
||||
if (buf.viewportY !== buf.baseY) {
|
||||
this.clearPredictions(); // user scrolled up
|
||||
return;
|
||||
}
|
||||
if (typeof buf.cursorX !== 'number' || typeof buf.cursorY !== 'number') return; // TTL will clean
|
||||
const anchor = this._anchor!;
|
||||
const now = performance.now();
|
||||
|
||||
// Off-row grace: transient cursor excursions (repaints park the cursor
|
||||
// elsewhere mid-frame) are tolerated; a sustained move means the composer
|
||||
// relocated or the user navigated, so predictions are stale.
|
||||
if (buf.cursorY !== anchor.row) {
|
||||
this._cursorOffRowSince ??= now;
|
||||
if (now - this._cursorOffRowSince > this._options.cursorGraceMs) {
|
||||
this.clearPredictions();
|
||||
return;
|
||||
}
|
||||
} else {
|
||||
this._cursorOffRowSince = null;
|
||||
}
|
||||
|
||||
// Confirm loop: PREFIX-ONLY, and only with the cursor advanced past the
|
||||
// record. Cell match alone is not enough: the predicted char may equal
|
||||
// pre-existing content (placeholder glyphs), and an identical in-place
|
||||
// tmux repaint must be a no-op (cells match snapshots, cursor unmoved).
|
||||
while (this._outstanding.length > 0) {
|
||||
const rec = this._outstanding[0];
|
||||
const cell = this._readCell(anchor.row, anchor.col + rec.offsetCells);
|
||||
if (cell === rec.char && buf.cursorY === anchor.row && buf.cursorX >= anchor.col + rec.offsetCells + rec.width) {
|
||||
this._outstanding.shift();
|
||||
removePredictionSpan(this._spans, rec.seq);
|
||||
this._confirmedTotal++;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Mismatch scan (two-pass rule): a half-parsed row on pass N is fully
|
||||
// redrawn a few ms later, so only content foreign on TWO consecutive
|
||||
// passes cascades. Blank cells are NEUTRAL, not foreign: codex clears its
|
||||
// placeholder on the first echo, and the blanks left under later
|
||||
// predictions are what "not yet echoed" looks like, not evidence of a
|
||||
// redraw (measured 2026-08-09; without this, fast typing over the
|
||||
// placeholder cascades exactly when RTT is high). TTL still bounds them.
|
||||
let dropFrom = -1;
|
||||
for (let i = 0; i < this._outstanding.length; i++) {
|
||||
const rec = this._outstanding[i];
|
||||
const cell = this._readCell(anchor.row, anchor.col + rec.offsetCells);
|
||||
if (cell !== rec.snapshot && cell !== rec.char && cell !== ' ') {
|
||||
rec.mismatches++;
|
||||
if (rec.mismatches >= 2) {
|
||||
dropFrom = i;
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
rec.mismatches = 0;
|
||||
}
|
||||
}
|
||||
if (dropFrom !== -1) this._dropFrom(dropFrom);
|
||||
|
||||
// TTL: the first stale record drops itself and everything after it.
|
||||
for (let i = 0; i < this._outstanding.length; i++) {
|
||||
if (now - this._outstanding[i].sentAt > this._options.ttlMs) {
|
||||
this._dropFrom(i);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (this._outstanding.length === 0) {
|
||||
this._resetRun();
|
||||
} else {
|
||||
this._armTtl();
|
||||
}
|
||||
}
|
||||
|
||||
private _dropFrom(index: number): void {
|
||||
const dropped = this._outstanding.splice(index);
|
||||
for (const rec of dropped) removePredictionSpan(this._spans, rec.seq);
|
||||
this._droppedTotal += dropped.length;
|
||||
}
|
||||
|
||||
private _resetRun(): void {
|
||||
this._anchor = null;
|
||||
this._cursorOffRowSince = null;
|
||||
if (this._ttlTimer !== null) {
|
||||
clearTimeout(this._ttlTimer);
|
||||
this._ttlTimer = null;
|
||||
}
|
||||
}
|
||||
|
||||
private _armTtl(): void {
|
||||
if (this._ttlTimer !== null) return;
|
||||
const oldest = this._outstanding[0];
|
||||
if (!oldest) return;
|
||||
const delay = Math.max(0, oldest.sentAt + this._options.ttlMs - performance.now()) + 1;
|
||||
this._ttlTimer = setTimeout(() => {
|
||||
this._ttlTimer = null;
|
||||
this._safeReconcile();
|
||||
this._armTtl();
|
||||
}, delay);
|
||||
}
|
||||
|
||||
// ─── Cell access ──────────────────────────────────────────────────
|
||||
|
||||
private _hasGetCell(): boolean {
|
||||
const buf = this._terminal?.buffer.active;
|
||||
if (!buf) return false;
|
||||
const line = buf.getLine(buf.baseY + (buf.cursorY ?? 0));
|
||||
return typeof line?.getCell === 'function';
|
||||
}
|
||||
|
||||
/** Read one cell's chars at (viewport-relative row, col); '' -> ' '. */
|
||||
private _readCell(row: number, col: number): string {
|
||||
const buf = this._terminal!.buffer.active;
|
||||
const line = buf.getLine(buf.baseY + row);
|
||||
if (!line) return ' ';
|
||||
if (typeof line.getCell === 'function') {
|
||||
const chars = line.getCell(col)?.getChars() ?? '';
|
||||
return chars === '' ? ' ' : chars;
|
||||
}
|
||||
// ASCII fallback: code-unit index, misaligns after wide columns, which is
|
||||
// why width-2 predictions are suppressed without getCell.
|
||||
const text = line.translateToString(true);
|
||||
return text[col] ?? ' ';
|
||||
}
|
||||
|
||||
// ─── Font ─────────────────────────────────────────────────────────
|
||||
|
||||
/** Same recipe as the buffer addon's _cacheFont (kept private on purpose:
|
||||
* zerolag-input-addon.ts must stay untouched by this feature). */
|
||||
private _readFontStyle(): void {
|
||||
const t = this._terminal;
|
||||
if (!t) return;
|
||||
this._font.fontFamily = t.options.fontFamily || 'monospace';
|
||||
this._font.fontSize = (t.options.fontSize || 14) + 'px';
|
||||
this._font.fontWeight = String(t.options.fontWeight || 'normal');
|
||||
this._font.backgroundColor = this._options.backgroundColor ?? t.options.theme?.background ?? DEFAULT_BG;
|
||||
this._font.color = this._options.foregroundColor ?? t.options.theme?.foreground ?? DEFAULT_FG;
|
||||
this._font.letterSpacing = '';
|
||||
const rows = t.element?.querySelector('.xterm-rows');
|
||||
if (rows) {
|
||||
const cs = getComputedStyle(rows);
|
||||
this._font.letterSpacing = cs.letterSpacing;
|
||||
if (!this._options.foregroundColor && cs.color) this._font.color = cs.color;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -6,55 +6,50 @@ import type { XtermTerminal, PromptFinder, PromptPosition } from './types.js';
|
||||
*
|
||||
* @returns The prompt position (viewport-relative), or `null` if not found.
|
||||
*/
|
||||
export function findPrompt(
|
||||
terminal: XtermTerminal,
|
||||
finder: PromptFinder,
|
||||
): PromptPosition | null {
|
||||
try {
|
||||
const buffer = terminal.buffer.active;
|
||||
const viewportTop = buffer.viewportY;
|
||||
export function findPrompt(terminal: XtermTerminal, finder: PromptFinder): PromptPosition | null {
|
||||
try {
|
||||
const buffer = terminal.buffer.active;
|
||||
const viewportTop = buffer.viewportY;
|
||||
|
||||
switch (finder.type) {
|
||||
case 'character': {
|
||||
for (let row = terminal.rows - 1; row >= 0; row--) {
|
||||
const line = buffer.getLine(viewportTop + row);
|
||||
if (!line) continue;
|
||||
const text = line.translateToString(true);
|
||||
const idx = text.lastIndexOf(finder.char);
|
||||
if (idx >= 0) return { row, col: idx };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
case 'regex': {
|
||||
// Create a fresh non-global regex to avoid lastIndex mutation
|
||||
// and ensure .match() returns a single result with .index
|
||||
const pattern = finder.pattern;
|
||||
const safePattern = pattern.global
|
||||
? new RegExp(pattern.source, pattern.flags.replace('g', ''))
|
||||
: pattern;
|
||||
for (let row = terminal.rows - 1; row >= 0; row--) {
|
||||
const line = buffer.getLine(viewportTop + row);
|
||||
if (!line) continue;
|
||||
const text = line.translateToString(true);
|
||||
const match = text.match(safePattern);
|
||||
if (match) {
|
||||
const col = match.index ?? 0;
|
||||
return { row, col };
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
case 'custom':
|
||||
return finder.find(terminal);
|
||||
|
||||
default:
|
||||
return null;
|
||||
switch (finder.type) {
|
||||
case 'character': {
|
||||
for (let row = terminal.rows - 1; row >= 0; row--) {
|
||||
const line = buffer.getLine(viewportTop + row);
|
||||
if (!line) continue;
|
||||
const text = line.translateToString(true);
|
||||
const idx = text.lastIndexOf(finder.char);
|
||||
if (idx >= 0) return { row, col: idx };
|
||||
}
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
|
||||
case 'regex': {
|
||||
// Create a fresh non-global regex to avoid lastIndex mutation
|
||||
// and ensure .match() returns a single result with .index
|
||||
const pattern = finder.pattern;
|
||||
const safePattern = pattern.global ? new RegExp(pattern.source, pattern.flags.replace('g', '')) : pattern;
|
||||
for (let row = terminal.rows - 1; row >= 0; row--) {
|
||||
const line = buffer.getLine(viewportTop + row);
|
||||
if (!line) continue;
|
||||
const text = line.translateToString(true);
|
||||
const match = text.match(safePattern);
|
||||
if (match) {
|
||||
const col = match.index ?? 0;
|
||||
return { row, col };
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
case 'custom':
|
||||
return finder.find(terminal);
|
||||
|
||||
default:
|
||||
return null;
|
||||
}
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -65,19 +60,15 @@ export function findPrompt(
|
||||
* @param offset - Characters to skip after the prompt marker (e.g., 2 for "> ")
|
||||
* @returns The text after the prompt, trimmed. Empty string if nothing found.
|
||||
*/
|
||||
export function readTextAfterPrompt(
|
||||
terminal: XtermTerminal,
|
||||
prompt: PromptPosition,
|
||||
offset: number,
|
||||
): string {
|
||||
try {
|
||||
const buffer = terminal.buffer.active;
|
||||
const absRow = buffer.viewportY + prompt.row;
|
||||
const line = buffer.getLine(absRow);
|
||||
if (!line) return '';
|
||||
const lineText = line.translateToString(true);
|
||||
return lineText.slice(prompt.col + offset).trimEnd();
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
export function readTextAfterPrompt(terminal: XtermTerminal, prompt: PromptPosition, offset: number): string {
|
||||
try {
|
||||
const buffer = terminal.buffer.active;
|
||||
const absRow = buffer.viewportY + prompt.row;
|
||||
const line = buffer.getLine(absRow);
|
||||
if (!line) return '';
|
||||
const lineText = line.translateToString(true);
|
||||
return lineText.slice(prompt.col + offset).trimEnd();
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,9 +22,15 @@ export interface XtermTerminal {
|
||||
readonly active: {
|
||||
readonly viewportY: number;
|
||||
readonly baseY: number;
|
||||
/** Cursor column (0-based). Used by PredictiveEchoAddon. */
|
||||
readonly cursorX?: number;
|
||||
/** Cursor row, relative to baseY (0-based). Used by PredictiveEchoAddon. */
|
||||
readonly cursorY?: number;
|
||||
getLine(y: number):
|
||||
| {
|
||||
translateToString(trimRight?: boolean): string;
|
||||
/** Cell access (xterm public API). Optional: mocks/exotic hosts may omit it. */
|
||||
getCell?(x: number): { getChars(): string; getWidth(): number } | undefined;
|
||||
}
|
||||
| undefined;
|
||||
};
|
||||
@@ -34,6 +40,10 @@ export interface XtermTerminal {
|
||||
getStringCellWidth(str: string): number;
|
||||
activeVersion?: string;
|
||||
};
|
||||
/** Fires after the parser finishes a write chunk. Used by PredictiveEchoAddon. */
|
||||
onWriteParsed?(cb: () => void): { dispose(): void };
|
||||
/** Fires on terminal resize. Used by PredictiveEchoAddon. */
|
||||
onResize?(cb: (size: { cols: number; rows: number }) => void): { dispose(): void };
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -6,122 +6,125 @@ import type { XtermTerminal } from '../src/types.js';
|
||||
let cleanups: (() => void)[] = [];
|
||||
|
||||
afterEach(() => {
|
||||
for (const fn of cleanups) fn();
|
||||
cleanups = [];
|
||||
for (const fn of cleanups) fn();
|
||||
cleanups = [];
|
||||
});
|
||||
|
||||
describe('getCellDimensions', () => {
|
||||
describe('v5 private API (mock _core._renderService)', () => {
|
||||
it('returns cell width and height from css.cell', () => {
|
||||
const mock = createMockTerminal({ cellWidth: 8.4, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
expect(dims!.width).toBe(8.4);
|
||||
expect(dims!.height).toBe(19);
|
||||
});
|
||||
|
||||
it('returns charTop from device.char.top divided by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 8, cellHeight: 19,
|
||||
deviceCharTop: 2,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// DPR=1 in jsdom, so charTop = 2 / 1 = 2
|
||||
expect(dims!.charTop).toBe(2);
|
||||
});
|
||||
|
||||
it('returns charHeight from device.char.height divided by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 8, cellHeight: 19,
|
||||
deviceCharHeight: 16,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// DPR=1, so charHeight = 16 / 1 = 16
|
||||
expect(dims!.charHeight).toBe(16);
|
||||
});
|
||||
|
||||
it('defaults charTop to 0 when device.char not present', () => {
|
||||
// Default mock has deviceCharTop=0
|
||||
const mock = createMockTerminal({ cellWidth: 8, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims!.charTop).toBe(0);
|
||||
});
|
||||
|
||||
it('defaults charHeight to cellH when device.char.height not set', () => {
|
||||
// Default mock has deviceCharHeight=cellH
|
||||
const mock = createMockTerminal({ cellWidth: 8, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims!.charHeight).toBe(19);
|
||||
});
|
||||
describe('v5 private API (mock _core._renderService)', () => {
|
||||
it('returns cell width and height from css.cell', () => {
|
||||
const mock = createMockTerminal({ cellWidth: 8.4, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
expect(dims!.width).toBe(8.4);
|
||||
expect(dims!.height).toBe(19);
|
||||
});
|
||||
|
||||
describe('DPR simulation', () => {
|
||||
const originalDPR = globalThis.devicePixelRatio;
|
||||
|
||||
beforeEach(() => {
|
||||
// Set DPR=2 to test division
|
||||
Object.defineProperty(globalThis, 'devicePixelRatio', {
|
||||
value: 2,
|
||||
writable: true,
|
||||
configurable: true,
|
||||
});
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
Object.defineProperty(globalThis, 'devicePixelRatio', {
|
||||
value: originalDPR,
|
||||
writable: true,
|
||||
configurable: true,
|
||||
});
|
||||
});
|
||||
|
||||
it('divides device.char.top by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 16, cellHeight: 38,
|
||||
deviceCharTop: 4,
|
||||
deviceCharHeight: 32,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// charTop = 4 / 2 = 2
|
||||
expect(dims!.charTop).toBe(2);
|
||||
// charHeight = 32 / 2 = 16
|
||||
expect(dims!.charHeight).toBe(16);
|
||||
});
|
||||
it('returns charTop from device.char.top divided by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 8,
|
||||
cellHeight: 19,
|
||||
deviceCharTop: 2,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// DPR=1 in jsdom, so charTop = 2 / 1 = 2
|
||||
expect(dims!.charTop).toBe(2);
|
||||
});
|
||||
|
||||
describe('null cases', () => {
|
||||
it('returns null for terminal without _core', () => {
|
||||
const terminal = {
|
||||
element: document.createElement('div'),
|
||||
cols: 80,
|
||||
rows: 24,
|
||||
options: {},
|
||||
buffer: { active: { viewportY: 0, baseY: 0, getLine: () => undefined } },
|
||||
} as unknown as XtermTerminal;
|
||||
const dims = getCellDimensions(terminal);
|
||||
expect(dims).toBeNull();
|
||||
});
|
||||
|
||||
it('returns null for terminal with no dimensions', () => {
|
||||
const terminal = {
|
||||
element: document.createElement('div'),
|
||||
cols: 80,
|
||||
rows: 24,
|
||||
options: {},
|
||||
buffer: { active: { viewportY: 0, baseY: 0, getLine: () => undefined } },
|
||||
_core: { _renderService: {} },
|
||||
} as unknown as XtermTerminal;
|
||||
const dims = getCellDimensions(terminal);
|
||||
expect(dims).toBeNull();
|
||||
});
|
||||
it('returns charHeight from device.char.height divided by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 8,
|
||||
cellHeight: 19,
|
||||
deviceCharHeight: 16,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// DPR=1, so charHeight = 16 / 1 = 16
|
||||
expect(dims!.charHeight).toBe(16);
|
||||
});
|
||||
|
||||
it('defaults charTop to 0 when device.char not present', () => {
|
||||
// Default mock has deviceCharTop=0
|
||||
const mock = createMockTerminal({ cellWidth: 8, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims!.charTop).toBe(0);
|
||||
});
|
||||
|
||||
it('defaults charHeight to cellH when device.char.height not set', () => {
|
||||
// Default mock has deviceCharHeight=cellH
|
||||
const mock = createMockTerminal({ cellWidth: 8, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims!.charHeight).toBe(19);
|
||||
});
|
||||
});
|
||||
|
||||
describe('DPR simulation', () => {
|
||||
const originalDPR = globalThis.devicePixelRatio;
|
||||
|
||||
beforeEach(() => {
|
||||
// Set DPR=2 to test division
|
||||
Object.defineProperty(globalThis, 'devicePixelRatio', {
|
||||
value: 2,
|
||||
writable: true,
|
||||
configurable: true,
|
||||
});
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
Object.defineProperty(globalThis, 'devicePixelRatio', {
|
||||
value: originalDPR,
|
||||
writable: true,
|
||||
configurable: true,
|
||||
});
|
||||
});
|
||||
|
||||
it('divides device.char.top by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 16,
|
||||
cellHeight: 38,
|
||||
deviceCharTop: 4,
|
||||
deviceCharHeight: 32,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// charTop = 4 / 2 = 2
|
||||
expect(dims!.charTop).toBe(2);
|
||||
// charHeight = 32 / 2 = 16
|
||||
expect(dims!.charHeight).toBe(16);
|
||||
});
|
||||
});
|
||||
|
||||
describe('null cases', () => {
|
||||
it('returns null for terminal without _core', () => {
|
||||
const terminal = {
|
||||
element: document.createElement('div'),
|
||||
cols: 80,
|
||||
rows: 24,
|
||||
options: {},
|
||||
buffer: { active: { viewportY: 0, baseY: 0, getLine: () => undefined } },
|
||||
} as unknown as XtermTerminal;
|
||||
const dims = getCellDimensions(terminal);
|
||||
expect(dims).toBeNull();
|
||||
});
|
||||
|
||||
it('returns null for terminal with no dimensions', () => {
|
||||
const terminal = {
|
||||
element: document.createElement('div'),
|
||||
cols: 80,
|
||||
rows: 24,
|
||||
options: {},
|
||||
buffer: { active: { viewportY: 0, baseY: 0, getLine: () => undefined } },
|
||||
_core: { _renderService: {} },
|
||||
} as unknown as XtermTerminal;
|
||||
const dims = getCellDimensions(terminal);
|
||||
expect(dims).toBeNull();
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,188 @@
|
||||
/**
|
||||
* @vitest-environment jsdom
|
||||
*
|
||||
* Layer 2 (the load-bearing suite): the REAL algorithm against the REAL xterm
|
||||
* parser, fed by fixtures recorded from real codex 0.147 through the
|
||||
* production pipeline (tmux + the codex full strip). See
|
||||
* scripts/dev/record-codex-frames.mjs in the consuming repo.
|
||||
*
|
||||
* Every replay ends with the convergence invariant: predictions never outlive
|
||||
* their run (outstanding 0, span container empty).
|
||||
*/
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { PredictiveEchoAddon } from '../src/predictive-echo-addon.js';
|
||||
import {
|
||||
CELL_H,
|
||||
CELL_W,
|
||||
classifyPredictInput,
|
||||
codexComposerGate,
|
||||
createReplayTerminal,
|
||||
loadFixture,
|
||||
type ReplayTerminal,
|
||||
} from './replay-helpers.js';
|
||||
|
||||
async function flushMicrotasks() {
|
||||
await Promise.resolve();
|
||||
await Promise.resolve();
|
||||
}
|
||||
|
||||
function sleep(ms: number) {
|
||||
return new Promise((r) => setTimeout(r, ms));
|
||||
}
|
||||
|
||||
interface KeyEvent {
|
||||
key: string;
|
||||
kind: ReturnType<typeof classifyPredictInput>;
|
||||
painted: boolean;
|
||||
spansAfter: number;
|
||||
}
|
||||
|
||||
function assertSpansInGrid(rt: ReplayTerminal) {
|
||||
for (const s of rt.spans()) {
|
||||
const left = parseFloat(s.style.left);
|
||||
const width = parseFloat(s.style.width);
|
||||
const top = parseFloat(s.style.top);
|
||||
expect(left + width).toBeLessThanOrEqual(rt.hybrid.cols * CELL_W);
|
||||
expect(top).toBeLessThanOrEqual((rt.hybrid.rows - 1) * CELL_H);
|
||||
expect(left).toBeGreaterThanOrEqual(0);
|
||||
expect(top).toBeGreaterThanOrEqual(0);
|
||||
}
|
||||
}
|
||||
|
||||
async function replay(name: string) {
|
||||
const { meta, lines } = loadFixture(name);
|
||||
const rt = createReplayTerminal(meta.cols, meta.rows);
|
||||
const addon = new PredictiveEchoAddon({ predictWhen: codexComposerGate });
|
||||
addon.activate(rt.hybrid);
|
||||
|
||||
const events: KeyEvent[] = [];
|
||||
for (const line of lines) {
|
||||
if (line.keyAt) {
|
||||
const kind = classifyPredictInput(line.data);
|
||||
let painted = false;
|
||||
if (kind === 'char') painted = addon.predictChar(line.data);
|
||||
else if (kind === 'backspace') addon.predictBackspace();
|
||||
else addon.clearPredictions(); // 'clear' AND 'text', like the terminal-ui hook
|
||||
// Span/record parity and grid bounds hold at every step
|
||||
expect(rt.spanCount()).toBe(addon.state.outstanding);
|
||||
assertSpansInGrid(rt);
|
||||
events.push({ key: line.data, kind, painted, spansAfter: rt.spanCount() });
|
||||
} else {
|
||||
await rt.write(line.data);
|
||||
await flushMicrotasks();
|
||||
}
|
||||
}
|
||||
return { rt, addon, events, meta };
|
||||
}
|
||||
|
||||
/** Convergence invariant: after the last chunk + reconcile (+ TTL if needed),
|
||||
* nothing outlives the run. */
|
||||
async function converge(rt: ReplayTerminal, addon: PredictiveEchoAddon) {
|
||||
addon.reconcile();
|
||||
if (addon.state.outstanding > 0) {
|
||||
await sleep(1100); // ttlMs default
|
||||
addon.reconcile();
|
||||
}
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(rt.spanCount()).toBe(0);
|
||||
}
|
||||
|
||||
describe('codex replay', () => {
|
||||
it('type-hello: all 5 predictions confirm, zero drops, composer converges', async () => {
|
||||
const { rt, addon, events } = await replay('type-hello');
|
||||
const chars = events.filter((e) => e.kind === 'char');
|
||||
expect(chars).toHaveLength(5);
|
||||
expect(chars.every((e) => e.painted)).toBe(true);
|
||||
await converge(rt, addon);
|
||||
expect(addon.state.confirmedTotal).toBe(5);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
expect(rt.cursorRowText()).toBe('› hello');
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('slash-picker: "/" and filter chars confirm; no ghosts while picker rows redraw', async () => {
|
||||
const { rt, addon, events } = await replay('slash-picker');
|
||||
const chars = events.filter((e) => e.kind === 'char');
|
||||
expect(chars.map((e) => e.key)).toEqual(['/', 'm', 'o']);
|
||||
expect(chars.every((e) => e.painted)).toBe(true);
|
||||
await converge(rt, addon);
|
||||
expect(addon.state.confirmedTotal).toBe(3);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('wrap: predictions stay inside the grid, continuation rows fall back to real echo, buffer converges', async () => {
|
||||
const { rt, addon, events } = await replay('wrap');
|
||||
// The gate goes false once the cursor is on a wrapped continuation row
|
||||
// (2-space indent, no "› "): a tail of keystrokes must be suppressed.
|
||||
const chars = events.filter((e) => e.kind === 'char');
|
||||
expect(chars.some((e) => !e.painted)).toBe(true);
|
||||
expect(chars.some((e) => e.painted)).toBe(true);
|
||||
await converge(rt, addon);
|
||||
// The composer content is exactly what was typed (word-wrapped)
|
||||
const b = rt.term.buffer.active;
|
||||
const cursorRow = b.cursorY;
|
||||
expect(rt.rowText(cursorRow).trim()).toBe('this line twice over');
|
||||
expect(rt.rowText(cursorRow - 1)).toMatch(/^› the quick brown fox/);
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('streaming-burst: typed predictions confirm; the re-rendered composer keeps its signature', async () => {
|
||||
const { rt, addon, events } = await replay('streaming-burst');
|
||||
const chars = events.filter((e) => e.kind === 'char');
|
||||
expect(chars).toHaveLength(5); // "hello" (the \r is kind 'clear')
|
||||
await converge(rt, addon);
|
||||
expect(addon.state.confirmedTotal).toBe(5);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
// After the 401 burst codex re-renders a fresh composer at the cursor
|
||||
expect(rt.cursorRowText()).toMatch(/^› /);
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('streaming-real: mid-stream typing survives real baseY growth (recorded with real auth)', async () => {
|
||||
// The one shape the fake-key lab cannot produce: a genuine model reply
|
||||
// streaming above the pinned composer pushes lines into history, so
|
||||
// baseY GROWS while predictions are outstanding: the no-drop-on-baseY
|
||||
// rule against reality instead of a synthetic scroll.
|
||||
const { rt, addon, events } = await replay('streaming-real');
|
||||
expect(rt.term.buffer.active.baseY).toBeGreaterThan(0); // history really grew
|
||||
const midStream = events.filter((e) => e.kind === 'char' && ['a', 'b', 'c'].includes(e.key));
|
||||
expect(midStream.length).toBe(3);
|
||||
expect(midStream.some((e) => e.painted)).toBe(true); // predictions ran mid-stream
|
||||
await converge(rt, addon);
|
||||
expect(rt.cursorRowText()).toBe('› abc'); // the mid-stream chars landed intact
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('paste-bracketed: typed chars confirm, the paste clears predictions, content intact', async () => {
|
||||
const { rt, addon, events } = await replay('paste-bracketed');
|
||||
const paste = events.find((e) => e.key.startsWith('\x1b[200~'))!;
|
||||
expect(paste.kind).toBe('clear');
|
||||
expect(paste.spansAfter).toBe(0);
|
||||
await converge(rt, addon);
|
||||
expect(addon.state.confirmedTotal).toBe(2); // 'a', 'b'
|
||||
expect(rt.cursorRowText()).toContain('abXYZpasted');
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('trust-modal: the predictWhen gate paints ZERO spans on the modal (ghost eliminator)', async () => {
|
||||
const { rt, addon, events } = await replay('trust-modal');
|
||||
const x = events.find((e) => e.key === 'x')!;
|
||||
expect(x.painted).toBe(false);
|
||||
expect(x.spansAfter).toBe(0);
|
||||
expect(events.every((e) => e.spansAfter === 0)).toBe(true);
|
||||
await converge(rt, addon);
|
||||
expect(addon.state.confirmedTotal).toBe(0);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
// The transition landed on the real composer afterwards
|
||||
expect(rt.cursorRowText()).toMatch(/^› /);
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
});
|
||||
@@ -0,0 +1,28 @@
|
||||
{"scenario":"paste-bracketed","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:51:11.762Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":0,"data":"\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":45,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-bWhHjh\u001b(B\u001b[m$ "}
|
||||
{"delayMs":638,"data":"exec codex\r\n"}
|
||||
{"delayMs":420,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":182,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":5,"data":"\r\n\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2;30r\u001b[2;1H\u001bM\u001bM\u001bM\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[1;30r\u001b[4;1H\u001b(B\u001b[m"}
|
||||
{"delayMs":1,"data":" \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[6;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bWhHjh\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[14;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mSummarize rec\u001b(B\u001b[m\u001b[2ment commits\u001b[16;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bWhHjh\u001b[14;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":7,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;27H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":21,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;27H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":8,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;27H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":159,"data":"\u001b[6;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mSummarize recent commits\u001b[9;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bWhHjh\u001b[7;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":21,"data":"\u001b[5;30r\u001b[5;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\r\n\u001b[2m╭─────────────────────────────────────────────────╮\u001b[1;30r\u001b[7;1H\u001b(B\u001b[m\u001b[2m│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bWhHjh\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[18;27H\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[24C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[24C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":3487,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[32m\u001b[1m\u001b[Harkon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-bWhHjh\u001b(B\u001b[m$ exec codex\u001b[K\u001b[33m\r\n⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\u001b[39m\u001b[K\r\n \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\u001b[K\r\n \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[39m\u001b[K\r\n\u001b[K\u001b[2m\r\n╭─────────────────────────────────────────────────╮\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bWhHjh\u001b[2m │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n╰─────────────────────────────────────────────────╯\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\u001b[K\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\u001b[K\r\n reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[1m\r\n›\u001b(B\u001b[m\u001b[1X\u001b[2m\u001b[CSummarize recent commits\u001b(B\u001b[m\u001b[K\r\n\u001b[K\u001b[20;2H\u001b[1K\u001b[38;5;223m\u001b[Cgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bWhHjh\u001b[39m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[18;3H"}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"delayMs":207,"data":"a\u001b[K\u001b[20;80H\u001b[K\u001b[18;4H"}
|
||||
{"keyAt":true,"data":"b"}
|
||||
{"delayMs":91,"data":"b\u001b[K\u001b[20;80H\u001b[K\u001b[18;5H"}
|
||||
{"keyAt":true,"data":"\u001b[200~XYZpasted\u001b[201~"}
|
||||
{"delayMs":383,"data":"XYZpasted\u001b[K\u001b[20;80H\u001b[K\u001b[18;14H"}
|
||||
@@ -0,0 +1,32 @@
|
||||
{"scenario":"slash-picker","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:50:42.069Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":0,"data":"\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":37,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-bw9Uto\u001b(B\u001b[m$ "}
|
||||
{"delayMs":647,"data":"exec codex\r\n"}
|
||||
{"delayMs":437,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":183,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":8,"data":"\r\n\u001b[J\u001b[A\u001b[K\u001b[2;30r\u001b[2;1H\u001bM\u001bM\u001bM\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\r\n\u001b[39m \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[6;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bw9Uto\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[14;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[16;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bw9Uto\u001b[1;30r\u001b[14;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":9,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":12,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":10,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":157,"data":"\u001b[6;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[9;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bw9Uto\u001b[7;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":20,"data":"\u001b[5;30r\u001b[5;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\r\n\u001b[2m╭─────────────────────────────────────────────────╮\u001b[1;30r\u001b[7;1H\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bw9Uto\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[18;39H\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[36C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[36C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":3476,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[32m\u001b[1m\u001b[Harkon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-bw9Uto\u001b(B\u001b[m$ exec codex\u001b[K\u001b[33m\r\n⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\u001b[39m\u001b[K\r\n \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\u001b[K\r\n \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[39m\u001b[K\r\n\u001b[K\u001b[2m\r\n╭─────────────────────────────────────────────────╮\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bw9Uto\u001b[2m │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n╰─────────────────────────────────────────────────╯\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\u001b[K\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\u001b[K\r\n reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[1m\r\n›\u001b(B\u001b[m\u001b[1X\u001b[2m\u001b[CUse /skills to list available skills\u001b(B\u001b[m\u001b[K\r\n\u001b[K\u001b[20;2H\u001b[1K\u001b[38;5;223m\u001b[Cgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bw9Uto\u001b[39m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[18;3H"}
|
||||
{"keyAt":true,"data":"/"}
|
||||
{"delayMs":207,"data":"\u001b[17;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":1,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m/\u001b[20;3H\u001b[36m\u001b[1m/model choose what model and reasoning effort to use\u001b[21;3H\u001b(B\u001b[m/fast\u001b[10C\u001b[2m1.5x speed, increased usage\u001b[22;3H\u001b(B\u001b[m/ide\u001b[11C\u001b[2minclude current selection, open files, and other context from your IDE\u001b[23;3H\u001b(B\u001b[m/permissions\u001b[3C\u001b[2mchoose what Codex is allowed to do\u001b[24;3H\u001b(B\u001b[m/keymap\u001b[8C\u001b[2mremap TUI shortcuts\u001b[25;3H\u001b(B\u001b[m/vim\u001b[11C\u001b[2mtoggle Vim mode for the composer\u001b[26;3H\u001b(B\u001b[m/experimental\u001b[2C\u001b[2mtoggle experimental features\u001b[27;3H\u001b(B\u001b[m/approve\u001b[7C\u001b[2mapprove one retry of a recent auto-review denial\u001b[18;4H\u001b(B\u001b[m"}
|
||||
{"keyAt":true,"data":"m"}
|
||||
{"delayMs":398,"data":"\u001b[17;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":1,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m/m\u001b[20;3H\u001b[36m\u001b[1m/model choose what model and reasoning effort to use\u001b[21;3H\u001b(B\u001b[m/\u001b[1mm\u001b(B\u001b[memories\u001b[2C\u001b[2mconfigure memory use and generation\u001b[22;3H\u001b(B\u001b[m/\u001b[1mm\u001b(B\u001b[mention\u001b[3C\u001b[2mmention a file\u001b[23;3H\u001b(B\u001b[m/\u001b[1mm\u001b(B\u001b[mcp\u001b[7C\u001b[2mlist configured MCP tools; use /mcp verbose for details\u001b[18;5H\u001b(B\u001b[m"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":148,"data":"\u001b[17;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m/mo\u001b[20;3H\u001b[36m\u001b[1m/model choose what model and reasoning effort to use\u001b[18;6H\u001b(B\u001b[m"}
|
||||
{"keyAt":true,"data":"\u001b"}
|
||||
@@ -0,0 +1,233 @@
|
||||
{"scenario":"streaming-burst","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:51:03.828Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":0,"data":"\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":39,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b(B\u001b[m$ "}
|
||||
{"delayMs":635,"data":"exec codex\r\n"}
|
||||
{"delayMs":439,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":184,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":9,"data":"\r\n\u001b[J\u001b[A\u001b[K\u001b[2;30r\u001b[2;1H\u001bM\u001bM\u001bM\u001b[1;30r\u001b[2;1H\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\r\n\u001b(B\u001b[m \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[6;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-ruT16A\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[14;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[16;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b[14;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":8,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":8,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":8,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":158,"data":"\u001b[6;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[9;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b[7;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":21,"data":"\u001b[5;30r\u001b[5;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\r\n\u001b[2m╭─────────────────────────────────────────────────╮\u001b[1;30r\u001b[7;1H\u001b(B\u001b[m\u001b[2m│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-ruT16A\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[18;39H\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[36C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":3486,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[32m\u001b[1m\u001b[Harkon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b(B\u001b[m$ exec codex\u001b[K\u001b[33m\r\n⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\u001b[39m\u001b[K\r\n \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\u001b[K\r\n \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[39m\u001b[K\r\n\u001b[K\u001b[2m\r\n╭─────────────────────────────────────────────────╮\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-ruT16A\u001b[2m │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n╰─────────────────────────────────────────────────╯\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\u001b[K\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\u001b[K\r\n reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[1m\r\n›\u001b(B\u001b[m\u001b[1X\u001b[2m\u001b[CUse /skills to list available skills\u001b(B\u001b[m\u001b[K\r\n\u001b[K\u001b[20;2H\u001b[1K\u001b[38;5;223m\u001b[Cgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b[39m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[18;3H"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":199,"data":"h\u001b[K\u001b[20;80H\u001b[K\u001b[18;4H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":40,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;5H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":40,"data":"l\u001b[K\u001b[20;80H\u001b[K\u001b[18;6H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":40,"data":"l\u001b[K\u001b[20;80H\u001b[K\u001b[18;7H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":40,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;8H"}
|
||||
{"keyAt":true,"data":"\r"}
|
||||
{"delayMs":281,"data":"\u001b[16;30r\u001b[16;1H\u001bM\u001bM\u001bM\u001bM\u001b[1;30r\u001b[18;1H"}
|
||||
{"delayMs":0,"data":"\u001b[1m\u001b[2m› \u001b(B\u001b[mhello\r\n"}
|
||||
{"delayMs":0,"data":"\u001b[22;3H\u001b[2mUse /skills to list available skills\u001b(B\u001b[m\u001b[K\u001b[24;80H\u001b[K\u001b[22;3H"}
|
||||
{"delayMs":12,"data":"\u001b[36C\u001b[K\u001b[24;80H\u001b[K\u001b[22;3H"}
|
||||
{"delayMs":6,"data":"\u001b[36C\u001b[K\u001b[24;80H\u001b[K\u001b[22;3H"}
|
||||
{"delayMs":6,"data":"\u001b[36C\u001b[K\u001b[24;80H\u001b[K\u001b[22;3H"}
|
||||
{"delayMs":118,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":1,"data":"\r\n•\u001b[C\u001b[2mWorking\u001b[C(0s • esc to interrupt)\u001b[24;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[26;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b[24;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[3AW\u001b[30C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[21;1H\u001b[2m◦\u001b[C\u001b(B\u001b[m\u001b[1mW\u001b(B\u001b[mo\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;4H\u001b[1mo\u001b(B\u001b[mr\u001b[28C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;5H\u001b[1mr\u001b(B\u001b[mk\u001b[27C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;6H\u001b[1mk\u001b(B\u001b[mi\u001b[26C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;7H\u001b[1mi\u001b(B\u001b[mn\u001b[25C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":34,"data":"\u001b[3AW\u001b[4C\u001b[1mn\u001b(B\u001b[mg\u001b[24C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;12H\u001b[2m1\u001b(B\u001b[m\u001b[21C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":19,"data":"\u001b[21;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":1,"data":"\r\n\u001b[2m◦\u001b[CReconne\u001b(B\u001b[mc\u001b[1mting.\u001b(B\u001b[m.\u001b[2m. 2/5\u001b[C(1s • esc to interrupt)\r\n └ Unexpected status 401 Unauthorized: {\r\n \"error\": {\r\n \"message\": \"Incorre, url: wss://api.openai.com/v1/responses, cf-ray: a2831cf59baa039d-ZRH,…\u001b[27;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[29;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b[27;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":33,"data":"\u001b[21;10H\u001b[2mc\u001b(B\u001b[mt\u001b[4C\u001b[1m.\u001b(B\u001b[m.\u001b[28C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;11H\u001b[2mt\u001b(B\u001b[mi\u001b[4C\u001b[1m.\u001b(B\u001b[m \u001b[27C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;12H\u001b[2mi\u001b(B\u001b[mn\u001b[4C\u001b[1m \u001b(B\u001b[m2\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H•\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;13H\u001b[2mn\u001b(B\u001b[mg\u001b[4C\u001b[1m2\u001b(B\u001b[m/\u001b[25C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;14H\u001b[2mg\u001b(B\u001b[m.\u001b[4C\u001b[1m/\u001b(B\u001b[m5\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":36,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[?25l\u001b[?12l\u001b[?25h\u001b[27;3H"}
|
||||
{"delayMs":31,"data":"\u001b[21;15H\u001b[2m.\u001b(B\u001b[m.\u001b[4C\u001b[1m5\u001b(B\u001b[m\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;16H\u001b[2m.\u001b(B\u001b[m.\u001b[28C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;17H\u001b[2m.\u001b(B\u001b[m \u001b[27C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":35,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;18H\u001b[2m \u001b(B\u001b[m2\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;19H\u001b[2m2\u001b(B\u001b[m/\u001b[25C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;20H\u001b[2m/\u001b(B\u001b[m5\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;21H\u001b[2m5\u001b(B\u001b[m\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H\u001b[2m◦\u001b[27;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":21,"data":"\u001b[21;19H\u001b[2m3\u001b(B\u001b[m\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;85H\u001b[2maca388822\u001b(B\u001b[m\u001b[6C\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;24H\u001b[2m2\u001b(B\u001b[m\u001b[21C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[6AR\u001b[42C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H•\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[6A\u001b[1mR\u001b(B\u001b[me\u001b[41C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":33,"data":"\u001b[21;4H\u001b[1me\u001b(B\u001b[mc\u001b[40C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;5H\u001b[1mc\u001b(B\u001b[mo\u001b[39C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;6H\u001b[1mo\u001b(B\u001b[mn\u001b[38C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;7H\u001b[1mn\u001b(B\u001b[mn\u001b[37C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[6AR\u001b[4C\u001b[1mn\u001b(B\u001b[me\u001b[36C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[6A\u001b[2mR\u001b(B\u001b[me\u001b[4C\u001b[1me\u001b(B\u001b[mc\u001b[35C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;4H\u001b[2me\u001b(B\u001b[mc\u001b[4C\u001b[1mc\u001b(B\u001b[mt\u001b[34C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;5H\u001b[2mc\u001b(B\u001b[mo\u001b[4C\u001b[1mt\u001b(B\u001b[mi\u001b[33C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;6H\u001b[2mo\u001b(B\u001b[mn\u001b[4C\u001b[1mi\u001b(B\u001b[mn\u001b[32C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;7H\u001b[2mn\u001b(B\u001b[mn\u001b[4C\u001b[1mn\u001b(B\u001b[mg\u001b[31C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H\u001b[2m◦\u001b[6Cn\u001b(B\u001b[me\u001b[4C\u001b[1mg\u001b(B\u001b[m.\u001b[8C\u001b[2m3\u001b[27;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;9H\u001b[2me\u001b(B\u001b[mc\u001b[4C\u001b[1m.\u001b(B\u001b[m.\u001b[29C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;10H\u001b[2mc\u001b(B\u001b[mt\u001b[4C\u001b[1m.\u001b(B\u001b[m.\u001b[28C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":25,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;11H\u001b[2mt\u001b(B\u001b[mi\u001b[4C\u001b[1m.\u001b(B\u001b[m \u001b[2m4\u001b(B\u001b[m\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;27H\u001b[2m, url: ws\u001b[C:/\u001b[Capi.openai.com/v1/responses, cf-ray: a2831d0298dca625-ZRH,\u001b(B\u001b[m\u001b[C\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[24;98H\u001b[2m…\u001b[27;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;12H\u001b[2mi\u001b(B\u001b[mn\u001b[4C\u001b[1m \u001b(B\u001b[m4\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;13H\u001b[2mn\u001b(B\u001b[mg\u001b[4C\u001b[1m4\u001b(B\u001b[m/\u001b[25C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;14H\u001b[2mg\u001b(B\u001b[m.\u001b[4C\u001b[1m/\u001b(B\u001b[m5\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;15H\u001b[2m.\u001b(B\u001b[m.\u001b[4C\u001b[1m5\u001b(B\u001b[m\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;16H\u001b[2m.\u001b(B\u001b[m.\u001b[28C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;17H\u001b[2m.\u001b(B\u001b[m \u001b[27C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;18H\u001b[2m \u001b(B\u001b[m4\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;19H\u001b[2m4\u001b(B\u001b[m/\u001b[25C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;20H\u001b[2m/\u001b(B\u001b[m5\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H•\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;21H\u001b[2m5\u001b(B\u001b[m\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":1,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;24H\u001b[2m4\u001b(B\u001b[m\u001b[21C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H\u001b[2m◦\u001b[27;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[6AR\u001b[42C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[6A\u001b[1mR\u001b(B\u001b[me\u001b[41C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;4H\u001b[1me\u001b(B\u001b[mc\u001b[40C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;5H\u001b[1mc\u001b(B\u001b[mo\u001b[39C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;6H\u001b[1mo\u001b(B\u001b[mn\u001b[38C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;7H\u001b[1mn\u001b(B\u001b[mn\u001b[37C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[6AR\u001b[4C\u001b[1mn\u001b(B\u001b[me\u001b[36C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[6A\u001b[2mR\u001b(B\u001b[me\u001b[4C\u001b[1me\u001b(B\u001b[mc\u001b[35C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H•\u001b[2C\u001b[2me\u001b(B\u001b[mc\u001b[4C\u001b[1mc\u001b(B\u001b[mt\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;5H\u001b[2mc\u001b(B\u001b[mo\u001b[4C\u001b[1mt\u001b(B\u001b[mi\u001b[33C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
@@ -0,0 +1,154 @@
|
||||
{"scenario":"streaming-real","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T09:31:57.351Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":1,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":35,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b(B\u001b[m$ "}
|
||||
{"delayMs":647,"data":"exec codex\r\n"}
|
||||
{"delayMs":479,"data":"\u001b[30d\n\u001b[K\u001b[2d\u001b[J\u001b[H\u001b[K\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":2,"data":">\u001b[C\u001b[1mYou are in \u001b(B\u001b[m/home/arkon/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[3;3H\u001b[33mNote: You’re in a subdirectory of a Git project. Trusting will apply to the repository root:\u001b[4;3H/home/arkon/default/claudeman\u001b[6;3H\u001b[39mDo\u001b[Cyou\u001b[Ctrust\u001b[Cthe\u001b[Ccontents\u001b[Cof\u001b[Cthis\u001b[Cdirectory?\u001b[CWorking\u001b[Cwith\u001b[Cuntrusted\u001b[Ccontents\u001b[Ccomes\u001b[Cwith\u001b[Chigher\u001b[7;3Hrisk\u001b[Cof\u001b[Cprompt\u001b[Cinjection.\u001b[CTrusting\u001b[Cthe\u001b[Cdirectory\u001b[Callows\u001b[Cproject-local\u001b[Cconfig,\u001b[Chooks,\u001b[Cand\u001b[Cexec\u001b[8;3Hpolicies\u001b[Cto\u001b[Cload.\u001b[10;1H\u001b[36m› 1. Yes, continue\u001b[11;3H\u001b[39m2.\u001b[CNo,\u001b[Cquit\u001b[13;3H\u001b[2mPress enter to continue\u001b[?25l\u001b(B\u001b[m"}
|
||||
{"delayMs":3830,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[13;26H\u001b[?25l"}
|
||||
{"delayMs":1,"data":"\u001b[H>\u001b[1X\u001b[1m\u001b[CYou are in \u001b(B\u001b[m/home/arkon/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[K\r\n\u001b[K\u001b[3;2H\u001b[1K\u001b[33m\u001b[CNote: You’re in a subdirectory of a Git project. Trusting will apply to the repository root:\u001b[39m\u001b[K\u001b[4;2H\u001b[1K\u001b[33m\u001b[C/home/arkon/default/claudeman\u001b[39m\u001b[K\r\n\u001b[K\u001b[6;2H\u001b[1K\u001b[CDo\u001b[1X\u001b[Cyou\u001b[1X\u001b[Ctrust\u001b[1X\u001b[Cthe\u001b[1X\u001b[Ccontents\u001b[1X\u001b[Cof\u001b[1X\u001b[Cthis\u001b[1X\u001b[Cdirectory?\u001b[1X\u001b[CWorking\u001b[1X\u001b[Cwith\u001b[1X\u001b[Cuntrusted\u001b[1X\u001b[Ccontents\u001b[1X\u001b[Ccomes\u001b[1X\u001b[Cwith\u001b[1X\u001b[Chigher\u001b[K\u001b[7;2H\u001b[1K\u001b[Crisk\u001b[1X\u001b[Cof\u001b[1X\u001b[Cprompt\u001b[1X\u001b[Cinjection.\u001b[1X\u001b[CTrusting\u001b[1X\u001b[Cthe\u001b[1X\u001b[Cdirectory\u001b[1X\u001b[Callows\u001b[1X\u001b[Cproject-local\u001b[1X\u001b[Cconfig,\u001b[1X\u001b[Chooks,\u001b[1X\u001b[Cand\u001b[1X\u001b[Cexec\u001b[K\u001b[8;2H\u001b[1K\u001b[Cpolicies\u001b[1X\u001b[Cto\u001b[1X\u001b[Cload.\u001b[K\r\n\u001b[K\u001b[36m\r\n› 1. Yes, continue\u001b[39m\u001b[K\u001b[11;2H\u001b[1K\u001b[C2.\u001b[1X\u001b[CNo,\u001b[1X\u001b[Cquit\u001b[K\r\n\u001b[K\u001b[13;2H\u001b[1K\u001b[2m\u001b[CPress enter to continue\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[13;26H"}
|
||||
{"keyAt":true,"data":"\r"}
|
||||
{"delayMs":235,"data":"\u001b[2;1H\u001b[J\u001b[H\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001bM\u001bM\u001bM\r\n\u001b[33m⚠\u001b[39m\u001b[1;3r\u001b[3;1H\n\u001b[1;2H\u001b[33m Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\r\n\u001b[K\u001b[1;30r\u001b[3;1H"}
|
||||
{"delayMs":1,"data":" \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[5;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-SFpno1\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills t\u001b(B\u001b[m\u001b[2mo list available skills\u001b[15;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-terra default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[13;3H\u001b[?12l\u001b[?25h\u001b(B\u001b[m"}
|
||||
{"delayMs":10,"data":"\u001b[5;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[13;39H\u001b[K\u001b[15;82H\u001b[K\u001b[13;3H"}
|
||||
{"delayMs":12,"data":"\u001b[5;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[13;39H\u001b[K\u001b[15;82H\u001b[K\u001b[13;3H"}
|
||||
{"delayMs":208,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[5;1H\u001b[J\u001b[A\u001b[K\u001b[4;30r\u001b[4;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001b[1;30r\u001b[5;1H"}
|
||||
{"delayMs":0,"data":"\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ model: \u001b(B\u001b[mgpt-5.6-terra\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-SFpno1\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[12;1H\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":" \u001b[1mTip:\u001b(B\u001b[m \u001b[3mNew\u001b(B\u001b[m For a limited time, Codex is included in your plan for free – let’s build together.\u001b[14;1H•\u001b[C\u001b[2mBooting MCP server: codex_apps\u001b[C(0s • esc to interrupt)\u001b[17;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[19;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-terra default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[17;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":19,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":1,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":33,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":1,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":33,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":33,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":28,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":34,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[3AB\u001b[53C\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":2,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":5,"data":"\u001b[14;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":1,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[17;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-terra default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[15;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":279,"data":"\u001b[36C\u001b[K\u001b[17;82H\u001b[K\u001b[15;3H"}
|
||||
{"delayMs":86,"data":"\u001b[36C\u001b[K\u001b[17;82H\u001b[K\u001b[15;3H"}
|
||||
{"delayMs":71,"data":"\u001b[36C\u001b[K\u001b[17;82H\u001b[K\u001b[15;3H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"keyAt":true,"data":"p"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"keyAt":true,"data":"y"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"w"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"keyAt":true,"data":"g"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"w"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"keyAt":true,"data":"d"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":2010,"data":"reply with the single word hello\u001b[K\u001b[17;82H\u001b[K\u001b[15;35H"}
|
||||
{"keyAt":true,"data":"\r"}
|
||||
{"delayMs":382,"data":"\u001b[13;30r\u001b[13;1H\u001bM\u001bM\u001bM\u001bM\u001b[1;30r\u001b[15;1H"}
|
||||
{"delayMs":0,"data":"\u001b[1m\u001b[2m› \u001b(B\u001b[mreply with the single word hello\r\n"}
|
||||
{"delayMs":0,"data":"\u001b[19;3H\u001b[2mUse /skills to list available skills\u001b(B\u001b[m\u001b[K\u001b[21;82H\u001b[K\u001b[19;3H"}
|
||||
{"delayMs":0,"data":"\u001b[36C\u001b[K\u001b[21;82H\u001b[K\u001b[19;3H"}
|
||||
{"delayMs":20,"data":"\u001b[36C\u001b[K\u001b[21;82H\u001b[K\u001b[19;3H"}
|
||||
{"delayMs":8,"data":"\u001b[36C\u001b[K\u001b[21;82H\u001b[K\u001b[19;3H"}
|
||||
{"delayMs":78,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\r\n•\u001b[C\u001b[2mWor\u001b(B\u001b[mk\u001b[1ming\u001b[C\u001b(B\u001b[m\u001b[2m(0s • esc to interrupt)\u001b[21;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[23;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-terra default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[21;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;6H\u001b[2mk\u001b(B\u001b[mi\u001b[26C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":35,"data":"\u001b[18;7H\u001b[2mi\u001b(B\u001b[mn\u001b[25C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;8H\u001b[2mn\u001b(B\u001b[mg\u001b[24C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;9H\u001b[2mg\u001b(B\u001b[m\u001b[24C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[18;1H\u001b[2m◦\u001b[21;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":32,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;12H\u001b[2m1\u001b(B\u001b[m\u001b[21C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":32,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[18;1H•\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[3AW\u001b[30C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[3A\u001b[1mW\u001b(B\u001b[mo\u001b[29C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;4H\u001b[1mo\u001b(B\u001b[mr\u001b[28C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;5H\u001b[1mr\u001b(B\u001b[mk\u001b[27C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;6H\u001b[1mk\u001b(B\u001b[mi\u001b[26C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":32,"data":"\u001b[18;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[17;30r\u001b[17;1H\u001bM\u001bM\u001b[1;30r\u001b[18;1H"}
|
||||
{"delayMs":0,"data":"\u001b[2m• \u001b(B\u001b[mhello\u001b[21;1H\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[23;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-terra default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[21;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":25,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[36C\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":6,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":3,"data":"\u001b[36C\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"delayMs":2252,"data":"a\u001b[K\u001b[23;82H\u001b[K\u001b[21;4H"}
|
||||
{"keyAt":true,"data":"b"}
|
||||
{"delayMs":121,"data":"b\u001b[K\u001b[23;82H\u001b[K\u001b[21;5H"}
|
||||
{"keyAt":true,"data":"c"}
|
||||
{"delayMs":121,"data":"c\u001b[K\u001b[23;82H\u001b[K\u001b[21;6H"}
|
||||
@@ -0,0 +1,28 @@
|
||||
{"scenario":"trust-modal","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:51:20.960Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":0,"data":"\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":28,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-X4gHpE\u001b(B\u001b[m$ "}
|
||||
{"delayMs":654,"data":"exec codex\r\n"}
|
||||
{"delayMs":486,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":168,"data":"\u001b[30d\n\u001b[K\u001b[2d\u001b[J\u001b[H\u001b[K"}
|
||||
{"delayMs":2,"data":">\u001b[C\u001b[1mYou are in \u001b(B\u001b[m/home/arkon/default/claudeman-predictive/tmp/codexrec-work-X4gHpE\u001b[3;3H\u001b[33mNote: You’re in a subdirectory of a Git project. Trusting will apply to the repository root:\u001b[4;3H/home/arkon/default/claudeman\u001b[6;3H\u001b[39mDo\u001b[Cyou\u001b[Ctrust\u001b[Cthe\u001b[Ccontents\u001b[Cof\u001b[Cthis\u001b[Cdirectory?\u001b[CWorking\u001b[Cwith\u001b[Cuntrusted\u001b[Ccontents\u001b[Ccomes\u001b[Cwith\u001b[Chigher\u001b[7;3Hrisk\u001b[Cof\u001b[Cprompt\u001b[Cinjection.\u001b[CTrusting\u001b[Cthe\u001b[Cdirectory\u001b[Callows\u001b[Cproject-local\u001b[Cconfig,\u001b[Chooks,\u001b[Cand\u001b[Cexec\u001b[8;3Hpolicies\u001b[Cto\u001b[Cload.\u001b[10;1H\u001b[36m› 1. Yes, continue\u001b[11;3H\u001b[39m2.\u001b[CNo,\u001b[Cquit\u001b[13;3H\u001b[2mPress enter to continue\u001b[?25l\u001b(B\u001b[m"}
|
||||
{"delayMs":3659,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[13;26H\u001b[?25l"}
|
||||
{"delayMs":0,"data":"\u001b[H>\u001b[1X\u001b[1m\u001b[CYou are in \u001b(B\u001b[m/home/arkon/default/claudeman-predictive/tmp/codexrec-work-X4gHpE\u001b[K\r\n\u001b[K\u001b[3;2H\u001b[1K\u001b[33m\u001b[CNote: You’re in a subdirectory of a Git project. Trusting will apply to the repository root:\u001b[39m\u001b[K\u001b[4;2H\u001b[1K\u001b[33m\u001b[C/home/arkon/default/claudeman\u001b[39m\u001b[K\r\n\u001b[K\u001b[6;2H\u001b[1K\u001b[CDo\u001b[1X\u001b[Cyou\u001b[1X\u001b[Ctrust\u001b[1X\u001b[Cthe\u001b[1X\u001b[Ccontents\u001b[1X\u001b[Cof\u001b[1X\u001b[Cthis\u001b[1X\u001b[Cdirectory?\u001b[1X\u001b[CWorking\u001b[1X\u001b[Cwith\u001b[1X\u001b[Cuntrusted\u001b[1X\u001b[Ccontents\u001b[1X\u001b[Ccomes\u001b[1X\u001b[Cwith\u001b[1X\u001b[Chigher\u001b[K\u001b[7;2H\u001b[1K\u001b[Crisk\u001b[1X\u001b[Cof\u001b[1X\u001b[Cprompt\u001b[1X\u001b[Cinjection.\u001b[1X\u001b[CTrusting\u001b[1X\u001b[Cthe\u001b[1X\u001b[Cdirectory\u001b[1X\u001b[Callows\u001b[1X\u001b[Cproject-local\u001b[1X\u001b[Cconfig,\u001b[1X\u001b[Chooks,\u001b[1X\u001b[Cand\u001b[1X\u001b[Cexec\u001b[K\u001b[8;2H\u001b[1K\u001b[Cpolicies\u001b[1X\u001b[Cto\u001b[1X\u001b[Cload.\u001b[K\r\n\u001b[K\u001b[36m\r\n› 1. Yes, continue\u001b[39m\u001b[K\u001b[11;2H\u001b[1K\u001b[C2.\u001b[1X\u001b[CNo,\u001b[1X\u001b[Cquit\u001b[K\r\n\u001b[K\u001b[13;2H\u001b[1K\u001b[2m\u001b[CPress enter to continue\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[13;26H"}
|
||||
{"keyAt":true,"data":"x"}
|
||||
{"delayMs":188,"data":"\u001b[1;79H\u001b[K\u001b[3;95H\u001b[K\u001b[4;32H\u001b[K\u001b[6;97H\u001b[K\u001b[7;96H\u001b[K\u001b[8;20H\u001b[K\u001b[10;19H\u001b[K\u001b[11;14H\u001b[K\u001b[13;26H\u001b[K\u001b[30;2H"}
|
||||
{"keyAt":true,"data":"\r"}
|
||||
{"delayMs":849,"data":"\u001b[2;1H\u001b[J\u001b[H\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001bM\u001bM\u001bM\r\n\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b(B\u001b[m\u001b[1;3r\u001b[3;1H\n\u001b[A \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\r\n\u001b[K\u001b[1;30r\u001b[3;1H"}
|
||||
{"delayMs":2,"data":" \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[5;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-X4gHpE\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mImprove docum\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2mentation in @filename\u001b[15;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-X4gHpE\u001b[13;3H\u001b[?12l\u001b[?25h\u001b(B\u001b[m"}
|
||||
{"delayMs":7,"data":"\u001b[5;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[13;37H\u001b[K\u001b[15;80H\u001b[K\u001b[13;3H"}
|
||||
{"delayMs":13,"data":"\u001b[5;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[13;37H\u001b[K\u001b[15;80H\u001b[K\u001b[13;3H"}
|
||||
{"delayMs":165,"data":"\u001b[5;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mImprove documentation in @filename\u001b[8;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-X4gHpE\u001b[6;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":1,"data":"\u001b[34C\u001b[K\u001b[8;80H\u001b[K\u001b[6;3H"}
|
||||
{"delayMs":24,"data":"\u001b[4;30r\u001b[4;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\r\n\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b[1;30r\u001b[7;1H\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ │\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-X4gHpE\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[12;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[17;37H\u001b[K\u001b[19;80H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":1,"data":"\u001b[34C\u001b[K\u001b[19;80H\u001b[K\u001b[17;3H"}
|
||||
@@ -0,0 +1,33 @@
|
||||
{"scenario":"type-hello","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:50:33.854Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":1,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":32,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-tXbGez\u001b(B\u001b[m$ "}
|
||||
{"delayMs":651,"data":"exec codex\r\n"}
|
||||
{"delayMs":403,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":189,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":7,"data":"\r\n\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":4,"data":"\u001b[2;30r\u001b[2;1H\u001bM\u001bM\u001bM\u001b[1;30r\u001b[2;1H\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\r\n\u001b[39m \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[6;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-tXbGez\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[14;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mImprove documentation in @filename\u001b[16;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-tXbGez\u001b[14;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":6,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;37H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":25,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;37H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":8,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;37H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":185,"data":"\u001b[6;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[5;30r\u001b[5;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001b[1;30r\u001b[5;1H"}
|
||||
{"delayMs":0,"data":"\r\n\u001b[2m╭─────────────────────────────────────────────────╮\r\n\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-tXbGez\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n"}
|
||||
{"delayMs":0,"data":" produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[18;1H\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mImprove documentation in @filename\u001b[20;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-tXbGez\u001b[18;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[34C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[34C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":3484,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[32m\u001b[1m\u001b[Harkon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-tXbGez\u001b(B\u001b[m$ exec codex\u001b[K\u001b[33m\r\n⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\u001b[39m\u001b[K\r\n \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\u001b[K\r\n \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[39m\u001b[K\r\n\u001b[K\u001b[2m\r\n╭─────────────────────────────────────────────────╮\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-tXbGez\u001b[2m │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n╰─────────────────────────────────────────────────╯\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\u001b[K\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\u001b[K\r\n reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[1m\r\n›\u001b(B\u001b[m\u001b[1X\u001b[2m\u001b[CImprove documentation in @filename\u001b(B\u001b[m\u001b[K\r\n\u001b[K\u001b[20;2H\u001b[1K\u001b[38;5;223m\u001b[Cgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-tXbGez\u001b[39m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[18;3H"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":208,"data":"h\u001b[K\u001b[20;80H\u001b[K\u001b[18;4H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":93,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;5H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":89,"data":"l\u001b[K\u001b[20;80H\u001b[K\u001b[18;6H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":92,"data":"l\u001b[K\u001b[20;80H\u001b[K\u001b[18;7H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":90,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;8H"}
|
||||
@@ -0,0 +1,250 @@
|
||||
{"scenario":"wrap","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:50:52.462Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":0,"data":"\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":32,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b(B\u001b[m$ "}
|
||||
{"delayMs":650,"data":"exec codex\r\n"}
|
||||
{"delayMs":437,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":181,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":4,"data":"\r\n\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2;30r\u001b[2;1H\u001bM\u001bM\u001bM\u001b[1;30r\u001b[2;1H"}
|
||||
{"delayMs":0,"data":"\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\r\n\u001b(B\u001b[m"}
|
||||
{"delayMs":1,"data":" \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[6;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-VGU83J\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[14;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mWrite tests for @filename\u001b[16;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b[14;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":6,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;28H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":19,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;28H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":6,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;28H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":160,"data":"\u001b[6;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[5;30r\u001b[5;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001b[1;30r\u001b[5;1H"}
|
||||
{"delayMs":0,"data":"\r\n\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-VGU83J\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[18;1H\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mWrite tests for @filename\u001b[20;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b[18;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":18,"data":"\u001b[25C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":1,"data":"\u001b[25C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":3478,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[32m\u001b[1m\u001b[Harkon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b(B\u001b[m$ exec codex\u001b[K\u001b[33m\r\n⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\u001b[39m\u001b[K\r\n \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\u001b[K\r\n \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[39m\u001b[K\r\n\u001b[K\u001b[2m\r\n╭─────────────────────────────────────────────────╮\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-VGU83J\u001b[2m │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n╰─────────────────────────────────────────────────╯\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\u001b[K\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\u001b[K\r\n reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[1m\r\n›\u001b(B\u001b[m\u001b[1X\u001b[2m\u001b[CWrite tests for @filename\u001b(B\u001b[m\u001b[K\r\n\u001b[K\u001b[20;2H\u001b[1K\u001b[38;5;223m\u001b[Cgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b[39m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[18;3H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"delayMs":232,"data":"t\u001b[K\u001b[20;80H\u001b[K\u001b[18;4H"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":27,"data":"h\u001b[K\u001b[20;80H\u001b[K\u001b[18;5H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":27,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;6H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;7H"}
|
||||
{"keyAt":true,"data":"q"}
|
||||
{"delayMs":26,"data":"q\u001b[K\u001b[20;80H\u001b[K\u001b[18;8H"}
|
||||
{"keyAt":true,"data":"u"}
|
||||
{"delayMs":17,"data":"u\u001b[K\u001b[20;80H\u001b[K\u001b[18;9H"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"delayMs":28,"data":"i\u001b[K\u001b[20;80H\u001b[K\u001b[18;10H"}
|
||||
{"keyAt":true,"data":"c"}
|
||||
{"delayMs":27,"data":"c\u001b[K\u001b[20;80H\u001b[K\u001b[18;11H"}
|
||||
{"keyAt":true,"data":"k"}
|
||||
{"delayMs":27,"data":"k\u001b[K\u001b[20;80H\u001b[K\u001b[18;12H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"b"}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;13H"}
|
||||
{"delayMs":16,"data":"b\u001b[K\u001b[20;80H\u001b[K\u001b[18;14H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"delayMs":30,"data":"r\u001b[K\u001b[20;80H\u001b[K\u001b[18;15H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":27,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;16H"}
|
||||
{"keyAt":true,"data":"w"}
|
||||
{"delayMs":27,"data":"w\u001b[K\u001b[20;80H\u001b[K\u001b[18;17H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":27,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;18H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"f"}
|
||||
{"delayMs":45,"data":"\u001b[Cf\u001b[K\u001b[20;80H\u001b[K\u001b[18;20H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":27,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;21H"}
|
||||
{"keyAt":true,"data":"x"}
|
||||
{"delayMs":26,"data":"x\u001b[K\u001b[20;80H\u001b[K\u001b[18;22H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;23H"}
|
||||
{"keyAt":true,"data":"j"}
|
||||
{"keyAt":true,"data":"u"}
|
||||
{"delayMs":27,"data":"j\u001b[K\u001b[20;80H\u001b[K\u001b[18;24H"}
|
||||
{"keyAt":true,"data":"m"}
|
||||
{"delayMs":46,"data":"um\u001b[K\u001b[20;80H\u001b[K\u001b[18;26H"}
|
||||
{"keyAt":true,"data":"p"}
|
||||
{"delayMs":26,"data":"p\u001b[K\u001b[20;80H\u001b[K\u001b[18;27H"}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"delayMs":28,"data":"s\u001b[K\u001b[20;80H\u001b[K\u001b[18;28H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;29H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":16,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;30H"}
|
||||
{"keyAt":true,"data":"v"}
|
||||
{"delayMs":31,"data":"v\u001b[K\u001b[20;80H\u001b[K\u001b[18;31H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":26,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;32H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"delayMs":28,"data":"r\u001b[K\u001b[20;80H\u001b[K\u001b[18;33H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":26,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;34H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":45,"data":"th\u001b[K\u001b[20;80H\u001b[K\u001b[18;36H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":27,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;37H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;38H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":27,"data":"l\u001b[K\u001b[20;80H\u001b[K\u001b[18;39H"}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"keyAt":true,"data":"z"}
|
||||
{"delayMs":28,"data":"a\u001b[K\u001b[20;80H\u001b[K\u001b[18;40H"}
|
||||
{"keyAt":true,"data":"y"}
|
||||
{"delayMs":26,"data":"z\u001b[K\u001b[20;80H\u001b[K\u001b[18;41H"}
|
||||
{"delayMs":17,"data":"y\u001b[K\u001b[20;80H\u001b[K\u001b[18;42H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":29,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;43H"}
|
||||
{"keyAt":true,"data":"d"}
|
||||
{"delayMs":27,"data":"d\u001b[K\u001b[20;80H\u001b[K\u001b[18;44H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":27,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;45H"}
|
||||
{"keyAt":true,"data":"g"}
|
||||
{"delayMs":27,"data":"g\u001b[K\u001b[20;80H\u001b[K\u001b[18;46H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"delayMs":45,"data":"\u001b[Ca\u001b[K\u001b[20;80H\u001b[K\u001b[18;48H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":26,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;49H"}
|
||||
{"keyAt":true,"data":"d"}
|
||||
{"delayMs":28,"data":"d\u001b[K\u001b[20;80H\u001b[K\u001b[18;50H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":26,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;51H"}
|
||||
{"keyAt":true,"data":"k"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":27,"data":"k\u001b[20;80H\u001b[K\u001b[18;52H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":26,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;53H"}
|
||||
{"delayMs":17,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;54H"}
|
||||
{"keyAt":true,"data":"p"}
|
||||
{"delayMs":28,"data":"p\u001b[K\u001b[20;80H\u001b[K\u001b[18;55H"}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"delayMs":27,"data":"s\u001b[K\u001b[20;80H\u001b[K\u001b[18;56H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;57H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"keyAt":true,"data":"u"}
|
||||
{"delayMs":27,"data":"r\u001b[K\u001b[20;80H\u001b[K\u001b[18;58H"}
|
||||
{"delayMs":17,"data":"u\u001b[K\u001b[20;80H\u001b[K\u001b[18;59H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":29,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;60H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":26,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;61H"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"delayMs":28,"data":"i\u001b[K\u001b[20;80H\u001b[K\u001b[18;62H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":26,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;63H"}
|
||||
{"keyAt":true,"data":"g"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":45,"data":"g\u001b[K\u001b[20;80H\u001b[K\u001b[18;65H"}
|
||||
{"keyAt":true,"data":"u"}
|
||||
{"delayMs":28,"data":"u\u001b[K\u001b[20;80H\u001b[K\u001b[18;66H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":27,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;67H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"delayMs":27,"data":"t\u001b[K\u001b[20;80H\u001b[K\u001b[18;68H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":45,"data":"il\u001b[K\u001b[20;80H\u001b[K\u001b[18;70H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":28,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;71H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"delayMs":26,"data":"t\u001b[K\u001b[20;80H\u001b[K\u001b[18;72H"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":28,"data":"h\u001b[K\u001b[20;80H\u001b[K\u001b[18;73H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":45,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;75H"}
|
||||
{"keyAt":true,"data":"c"}
|
||||
{"delayMs":27,"data":"c\u001b[K\u001b[20;80H\u001b[K\u001b[18;76H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":28,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;77H"}
|
||||
{"keyAt":true,"data":"m"}
|
||||
{"keyAt":true,"data":"p"}
|
||||
{"delayMs":27,"data":"m\u001b[K\u001b[20;80H\u001b[K\u001b[18;78H"}
|
||||
{"delayMs":16,"data":"p\u001b[K\u001b[20;80H\u001b[K\u001b[18;79H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":30,"data":"o\u001b[K\u001b[2B\u001b[K\u001b[2A"}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"delayMs":27,"data":"s\u001b[K\u001b[20;80H\u001b[K\u001b[18;81H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":27,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;82H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"delayMs":27,"data":"r\u001b[K\u001b[20;80H\u001b[K\u001b[18;83H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":16,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;84H"}
|
||||
{"keyAt":true,"data":"b"}
|
||||
{"delayMs":30,"data":"b\u001b[K\u001b[20;80H\u001b[K\u001b[18;85H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":27,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;86H"}
|
||||
{"keyAt":true,"data":"x"}
|
||||
{"delayMs":27,"data":"x\u001b[K\u001b[20;80H\u001b[K\u001b[18;87H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":26,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;88H"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":17,"data":"h\u001b[K\u001b[20;80H\u001b[K\u001b[18;89H"}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"delayMs":28,"data":"a\u001b[K\u001b[20;80H\u001b[K\u001b[18;90H"}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"delayMs":27,"data":"s\u001b[K\u001b[20;80H\u001b[K\u001b[18;91H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":28,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;92H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"delayMs":26,"data":"t\u001b[K\u001b[20;80H\u001b[K\u001b[18;93H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;94H"}
|
||||
{"delayMs":16,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;95H"}
|
||||
{"keyAt":true,"data":"w"}
|
||||
{"delayMs":30,"data":"w\u001b[K\u001b[20;80H\u001b[K\u001b[18;96H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"delayMs":27,"data":"r\u001b[K\u001b[20;80H\u001b[K\u001b[18;97H"}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"delayMs":26,"data":"a\u001b[K\u001b[20;80H\u001b[K\u001b[18;98H"}
|
||||
{"keyAt":true,"data":"p"}
|
||||
{"delayMs":27,"data":"p\u001b[K\u001b[20;80H\u001b[K\u001b[18;99H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[17;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"delayMs":1,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[mthe\u001b[Cquick\u001b[Cbrown\u001b[Cfox\u001b[Cjumps\u001b[Cover\u001b[Cthe\u001b[Clazy\u001b[Cdog\u001b[Cand\u001b[Ckeeps\u001b[Crunning\u001b[Cuntil\u001b[Cthe\u001b[Ccomposer\u001b[Cbox\u001b[Chas\u001b[Cto\u001b[Cwrap\u001b[21;3H\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b[19;3H\u001b(B\u001b[m"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":42,"data":"\u001b[18;99H\u001b[K\u001b[19;3Hth\u001b[21;80H\u001b[K\u001b[19;5H"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"delayMs":29,"data":"\u001b[18;99H\u001b[K\u001b[19;5Hi\u001b[K\u001b[21;80H\u001b[K\u001b[19;6H"}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"delayMs":27,"data":"\u001b[18;99H\u001b[K\u001b[19;6Hs\u001b[K\u001b[21;80H\u001b[K\u001b[19;7H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[18;99H\u001b[K\u001b[19;7H\u001b[K\u001b[21;80H\u001b[K\u001b[19;8H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":26,"data":"\u001b[18;99H\u001b[K\u001b[19;8Hl\u001b[K\u001b[21;80H\u001b[K\u001b[19;9H"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":45,"data":"\u001b[18;99H\u001b[K\u001b[19;9Hin\u001b[K\u001b[21;80H\u001b[K\u001b[19;11H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":27,"data":"\u001b[18;99H\u001b[K\u001b[19;11He\u001b[K\u001b[21;80H\u001b[K\u001b[19;12H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[18;99H\u001b[K\u001b[19;12H\u001b[K\u001b[21;80H\u001b[K\u001b[19;13H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"delayMs":27,"data":"\u001b[18;99H\u001b[K\u001b[19;13Ht\u001b[K\u001b[21;80H\u001b[K\u001b[19;14H"}
|
||||
{"keyAt":true,"data":"w"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"delayMs":45,"data":"\u001b[18;99H\u001b[K\u001b[19;14Hwi\u001b[K\u001b[21;80H\u001b[K\u001b[19;16H"}
|
||||
{"keyAt":true,"data":"c"}
|
||||
{"delayMs":28,"data":"\u001b[18;99H\u001b[K\u001b[19;16Hc\u001b[K\u001b[21;80H\u001b[K\u001b[19;17H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":26,"data":"\u001b[18;99H\u001b[K\u001b[19;17He\u001b[K\u001b[21;80H\u001b[K\u001b[19;18H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":28,"data":"\u001b[18;99H\u001b[K\u001b[19;18H\u001b[K\u001b[21;80H\u001b[K\u001b[19;19H"}
|
||||
{"keyAt":true,"data":"v"}
|
||||
{"delayMs":26,"data":"\u001b[18;99H\u001b[K\u001b[19;19Ho\u001b[K\u001b[21;80H\u001b[K\u001b[19;20H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":26,"data":"\u001b[18;99H\u001b[K\u001b[19;20Hv\u001b[K\u001b[21;80H\u001b[K\u001b[19;21H"}
|
||||
{"delayMs":17,"data":"\u001b[18;99H\u001b[K\u001b[19;21He\u001b[K\u001b[21;80H\u001b[K\u001b[19;22H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"delayMs":29,"data":"\u001b[18;99H\u001b[K\u001b[19;22Hr\u001b[K\u001b[21;80H\u001b[K\u001b[19;23H"}
|
||||
@@ -3,129 +3,204 @@
|
||||
*
|
||||
* Creates a minimal Terminal-like object that satisfies the addon's
|
||||
* requirements without needing a real xterm.js instance or DOM renderer.
|
||||
*
|
||||
* PredictiveEchoAddon additions (all ADDITIVE, existing tests unchanged):
|
||||
* mutable cursor via setCursor(), wide-char-aware getCell() on mock lines,
|
||||
* onWriteParsed/onResize emitters with fire* triggers, and opt-outs for
|
||||
* getCell support and the emitters (getCellSupport / emitters options).
|
||||
*/
|
||||
import { charCellWidth } from '../src/overlay-renderer.js';
|
||||
|
||||
interface MockLine {
|
||||
translateToString(_trimRight?: boolean): string;
|
||||
translateToString(_trimRight?: boolean): string;
|
||||
getCell?(x: number): { getChars(): string; getWidth(): number } | undefined;
|
||||
}
|
||||
|
||||
interface MockBufferOptions {
|
||||
lines: string[];
|
||||
viewportY?: number;
|
||||
baseY?: number;
|
||||
cursorX?: number;
|
||||
cursorY?: number;
|
||||
lines: string[];
|
||||
viewportY?: number;
|
||||
baseY?: number;
|
||||
cursorX?: number;
|
||||
cursorY?: number;
|
||||
}
|
||||
|
||||
interface MockTerminalOptions {
|
||||
buffer?: MockBufferOptions;
|
||||
cols?: number;
|
||||
rows?: number;
|
||||
fontFamily?: string;
|
||||
fontSize?: number;
|
||||
fontWeight?: string | number;
|
||||
theme?: {
|
||||
background?: string;
|
||||
foreground?: string;
|
||||
cursor?: string;
|
||||
};
|
||||
cellWidth?: number;
|
||||
cellHeight?: number;
|
||||
/** Device-pixel char top offset (for charTop calculation). Default: 0 */
|
||||
deviceCharTop?: number;
|
||||
/** Device-pixel char height (for charHeight calculation). Default: cellHeight * dpr */
|
||||
deviceCharHeight?: number;
|
||||
buffer?: MockBufferOptions;
|
||||
cols?: number;
|
||||
rows?: number;
|
||||
fontFamily?: string;
|
||||
fontSize?: number;
|
||||
fontWeight?: string | number;
|
||||
theme?: {
|
||||
background?: string;
|
||||
foreground?: string;
|
||||
cursor?: string;
|
||||
};
|
||||
cellWidth?: number;
|
||||
cellHeight?: number;
|
||||
/** Device-pixel char top offset (for charTop calculation). Default: 0 */
|
||||
deviceCharTop?: number;
|
||||
/** Device-pixel char height (for charHeight calculation). Default: cellHeight * dpr */
|
||||
deviceCharHeight?: number;
|
||||
/** Provide getCell() on mock lines (PredictiveEchoAddon). Default: true */
|
||||
getCellSupport?: boolean;
|
||||
/** Provide onWriteParsed/onResize emitters (PredictiveEchoAddon). Default: true */
|
||||
emitters?: boolean;
|
||||
}
|
||||
|
||||
/** Column-indexed cell access over a plain string, wide-char aware. */
|
||||
function cellAt(text: string, col: number): { getChars(): string; getWidth(): number } {
|
||||
let c = 0;
|
||||
for (const ch of text) {
|
||||
const w = charCellWidth(null, ch);
|
||||
if (col === c) return { getChars: () => ch, getWidth: () => w };
|
||||
if (w === 2 && col === c + 1) return { getChars: () => '', getWidth: () => 0 };
|
||||
c += w;
|
||||
}
|
||||
return { getChars: () => '', getWidth: () => 1 };
|
||||
}
|
||||
|
||||
export function createMockTerminal(opts: MockTerminalOptions = {}) {
|
||||
const bufOpts = opts.buffer ?? { lines: ['$ '] };
|
||||
const lines = bufOpts.lines;
|
||||
const viewportY = bufOpts.viewportY ?? 0;
|
||||
const baseY = bufOpts.baseY ?? viewportY;
|
||||
const cols = opts.cols ?? 80;
|
||||
const rows = opts.rows ?? Math.max(lines.length, 24);
|
||||
const cellW = opts.cellWidth ?? 8.4;
|
||||
const cellH = opts.cellHeight ?? 17;
|
||||
const bufOpts = opts.buffer ?? { lines: ['$ '] };
|
||||
const viewportY = bufOpts.viewportY ?? 0;
|
||||
const baseY = bufOpts.baseY ?? viewportY;
|
||||
const cols = opts.cols ?? 80;
|
||||
const rows = opts.rows ?? Math.max(bufOpts.lines.length, 24);
|
||||
const cellW = opts.cellWidth ?? 8.4;
|
||||
const cellH = opts.cellHeight ?? 17;
|
||||
const getCellSupport = opts.getCellSupport ?? true;
|
||||
const emitters = opts.emitters ?? true;
|
||||
|
||||
const mockLines: MockLine[] = lines.map((text) => ({
|
||||
translateToString: () => text,
|
||||
}));
|
||||
|
||||
// Create minimal DOM structure
|
||||
const element = document.createElement('div');
|
||||
element.className = 'terminal xterm';
|
||||
|
||||
const viewport = document.createElement('div');
|
||||
viewport.className = 'xterm-viewport';
|
||||
|
||||
const screen = document.createElement('div');
|
||||
screen.className = 'xterm-screen';
|
||||
screen.style.position = 'relative';
|
||||
|
||||
const xtermRows = document.createElement('div');
|
||||
xtermRows.className = 'xterm-rows';
|
||||
|
||||
element.appendChild(viewport);
|
||||
element.appendChild(screen);
|
||||
screen.appendChild(xtermRows);
|
||||
|
||||
// Append to document so getComputedStyle works
|
||||
document.body.appendChild(element);
|
||||
|
||||
const terminal = {
|
||||
element,
|
||||
cols,
|
||||
rows,
|
||||
options: {
|
||||
fontFamily: opts.fontFamily ?? 'monospace',
|
||||
fontSize: opts.fontSize ?? 14,
|
||||
fontWeight: opts.fontWeight ?? 'normal',
|
||||
theme: opts.theme ?? {},
|
||||
},
|
||||
buffer: {
|
||||
active: {
|
||||
viewportY,
|
||||
baseY,
|
||||
cursorX: bufOpts.cursorX ?? 0,
|
||||
cursorY: bufOpts.cursorY ?? 0,
|
||||
getLine: (absRow: number): MockLine | undefined => {
|
||||
return mockLines[absRow - viewportY];
|
||||
},
|
||||
},
|
||||
},
|
||||
_core: {
|
||||
_renderService: {
|
||||
dimensions: {
|
||||
css: {
|
||||
cell: { width: cellW, height: cellH },
|
||||
},
|
||||
device: {
|
||||
char: {
|
||||
top: opts.deviceCharTop ?? 0,
|
||||
height: opts.deviceCharHeight ?? cellH,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
// Simulate loadAddon
|
||||
loadAddon(addon: { activate: (t: unknown) => void }) {
|
||||
addon.activate(this);
|
||||
},
|
||||
const makeLine = (text: string): { line: MockLine; set(t: string): void } => {
|
||||
let current = text;
|
||||
const line: MockLine = {
|
||||
translateToString: () => current,
|
||||
};
|
||||
if (getCellSupport) {
|
||||
line.getCell = (x: number) => cellAt(current, x);
|
||||
}
|
||||
return { line, set: (t: string) => (current = t) };
|
||||
};
|
||||
|
||||
return {
|
||||
terminal,
|
||||
/** Update buffer lines for subsequent calls */
|
||||
setLines(newLines: string[]) {
|
||||
mockLines.length = 0;
|
||||
for (const text of newLines) {
|
||||
mockLines.push({ translateToString: () => text });
|
||||
}
|
||||
let mockLines = bufOpts.lines.map(makeLine);
|
||||
|
||||
// Create minimal DOM structure
|
||||
const element = document.createElement('div');
|
||||
element.className = 'terminal xterm';
|
||||
|
||||
const viewport = document.createElement('div');
|
||||
viewport.className = 'xterm-viewport';
|
||||
|
||||
const screen = document.createElement('div');
|
||||
screen.className = 'xterm-screen';
|
||||
screen.style.position = 'relative';
|
||||
|
||||
const xtermRows = document.createElement('div');
|
||||
xtermRows.className = 'xterm-rows';
|
||||
|
||||
element.appendChild(viewport);
|
||||
element.appendChild(screen);
|
||||
screen.appendChild(xtermRows);
|
||||
|
||||
// Append to document so getComputedStyle works
|
||||
document.body.appendChild(element);
|
||||
|
||||
const writeParsedCbs = new Set<() => void>();
|
||||
const resizeCbs = new Set<(s: { cols: number; rows: number }) => void>();
|
||||
|
||||
const terminal = {
|
||||
element,
|
||||
cols,
|
||||
rows,
|
||||
options: {
|
||||
fontFamily: opts.fontFamily ?? 'monospace',
|
||||
fontSize: opts.fontSize ?? 14,
|
||||
fontWeight: opts.fontWeight ?? 'normal',
|
||||
theme: opts.theme ?? {},
|
||||
},
|
||||
buffer: {
|
||||
active: {
|
||||
viewportY,
|
||||
baseY,
|
||||
cursorX: bufOpts.cursorX ?? 0,
|
||||
cursorY: bufOpts.cursorY ?? 0,
|
||||
getLine: (absRow: number): MockLine | undefined => {
|
||||
return mockLines[absRow - viewportY]?.line;
|
||||
},
|
||||
/** Clean up DOM */
|
||||
cleanup() {
|
||||
element.remove();
|
||||
},
|
||||
},
|
||||
_core: {
|
||||
_renderService: {
|
||||
dimensions: {
|
||||
css: {
|
||||
cell: { width: cellW, height: cellH },
|
||||
},
|
||||
device: {
|
||||
char: {
|
||||
top: opts.deviceCharTop ?? 0,
|
||||
height: opts.deviceCharHeight ?? cellH,
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
},
|
||||
},
|
||||
...(emitters
|
||||
? {
|
||||
onWriteParsed(cb: () => void) {
|
||||
writeParsedCbs.add(cb);
|
||||
return { dispose: () => writeParsedCbs.delete(cb) };
|
||||
},
|
||||
onResize(cb: (s: { cols: number; rows: number }) => void) {
|
||||
resizeCbs.add(cb);
|
||||
return { dispose: () => resizeCbs.delete(cb) };
|
||||
},
|
||||
}
|
||||
: {}),
|
||||
// Simulate loadAddon
|
||||
loadAddon(addon: { activate: (t: unknown) => void }) {
|
||||
addon.activate(this);
|
||||
},
|
||||
};
|
||||
|
||||
return {
|
||||
terminal,
|
||||
/** Update buffer lines for subsequent calls */
|
||||
setLines(newLines: string[]) {
|
||||
mockLines = newLines.map(makeLine);
|
||||
},
|
||||
/** Update one line's text in place (PredictiveEchoAddon echo simulation) */
|
||||
setLine(index: number, text: string) {
|
||||
mockLines[index]?.set(text);
|
||||
},
|
||||
/** Move the mock cursor (PredictiveEchoAddon) */
|
||||
setCursor(x: number, y: number) {
|
||||
terminal.buffer.active.cursorX = x;
|
||||
terminal.buffer.active.cursorY = y;
|
||||
},
|
||||
/** Set scroll state (viewportY / baseY) */
|
||||
setScroll(newViewportY: number, newBaseY: number) {
|
||||
terminal.buffer.active.viewportY = newViewportY;
|
||||
terminal.buffer.active.baseY = newBaseY;
|
||||
},
|
||||
/** Fire the onWriteParsed emitter (PredictiveEchoAddon reconcile trigger) */
|
||||
fireWriteParsed() {
|
||||
for (const cb of [...writeParsedCbs]) cb();
|
||||
},
|
||||
/** Fire the onResize emitter */
|
||||
fireResize(newCols = cols, newRows = rows) {
|
||||
for (const cb of [...resizeCbs]) cb({ cols: newCols, rows: newRows });
|
||||
},
|
||||
/** Number of live onWriteParsed listeners (dispose assertions) */
|
||||
writeParsedListenerCount() {
|
||||
return writeParsedCbs.size;
|
||||
},
|
||||
/** Number of live onResize listeners (dispose assertions) */
|
||||
resizeListenerCount() {
|
||||
return resizeCbs.size;
|
||||
},
|
||||
/** Clean up DOM */
|
||||
cleanup() {
|
||||
element.remove();
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
/**
|
||||
* @vitest-environment jsdom
|
||||
*
|
||||
* prediction-renderer unit tests: span geometry math, seam-cover height,
|
||||
* ligature suppression, incremental add/remove keyed by seq, and geometry
|
||||
* stability under a non-1 devicePixelRatio (all dims are CSS px).
|
||||
*/
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest';
|
||||
import { addPredictionSpan, clearAllSpans, removePredictionSpan } from '../src/prediction-renderer.js';
|
||||
import type { CellDimensions, FontStyle } from '../src/types.js';
|
||||
|
||||
const dims: CellDimensions = { width: 9, height: 18, charTop: 1, charHeight: 16 };
|
||||
const font: FontStyle = {
|
||||
fontFamily: 'monospace',
|
||||
fontSize: '14px',
|
||||
fontWeight: 'normal',
|
||||
color: '#e0e0e0',
|
||||
backgroundColor: '#101010',
|
||||
letterSpacing: '0.5px',
|
||||
};
|
||||
|
||||
function makeContainer() {
|
||||
const el = document.createElement('div');
|
||||
document.body.appendChild(el);
|
||||
return el;
|
||||
}
|
||||
|
||||
function span(container: HTMLElement, map: Map<number, HTMLSpanElement>, over: Record<string, unknown> = {}) {
|
||||
addPredictionSpan(container, map, {
|
||||
seq: 1,
|
||||
row: 3,
|
||||
col: 5,
|
||||
char: 'x',
|
||||
width: 1,
|
||||
dims,
|
||||
font,
|
||||
underline: false,
|
||||
...over,
|
||||
} as never);
|
||||
return map.get((over.seq as number) ?? 1)!;
|
||||
}
|
||||
|
||||
describe('prediction-renderer', () => {
|
||||
afterEach(() => {
|
||||
document.body.innerHTML = '';
|
||||
vi.unstubAllGlobals();
|
||||
});
|
||||
|
||||
it('positions a width-1 span on the exact cell grid', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map);
|
||||
expect(s.style.left).toBe(`${5 * 9}px`);
|
||||
expect(s.style.top).toBe(`${3 * 18}px`);
|
||||
expect(s.style.width).toBe(`${9}px`);
|
||||
expect(s.textContent).toBe('x');
|
||||
});
|
||||
|
||||
it('positions a width-2 span across two cells', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map, { char: '你', width: 2 });
|
||||
expect(s.style.width).toBe(`${2 * 9}px`);
|
||||
});
|
||||
|
||||
it('covers the row seam: height is cellH+1 with line-height cellH', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map);
|
||||
expect(s.style.height).toBe(`${18 + 1}px`);
|
||||
expect(s.style.lineHeight).toBe('18px');
|
||||
});
|
||||
|
||||
it('disables ligatures and pointer events, applies font + letter-spacing', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map);
|
||||
expect(s.style.cssText).toContain("'liga' 0");
|
||||
expect(s.style.cssText).toContain("'calt' 0");
|
||||
expect(s.style.pointerEvents).toBe('none');
|
||||
expect(s.style.fontFamily).toBe('monospace');
|
||||
expect(s.style.letterSpacing).toBe('0.5px');
|
||||
expect(s.style.textAlign).toBe('center');
|
||||
});
|
||||
|
||||
it('paints an opaque background over only its own cells', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map);
|
||||
expect(['#101010', 'rgb(16, 16, 16)']).toContain(s.style.backgroundColor);
|
||||
// Background is bounded by the span's own width, never a full row
|
||||
expect(s.style.width).toBe('9px');
|
||||
});
|
||||
|
||||
it('underline renders only when requested', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const container = makeContainer();
|
||||
const plain = span(container, map, { seq: 1 });
|
||||
const lined = span(container, map, { seq: 2, underline: true });
|
||||
expect(plain.style.textDecoration).toBe('');
|
||||
expect(lined.style.textDecoration).toBe('underline');
|
||||
});
|
||||
|
||||
it('adds and removes incrementally, keyed by seq', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const container = makeContainer();
|
||||
span(container, map, { seq: 1 });
|
||||
span(container, map, { seq: 2, col: 6 });
|
||||
span(container, map, { seq: 3, col: 7 });
|
||||
expect(container.children).toHaveLength(3);
|
||||
|
||||
removePredictionSpan(map, 2);
|
||||
expect(container.children).toHaveLength(2);
|
||||
expect(map.has(2)).toBe(false);
|
||||
expect(map.has(1)).toBe(true);
|
||||
expect(map.has(3)).toBe(true);
|
||||
|
||||
removePredictionSpan(map, 999); // unknown seq: no-op
|
||||
expect(container.children).toHaveLength(2);
|
||||
});
|
||||
|
||||
it('clearAllSpans empties both the DOM and the map', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const container = makeContainer();
|
||||
span(container, map, { seq: 1 });
|
||||
span(container, map, { seq: 2, col: 6 });
|
||||
clearAllSpans(map);
|
||||
expect(container.children).toHaveLength(0);
|
||||
expect(map.size).toBe(0);
|
||||
});
|
||||
|
||||
it('geometry is stable under devicePixelRatio 2 (dims are CSS px)', () => {
|
||||
vi.stubGlobal('devicePixelRatio', 2);
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map);
|
||||
expect(s.style.left).toBe(`${5 * 9}px`);
|
||||
expect(s.style.top).toBe(`${3 * 18}px`);
|
||||
expect(s.style.width).toBe('9px');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,538 @@
|
||||
/**
|
||||
* @vitest-environment jsdom
|
||||
*
|
||||
* PredictiveEchoAddon unit tests: the algorithm laws (anchoring, prefix-only
|
||||
* confirmation with cursor advance, two-pass mismatch cascade with neutral
|
||||
* blanks, TTL, off-row grace, gates) and lifecycle safety.
|
||||
*
|
||||
* Timer-based cases fake `performance` explicitly: the addon clocks
|
||||
* sentAt/TTL/grace with performance.now(), which vitest does NOT fake by
|
||||
* default.
|
||||
*/
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
import { PredictiveEchoAddon } from '../src/predictive-echo-addon.js';
|
||||
import { createMockTerminal } from './helpers.js';
|
||||
|
||||
const TIMER_CONFIG = {
|
||||
toFake: ['setTimeout', 'clearTimeout', 'setInterval', 'clearInterval', 'Date', 'performance'] as const,
|
||||
};
|
||||
|
||||
/** Composer-like buffer: `› ` marker + placeholder, cursor at col 2 row 0. */
|
||||
function composerMock(opts: Parameters<typeof createMockTerminal>[0] = {}) {
|
||||
return createMockTerminal({
|
||||
buffer: { lines: ['› Use /skills to list', '', ''], cursorX: 2, cursorY: 0 },
|
||||
...opts,
|
||||
});
|
||||
}
|
||||
|
||||
function spansOf(mock: ReturnType<typeof createMockTerminal>): HTMLSpanElement[] {
|
||||
const screen = mock.terminal.element.querySelector('.xterm-screen')!;
|
||||
return Array.from(screen.querySelectorAll('[data-predictive-echo] span')) as HTMLSpanElement[];
|
||||
}
|
||||
|
||||
async function flushMicrotasks() {
|
||||
await Promise.resolve();
|
||||
await Promise.resolve();
|
||||
}
|
||||
|
||||
describe('PredictiveEchoAddon', () => {
|
||||
let mock: ReturnType<typeof createMockTerminal>;
|
||||
let addon: PredictiveEchoAddon;
|
||||
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers(TIMER_CONFIG);
|
||||
mock = composerMock();
|
||||
addon = new PredictiveEchoAddon();
|
||||
addon.activate(mock.terminal as never);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
addon.dispose();
|
||||
mock.cleanup();
|
||||
vi.useRealTimers();
|
||||
});
|
||||
|
||||
it('paints a span at the cursor cell and returns true', () => {
|
||||
expect(addon.predictChar('h')).toBe(true);
|
||||
const spans = spansOf(mock);
|
||||
expect(spans).toHaveLength(1);
|
||||
expect(spans[0].textContent).toBe('h');
|
||||
expect(spans[0].style.left).toBe(`${2 * 8.4}px`);
|
||||
expect(spans[0].style.top).toBe('0px');
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
});
|
||||
|
||||
it('stacks predictions at anchor+cumulative width while the cursor is unmoved', () => {
|
||||
addon.predictChar('h');
|
||||
addon.predictChar('e');
|
||||
addon.predictChar('y');
|
||||
const spans = spansOf(mock);
|
||||
expect(spans.map((s) => s.style.left)).toEqual([`${2 * 8.4}px`, `${3 * 8.4}px`, `${4 * 8.4}px`]);
|
||||
expect(addon.state.anchor).toEqual({ row: 0, col: 2 });
|
||||
});
|
||||
|
||||
it('re-anchors at the new cursor once outstanding drains to zero', async () => {
|
||||
addon.predictChar('h');
|
||||
mock.setLine(0, '› h');
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(addon.state.anchor).toBeNull();
|
||||
|
||||
addon.predictChar('i');
|
||||
expect(addon.state.anchor).toEqual({ row: 0, col: 3 });
|
||||
expect(spansOf(mock)[0].style.left).toBe(`${3 * 8.4}px`);
|
||||
});
|
||||
|
||||
it('inline reconcile inside predictChar absorbs an echo that landed between keystrokes', () => {
|
||||
addon.predictChar('h');
|
||||
// Echo lands but no onWriteParsed fires before the next keystroke
|
||||
mock.setLine(0, '› h');
|
||||
mock.setCursor(3, 0);
|
||||
expect(addon.predictChar('i')).toBe(true);
|
||||
// 'h' confirmed inline; 'i' anchored at the advanced cursor, not stacked
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.state.anchor).toEqual({ row: 0, col: 3 });
|
||||
});
|
||||
|
||||
it('confirms and removes exactly the echoed prefix (cell match + cursor advance)', async () => {
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
addon.predictChar('c');
|
||||
mock.setLine(0, '› ab');
|
||||
mock.setCursor(4, 0); // advanced past 'a' and 'b' only
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.confirmedTotal).toBe(2);
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(spansOf(mock).map((s) => s.textContent)).toEqual(['c']);
|
||||
});
|
||||
|
||||
it('partial confirmation never moves remaining spans (no jitter)', async () => {
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
const bLeft = spansOf(mock)[1].style.left;
|
||||
mock.setLine(0, '› a');
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(spansOf(mock)).toHaveLength(1);
|
||||
expect(spansOf(mock)[0].style.left).toBe(bLeft);
|
||||
});
|
||||
|
||||
it('does NOT confirm when the cell matches but the cursor has not advanced (in-place repaint)', async () => {
|
||||
// Predict 'U' over the placeholder whose cell already shows 'U'
|
||||
addon.predictChar('U');
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
// tmux repaints the identical row; cursor stays at the anchor
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(addon.state.confirmedTotal).toBe(0);
|
||||
});
|
||||
|
||||
it('does NOT confirm or drop when the predicted char equals the pre-existing snapshot', async () => {
|
||||
addon.predictChar('U');
|
||||
// Several passes over the unchanged placeholder: no confirm, no cascade
|
||||
for (let i = 0; i < 4; i++) {
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
}
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
});
|
||||
|
||||
it('one transient mismatch survives; a persistent foreign cell cascades (two-pass rule)', async () => {
|
||||
addon.predictChar('a');
|
||||
mock.setLine(0, '› Z'); // foreign non-blank at the predicted cell
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1); // pass 1: survives
|
||||
|
||||
// Transient recovery resets the counter
|
||||
mock.setLine(0, '› Use /skills to list');
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
mock.setLine(0, '› Z');
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1); // count restarted, pass 1 again
|
||||
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(0); // pass 2: cascaded
|
||||
expect(addon.state.droppedTotal).toBe(1);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('blank cells are neutral: placeholder cleared under predictions does not cascade', async () => {
|
||||
// Predict over placeholder text, then codex clears the placeholder on
|
||||
// first echo: later cells become blank, which must NOT count as
|
||||
// foreign (measured behavior; without this, fast typing over the
|
||||
// placeholder drops exactly when RTT is high).
|
||||
addon.predictChar('h');
|
||||
addon.predictChar('i');
|
||||
mock.setLine(0, '› h'); // 'h' echoed; placeholder gone; 'i' cell now blank
|
||||
mock.setCursor(3, 0);
|
||||
for (let i = 0; i < 4; i++) {
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
}
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.state.outstanding).toBe(1); // 'i' still pending, TTL-bounded
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
});
|
||||
|
||||
it('mismatch cascade drops the record and all later ones, earlier confirmed stay gone', async () => {
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
addon.predictChar('c');
|
||||
mock.setLine(0, '› aXX'); // 'a' echoed; foreign 'X' under 'b' and 'c'
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.state.droppedTotal).toBe(2);
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('TTL expiry drops predictions and leaves no timers armed (fake timers)', () => {
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
expect(vi.getTimerCount()).toBe(1);
|
||||
vi.advanceTimersByTime(1100);
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(addon.state.droppedTotal).toBe(2);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
expect(vi.getTimerCount()).toBe(0);
|
||||
});
|
||||
|
||||
it('TTL timer re-arms for remaining records after a partial confirm', async () => {
|
||||
addon.predictChar('a'); // t=0, deadline ~1001
|
||||
vi.advanceTimersByTime(600);
|
||||
addon.predictChar('b'); // t=600, deadline ~1601
|
||||
// Echo confirms 'a' before its TTL; 'b' remains
|
||||
mock.setLine(0, '› a');
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
vi.advanceTimersByTime(450); // t=1050: a's timer fired, b (age 450) survives
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(vi.getTimerCount()).toBe(1); // re-armed for b
|
||||
vi.advanceTimersByTime(600); // t=1650: b expired
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(vi.getTimerCount()).toBe(0);
|
||||
});
|
||||
|
||||
it('cursor off anchor row within grace keeps predictions; sustained off-row drops all', async () => {
|
||||
addon.predictChar('a');
|
||||
mock.setCursor(0, 5);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1); // transient excursion tolerated
|
||||
|
||||
vi.advanceTimersByTime(200); // > cursorGraceMs (150)
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('viewportY !== baseY clears predictions (scrolled up)', async () => {
|
||||
addon.predictChar('a');
|
||||
mock.setScroll(0, 5); // user scrolled: viewport pinned above baseY
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
// And no new predictions while scrolled
|
||||
expect(addon.predictChar('b')).toBe(false);
|
||||
});
|
||||
|
||||
it('maxPending: the 33rd predictChar returns false', () => {
|
||||
for (let i = 0; i < 32; i++) {
|
||||
expect(addon.predictChar('x')).toBe(true);
|
||||
}
|
||||
expect(addon.predictChar('y')).toBe(false);
|
||||
expect(addon.state.outstanding).toBe(32);
|
||||
});
|
||||
|
||||
it('edge margin: a prediction landing within edgeMarginCells of cols returns false', () => {
|
||||
mock.setCursor(75, 0); // cols 80, margin 4: col 75 + 1 <= 76 allowed
|
||||
expect(addon.predictChar('a')).toBe(true);
|
||||
// Next lands at col 76: 77 > 76 suppressed
|
||||
expect(addon.predictChar('b')).toBe(false);
|
||||
});
|
||||
|
||||
it('predictWhen gate false suppresses painting, predictChar just returns false', () => {
|
||||
addon.setPredictWhen(() => false);
|
||||
expect(addon.predictChar('a')).toBe(false);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('setPredictWhen(null) removes the gate at runtime', () => {
|
||||
addon.setPredictWhen(() => false);
|
||||
expect(addon.predictChar('a')).toBe(false);
|
||||
addon.setPredictWhen(null);
|
||||
expect(addon.predictChar('a')).toBe(true);
|
||||
});
|
||||
|
||||
it('multi-codepoint graphemes and control chars return false', () => {
|
||||
for (const bad of ['ab', '\x1b', '\x03', '\r', '\n', '\t', '\x7f', '👨👩👧', '']) {
|
||||
expect(addon.predictChar(bad)).toBe(false);
|
||||
}
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
// Single astral emoji IS a single codepoint: predicted (width 2)
|
||||
expect(addon.predictChar('😀')).toBe(true);
|
||||
});
|
||||
|
||||
it('CJK: 2-cell span, next prediction offsets by 2, confirm reads the leading cell', async () => {
|
||||
expect(addon.predictChar('你')).toBe(true);
|
||||
const first = spansOf(mock)[0];
|
||||
expect(first.style.width).toBe(`${2 * 8.4}px`);
|
||||
addon.predictChar('a');
|
||||
expect(spansOf(mock)[1].style.left).toBe(`${4 * 8.4}px`); // 2 + width 2
|
||||
|
||||
mock.setLine(0, '› 你');
|
||||
mock.setCursor(4, 0); // advanced past the wide char
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
});
|
||||
|
||||
it('getCell-less terminal: ASCII fallback works, wide chars suppressed', () => {
|
||||
const bare = createMockTerminal({
|
||||
buffer: { lines: ['› ', ''], cursorX: 2, cursorY: 0 },
|
||||
getCellSupport: false,
|
||||
});
|
||||
const a = new PredictiveEchoAddon();
|
||||
a.activate(bare.terminal as never);
|
||||
expect(a.predictChar('x')).toBe(true);
|
||||
expect(a.predictChar('你')).toBe(false);
|
||||
a.dispose();
|
||||
bare.cleanup();
|
||||
});
|
||||
|
||||
it("'' and ' ' cell reads are equivalent for snapshot and confirm", async () => {
|
||||
// Snapshot beyond the line text reads '' -> normalized ' '
|
||||
mock.setLine(0, '› ');
|
||||
addon.predictChar('a'); // snapshot at col 2 is '' -> ' '
|
||||
// A repaint that writes explicit spaces must not count as foreign
|
||||
mock.setLine(0, '› ');
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
});
|
||||
|
||||
it('predictBackspace pops newest, returns false when empty, never touches confirmed', async () => {
|
||||
expect(addon.predictBackspace()).toBe(false);
|
||||
addon.reconcile(); // the empty pop armed the anchor hold; release it
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
expect(addon.predictBackspace()).toBe(true);
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(spansOf(mock).map((s) => s.textContent)).toEqual(['a']);
|
||||
|
||||
mock.setLine(0, '› a');
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.predictBackspace()).toBe(false); // confirmed text is not popped
|
||||
});
|
||||
|
||||
it('clearPredictions empties the container, resets anchor, cancels the timer', () => {
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
expect(vi.getTimerCount()).toBe(1);
|
||||
addon.clearPredictions();
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(addon.state.anchor).toBeNull();
|
||||
expect(vi.getTimerCount()).toBe(0);
|
||||
});
|
||||
|
||||
it('onWriteParsed reconcile is debounced to one pass per burst', async () => {
|
||||
addon.predictChar('a');
|
||||
mock.setLine(0, '› Z'); // foreign cell: each PASS increments mismatches
|
||||
mock.fireWriteParsed();
|
||||
mock.fireWriteParsed();
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
// Three synchronous fires coalesced into ONE pass: not dropped yet
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(0); // second pass cascades
|
||||
});
|
||||
|
||||
it('onResize clears predictions (cell geometry changed)', () => {
|
||||
addon.predictChar('a');
|
||||
mock.fireResize(120, 40);
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('works without onWriteParsed via manual reconcile()', () => {
|
||||
const bare = composerMock({ emitters: false });
|
||||
const a = new PredictiveEchoAddon();
|
||||
a.activate(bare.terminal as never);
|
||||
a.predictChar('h');
|
||||
bare.setLine(0, '› h');
|
||||
bare.setCursor(3, 0);
|
||||
a.reconcile();
|
||||
expect(a.state.confirmedTotal).toBe(1);
|
||||
expect(a.state.outstanding).toBe(0);
|
||||
a.dispose();
|
||||
bare.cleanup();
|
||||
});
|
||||
|
||||
it('dispose unhooks listeners and removes the container', () => {
|
||||
expect(mock.writeParsedListenerCount()).toBe(1);
|
||||
expect(mock.resizeListenerCount()).toBe(1);
|
||||
addon.predictChar('a');
|
||||
addon.dispose();
|
||||
expect(mock.writeParsedListenerCount()).toBe(0);
|
||||
expect(mock.resizeListenerCount()).toBe(0);
|
||||
const screen = mock.terminal.element.querySelector('.xterm-screen')!;
|
||||
expect(screen.querySelector('[data-predictive-echo]')).toBeNull();
|
||||
expect(vi.getTimerCount()).toBe(0);
|
||||
});
|
||||
|
||||
it('every public method is safe before activate and after dispose', () => {
|
||||
const fresh = new PredictiveEchoAddon();
|
||||
expect(fresh.predictChar('a')).toBe(false);
|
||||
expect(fresh.predictBackspace()).toBe(false);
|
||||
fresh.clearPredictions();
|
||||
fresh.reconcile();
|
||||
fresh.refreshFont();
|
||||
fresh.setPredictWhen(() => true);
|
||||
expect(fresh.hasPredictions).toBe(false);
|
||||
expect(fresh.state.outstanding).toBe(0);
|
||||
|
||||
addon.dispose();
|
||||
expect(addon.predictChar('a')).toBe(false);
|
||||
expect(addon.predictBackspace()).toBe(false);
|
||||
addon.clearPredictions();
|
||||
addon.reconcile();
|
||||
addon.refreshFont();
|
||||
expect(addon.hasPredictions).toBe(false);
|
||||
});
|
||||
|
||||
it('hostile terminal stubs never propagate exceptions', () => {
|
||||
const hostile = {
|
||||
element: document.createElement('div'),
|
||||
cols: 80,
|
||||
rows: 24,
|
||||
options: {},
|
||||
buffer: {
|
||||
active: {
|
||||
viewportY: 0,
|
||||
baseY: 0,
|
||||
cursorX: 0,
|
||||
cursorY: 0,
|
||||
getLine: () => {
|
||||
throw new Error('boom');
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
const a = new PredictiveEchoAddon();
|
||||
expect(() => a.activate(hostile as never)).not.toThrow();
|
||||
expect(a.predictChar('x')).toBe(false); // getLine throws inside -> caught
|
||||
expect(() => a.reconcile()).not.toThrow();
|
||||
a.dispose();
|
||||
|
||||
// Terminal with no render dimensions: addon inert, no throws
|
||||
const dimless = composerMock();
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
delete (dimless.terminal as any)._core;
|
||||
const b = new PredictiveEchoAddon();
|
||||
b.activate(dimless.terminal as never);
|
||||
expect(b.predictChar('x')).toBe(false);
|
||||
b.dispose();
|
||||
dimless.cleanup();
|
||||
});
|
||||
|
||||
it('underlinePredictions styles spans; refreshFont re-reads the rendered color', () => {
|
||||
const themed = composerMock({ theme: { foreground: '#aabbcc', background: '#112233' } });
|
||||
// The recipe prefers the computed .xterm-rows color (what xterm really
|
||||
// renders with); give the mock rows an explicit color like a real skin.
|
||||
const rows = themed.terminal.element.querySelector('.xterm-rows') as HTMLElement;
|
||||
rows.style.color = 'rgb(170, 187, 204)';
|
||||
const a = new PredictiveEchoAddon({ underlinePredictions: true });
|
||||
a.activate(themed.terminal as never);
|
||||
a.predictChar('u');
|
||||
const span = themed.terminal.element.querySelector('.xterm-screen span') as HTMLSpanElement;
|
||||
expect(span.style.textDecoration).toBe('underline');
|
||||
expect(span.style.color).toBe('rgb(170, 187, 204)');
|
||||
|
||||
rows.style.color = 'rgb(255, 0, 0)'; // skin change
|
||||
a.refreshFont();
|
||||
a.clearPredictions();
|
||||
a.reconcile(); // release the anchor hold armed by the clear
|
||||
a.predictChar('v');
|
||||
const span2 = themed.terminal.element.querySelector('.xterm-screen span') as HTMLSpanElement;
|
||||
expect(span2.style.color).toBe('rgb(255, 0, 0)');
|
||||
a.dispose();
|
||||
themed.cleanup();
|
||||
});
|
||||
|
||||
it('anchor hold: backspace into echoed text suppresses prediction until a write parses', async () => {
|
||||
// \x7f went to the wire with nothing outstanding: the cursor will move
|
||||
// in a way the display has not shown, so anchoring now paints one cell
|
||||
// off (review finding: "tehh" ghosts on backspace-then-retype at RTT)
|
||||
expect(addon.predictBackspace()).toBe(false);
|
||||
expect(addon.predictChar('x')).toBe(false);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
mock.fireWriteParsed(); // the display caught up
|
||||
await flushMicrotasks();
|
||||
expect(addon.predictChar('x')).toBe(true);
|
||||
});
|
||||
|
||||
it('anchor hold: clearPredictions suppresses until a write parses (or manual reconcile)', async () => {
|
||||
addon.predictChar('a');
|
||||
addon.clearPredictions(); // consumer saw Enter/Esc/arrow/paste
|
||||
expect(addon.predictChar('b')).toBe(false);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.predictChar('b')).toBe(true);
|
||||
});
|
||||
|
||||
it('anchor hold: the inline predictChar reconcile does NOT release it', () => {
|
||||
addon.clearPredictions();
|
||||
// Several keystrokes in a row before any echo: all suppressed, because
|
||||
// predictChar's inline pass must not count as the display catching up
|
||||
expect(addon.predictChar('a')).toBe(false);
|
||||
expect(addon.predictChar('b')).toBe(false);
|
||||
addon.reconcile(); // public/manual pass IS the caught-up contract
|
||||
expect(addon.predictChar('c')).toBe(true);
|
||||
});
|
||||
|
||||
it('state getter reports outstanding/confirmedTotal/droppedTotal/anchor', async () => {
|
||||
expect(addon.state).toEqual({ outstanding: 0, confirmedTotal: 0, droppedTotal: 0, anchor: null });
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
expect(addon.state.outstanding).toBe(2);
|
||||
expect(addon.state.anchor).toEqual({ row: 0, col: 2 });
|
||||
expect(addon.hasPredictions).toBe(true);
|
||||
|
||||
mock.setLine(0, '› a');
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
addon.clearPredictions();
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.state.droppedTotal).toBe(1);
|
||||
expect(addon.hasPredictions).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,125 @@
|
||||
/**
|
||||
* @vitest-environment jsdom
|
||||
*
|
||||
* Layer 3: seeded property fuzz against the REAL xterm parser. Random
|
||||
* interleavings of predictions, backspaces, clears, echo writes (correct,
|
||||
* partial, foreign), screen clears, scrolls and cursor jumps; invariants
|
||||
* checked after EVERY op:
|
||||
* 1. span count === outstanding record count, every span inside the grid
|
||||
* 2. no public method throws
|
||||
* 3. eventual convergence: after the run settles (TTL elapse + reconcile),
|
||||
* outstanding === 0 and the span container is empty
|
||||
*
|
||||
* Reproduce a failure with FUZZ_SEED=<seed> FUZZ_ITERS=<n> npx vitest run
|
||||
* test/predictive-echo-fuzz.test.ts (the failing seed+iter is in the
|
||||
* assertion message).
|
||||
*/
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { PredictiveEchoAddon } from '../src/predictive-echo-addon.js';
|
||||
import { CELL_H, CELL_W, createReplayTerminal } from './replay-helpers.js';
|
||||
|
||||
const SEED = Number(process.env.FUZZ_SEED ?? 1337);
|
||||
const TOTAL_ITERS = Number(process.env.FUZZ_ITERS ?? 500);
|
||||
const BATCHES = 4;
|
||||
const TTL_MS = 5;
|
||||
|
||||
function mulberry32(seed: number) {
|
||||
let a = seed >>> 0;
|
||||
return () => {
|
||||
a |= 0;
|
||||
a = (a + 0x6d2b79f5) | 0;
|
||||
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
||||
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
||||
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
||||
};
|
||||
}
|
||||
|
||||
const ALPHABET = [...'abcdefghij XZ!?', '你', '好', '😀'];
|
||||
|
||||
function sleep(ms: number) {
|
||||
return new Promise((r) => setTimeout(r, ms));
|
||||
}
|
||||
|
||||
async function fuzzIteration(iter: number, label: string) {
|
||||
const rand = mulberry32(SEED + iter);
|
||||
const rt = createReplayTerminal(60, 12);
|
||||
const addon = new PredictiveEchoAddon({ ttlMs: TTL_MS });
|
||||
addon.activate(rt.hybrid);
|
||||
const ctx = `${label} seed=${SEED} iter=${iter}`;
|
||||
|
||||
// Park the cursor mid-screen like a composer would
|
||||
await rt.write('\x1b[6;3H');
|
||||
|
||||
const ops = 4 + Math.floor(rand() * 12);
|
||||
for (let i = 0; i < ops; i++) {
|
||||
const r = rand();
|
||||
if (r < 0.35) {
|
||||
addon.predictChar(ALPHABET[Math.floor(rand() * ALPHABET.length)]);
|
||||
} else if (r < 0.43) {
|
||||
addon.predictBackspace();
|
||||
} else if (r < 0.48) {
|
||||
addon.clearPredictions();
|
||||
} else if (r < 0.62) {
|
||||
// Correct-ish echo: write a run of random chars at the anchor and
|
||||
// leave the cursor advanced (confirms whatever happens to match)
|
||||
const a = addon.state.anchor;
|
||||
if (a) {
|
||||
const n = 1 + Math.floor(rand() * 3);
|
||||
let text = '';
|
||||
for (let k = 0; k < n; k++) text += ALPHABET[Math.floor(rand() * ALPHABET.length)];
|
||||
await rt.write(`\x1b[${a.row + 1};${a.col + 1}H${text}`);
|
||||
}
|
||||
} else if (r < 0.72) {
|
||||
// Foreign rewrite across the anchor row
|
||||
await rt.write(`\x1b[6;1H${'Q'.repeat(1 + Math.floor(rand() * 20))}`);
|
||||
} else if (r < 0.8) {
|
||||
// Scroll: newlines at the bottom push history
|
||||
await rt.write(`\x1b[12;1H${'\r\n'.repeat(1 + Math.floor(rand() * 3))}`);
|
||||
} else if (r < 0.85) {
|
||||
await rt.write('\x1b[2J\x1b[H'); // clear screen + home
|
||||
} else if (r < 0.95) {
|
||||
addon.reconcile();
|
||||
} else {
|
||||
// Cursor jump
|
||||
const row = 1 + Math.floor(rand() * 12);
|
||||
const col = 1 + Math.floor(rand() * 60);
|
||||
await rt.write(`\x1b[${row};${col}H`);
|
||||
}
|
||||
await Promise.resolve(); // flush the debounced reconcile microtask
|
||||
|
||||
// Invariant 1: span/record parity + grid bounds, after every op
|
||||
expect(rt.spanCount(), ctx).toBe(addon.state.outstanding);
|
||||
for (const s of rt.spans()) {
|
||||
const left = parseFloat(s.style.left);
|
||||
const width = parseFloat(s.style.width);
|
||||
const top = parseFloat(s.style.top);
|
||||
expect(left + width, ctx).toBeLessThanOrEqual(60 * CELL_W);
|
||||
expect(top, ctx).toBeLessThanOrEqual(11 * CELL_H);
|
||||
expect(left, ctx).toBeGreaterThanOrEqual(0);
|
||||
}
|
||||
}
|
||||
|
||||
// Invariant 3: eventual convergence via echo/TTL, never via dispose
|
||||
if (addon.state.outstanding > 0) {
|
||||
await sleep(TTL_MS + 15);
|
||||
addon.reconcile();
|
||||
}
|
||||
expect(addon.state.outstanding, ctx).toBe(0);
|
||||
expect(rt.spanCount(), ctx).toBe(0);
|
||||
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}
|
||||
|
||||
describe(`predictive echo fuzz (${TOTAL_ITERS} iterations, seed ${SEED})`, () => {
|
||||
const perBatch = Math.ceil(TOTAL_ITERS / BATCHES);
|
||||
for (let b = 0; b < BATCHES; b++) {
|
||||
it(`batch ${b + 1}/${BATCHES}`, async () => {
|
||||
const start = b * perBatch;
|
||||
const end = Math.min(start + perBatch, TOTAL_ITERS);
|
||||
for (let iter = start; iter < end; iter++) {
|
||||
await fuzzIteration(iter, `batch${b + 1}`);
|
||||
}
|
||||
}, 60000);
|
||||
}
|
||||
});
|
||||
@@ -4,159 +4,155 @@ import { findPrompt, readTextAfterPrompt } from '../src/prompt-finder.js';
|
||||
import type { XtermTerminal, PromptFinder } from '../src/types.js';
|
||||
|
||||
function term(lines: string[]) {
|
||||
return createMockTerminal({ buffer: { lines } });
|
||||
return createMockTerminal({ buffer: { lines } });
|
||||
}
|
||||
|
||||
describe('findPrompt', () => {
|
||||
describe('character strategy', () => {
|
||||
it('finds $ prompt at column 0', () => {
|
||||
const { terminal, cleanup } = term(['output line', '$ ls -la']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 1, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('finds > prompt', () => {
|
||||
const { terminal, cleanup } = term(['> hello']);
|
||||
const finder: PromptFinder = { type: 'character', char: '>' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('finds prompt with prefix (user@host)', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ command']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 11 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('scans bottom-up and returns lowest match', () => {
|
||||
const { terminal, cleanup } = term([
|
||||
'$ old prompt',
|
||||
'output',
|
||||
'$ current prompt',
|
||||
]);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 2, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('returns null when no prompt found', () => {
|
||||
const { terminal, cleanup } = term(['no prompt here', 'or here']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('finds Unicode prompt character', () => {
|
||||
const { terminal, cleanup } = term(['\u276f hello']);
|
||||
const finder: PromptFinder = { type: 'character', char: '\u276f' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
describe('character strategy', () => {
|
||||
it('finds $ prompt at column 0', () => {
|
||||
const { terminal, cleanup } = term(['output line', '$ ls -la']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 1, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
describe('regex strategy', () => {
|
||||
it('finds regex prompt', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~/dir$ ls']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(15);
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('matches complex PS1 patterns', () => {
|
||||
const { terminal, cleanup } = term(['(venv) user % cmd']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /%/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(12);
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('returns null on no match', () => {
|
||||
const { terminal, cleanup } = term(['just output']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$\s*$/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('handles global flag safely (strips g to avoid lastIndex)', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ cmd']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$/g };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(11);
|
||||
// Call again — should return same result (no lastIndex drift)
|
||||
const pos2 = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos2).toEqual(pos);
|
||||
cleanup();
|
||||
});
|
||||
it('finds > prompt', () => {
|
||||
const { terminal, cleanup } = term(['> hello']);
|
||||
const finder: PromptFinder = { type: 'character', char: '>' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
describe('custom strategy', () => {
|
||||
it('uses custom finder function', () => {
|
||||
const { terminal, cleanup } = term(['anything']);
|
||||
const finder: PromptFinder = {
|
||||
type: 'custom',
|
||||
find: () => ({ row: 5, col: 10 }),
|
||||
};
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 5, col: 10 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('handles null from custom finder', () => {
|
||||
const { terminal, cleanup } = term(['anything']);
|
||||
const finder: PromptFinder = {
|
||||
type: 'custom',
|
||||
find: () => null,
|
||||
};
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
it('finds prompt with prefix (user@host)', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ command']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 11 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('scans bottom-up and returns lowest match', () => {
|
||||
const { terminal, cleanup } = term(['$ old prompt', 'output', '$ current prompt']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 2, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('returns null when no prompt found', () => {
|
||||
const { terminal, cleanup } = term(['no prompt here', 'or here']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('finds Unicode prompt character', () => {
|
||||
const { terminal, cleanup } = term(['\u276f hello']);
|
||||
const finder: PromptFinder = { type: 'character', char: '\u276f' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
});
|
||||
|
||||
describe('regex strategy', () => {
|
||||
it('finds regex prompt', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~/dir$ ls']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(15);
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('matches complex PS1 patterns', () => {
|
||||
const { terminal, cleanup } = term(['(venv) user % cmd']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /%/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(12);
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('returns null on no match', () => {
|
||||
const { terminal, cleanup } = term(['just output']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$\s*$/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('handles global flag safely (strips g to avoid lastIndex)', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ cmd']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$/g };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(11);
|
||||
// Call again — should return same result (no lastIndex drift)
|
||||
const pos2 = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos2).toEqual(pos);
|
||||
cleanup();
|
||||
});
|
||||
});
|
||||
|
||||
describe('custom strategy', () => {
|
||||
it('uses custom finder function', () => {
|
||||
const { terminal, cleanup } = term(['anything']);
|
||||
const finder: PromptFinder = {
|
||||
type: 'custom',
|
||||
find: () => ({ row: 5, col: 10 }),
|
||||
};
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 5, col: 10 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('handles null from custom finder', () => {
|
||||
const { terminal, cleanup } = term(['anything']);
|
||||
const finder: PromptFinder = {
|
||||
type: 'custom',
|
||||
find: () => null,
|
||||
};
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe('readTextAfterPrompt', () => {
|
||||
it('reads text after prompt with offset', () => {
|
||||
const { terminal, cleanup } = term(['$ hello world']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('hello world');
|
||||
cleanup();
|
||||
});
|
||||
it('reads text after prompt with offset', () => {
|
||||
const { terminal, cleanup } = term(['$ hello world']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('hello world');
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('returns empty string for empty prompt line', () => {
|
||||
const { terminal, cleanup } = term(['$ ']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('');
|
||||
cleanup();
|
||||
});
|
||||
it('returns empty string for empty prompt line', () => {
|
||||
const { terminal, cleanup } = term(['$ ']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('');
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('trims trailing whitespace', () => {
|
||||
const { terminal, cleanup } = term(['$ hello ']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('hello');
|
||||
cleanup();
|
||||
});
|
||||
it('trims trailing whitespace', () => {
|
||||
const { terminal, cleanup } = term(['$ hello ']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('hello');
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('handles offset for complex prompts', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ ls -la']);
|
||||
const prompt = { row: 0, col: 11 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('ls -la');
|
||||
cleanup();
|
||||
});
|
||||
it('handles offset for complex prompts', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ ls -la']);
|
||||
const prompt = { row: 0, col: 11 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('ls -la');
|
||||
cleanup();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
/** Vite `?raw` imports used by replay-helpers.ts (fixture JSONL as strings). */
|
||||
declare module '*.jsonl?raw' {
|
||||
const content: string;
|
||||
export default content;
|
||||
}
|
||||
@@ -0,0 +1,173 @@
|
||||
/**
|
||||
* Replay-test helpers: a structural hybrid terminal whose buffer, cursor and
|
||||
* onWriteParsed delegate to a REAL @xterm/headless Terminal (so fixtures run
|
||||
* through the real parser), while `element` is a jsdom div the addon can
|
||||
* paint spans into. Works because XtermTerminal is structurally typed.
|
||||
*
|
||||
* Also carries the test-side mirror of Codeman's classifyPredictInput() and
|
||||
* codex composer gate (the real ones live in terminal-ui.js and are pinned by
|
||||
* the repo's Layer 4 vm tests; keep the two in sync).
|
||||
*/
|
||||
import { Terminal } from '@xterm/headless';
|
||||
import type { XtermTerminal } from '../src/types.js';
|
||||
// ?raw imports keep the jsdom environment free of node: builtins
|
||||
import pasteBracketed from './fixtures/codex/paste-bracketed.jsonl?raw';
|
||||
import slashPicker from './fixtures/codex/slash-picker.jsonl?raw';
|
||||
import streamingBurst from './fixtures/codex/streaming-burst.jsonl?raw';
|
||||
import streamingReal from './fixtures/codex/streaming-real.jsonl?raw';
|
||||
import trustModal from './fixtures/codex/trust-modal.jsonl?raw';
|
||||
import typeHello from './fixtures/codex/type-hello.jsonl?raw';
|
||||
import wrap from './fixtures/codex/wrap.jsonl?raw';
|
||||
|
||||
const FIXTURES: Record<string, string> = {
|
||||
'paste-bracketed': pasteBracketed,
|
||||
'slash-picker': slashPicker,
|
||||
'streaming-burst': streamingBurst,
|
||||
'streaming-real': streamingReal,
|
||||
'trust-modal': trustModal,
|
||||
'type-hello': typeHello,
|
||||
wrap,
|
||||
};
|
||||
|
||||
export const CELL_W = 9;
|
||||
export const CELL_H = 18;
|
||||
|
||||
export interface FixtureLine {
|
||||
delayMs?: number;
|
||||
keyAt?: boolean;
|
||||
data: string;
|
||||
}
|
||||
|
||||
export interface FixtureMeta {
|
||||
scenario: string;
|
||||
cols: number;
|
||||
rows: number;
|
||||
codexVersion: string;
|
||||
recordedAt: string;
|
||||
}
|
||||
|
||||
export function loadFixture(name: string): { meta: FixtureMeta; lines: FixtureLine[] } {
|
||||
const content = FIXTURES[name];
|
||||
if (!content) throw new Error(`unknown fixture ${name}`);
|
||||
const raw = content
|
||||
.trim()
|
||||
.split('\n')
|
||||
.map((l) => JSON.parse(l));
|
||||
return { meta: raw[0] as FixtureMeta, lines: raw.slice(1) as FixtureLine[] };
|
||||
}
|
||||
|
||||
export interface ReplayTerminal {
|
||||
hybrid: XtermTerminal;
|
||||
term: Terminal;
|
||||
write(data: string): Promise<void>;
|
||||
cursorRowText(): string;
|
||||
rowText(viewportRow: number): string;
|
||||
spanCount(): number;
|
||||
spans(): HTMLSpanElement[];
|
||||
cleanup(): void;
|
||||
}
|
||||
|
||||
export function createReplayTerminal(cols: number, rows: number): ReplayTerminal {
|
||||
const term = new Terminal({ cols, rows, scrollback: 2000, allowProposedApi: true });
|
||||
|
||||
const element = document.createElement('div');
|
||||
element.className = 'terminal xterm';
|
||||
const screen = document.createElement('div');
|
||||
screen.className = 'xterm-screen';
|
||||
const rowsEl = document.createElement('div');
|
||||
rowsEl.className = 'xterm-rows';
|
||||
element.appendChild(screen);
|
||||
screen.appendChild(rowsEl);
|
||||
document.body.appendChild(element);
|
||||
|
||||
const hybrid = {
|
||||
element,
|
||||
get cols() {
|
||||
return term.cols;
|
||||
},
|
||||
get rows() {
|
||||
return term.rows;
|
||||
},
|
||||
options: { fontFamily: 'monospace', fontSize: 14, fontWeight: 'normal', theme: {} },
|
||||
buffer: {
|
||||
active: {
|
||||
get viewportY() {
|
||||
return term.buffer.active.viewportY;
|
||||
},
|
||||
get baseY() {
|
||||
return term.buffer.active.baseY;
|
||||
},
|
||||
get cursorX() {
|
||||
return term.buffer.active.cursorX;
|
||||
},
|
||||
get cursorY() {
|
||||
return term.buffer.active.cursorY;
|
||||
},
|
||||
getLine: (y: number) => term.buffer.active.getLine(y),
|
||||
},
|
||||
},
|
||||
onWriteParsed: (cb: () => void) => term.onWriteParsed(cb),
|
||||
onResize: (cb: (s: { cols: number; rows: number }) => void) => term.onResize(cb),
|
||||
_core: {
|
||||
_renderService: {
|
||||
dimensions: {
|
||||
css: { cell: { width: CELL_W, height: CELL_H } },
|
||||
device: { char: { top: 0, height: CELL_H } },
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
return {
|
||||
hybrid: hybrid as unknown as XtermTerminal,
|
||||
term,
|
||||
write: (data: string) => new Promise<void>((resolve) => term.write(data, () => resolve())),
|
||||
cursorRowText() {
|
||||
const b = term.buffer.active;
|
||||
return b.getLine(b.baseY + b.cursorY)?.translateToString(true) ?? '';
|
||||
},
|
||||
rowText(viewportRow: number) {
|
||||
const b = term.buffer.active;
|
||||
return b.getLine(b.baseY + viewportRow)?.translateToString(true) ?? '';
|
||||
},
|
||||
spanCount() {
|
||||
return element.querySelectorAll('[data-predictive-echo] span').length;
|
||||
},
|
||||
spans() {
|
||||
return Array.from(element.querySelectorAll('[data-predictive-echo] span')) as HTMLSpanElement[];
|
||||
},
|
||||
cleanup() {
|
||||
term.dispose();
|
||||
element.remove();
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Codeman-side mirrors (keep in sync with terminal-ui.js) ────────────
|
||||
|
||||
/** Mirror of window.CodemanTerminalInput.classifyPredictInput. */
|
||||
export function classifyPredictInput(data: string): 'char' | 'backspace' | 'clear' | 'text' {
|
||||
const cps = Array.from(data);
|
||||
if (cps.length === 1) {
|
||||
const cp = cps[0].codePointAt(0)!;
|
||||
if (cp === 0x7f) return 'backspace';
|
||||
if (cp >= 0x20) return 'char';
|
||||
return 'clear';
|
||||
}
|
||||
if (data.charCodeAt(0) === 0x1b) return 'clear';
|
||||
if (data.charCodeAt(0) >= 0x20) return 'text';
|
||||
return 'clear';
|
||||
}
|
||||
|
||||
/** Mirror of the codex composer-row gate (CODEX_COMPOSER_ROW_RE). */
|
||||
export const CODEX_COMPOSER_ROW_RE = /^› /;
|
||||
|
||||
export function codexComposerGate(terminal: XtermTerminal): boolean {
|
||||
try {
|
||||
const buf = terminal.buffer.active;
|
||||
const line = buf.getLine(buf.baseY + (buf.cursorY ?? 0));
|
||||
return !!line && CODEX_COMPOSER_ROW_RE.test(line.translateToString(true));
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -26,6 +26,13 @@ export default defineConfig([
|
||||
' this.activate(terminal);',
|
||||
' }',
|
||||
' };',
|
||||
' window.PredictiveEchoAddon=XtermZerolagInput.PredictiveEchoAddon;',
|
||||
' window.PredictiveEchoOverlay=class extends XtermZerolagInput.PredictiveEchoAddon{',
|
||||
' constructor(terminal){',
|
||||
' super({});',
|
||||
' this.activate(terminal);',
|
||||
' }',
|
||||
' };',
|
||||
'}',
|
||||
].join('\n'),
|
||||
},
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user