diff --git a/.changeset/release-thanks.md b/.changeset/release-thanks.md index fc475716..2df367b5 100644 --- a/.changeset/release-thanks.md +++ b/.changeset/release-thanks.md @@ -7,3 +7,4 @@ - @irisitymichaelgrundberg for three terminal fixes in one release: keeping the output a pane capture could not contain (#436), replaying a capture at the geometry it was taken at (#435, five rounds and a Playwright suite that fails against the merge base), and trimming the padding out of a copied selection (#451), where the scan-instead-of-regex call avoided a 2.9s freeze nobody would have traced back to a copy. - @timkjr for a first contribution that found a real silent failure: the Instance count stepper next to the Run button had only ever applied to Claude, so on the other eight run modes it launched one session and said nothing (#454). - @Randalix for Wake-on-LAN on remote hosts (#439), built and live-tested against a real sleeping machine, and for reading the whole diff again between rounds rather than only the parts that were asked about. +- @opticon454 for turning #393's backend-only custom model endpoints into the whole feature (#430), and for validating it against a real llama-swap box rather than against the tests: the `/props` versus `/running` context discrepancy and the DeepSeek `/v1` root cause were both tracked down to the SDK source instead of guessed at. diff --git a/CLAUDE.md b/CLAUDE.md index e32b1407..88a7d4b3 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -383,11 +383,11 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L ### SSE Event Registry -160 event constants in `src/web/sse-events.ts` (backend) and `SSE_EVENTS` in `constants.js` (frontend). **Both must be kept in sync**, and `test/sse-registry-parity.test.ts` is the guard that pins it (currently exactly in sync, 160 = 160, no drift either direction). ⚠️ `hook:agent_working` is the one hook event with no Claude Code hook behind it — the DeepSeek status bridge reports it (see External CLI modes). The backend file's `@fileoverview` carries the per-category breakdown, including the two Web tab events. +161 event constants in `src/web/sse-events.ts` (backend) and `SSE_EVENTS` in `constants.js` (frontend). **Both must be kept in sync**, and `test/sse-registry-parity.test.ts` is the guard that pins it (currently exactly in sync, 161 = 161, no drift either direction). ⚠️ `hook:agent_working` is the one hook event with no Claude Code hook behind it — the DeepSeek status bridge reports it (see External CLI modes). The backend file's `@fileoverview` carries the per-category breakdown, including the two Web tab events. ### API Routes -~235 handlers across 27 route files in `src/web/routes/`: system (56), sessions (37), cases (34), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), custom-model (5), reboot-restore (3), me (2), teams (2), tab-layout (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details. +~236 handlers across 27 route files in `src/web/routes/`: system (56), sessions (37), cases (34), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), custom-model (6), reboot-restore (3), me (2), teams (2), tab-layout (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details. **HTTP contract** (stable since 0.9.x, see `docs/versioning-policy.md`; full envelope/status/error-code/SSE spec in `docs/api-reference.md`): responses use the `ApiResponse` envelope — `{ success: true, data? }` or `{ success: false, error, errorCode }` (`src/types/api.ts`). `/api/v1/*` is a versioned alias of `/api/*` (URL rewrite in `server.ts`). diff --git a/src/custom-model-hosts.ts b/src/custom-model-hosts.ts index 61ef75f7..bd90cef7 100644 --- a/src/custom-model-hosts.ts +++ b/src/custom-model-hosts.ts @@ -68,7 +68,7 @@ export interface CustomModelHost { * llama.cpp") — a hand-configured profile's own description has no such figure and * correctly gets no entry, never a guess. Used only to label the Run-menu picker's * "loading model" banner with a rough, unmeasured expected-time estimate - * (`estimateModelLoad()` in session-ui.js) — never a guarantee, and never anything a + * (the Run-menu picker's loading banner in session-ui.js) — never a guarantee, and never anything a * server-side check relies on. */ modelSizesGB?: Record; diff --git a/src/web/routes/custom-model-routes.ts b/src/web/routes/custom-model-routes.ts index 8c49b046..fc86d883 100644 --- a/src/web/routes/custom-model-routes.ts +++ b/src/web/routes/custom-model-routes.ts @@ -416,6 +416,14 @@ function parseBackendLogDataEvent(dataLine: string): string | undefined { * line (`\n\n`), buffered the same way `/running`'s NDJSON-shaped siblings buffer partial * chunks — a frame split across two `reader.read()` calls must not be parsed early. */ +/** + * Cap on the unparsed remainder held between reads of the backend log stream. One + * SSE frame is a status line, so this is orders of magnitude more than a real frame + * needs; it exists so a server that never emits a frame boundary cannot grow the + * buffer without bound for the life of the connection. + */ +const MAX_LOG_TAIL_BUFFER_CHARS = 64 * 1024; + async function pumpLlamaSwapLogTail( host: Pick, entry: LlamaSwapLogTail @@ -435,6 +443,12 @@ async function pumpLlamaSwapLogTail( buffer += decoder.decode(value, { stream: true }); const frames = buffer.split('\n\n'); buffer = frames.pop() ?? ''; + // The remainder only shrinks at a frame boundary, so a server that streams + // without `\n\n` (or one very long frame) would grow it for as long as the + // connection is held, which is indefinitely by design. Past the cap the + // partial frame cannot become a useful log line anyway, so drop it and + // resynchronise on the next boundary rather than buffering forever. + if (buffer.length > MAX_LOG_TAIL_BUFFER_CHARS) buffer = ''; for (const frame of frames) { const dataLine = frame.split('\n').find((l) => l.startsWith('data:')); if (!dataLine) continue; @@ -807,7 +821,7 @@ export function registerCustomModelRoutes(app: FastifyInstance): void { return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed'); } const status = await getLlamaSwapStatus(host); - // Only worth tailing /logs once llama-swap is actually confirmed — a plain + // Only worth tailing /api/events once llama-swap is actually confirmed — a plain // llama.cpp/OpenAI-compatible server has no such endpoint at all. const logLine = status.isLlamaSwap ? getLatestLlamaSwapLogLine(host) : undefined; // `cmd` (the literal llama-server launch line, which can carry model paths and diff --git a/src/web/server.ts b/src/web/server.ts index 96bcf0e9..40161727 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -2837,7 +2837,7 @@ export class WebServer extends EventEmitter { .catch((err) => { console.error('[custom-model] swap-displacement check failed:', getErrorMessage(err)); }); - // Same cadence, unrelated concern: close any /logs tail (see + // Same cadence, unrelated concern: close any /api/events tail (see // getLatestLlamaSwapLogLine) nothing has polled in a while, so a loading banner // that finished (or was abandoned) doesn't leave a connection open forever. pruneIdleLlamaSwapLogTails();