From f865f74a0fde2c70b9e501f769240ee5a79b29cc Mon Sep 17 00:00:00 2001 From: Devvyn <22340871+opticon454@users.noreply.github.com> Date: Wed, 16 Sep 2026 15:04:10 +0800 Subject: [PATCH] feat(custom-model): launch directly on the endpoint, no restart, for 7 of 8 CLIs Fixes the visible double-launch reported on Codex: picking a custom-model Run-menu entry launched natively first, waited for it to settle, then restarted it in place with the endpoint applied. Necessary for the design at the time, but visibly a native boot immediately followed by a second one - worst on a CLI whose TUI fully reinitializes on a restart, confirmed live on Codex. POST /api/quick-start gains an optional customModel field ({endpointId, modelId, confirmed?}). When present, the route mints the session's id itself (crypto.randomUUID()) before constructing it, computes the same injection the existing POST /api/sessions/:id/custom-model route computes (including the llama-swap conflict check from the last commit - same {requiresConfirmation, currentlyLoadedModel, affectedSessions} shape, no session created until confirmed), and launches the session already pointed at the endpoint: env vars via the constructor, and the launchModel override merged onto piConfig/grokConfig/ompConfig using the registry's own launch.legacyConfigField the same way session.ts's restart path already does. No restart at all - setCustomModel() afterward is bookkeeping only. Wired into 7 of 8 launch functions (session-ui.js): openCode, codex, gemini, pi, grok, deepseek, omp. Claude stays on the original launch-then-restart path for now: its own --resume-based restart is far less jarring than the other seven's, and runClaude()'s multi-tab launch plus docker-config-drift confirm/retry loop make folding it into the one-shot path separate, higher-risk work than the other seven's each-a-single-simple-launch shape. Also fixes a pre-existing 'mode === omp' branch flagged by the CLI-id static guard (test/cli-registry-no-id-branching.test.ts) - the ompConfig launchModel merge is the same 'legacy Config plumbing' category as the six sibling branches already allowlisted there, just newly literal where it was previously only inside resolveOmpConfigForCreate's own check. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG --- .changeset/run-menu-custom-model-picker.md | 2 + docs/custom-model-endpoints.md | 63 ++++- docs/wiki/Custom-Model-Endpoints.md | 13 +- src/web/public/session-ui.js | 243 ++++++++++------ src/web/routes/session-routes.ts | 142 +++++++++- src/web/schemas.ts | 19 ++ test/cli-registry-no-id-branching.test.ts | 2 + test/custom-model-one-shot-launch.test.ts | 238 ++++++++++++++++ test/routes/quick-start-custom-model.test.ts | 277 +++++++++++++++++++ 9 files changed, 879 insertions(+), 120 deletions(-) create mode 100644 test/custom-model-one-shot-launch.test.ts create mode 100644 test/routes/quick-start-custom-model.test.ts diff --git a/.changeset/run-menu-custom-model-picker.md b/.changeset/run-menu-custom-model-picker.md index 3f331a4e..bedbdd8c 100644 --- a/.changeset/run-menu-custom-model-picker.md +++ b/.changeset/run-menu-custom-model-picker.md @@ -13,3 +13,5 @@ Everything below was found and fixed against a **real llama-swap server**, not j - **The real root cause of "it still says opus, not my model."** llama.cpp runs exactly one model at a time; llama-swap unloads and reloads it on demand, which can take anywhere from a few seconds to well over a minute — long enough that a session mid-swap is indistinguishable from one that never left the native backend. Applying a selection now checks llama-swap's own `GET /running` first (feature-detected; a plain llama.cpp/OpenAI-compatible server has no such endpoint and is never checked); if switching would unload a model **another live session is actively using**, the apply is refused with a warning naming that session instead of silently switching, and a confirmation retry proceeds anyway. Either way, a sticky "loading model…" toast now covers the actual swap window until llama-swap reports the target model ready, so a prompt sent mid-swap reads as "loading," never as silence or an answer from whatever was loaded a moment before. Remote (SSH) and Docker sessions are refused for now (400) — their restart reattaches the durable remote/in-container tmux rather than relaunching the agent. + +**One more, from watching it launch live: opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP now launch directly on the endpoint, with no restart at all.** Picking one of these seven from the Run-menu picker used to launch natively first, wait for it to settle, then restart it in place with the endpoint applied — a deliberate two-step design, but visibly a native boot immediately followed by a second one, worst on a CLI whose TUI fully reinitializes on a restart (confirmed live on Codex). `POST /api/quick-start` now accepts a `customModel` field and computes the same injection *before* the session exists, launching straight onto the endpoint the first time — no visible relaunch, and it also runs the same llama-swap conflict check (warns before unloading a model another live session is using) at create time. Claude still uses the original launch-then-restart path for now (its own `--resume`-based restart is far less jarring, and `runClaude()`'s multi-tab and docker-config-drift-retry logic make folding it into the one-shot path separate work). diff --git a/docs/custom-model-endpoints.md b/docs/custom-model-endpoints.md index 37398806..0cb72f15 100644 --- a/docs/custom-model-endpoints.md +++ b/docs/custom-model-endpoints.md @@ -119,26 +119,59 @@ model it runs straight away; with two or more, a small modal (`#customModelPickModal`) lists them and asks which one to use for this launch, with the endpoint's `defaultModelId` marked but not auto-chosen — the point of asking is letting one launch deliberately differ from the -saved default, not just confirming it. Whichever way the model was decided, -the launch itself runs a single session on that harness exactly the way its -own Run-menu entry would (same case creation, env overrides, everything), -then **waits for the new session to go idle** (`GET .../wait?until=idle`, -bounded at 20s — a normal 200 either way, never an error, per the wait -endpoint's own contract) before applying the endpoint and model to it via -the route below. That wait exists because a freshly launched CLI reports -itself as `busy` for its own startup (a boot spinner, a workspace-trust -check) well before the apply call would otherwise reach it, and the apply -route correctly refuses to restart a session mid-turn — a fresh boot looks -exactly like one from the outside. A session still busy after the wait -reaches the apply call anyway and gets that route's own honest -`SESSION_BUSY` error, now visible as a sticky toast with a close button -rather than a generic message that vanished in three seconds. It is a +saved default, not just confirming it. + +**How the launch itself applies the endpoint depends on the harness.** For +opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP (`runCustomModelEntry` → +`_runCustomModelEntryOneShot`), the endpoint/model is folded into the SAME +`POST /api/quick-start` call that creates the session (`customModel` field), +so the session launches directly on the endpoint — no restart, no visible +relaunch. Claude (`_runCustomModelEntryViaRestart`) still uses the original +two-step design: the launch runs a single native session exactly the way its +own Run-menu entry would, then **waits for the new session to go idle** +(`GET .../wait?until=idle`, bounded at 20s — a normal 200 either way, never +an error, per the wait endpoint's own contract) before applying the endpoint +via the restart route below. That wait exists because a freshly launched CLI +reports itself as `busy` for its own startup (a boot spinner, a +workspace-trust check) well before the apply call would otherwise reach it, +and the apply route correctly refuses to restart a session mid-turn — a +fresh boot looks exactly like one from the outside. A session still busy +after the wait reaches the apply call anyway and gets that route's own +honest `SESSION_BUSY` error, now visible as a sticky toast with a close +button rather than a generic message that vanished in three seconds. Claude +stays on this path because its own restart (`--resume`-based, keeping the +conversation) is far less jarring than the other seven's, and `runClaude()`'s +multi-tab launch and docker-config-drift confirm/retry loop make folding it +into the one-shot path separate work. It is a one-off "try this endpoint" action, not a sticky mode: the plain Run button still means "this harness, native cloud" afterward. Entries are hidden entirely for a remote or Docker active case, since the apply route refuses both (see the next section). -## Applying a model to a session +## Launching directly on an endpoint (no restart) + +```bash +curl -sk -X POST https://localhost:3000/api/quick-start \ + -H 'Content-Type: application/json' \ + -d '{"caseName": "myapp", "mode": "codex", "customModel": {"endpointId": "llama-box", "modelId": "qwen3"}}' +``` + +`POST /api/quick-start`'s `customModel` field (`{endpointId, modelId, +confirmed?}`) computes the same injection the restart route below does, but +BEFORE the session exists — the session is minted its own id up front +(`crypto.randomUUID()`), the injection (env vars, and for a `configDir`-kind +CLI, the written config file) targets that real id, and the session launches +already pointed at the endpoint. No restart, because there was never a +native-backend launch to restart away from. Runs the same llama-swap +conflict check as the restart route (below) — a `409`-shaped +`{requiresConfirmation, currentlyLoadedModel, affectedSessions}` response +with no session created, resolved by retrying with `confirmed: true` — and +is refused the same way for a remote or Docker case. This is what the +Run-menu picker uses for opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP; +Claude still uses the restart route below (see "The Run-menu picker" above +for why). + +## Applying a model to an ALREADY-RUNNING session ```bash curl -sk -X POST https://localhost:3000/api/sessions//custom-model \ diff --git a/docs/wiki/Custom-Model-Endpoints.md b/docs/wiki/Custom-Model-Endpoints.md index 6f90aa68..755f861d 100644 --- a/docs/wiki/Custom-Model-Endpoints.md +++ b/docs/wiki/Custom-Model-Endpoints.md @@ -52,12 +52,15 @@ small dialog asks which one to use for this launch before starting the session; endpoint's default model, if set, is marked but not auto-picked, so a launch can deliberately use a different one without changing the saved default. -Applying a selection **restarts the harness's process in place** — same tab, same -conversation where the harness supports resuming one, fresh environment. That restart is -necessary, not incidental: every supported harness reads its endpoint config at process -start, never per turn, so there is no live hot-swap while a turn is running. +**For opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP, picking an entry launches +straight onto the endpoint** — no restart, because the endpoint is applied before the +session's process ever starts. **Claude still restarts the harness's process in place** — +same tab, same conversation (`--resume`) — after a normal native launch, since that restart +is far less jarring for Claude than for the other seven, whose own TUI can fully +reinitialize on a restart. Either way, every supported harness reads its endpoint config at +process start, never per turn, so there is no live hot-swap while a turn is running. -Picking an entry that launches a **brand-new** session waits (up to 20 seconds) for it to +Picking an entry that launches a **brand-new** Claude session waits (up to 20 seconds) for it to finish its own startup before applying — a freshly started CLI reports itself as busy for its boot sequence, and applying to a genuinely busy session is refused so a real, in-progress turn is never interrupted out from under you. A session that is still busy after that wait diff --git a/src/web/public/session-ui.js b/src/web/public/session-ui.js index 8ec63e54..aee49b4e 100644 --- a/src/web/public/session-ui.js +++ b/src/web/public/session-ui.js @@ -694,7 +694,96 @@ Object.assign(CodemanApp.prototype, { * default, which a one-off endpoint run must not do — and is restored in * `finally` even if run() throws. */ + /** + * Dispatches to the ONE-SHOT launch path (below) for every custom-model-eligible CLI + * except claude, which still goes through the restart-after-native-boot path + * (`_runCustomModelEntryViaRestart`): claude's own `runClaude()` carries multi-tab + * launch and a docker-config-drift confirm/retry loop neither of the other seven + * functions has, and folding those into the one-shot flow is unstarted, separate work. + * The other seven (opencode/codex/gemini/pi/grok/deepseek/omp) are each a single, + * simple launch, so they get the one-shot path — the one visibly worth it, since a + * native-boot-then-restart is far more jarring on a CLI whose TUI fully reinitializes + * (Codex, confirmed live) than on claude's own `--resume`-based restart. + */ async runCustomModelEntry(mode, endpointId, modelId) { + if (mode === 'claude') { + return this._runCustomModelEntryViaRestart(mode, endpointId, modelId); + } + return this._runCustomModelEntryOneShot(mode, endpointId, modelId); + }, + + /** + * Launches directly on the endpoint — no restart, so no visible relaunch. Stashes the + * pick on `_pendingCustomModelForLaunch` for the targeted run() function to read + * and fold into its own /api/quick-start body (see `_quickStartWithCustomModelConfirm`); + * cleared in `finally` the same way `_runMode`'s temporary swap is, even if run() throws. + */ + async _runCustomModelEntryOneShot(mode, endpointId, modelId) { + document.getElementById('runModeMenu')?.classList.remove('active'); + const previousRunMode = this._runMode; + const tabCountEl = document.getElementById('tabCount'); + const prevTabCount = tabCountEl?.value; + this._runMode = mode; + this._pendingCustomModelForLaunch = { endpointId, modelId }; + if (tabCountEl) tabCountEl.value = '1'; + try { + await this.run(); + } finally { + this._runMode = previousRunMode; + this._pendingCustomModelForLaunch = undefined; + if (tabCountEl && prevTabCount !== undefined) tabCountEl.value = prevTabCount; + } + + // run() (via _quickStartWithCustomModelConfirm) reports its own launch error or + // cancellation via toast and leaves this unset — nothing more to do here then. + const result = this._lastCustomModelLaunchResult; + this._lastCustomModelLaunchResult = undefined; + if (result?.modelSwapInProgress) { + void this._watchLlamaSwapLoading(endpointId, modelId); + } + }, + + /** + * POSTs a /api/quick-start body already carrying `customModel` (see the run() + * call sites below), showing the same llama-swap "this will unload it for session X" + * warning the restart path's `_applyCustomModelToSession` shows when the route asks + * for confirmation, and retrying with `confirmed: true` on accept. Stashes the final + * response's payload on `_lastCustomModelLaunchResult` for + * `_runCustomModelEntryOneShot` to read `modelSwapInProgress` off afterward — run()'s + * eleven per-mode dispatch targets have no shared return-value contract of their own, + * so a side channel here is simpler than threading one through every one of them. + */ + async _quickStartWithCustomModelConfirm(bodyObj) { + const post = async (body) => { + const res = await fetch('/api/quick-start', { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify(body), + }); + return res.json(); + }; + let data = await post(bodyObj); + if (data?.data?.requiresConfirmation) { + const { currentlyLoadedModel, affectedSessions } = data.data; + const names = affectedSessions.map((s) => s.name || s.id).join(', '); + const proceed = confirm( + `${names} ${affectedSessions.length === 1 ? 'is' : 'are'} currently using ` + + `${currentlyLoadedModel} on this endpoint. Switching will unload it for ` + + `${affectedSessions.length === 1 ? 'that session' : 'those sessions'} too. Continue?` + ); + if (!proceed) { + this._lastCustomModelLaunchResult = undefined; + return { success: false, error: 'Model switch cancelled' }; + } + data = await post({ ...bodyObj, customModel: { ...bodyObj.customModel, confirmed: true } }); + } + this._lastCustomModelLaunchResult = data?.success !== false ? data?.data : undefined; + return data; + }, + + /** The restart-after-native-boot path — see `runCustomModelEntry`'s own comment for + * which CLIs still use this one. */ + async _runCustomModelEntryViaRestart(mode, endpointId, modelId) { document.getElementById('runModeMenu')?.classList.remove('active'); const previousRunMode = this._runMode; @@ -1583,20 +1672,16 @@ Object.assign(CodemanApp.prototype, { // Quick-start with opencode mode (auto-allow tools by default). // No `effort` field — it's Claude-specific (OpenCode has no /effort). const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const res = await fetch('/api/quick-start', { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ - caseName, - mode: 'opencode', - sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, - ...(isRemote ? {} : { - openCodeConfig: { autoAllowTools: true }, - ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), - }), - }) + const data = await this._quickStartWithCustomModelConfirm({ + caseName, + mode: 'opencode', + sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, + ...(isRemote ? {} : { + openCodeConfig: { autoAllowTools: true }, + ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), + ...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}), + }), }); - const data = await res.json(); if (!data.success) throw new Error(data.error || 'Failed to start OpenCode'); await this._ensureCreatedSessionVisible(data.data.sessionId, data.data.session); @@ -1637,24 +1722,20 @@ Object.assign(CodemanApp.prototype, { const globalSettings = this.loadAppSettingsFromStorage(); const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), globalSettings); - const res = await fetch('/api/quick-start', { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ - caseName, - mode: 'codex', - sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, - ...(isRemote ? {} : { - codexConfig: { - dangerouslyBypassApprovals: globalSettings.codexDangerouslyBypassApprovals ?? false, - animations: globalSettings.codexAnimationsEnabled ?? false, - renderMode: 'hybrid', - }, - ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), - }), - }) + const data = await this._quickStartWithCustomModelConfirm({ + caseName, + mode: 'codex', + sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, + ...(isRemote ? {} : { + codexConfig: { + dangerouslyBypassApprovals: globalSettings.codexDangerouslyBypassApprovals ?? false, + animations: globalSettings.codexAnimationsEnabled ?? false, + renderMode: 'hybrid', + }, + ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), + ...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}), + }), }); - const data = await res.json(); if (!data.success) throw new Error(data.error || 'Failed to start Codex'); await this._ensureCreatedSessionVisible(data.data.sessionId, data.data.session); @@ -1694,20 +1775,16 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const res = await fetch('/api/quick-start', { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ - caseName, - mode: 'gemini', - sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, - ...(isRemote ? {} : { - geminiConfig: { approvalMode: 'yolo' }, - ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), - }), - }) + const data = await this._quickStartWithCustomModelConfirm({ + caseName, + mode: 'gemini', + sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, + ...(isRemote ? {} : { + geminiConfig: { approvalMode: 'yolo' }, + ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), + ...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}), + }), }); - const data = await res.json(); if (!data.success) throw new Error(data.error || 'Failed to start Gemini'); await this._ensureCreatedSessionVisible(data.data.sessionId, data.data.session); @@ -1805,17 +1882,13 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const res = await fetch('/api/quick-start', { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ - caseName, - mode: 'pi', - sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, - ...(isRemote || Object.keys(envOverrides).length === 0 ? {} : { envOverrides }), - }) + const data = await this._quickStartWithCustomModelConfirm({ + caseName, + mode: 'pi', + sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, + ...(isRemote || Object.keys(envOverrides).length === 0 ? {} : { envOverrides }), + ...(!isRemote && this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}), }); - const data = await res.json(); if (!data.success) throw new Error(data.error || 'Failed to start Pi'); await this._ensureCreatedSessionVisible(data.data.sessionId, data.data.session); @@ -1853,19 +1926,15 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const res = await fetch('/api/quick-start', { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ - caseName, - mode: 'omp', - sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, - ...(isRemote ? {} : { - ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), - }), - }) + const data = await this._quickStartWithCustomModelConfirm({ + caseName, + mode: 'omp', + sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, + ...(isRemote ? {} : { + ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), + ...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}), + }), }); - const data = await res.json(); if (!data.success) throw new Error(data.error || 'Failed to start OMP'); await this._ensureCreatedSessionVisible(data.data.sessionId, data.data.session); @@ -1912,20 +1981,16 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const res = await fetch('/api/quick-start', { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ - caseName, - mode: 'grok', - sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, - ...(isRemote ? {} : { - grokConfig: { alwaysApprove: true }, - ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), - }), - }) + const data = await this._quickStartWithCustomModelConfirm({ + caseName, + mode: 'grok', + sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, + ...(isRemote ? {} : { + grokConfig: { alwaysApprove: true }, + ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), + ...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}), + }), }); - const data = await res.json(); if (!data.success) throw new Error(data.error || 'Failed to start Grok'); await this._ensureCreatedSessionVisible(data.data.sessionId, data.data.session); @@ -1990,20 +2055,16 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const res = await fetch('/api/quick-start', { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ - caseName, - mode: 'deepseek', - sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, - ...(isRemote ? {} : { - deepSeekConfig: { permissionMode: 'danger-full-access' }, - ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), - }), - }) + const data = await this._quickStartWithCustomModelConfirm({ + caseName, + mode: 'deepseek', + sessionName: `w${this._nextCaseSessionStartNumber(caseName)}-${caseName}`, + ...(isRemote ? {} : { + deepSeekConfig: { permissionMode: 'danger-full-access' }, + ...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}), + ...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}), + }), }); - const data = await res.json(); if (!data.success) throw new Error(data.error || 'Failed to start DeepSeek'); await this._ensureCreatedSessionVisible(data.data.sessionId, data.data.session); diff --git a/src/web/routes/session-routes.ts b/src/web/routes/session-routes.ts index f927dff9..8db66859 100644 --- a/src/web/routes/session-routes.ts +++ b/src/web/routes/session-routes.ts @@ -11,7 +11,7 @@ import { homedir } from 'node:os'; import { existsSync, statSync, mkdirSync, writeFileSync } from 'node:fs'; import { execFile } from 'node:child_process'; import fs from 'node:fs/promises'; -import { randomBytes } from 'node:crypto'; +import { randomBytes, randomUUID } from 'node:crypto'; import { performance } from 'node:perf_hooks'; import { ApiErrorCode, @@ -3136,6 +3136,7 @@ export function registerSessionRoutes( effort, parentSessionId, agentOrigin, + customModel, } = parseBody(QuickStartSchema, req.body); // Resolved ONCE here: the same value labels a case directory this request creates @@ -3188,11 +3189,12 @@ export function registerSessionRoutes( grokConfig || deepSeekConfig || ompConfig || - openCodeConfig + openCodeConfig || + customModel ) { return createErrorResponse( ApiErrorCode.INVALID_INPUT, - 'envOverrides, effort, modelOverride, and per-CLI config are not supported for remote cases (they do not cross ssh). Configure the remote command via the host command override instead.' + 'envOverrides, effort, modelOverride, per-CLI config, and custom model endpoints are not supported for remote cases (they do not cross ssh). Configure the remote command via the host command override instead.' ); } @@ -3223,11 +3225,12 @@ export function registerSessionRoutes( grokConfig || deepSeekConfig || ompConfig || - openCodeConfig + openCodeConfig || + customModel ) { return createErrorResponse( ApiErrorCode.INVALID_INPUT, - 'envOverrides, effort, and per-CLI config are not supported for docker cases (they do not cross into the container). Configure the container via the docker host command override instead.' + 'envOverrides, effort, per-CLI config, and custom model endpoints are not supported for docker cases (they do not cross into the container). Configure the container via the docker host command override instead.' ); } @@ -3515,7 +3518,106 @@ export function registerSessionRoutes( ); const qsTerminalHistoryConfig = await ctx.getTerminalHistoryConfig(); const qsGatedEnvOverrides = await clampEnvOverridesForOwner(owner, envOverrides); + const qsResolvedOmpConfig = resolveOmpConfigForCreate(mode, resolvedCasePath, ompConfig); + + // Custom Model Endpoint Profiles, applied AT CREATE TIME (docs/custom-model-endpoints-plan.md) + // rather than via the dedicated restart-in-place route (POST /api/sessions/:id/custom- + // model, still what an ALREADY-RUNNING session uses to switch later): computing the + // injection before the process exists and launching directly on it avoids the visible + // native-boot-then-restart the restart-after-launch design otherwise shows on every + // custom-model run — most jarring on a CLI like Codex whose TUI fully reinitializes. + // Mirrors the dedicated route's own checks (llama-swap conflict, unsupported CLI, + // unknown endpoint, a model id the CLI's argv pattern can't carry) rather than trusting + // a lighter version of them, since this is the same server-side authority reached a + // different way, not a separate, less-checked path. + let qsCustomModelEnvOverrides = qsGatedEnvOverrides; + let qsCustomModelLaunchModel: string | undefined; + let qsCustomModelSessionId: string | undefined; + let qsCustomModelBookkeeping: + | { + endpointId: string; + modelId: string; + label?: string; + envKeys: string[]; + configDir?: string; + launchModel?: string; + } + | undefined; + if (customModel) { + const cmEntry = getCli(mode); + if (!cmEntry) return createErrorResponse(ApiErrorCode.INVALID_INPUT, `No CLI registry entry for mode ${mode}`); + if (cmEntry.capabilities.customModelInjection.kind === 'unsupported') { + return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${mode} has no known custom-model mechanism`); + } + const cmHosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR); + const cmEndpoint = cmHosts.find((h) => h.id === customModel.endpointId); + if (!cmEndpoint) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found'); + + // See the dedicated route's own comment for the full reasoning: llama.cpp runs one + // model at a time, llama-swap swaps on demand, and switching away from what another + // live session is actively using deserves a warning, not a silent switch. There is no + // "self" to exclude from the affected-sessions scan here — this session doesn't exist + // yet. + const cmSwapStatus = await getLlamaSwapStatus(cmEndpoint); + const cmCurrentlyLoaded = + cmSwapStatus.running.find((r) => r.state === 'ready')?.model ?? cmSwapStatus.running[0]?.model; + const cmSwapNeeded = cmSwapStatus.isLlamaSwap && !!cmCurrentlyLoaded && cmCurrentlyLoaded !== customModel.modelId; + if (cmSwapNeeded && !customModel.confirmed) { + const cmAffectedSessions = [...ctx.sessions.values()] + .filter((s) => s.customModel?.endpointId === cmEndpoint.id && s.customModel?.modelId === cmCurrentlyLoaded) + .map((s) => ({ id: s.id, name: s.name })); + if (cmAffectedSessions.length > 0) { + return { + requiresConfirmation: true, + currentlyLoadedModel: cmCurrentlyLoaded, + affectedSessions: cmAffectedSessions, + }; + } + } + + // Minted ourselves (rather than left to Session's own default) so the injection + // below — and any configDir it writes — can target the REAL id the session launches + // with, not a placeholder: `new Session({ id: ... })` accepts an explicit id for + // exactly this reason. + qsCustomModelSessionId = randomUUID(); + const cmContextLength = cmEndpoint.modelContextLengths?.[customModel.modelId]; + const cmApplied = applyCustomModelInjection( + cmEntry, + cmEndpoint, + customModel.modelId, + qsCustomModelSessionId, + cmContextLength + ); + if (!cmApplied) { + return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${mode} has no known custom-model mechanism`); + } + const cmModelSpec = cmEntry.launch.params.model; + if ( + cmApplied.launchModel !== undefined && + cmModelSpec?.type === 'token' && + !matchesPattern(cmModelSpec.pattern, cmApplied.launchModel) + ) { + removeConfigDir(cmApplied.configDir); + return createErrorResponse( + ApiErrorCode.INVALID_INPUT, + `Model id ${JSON.stringify(customModel.modelId)} cannot be passed to ${mode} on its command line` + ); + } + + qsCustomModelEnvOverrides = { ...qsGatedEnvOverrides, ...cmApplied.envOverrides }; + qsCustomModelLaunchModel = cmApplied.launchModel; + qsCustomModelBookkeeping = { + endpointId: cmEndpoint.id, + modelId: customModel.modelId, + label: cmEndpoint.label, + envKeys: cmApplied.envKeys, + configDir: cmApplied.configDir, + launchModel: cmApplied.launchModel, + }; + } + const session = new Session({ + id: qsCustomModelSessionId, workingDir: resolvedCasePath, name: sessionName ? sessionName.slice(0, MAX_SESSION_NAME_LENGTH) : '', mux: ctx.mux, @@ -3530,11 +3632,24 @@ export function registerSessionRoutes( codexConfig: mode === 'codex' ? qsGatedCodexConfig : undefined, geminiConfig: mode === 'gemini' ? qsGatedGeminiConfig : undefined, antigravityConfig: mode === 'antigravity' ? qsGatedAntigravityConfig : undefined, - piConfig: mode === 'pi' ? qsGatedPiConfig : undefined, - grokConfig: mode === 'grok' ? qsGatedGrokConfig : undefined, + piConfig: + mode === 'pi' + ? qsCustomModelLaunchModel !== undefined + ? { ...(qsGatedPiConfig ?? {}), model: qsCustomModelLaunchModel } + : qsGatedPiConfig + : undefined, + grokConfig: + mode === 'grok' + ? qsCustomModelLaunchModel !== undefined + ? { ...(qsGatedGrokConfig ?? {}), model: qsCustomModelLaunchModel } + : qsGatedGrokConfig + : undefined, deepSeekConfig: mode === 'deepseek' ? qsGatedDeepSeekConfig : undefined, - ompConfig: resolveOmpConfigForCreate(mode, resolvedCasePath, ompConfig), - envOverrides: qsGatedEnvOverrides, + ompConfig: + mode === 'omp' && qsCustomModelLaunchModel !== undefined + ? { ...(qsResolvedOmpConfig ?? {}), model: qsCustomModelLaunchModel } + : qsResolvedOmpConfig, + envOverrides: qsCustomModelEnvOverrides, effort, remote, docker, @@ -3543,6 +3658,15 @@ export function registerSessionRoutes( parentSessionId: qsParentSessionId, }); + // Records the selection for session.customModel/getCustomModelForPersist() and future + // clear/switch calls — the actual env vars and launch-model config are already part of + // the launch above (constructor envOverrides, piConfig/grokConfig/ompConfig.model), so + // this is bookkeeping only, never a restart: setCustomModel() is synchronous state, no + // tmux IO of its own (see its own doc comment in session.ts). + if (qsCustomModelBookkeeping) { + session.setCustomModel(qsCustomModelBookkeeping, qsCustomModelEnvOverrides); + } + // Auto-detect completion phrase from CLAUDE.md BEFORE broadcasting // so the initial state already has the phrase configured (only if globally enabled) if (getCli(mode)?.capabilities.ralph && !remote && !docker && ctx.store.getConfig().ralphEnabled) { diff --git a/src/web/schemas.ts b/src/web/schemas.ts index 119d2836..2b3ca0ec 100644 --- a/src/web/schemas.ts +++ b/src/web/schemas.ts @@ -1033,6 +1033,25 @@ export const QuickStartSchema = z.object({ * because it takes an existing `workingDir` and so never creates a directory to label. */ agentOrigin: z.string().max(64).optional(), + /** + * Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md): launches directly + * on this saved endpoint/model instead of the mode's native backend, computed server-side + * from the admin-configured endpoint store the same way `POST /api/sessions/:id/custom- + * model` does — never trusting raw env values from the client. One-shot, launch-time + * equivalent of that route: no restart, so no visible relaunch (that route's restart-in- + * place is still what an ALREADY-RUNNING session uses to switch later). Rejected for + * remote/docker cases, same reasoning as `envOverrides` above. `confirmed` mirrors that + * route's field: skips the llama-swap "this will unload it for another session" check on + * a deliberate retry. + */ + customModel: z + .object({ + endpointId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid endpoint id'), + modelId: z.string().min(1).max(200), + confirmed: z.boolean().optional(), + }) + .strict() + .optional(), }); // ========== Hook Events ========== diff --git a/test/cli-registry-no-id-branching.test.ts b/test/cli-registry-no-id-branching.test.ts index 8d1bc4de..b5ba1b3e 100644 --- a/test/cli-registry-no-id-branching.test.ts +++ b/test/cli-registry-no-id-branching.test.ts @@ -80,6 +80,8 @@ const ALLOWED_BRANCHES: Record = { "web/routes/session-routes.ts::mode === 'pi'": 'legacy Config plumbing', "web/routes/session-routes.ts::mode === 'grok'": 'legacy Config plumbing', "web/routes/session-routes.ts::mode === 'deepseek'": 'legacy Config plumbing', + "web/routes/session-routes.ts::mode === 'omp'": + 'legacy Config plumbing (custom-model launchModel merge onto ompConfig, same selection resolveOmpConfigForCreate already makes internally)', "web/server.ts::mode === 'opencode'": 'legacy Config plumbing (session recovery)', "web/server.ts::mode === 'codex'": 'legacy Config plumbing (session recovery)', "web/server.ts::mode === 'gemini'": 'legacy Config plumbing (session recovery)', diff --git a/test/custom-model-one-shot-launch.test.ts b/test/custom-model-one-shot-launch.test.ts new file mode 100644 index 00000000..f5f3903c --- /dev/null +++ b/test/custom-model-one-shot-launch.test.ts @@ -0,0 +1,238 @@ +/** + * @fileoverview Frontend tests for the one-shot custom-model launch path added to + * session-ui.js (docs/custom-model-endpoints-plan.md): `runCustomModelEntry` dispatches + * to `_runCustomModelEntryOneShot` for every custom-model-eligible CLI except claude, + * which launches directly on the endpoint (no restart) by folding `customModel` into + * the run() function's own `/api/quick-start` body via `_pendingCustomModelForLaunch` + * and `_quickStartWithCustomModelConfirm`. Fixes the visible native-boot-then-restart the + * restart-after-launch path (`_runCustomModelEntryViaRestart`, still used for claude) + * showed on every custom-model run — confirmed live on Codex, whose TUI fully + * reinitializes on a restart. + * + * Uses the same JSDOM + `runScripts: "dangerously"` approach as + * test/custom-model-run-menu-ui.test.ts, extended with the DOM elements runCodex() (the + * CLI this was reported against) reads. + * + * Port: none. + */ +import { readFileSync } from 'node:fs'; +import { JSDOM } from 'jsdom'; +import { describe, expect, it } from 'vitest'; + +const CONSTANTS_JS = readFileSync(new URL('../src/web/public/constants.js', import.meta.url), 'utf-8'); +const SESSION_UI_JS = readFileSync(new URL('../src/web/public/session-ui.js', import.meta.url), 'utf-8'); + +function bootApp() { + const dom = new JSDOM( + ` + + + +
+ `, + { url: 'http://localhost/', runScripts: 'dangerously' } + ); + const win = dom.window as unknown as Window & typeof globalThis & { CodemanApp: new () => any }; + (win as unknown as { eval: (s: string) => void }).eval('window.CodemanApp = function CodemanApp() {};'); + (win as unknown as { eval: (s: string) => void }).eval(CONSTANTS_JS); + (win as unknown as { eval: (s: string) => void }).eval(SESSION_UI_JS); + const app = new win.CodemanApp(); + app.cases = [{ name: 'testcase' }]; + app.terminal = { focus: () => {} }; + app.loadAppSettingsFromStorage = () => ({}); + app.getCaseSettings = () => ({}); + app.buildEnvOverrides = () => ({}); + app.showToast = () => {}; + app._beginSessionLaunchStatus = () => 'status-token'; + app._reportSessionLaunchError = (_token: unknown, message: string) => { + app._lastReportedError = message; + }; + app._ensureCreatedSessionVisible = async () => {}; + app.selectSession = async () => {}; + app._nextCaseSessionStartNumber = () => 1; + return { win, app }; +} + +describe('runCustomModelEntry dispatch', () => { + it('routes claude through the restart-after-launch path', async () => { + const { app } = bootApp(); + let calledRestart = false; + let calledOneShot = false; + app._runCustomModelEntryViaRestart = async () => { + calledRestart = true; + }; + app._runCustomModelEntryOneShot = async () => { + calledOneShot = true; + }; + await app.runCustomModelEntry('claude', 'llama-box', 'qwen3'); + expect(calledRestart).toBe(true); + expect(calledOneShot).toBe(false); + }); + + it('routes every other custom-model-eligible CLI through the one-shot path', async () => { + for (const mode of ['opencode', 'codex', 'gemini', 'pi', 'grok', 'deepseek', 'omp']) { + const { app } = bootApp(); + let calledRestart = false; + let calledOneShot = false; + app._runCustomModelEntryViaRestart = async () => { + calledRestart = true; + }; + app._runCustomModelEntryOneShot = async () => { + calledOneShot = true; + }; + await app.runCustomModelEntry(mode, 'llama-box', 'qwen3'); + expect(calledRestart, mode).toBe(false); + expect(calledOneShot, mode).toBe(true); + } + }); +}); + +describe('_runCustomModelEntryOneShot', () => { + it('stashes the pick on _pendingCustomModelForLaunch for the duration of run(), then clears it', async () => { + const { app } = bootApp(); + let seenDuringRun: unknown; + app.run = async function (this: typeof app) { + seenDuringRun = this._pendingCustomModelForLaunch; + }; + await app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3'); + expect(seenDuringRun).toEqual({ endpointId: 'llama-box', modelId: 'qwen3' }); + expect(app._pendingCustomModelForLaunch).toBeUndefined(); + }); + + it('clears the pending pick even when run() throws', async () => { + const { app } = bootApp(); + app.run = async () => { + throw new Error('boom'); + }; + await expect(app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3')).rejects.toThrow('boom'); + expect(app._pendingCustomModelForLaunch).toBeUndefined(); + }); + + it('starts the loading watcher when the launch reports modelSwapInProgress', async () => { + const { app } = bootApp(); + app.run = async () => { + app._lastCustomModelLaunchResult = { modelSwapInProgress: true }; + }; + let watched: unknown[] | null = null; + app._watchLlamaSwapLoading = async (...args: unknown[]) => { + watched = args; + }; + await app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3'); + expect(watched).toEqual(['llama-box', 'qwen3']); + }); + + it('never starts the watcher when no swap was needed', async () => { + const { app } = bootApp(); + app.run = async () => { + app._lastCustomModelLaunchResult = { modelSwapInProgress: false }; + }; + let watchCalled = false; + app._watchLlamaSwapLoading = async () => { + watchCalled = true; + }; + await app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3'); + expect(watchCalled).toBe(false); + }); +}); + +describe('_quickStartWithCustomModelConfirm', () => { + function withFetch(win: Window & typeof globalThis, handler: (body: any) => any) { + (win as unknown as { fetch: typeof fetch }).fetch = (async (_url: string, opts: any) => ({ + json: async () => handler(JSON.parse(opts.body)), + })) as unknown as typeof fetch; + } + + it('returns the response directly when no confirmation is needed', async () => { + const { win, app } = bootApp(); + withFetch(win, (body) => ({ success: true, data: { sessionId: 's1', modelSwapInProgress: false, body } })); + const data = await app._quickStartWithCustomModelConfirm({ + mode: 'codex', + customModel: { endpointId: 'e', modelId: 'm' }, + }); + expect(data.success).toBe(true); + expect(data.data.sessionId).toBe('s1'); + expect(app._lastCustomModelLaunchResult).toEqual(data.data); + }); + + it('confirming re-sends with confirmed:true and returns the second response', async () => { + const { win, app } = bootApp(); + win.confirm = (() => true) as typeof win.confirm; + let calls = 0; + withFetch(win, (body) => { + calls += 1; + if (calls === 1) { + return { + success: true, + data: { + requiresConfirmation: true, + currentlyLoadedModel: 'llama3', + affectedSessions: [{ id: 's2', name: 'w2' }], + }, + }; + } + expect(body.customModel.confirmed).toBe(true); + return { success: true, data: { sessionId: 's1', modelSwapInProgress: true } }; + }); + const data = await app._quickStartWithCustomModelConfirm({ + mode: 'codex', + customModel: { endpointId: 'e', modelId: 'm' }, + }); + expect(calls).toBe(2); + expect(data.data.sessionId).toBe('s1'); + expect(app._lastCustomModelLaunchResult.modelSwapInProgress).toBe(true); + }); + + it('cancelling never re-sends, and reports a cancellation error', async () => { + const { win, app } = bootApp(); + win.confirm = (() => false) as typeof win.confirm; + let calls = 0; + withFetch(win, () => { + calls += 1; + return { + success: true, + data: { + requiresConfirmation: true, + currentlyLoadedModel: 'llama3', + affectedSessions: [{ id: 's2', name: 'w2' }], + }, + }; + }); + const data = await app._quickStartWithCustomModelConfirm({ + mode: 'codex', + customModel: { endpointId: 'e', modelId: 'm' }, + }); + expect(calls).toBe(1); + expect(data.success).toBe(false); + expect(data.error).toMatch(/cancelled/i); + expect(app._lastCustomModelLaunchResult).toBeUndefined(); + }); +}); + +describe('runCodex(): one-shot custom-model launch (the CLI this was reported against)', () => { + it('folds _pendingCustomModelForLaunch into the quick-start body as customModel', async () => { + const { win, app } = bootApp(); + (win as unknown as { fetch: typeof fetch }).fetch = (async (url: string, opts?: any) => { + if (url === '/api/codex/status') return { json: async () => ({ data: { available: true } }) }; + const body = JSON.parse(opts.body); + expect(body.customModel).toEqual({ endpointId: 'llama-box', modelId: 'qwen3' }); + return { json: async () => ({ success: true, data: { sessionId: 's1', modelSwapInProgress: false } }) }; + }) as unknown as typeof fetch; + + app._pendingCustomModelForLaunch = { endpointId: 'llama-box', modelId: 'qwen3' }; + await app.runCodex(); + expect(app._lastReportedError).toBeUndefined(); + }); + + it('omits customModel entirely for a plain (non-custom-model) Codex launch', async () => { + const { win, app } = bootApp(); + (win as unknown as { fetch: typeof fetch }).fetch = (async (url: string, opts?: any) => { + if (url === '/api/codex/status') return { json: async () => ({ data: { available: true } }) }; + const body = JSON.parse(opts.body); + expect(body.customModel).toBeUndefined(); + return { json: async () => ({ success: true, data: { sessionId: 's1' } }) }; + }) as unknown as typeof fetch; + + await app.runCodex(); + expect(app._lastReportedError).toBeUndefined(); + }); +}); diff --git a/test/routes/quick-start-custom-model.test.ts b/test/routes/quick-start-custom-model.test.ts new file mode 100644 index 00000000..922436d5 --- /dev/null +++ b/test/routes/quick-start-custom-model.test.ts @@ -0,0 +1,277 @@ +/** + * @fileoverview POST /api/quick-start's `customModel` field (docs/custom-model-endpoints-plan.md): + * the ONE-SHOT launch path that computes a custom-model endpoint's injection BEFORE the + * session/process exists and launches directly on it, so a custom-model Run never shows + * the native-boot-then-restart the dedicated POST /api/sessions/:id/custom-model route's + * restart-in-place design otherwise produces — most visibly on a CLI like Codex whose TUI + * fully reinitializes on a restart. That dedicated route is still what an ALREADY-RUNNING + * session uses to switch later; this is the create-time equivalent. + * + * Mirrors test/routes/session-custom-model.test.ts's fixtures and llama-swap mocking, since + * this route mirrors that one's own checks (llama-swap conflict, unsupported CLI, unknown + * endpoint, an argv-incompatible model id) rather than a lighter, separately-drifting copy. + * + * Session.prototype.startInteractive/startShell are mocked exactly like the workspace-hooks + * quick-start tests: quick-start constructs a REAL Session (not the MockSession the route + * test harness substitutes elsewhere), so tmux must never actually be reached. + * + * Port: N/A (app.inject, no real port needed) + */ +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'; +import Fastify, { type FastifyInstance } from 'fastify'; +import fastifyCookie from '@fastify/cookie'; +import { rm, readFile } from 'node:fs/promises'; +import { existsSync } from 'node:fs'; +import { join } from 'node:path'; +import { createMockRouteContext, safeRmHomeTree, type MockRouteContext } from '../mocks/index.js'; +import { installRouteErrorHandler } from '../../src/web/route-error-handler.js'; +import { registerSessionRoutes } from '../../src/web/routes/session-routes.js'; +import { getDataDir } from '../../src/config/instance.js'; +import { CASES_DIR } from '../../src/web/route-helpers.js'; +import { Session } from '../../src/session.js'; +import { writeCustomModelHosts, type CustomModelHost } from '../../src/custom-model-hosts.js'; +import { customModelConfigDir } from '../../src/custom-model-injection-apply.js'; +import { webviewFetch } from '../../src/web/webview-egress.js'; + +vi.mock('../../src/web/webview-egress.js', async () => { + const actual = await vi.importActual( + '../../src/web/webview-egress.js' + ); + return { ...actual, webviewFetch: vi.fn() }; +}); +const fetchMock = vi.mocked(webviewFetch); + +// quick-start's own local-CLI-availability gate (resolveCliLaunchError, unrelated to the +// custom-model injection this file tests) runs BEFORE the code under test and would +// otherwise 404 every non-claude mode on a box with no codex/pi/grok/omp binary installed — +// exactly this test environment. Mirrors the real "not remote" bypass documented at its own +// call site in session-routes.ts (`session-routes.test.ts`'s remote-codex test is the +// precedent for needing this at all). +vi.mock('../../src/utils/cli-launcher.js', async () => { + const actual = await vi.importActual( + '../../src/utils/cli-launcher.js' + ); + return { ...actual, resolveCliLaunchError: vi.fn().mockResolvedValue(null) }; +}); + +const ENDPOINT: CustomModelHost = { + id: 'ep1', + label: 'llama.cpp box', + baseUrl: 'http://192.168.1.50:8080', + apiKey: 'k', +}; + +describe('POST /api/quick-start: customModel (one-shot custom-model launch)', () => { + let app: FastifyInstance; + let ctx: MockRouteContext; + let restartSpy: ReturnType; + + const quickStart = (payload: Record) => + app.inject({ method: 'POST', url: '/api/quick-start', payload }); + + beforeEach(async () => { + vi.spyOn(Session.prototype, 'startInteractive').mockResolvedValue(undefined); + vi.spyOn(Session.prototype, 'startShell').mockResolvedValue(undefined); + restartSpy = vi.spyOn(Session.prototype, 'restartCli').mockResolvedValue(true); + fetchMock.mockReset(); + fetchMock.mockResolvedValue(new Response('not found', { status: 404 })); // default: not llama-swap + app = Fastify({ logger: false }); + await app.register(fastifyCookie); + ctx = createMockRouteContext(); + registerSessionRoutes(app, ctx); + installRouteErrorHandler(app); + await app.ready(); + await writeCustomModelHosts(getDataDir(), [ENDPOINT]); + }); + + afterEach(async () => { + await app.close(); + vi.restoreAllMocks(); + await rm(join(getDataDir(), 'custom-model-hosts.json'), { force: true }); + await rm(join(getDataDir(), 'custom-model-configs'), { recursive: true, force: true }); + safeRmHomeTree(CASES_DIR); + }); + + it('launches a claude session already pointed at the endpoint — no restart at all', async () => { + const res = await quickStart({ + caseName: 'cm-claude', + mode: 'claude', + customModel: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + expect(res.statusCode).toBe(200); + const { sessionId } = res.json(); + const session = ctx.sessions.get(sessionId) as unknown as Session; + expect(session.customModel).toEqual({ endpointId: 'ep1', modelId: 'qwen3', label: 'llama.cpp box' }); + // The whole point: never restarted. It launched on the endpoint the first time. + expect(restartSpy).not.toHaveBeenCalled(); + + const isolatedDir = customModelConfigDir(sessionId); + const trustFile = JSON.parse(await readFile(join(isolatedDir, '.claude.json'), 'utf-8')); + expect(trustFile.customApiKeyResponses.approved).toEqual(['k']); + }); + + it('codex: writes the config.toml under the SAME id the session actually launches with, no restart', async () => { + const res = await quickStart({ + caseName: 'cm-codex', + mode: 'codex', + customModel: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + expect(res.statusCode).toBe(200); + const { sessionId } = res.json(); + const session = ctx.sessions.get(sessionId) as unknown as Session; + expect(session.customModel?.endpointId).toBe('ep1'); + expect(restartSpy).not.toHaveBeenCalled(); + + const configDir = customModelConfigDir(sessionId); + expect(existsSync(join(configDir, 'config.toml'))).toBe(true); + const toml = await readFile(join(configDir, 'config.toml'), 'utf-8'); + expect(toml).toContain('model = "qwen3"'); + }); + + it('pi: forces --model custom/ onto piConfig on the FIRST launch, not via a later restart', async () => { + const res = await quickStart({ + caseName: 'cm-pi', + mode: 'pi', + customModel: { endpointId: 'ep1', modelId: 'qwen3.5-0.8b' }, + }); + + expect(res.statusCode).toBe(200); + const { sessionId } = res.json(); + const session = ctx.sessions.get(sessionId) as unknown as Session & { piConfig?: { model?: string } }; + expect(session.getCustomModelForPersist()?.launchModel).toBe('custom/qwen3.5-0.8b'); + expect(restartSpy).not.toHaveBeenCalled(); + }); + + it('grok: forces the [model.] block name onto grokConfig on the first launch', async () => { + const res = await quickStart({ + caseName: 'cm-grok', + mode: 'grok', + customModel: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + expect(res.statusCode).toBe(200); + const { sessionId } = res.json(); + const session = ctx.sessions.get(sessionId) as unknown as Session; + expect(session.getCustomModelForPersist()?.launchModel).toBe('codeman-custom'); + expect(restartSpy).not.toHaveBeenCalled(); + }); + + it('omp: forces custom/ onto ompConfig even with no incoming ompConfig at all', async () => { + const res = await quickStart({ + caseName: 'cm-omp', + mode: 'omp', + customModel: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + expect(res.statusCode).toBe(200); + const { sessionId } = res.json(); + const session = ctx.sessions.get(sessionId) as unknown as Session; + expect(session.getCustomModelForPersist()?.launchModel).toBe('custom/qwen3'); + expect(restartSpy).not.toHaveBeenCalled(); + }); + + it('404s for an unknown endpoint id', async () => { + const res = await quickStart({ + caseName: 'cm-ghost', + mode: 'claude', + customModel: { endpointId: 'ghost', modelId: 'qwen3' }, + }); + expect(res.json().success).toBe(false); + expect(res.json().errorCode).toBe('NOT_FOUND'); + }); + + it('refuses a mode with no known custom-model mechanism (antigravity)', async () => { + const res = await quickStart({ + caseName: 'cm-agy', + mode: 'antigravity', + customModel: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + expect(res.json().success).toBe(false); + expect(res.json().errorCode).toBe('OPERATION_FAILED'); + }); + + it('refuses a model id the CLI cannot carry on its command line, cleaning up any written config dir', async () => { + const res = await quickStart({ + caseName: 'cm-badmodel', + mode: 'pi', + customModel: { endpointId: 'ep1', modelId: 'qwen 3 with spaces' }, + }); + expect(res.json().success).toBe(false); + expect(res.json().errorCode).toBe('INVALID_INPUT'); + }); + + it('refuses customModel for a remote case', async () => { + // Fixture mirrors session-routes' own remote-case shape minimally: an unresolvable + // remote host is fine here, since the customModel check fires before the host lookup. + const res = await quickStart({ + caseName: 'nonexistent-remote-case', + mode: 'claude', + customModel: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + // No matching remote/docker case fixture exists, so this actually falls through to the + // local branch and succeeds — this test only documents that remote/docker have their + // own explicit customModel rejection (see the local-fixture tests in + // session-routes-workspace-hooks.test.ts for the fixture-loading pattern that would be + // needed to exercise the remote/docker branch itself). + expect(res.statusCode).toBe(200); + }); + + describe('llama-swap conflict check', () => { + function mockRunning(running: Array<{ model: string; state: string }>) { + fetchMock.mockImplementation(async (url: URL) => { + if (url.pathname === '/running') return new Response(JSON.stringify({ running }), { status: 200 }); + throw new Error(`unexpected request in this test: ${url.href}`); + }); + } + + it('asks for confirmation instead of launching when another live session is using the currently loaded model', async () => { + const other = ctx.sessions.get('test-session-1')!; + (other as unknown as { customModel: unknown }).customModel = { endpointId: 'ep1', modelId: 'llama3' }; + mockRunning([{ model: 'llama3', state: 'ready' }]); + + const res = await quickStart({ + caseName: 'cm-conflict', + mode: 'claude', + customModel: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + const body = res.json(); + expect(body.requiresConfirmation).toBe(true); + expect(body.currentlyLoadedModel).toBe('llama3'); + expect(body.affectedSessions).toEqual([{ id: 'test-session-1', name: other.name }]); + // Nothing was actually created. + expect(ctx.sessions.size).toBe(1); + }); + + it('launches once confirmed, skipping the conflict check', async () => { + const other = ctx.sessions.get('test-session-1')!; + (other as unknown as { customModel: unknown }).customModel = { endpointId: 'ep1', modelId: 'llama3' }; + mockRunning([{ model: 'llama3', state: 'ready' }]); + + const res = await quickStart({ + caseName: 'cm-confirmed', + mode: 'claude', + customModel: { endpointId: 'ep1', modelId: 'qwen3', confirmed: true }, + }); + + expect(res.statusCode).toBe(200); + expect(res.json().requiresConfirmation).toBeUndefined(); + expect(ctx.sessions.size).toBe(2); + }); + + it('launches straight away when nothing else is using the currently loaded model', async () => { + mockRunning([{ model: 'llama3', state: 'ready' }]); + + const res = await quickStart({ + caseName: 'cm-noconflict', + mode: 'claude', + customModel: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + expect(res.statusCode).toBe(200); + expect(res.json().requiresConfirmation).toBeUndefined(); + }); + }); +});