diff --git a/docs/custom-model-endpoints.md b/docs/custom-model-endpoints.md index 0cb72f15..25045d5f 100644 --- a/docs/custom-model-endpoints.md +++ b/docs/custom-model-endpoints.md @@ -239,6 +239,35 @@ earlier launch in the same isolated directory (`userID`, `numStartups`, earlier approved keys), and a missing or corrupt file is treated as empty rather than failing the apply. +**llama-swap gets two more fixes on top of the context-length/config-dir +ones above, both from watching a real switch live.** llama.cpp only ever +runs one model at a time; llama-swap swaps the backing process on demand, +which can take anywhere from a few seconds to well over a minute: + +- **The conflict check.** Both apply routes (the restart one here and the + one-shot `POST /api/quick-start` above) call llama-swap's own + `GET /running` first — feature-detected, so a plain llama.cpp/OpenAI- + compatible server (no such endpoint) is simply never checked. If a + *different* model is currently loaded and ready, and another **live + session's own selection** is using it, the apply returns + `{requiresConfirmation: true, currentlyLoadedModel, affectedSessions}` + instead of silently switching — nothing is applied or created yet. + Retrying with `confirmed: true` skips the check. Switching with nothing + else affected proceeds immediately; this is a warning about disrupting + another session, never a gate on the switch itself. +- **Actually starting the load.** llama-swap has no "switch model" admin + call — the only thing that starts a swap is a real inference request + naming the model, and confirmed live: applying a selection alone never + reached llama-swap at all (nothing in its own server logs), since nothing + had actually asked it to load anything yet. Both apply routes now also + send the smallest real request that will — `POST /v1/chat/ + completions` with `max_tokens: 1` and one throwaway message — whenever the + target model isn't already the one loaded and ready, fire-and-forget (its + response is never read; `GET /api/model-endpoints/:id/running-status`, + polled client-side, is what actually confirms readiness). The response + also carries `modelSwapInProgress: true` in that case, which is what + drives the Run-menu picker's own "loading model" status banner. + Clear back to the harness's native cloud default with: ```bash diff --git a/docs/wiki/Custom-Model-Endpoints.md b/docs/wiki/Custom-Model-Endpoints.md index 755f861d..96eb69d3 100644 --- a/docs/wiki/Custom-Model-Endpoints.md +++ b/docs/wiki/Custom-Model-Endpoints.md @@ -73,6 +73,16 @@ redirecting those hasn't landed yet, see below. The picker also only appears in **Run** dropdown; the phone home screen builds its own run picker separately and does not currently offer these entries. +**Against llama-swap, applying a selection also starts the actual model load, rather than +waiting on your first prompt to do it.** llama-swap has no "switch model" button of its own +— the only thing that starts a swap is a real request naming the model, and confirmed live: +just applying a selection never reached llama-swap's own logs at all until something asked +it to load. Picking an entry now also sends the smallest real request that will trigger +that load, in the background, the moment the target model isn't already loaded and ready — +which is what the prominent **"Loading ``… this can take a while"** banner +(centred on screen, not a corner toast — a real load can take well over a minute) is +actually watching for. + **Claude Code specifically gets two extra fixes applied automatically:** - Its discovered context length (see above) is passed through as diff --git a/src/web/routes/custom-model-routes.ts b/src/web/routes/custom-model-routes.ts index c75e40a4..5cdf36f2 100644 --- a/src/web/routes/custom-model-routes.ts +++ b/src/web/routes/custom-model-routes.ts @@ -228,6 +228,44 @@ export async function getLlamaSwapStatus( } } +/** + * Actually kicks off llama-swap's lazy model load, rather than waiting for the launched + * CLI's own first prompt to do it. llama-swap has no separate "switch model" admin + * endpoint — the ONLY thing that starts a swap is a real inference request naming the + * model (confirmed live: applying a selection alone never appeared in the llama-swap + * server's own logs; nothing had actually asked it to load anything). This sends the + * smallest real request that will — `max_tokens: 1`, one throwaway user message — to + * `${baseUrl}/v1/chat/completions`, the OpenAI-compatible endpoint every supported + * harness already points at. + * + * Deliberately fire-and-forget: the caller (the apply/create routes) returns to the + * client immediately, and the frontend's own polling (`GET .../running-status`) is what + * actually confirms readiness — this call's response is never read, just its side + * effect. No abort/timeout of its own either: a real load can take well over a minute for + * a large model, and this is a normal long-running Node process, so there is nothing to + * clean up by cutting it short. Errors are swallowed for the same reason `discoverModels`'s + * siblings swallow theirs — one endpoint's hiccup here is a nice-to-have that failed, not + * something worth surfacing as a request failure four layers up. + */ +export function triggerLlamaSwapLoad( + host: Pick, + modelId: string +): void { + const url = new URL(`${host.baseUrl.replace(/\/+$/, '')}/v1/chat/completions`); + webviewFetch(url, { + method: 'POST', + headers: { ...authHeaders(host), 'content-type': 'application/json' }, + body: JSON.stringify({ + model: modelId, + messages: [{ role: 'user', content: 'Hi' }], + max_tokens: 1, + stream: false, + }), + }).catch(() => { + // best-effort — see the doc comment above + }); +} + function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult): CustomModelHost { const { models, contextLengths } = result; const defaultModelId = host.defaultModelId && models.includes(host.defaultModelId) ? host.defaultModelId : undefined; diff --git a/src/web/routes/session-routes.ts b/src/web/routes/session-routes.ts index 8db66859..d33ca9b5 100644 --- a/src/web/routes/session-routes.ts +++ b/src/web/routes/session-routes.ts @@ -55,7 +55,7 @@ import { } from '../schemas.js'; import { readCustomModelHosts } from '../../custom-model-hosts.js'; import { applyCustomModelInjection, removeConfigDir } from '../../custom-model-injection-apply.js'; -import { getLlamaSwapStatus } from './custom-model-routes.js'; +import { getLlamaSwapStatus, triggerLlamaSwapLoad } from './custom-model-routes.js'; import { matchesPattern } from '../../config/cli-registry/patterns.js'; import { ownerLayoutKey } from '../../tab-layout-persistence.js'; import { TabLayoutValidationError } from '../../tab-layout.js'; @@ -1217,7 +1217,16 @@ export function registerSessionRoutes( // server has no such endpoint and reads as `isLlamaSwap: false` — nothing to check). const swapStatus = await getLlamaSwapStatus(endpoint); const currentlyLoaded = swapStatus.running.find((r) => r.state === 'ready')?.model ?? swapStatus.running[0]?.model; + // Distinct from targetReady below: this is ONLY about whether proceeding would evict a + // model another session is actively using — true even if nothing is loaded at all yet + // would be wrong here (nothing to evict), so this stays narrowly "a DIFFERENT model is + // currently ready". const swapNeeded = swapStatus.isLlamaSwap && !!currentlyLoaded && currentlyLoaded !== body.modelId; + // Whether the TARGET model itself is already the one loaded and ready — false whether + // nothing is loaded yet, a different model is loaded, or this one is loaded but still + // mid-load. Drives both the actual load trigger below and modelSwapInProgress in the + // response; deliberately broader than swapNeeded, which only gates the confirmation ask. + const targetReady = swapStatus.running.some((r) => r.model === body.modelId && r.state === 'ready'); // Only ask when switching would actually take the model away from another session // that is currently using it — never just because a swap is needed at all. `confirmed` @@ -1275,9 +1284,17 @@ export function registerSessionRoutes( removeConfigDir(previousConfigDir); } + // Actually kick off llama-swap's load now, rather than waiting on the restarted CLI's + // own first prompt to do it — confirmed live that applying a selection alone never + // reached the llama-swap server at all (nothing in its own logs), since llama-swap has + // no "switch model" admin call, only a real inference request naming the model. + if (swapStatus.isLlamaSwap && !targetReady) { + triggerLlamaSwapLoad(endpoint, body.modelId); + } + const restarted = await session.restartCli(); persistAndBroadcastSession(ctx, session); - return { customModel: session.customModel, restarted, modelSwapInProgress: swapNeeded }; + return { customModel: session.customModel, restarted, modelSwapInProgress: swapStatus.isLlamaSwap && !targetReady }; }); // ========== Delete Session ========== @@ -3533,6 +3550,7 @@ export function registerSessionRoutes( let qsCustomModelEnvOverrides = qsGatedEnvOverrides; let qsCustomModelLaunchModel: string | undefined; let qsCustomModelSessionId: string | undefined; + let qsCustomModelSwapInProgress = false; let qsCustomModelBookkeeping: | { endpointId: string; @@ -3562,6 +3580,11 @@ export function registerSessionRoutes( const cmCurrentlyLoaded = cmSwapStatus.running.find((r) => r.state === 'ready')?.model ?? cmSwapStatus.running[0]?.model; const cmSwapNeeded = cmSwapStatus.isLlamaSwap && !!cmCurrentlyLoaded && cmCurrentlyLoaded !== customModel.modelId; + // Broader than cmSwapNeeded (which only gates the confirmation ask above): true + // whenever the TARGET model isn't already loaded and ready, including when nothing + // is loaded at all yet. Drives the actual load trigger below. + const cmTargetReady = cmSwapStatus.running.some((r) => r.model === customModel.modelId && r.state === 'ready'); + qsCustomModelSwapInProgress = cmSwapStatus.isLlamaSwap && !cmTargetReady; if (cmSwapNeeded && !customModel.confirmed) { const cmAffectedSessions = [...ctx.sessions.values()] .filter((s) => s.customModel?.endpointId === cmEndpoint.id && s.customModel?.modelId === cmCurrentlyLoaded) @@ -3614,6 +3637,14 @@ export function registerSessionRoutes( configDir: cmApplied.configDir, launchModel: cmApplied.launchModel, }; + + // Actually kick off llama-swap's load now — see the dedicated apply route's own + // comment on triggerLlamaSwapLoad for why this can't just wait on the launched CLI's + // first prompt. Fired here, before the session is even created, so the load starts + // concurrently with Claude/Codex/etc. booting rather than after. + if (qsCustomModelSwapInProgress) { + triggerLlamaSwapLoad(cmEndpoint, customModel.modelId); + } } const session = new Session({ @@ -3761,6 +3792,7 @@ export function registerSessionRoutes( sessionId: session.id, casePath: resolvedCasePath, caseName, + ...(customModel ? { modelSwapInProgress: qsCustomModelSwapInProgress } : {}), }; } catch (err) { // Clean up session on error to prevent orphaned resources diff --git a/test/routes/quick-start-custom-model.test.ts b/test/routes/quick-start-custom-model.test.ts index 922436d5..60c2c896 100644 --- a/test/routes/quick-start-custom-model.test.ts +++ b/test/routes/quick-start-custom-model.test.ts @@ -274,4 +274,56 @@ describe('POST /api/quick-start: customModel (one-shot custom-model launch)', () expect(res.json().requiresConfirmation).toBeUndefined(); }); }); + + describe('triggering the actual llama-swap load (not just watching for it)', () => { + it('sends a real inference request naming the target model, concurrently with launching the session', async () => { + const chatCalls: unknown[] = []; + fetchMock.mockImplementation(async (url: URL, init?: { body?: unknown }) => { + if (url.pathname === '/running') { + return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 }); + } + if (url.pathname === '/v1/chat/completions') { + chatCalls.push(JSON.parse(init!.body as string)); + return new Response(JSON.stringify({ choices: [] }), { status: 200 }); + } + throw new Error(`unexpected request in this test: ${url.href}`); + }); + + const res = await quickStart({ + caseName: 'cm-trigger', + mode: 'claude', + customModel: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + await new Promise((resolve) => setTimeout(resolve, 0)); // let the fire-and-forget trigger settle + + expect(res.statusCode).toBe(200); + expect(res.json().modelSwapInProgress).toBe(true); + expect(chatCalls).toHaveLength(1); + expect(chatCalls[0]).toMatchObject({ model: 'qwen3', max_tokens: 1 }); + }); + + it('never sends a load-trigger request when the target model is already loaded and ready', async () => { + let chatCalled = false; + fetchMock.mockImplementation(async (url: URL) => { + if (url.pathname === '/running') { + return new Response(JSON.stringify({ running: [{ model: 'qwen3', state: 'ready' }] }), { status: 200 }); + } + if (url.pathname === '/v1/chat/completions') { + chatCalled = true; + return new Response(JSON.stringify({ choices: [] }), { status: 200 }); + } + throw new Error(`unexpected request in this test: ${url.href}`); + }); + + const res = await quickStart({ + caseName: 'cm-no-trigger', + mode: 'claude', + customModel: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + await new Promise((resolve) => setTimeout(resolve, 0)); + + expect(res.json().modelSwapInProgress).toBe(false); + expect(chatCalled).toBe(false); + }); + }); }); diff --git a/test/routes/session-custom-model.test.ts b/test/routes/session-custom-model.test.ts index 272aa7d3..2bfa7bcf 100644 --- a/test/routes/session-custom-model.test.ts +++ b/test/routes/session-custom-model.test.ts @@ -351,6 +351,89 @@ describe('POST /api/sessions/:id/custom-model', () => { }); }); + describe('triggering the actual llama-swap load (not just watching for it)', () => { + it('sends a real inference request naming the target model when it is not already loaded and ready', async () => { + const { app, ctx } = await setup(); + ctx.sessions.get('test-session-1')!.mode = 'claude'; + const chatCalls: unknown[] = []; + fetchMock.mockImplementation(async (url: URL, init?: { body?: unknown }) => { + if (url.pathname === '/running') { + return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 }); + } + if (url.pathname === '/v1/chat/completions') { + chatCalls.push(JSON.parse(init!.body as string)); + return new Response(JSON.stringify({ choices: [] }), { status: 200 }); + } + throw new Error(`unexpected request in this test: ${url.href}`); + }); + + await app.inject({ + method: 'POST', + url: '/api/sessions/test-session-1/custom-model', + payload: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + await new Promise((resolve) => setTimeout(resolve, 0)); // let the fire-and-forget trigger settle + + expect(chatCalls).toHaveLength(1); + expect(chatCalls[0]).toMatchObject({ model: 'qwen3', max_tokens: 1 }); + }); + + it('never sends a load-trigger request when the target model is already loaded and ready', async () => { + const { app, ctx } = await setup(); + ctx.sessions.get('test-session-1')!.mode = 'claude'; + let chatCalled = false; + fetchMock.mockImplementation(async (url: URL) => { + if (url.pathname === '/running') { + return new Response(JSON.stringify({ running: [{ model: 'qwen3', state: 'ready' }] }), { status: 200 }); + } + if (url.pathname === '/v1/chat/completions') { + chatCalled = true; + return new Response(JSON.stringify({ choices: [] }), { status: 200 }); + } + throw new Error(`unexpected request in this test: ${url.href}`); + }); + + await app.inject({ + method: 'POST', + url: '/api/sessions/test-session-1/custom-model', + payload: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + await new Promise((resolve) => setTimeout(resolve, 0)); + + expect(chatCalled).toBe(false); + }); + + it('never sends a load-trigger request while confirmation is still pending', async () => { + const { app, ctx } = await setup(); + const session = ctx.sessions.get('test-session-1')!; + session.mode = 'claude'; + const other = createMockSession('other-session'); + other.customModel = { endpointId: 'ep1', modelId: 'llama3' }; + ctx.sessions.set('other-session', other); + let chatCalled = false; + fetchMock.mockImplementation(async (url: URL) => { + if (url.pathname === '/running') { + return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 }); + } + if (url.pathname === '/v1/chat/completions') { + chatCalled = true; + return new Response(JSON.stringify({ choices: [] }), { status: 200 }); + } + throw new Error(`unexpected request in this test: ${url.href}`); + }); + + const res = await app.inject({ + method: 'POST', + url: '/api/sessions/test-session-1/custom-model', + payload: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + await new Promise((resolve) => setTimeout(resolve, 0)); + + expect(res.json().requiresConfirmation).toBe(true); + expect(chatCalled).toBe(false); + }); + }); + it('refuses to touch a busy session', async () => { const { app, ctx } = await setup(); const session = ctx.sessions.get('test-session-1')!;