diff --git a/src/web/public/panels-ui.js b/src/web/public/panels-ui.js index 20a08306..64835505 100644 --- a/src/web/public/panels-ui.js +++ b/src/web/public/panels-ui.js @@ -5544,6 +5544,11 @@ Object.assign(CodemanApp.prototype, { if (duration > 0) { dismissTimer = setTimeout(dismiss, duration); } + + // Most callers ignore this — a handle exists for a long-running toast a caller needs + // to update or dismiss itself once its own condition resolves (e.g. a "loading model" + // toast a poll loop dismisses once the model reports ready). + return { dismiss, setMessage: (text) => { msgSpan.textContent = text; } }; }, diff --git a/src/web/public/session-ui.js b/src/web/public/session-ui.js index 684a2798..8ec63e54 100644 --- a/src/web/public/session-ui.js +++ b/src/web/public/session-ui.js @@ -740,17 +740,96 @@ Object.assign(CodemanApp.prototype, { // unreachable" apart from "the CLI can't be redirected", "not one of the // discovered models", or "this is a Docker/remote session". Go through the // raw response here instead so a failure is diagnosable, not just present. - const res = await this._api(`/api/sessions/${sessionId}/custom-model`, { - method: 'POST', - body: { endpointId, modelId }, - }); - const data = res ? await res.json().catch(() => null) : null; - if (!data || data.success === false) { + let { ok, data, res } = await this._applyCustomModelToSession(sessionId, endpointId, modelId); + + // A success body comes back as {success:true, data:{...}} (server.ts's preSerialization + // envelope), but a route-level error is {success:false, error, errorCode} with no nested + // data — createErrorResponse() never wraps one. `payload` below is only ever meaningful + // once `data.success !== false`. + let payload = data?.success !== false ? data?.data : undefined; + + // llama-swap runs one model at a time: switching would unload it out from under + // another session actively using it. The route only asks when that's actually true + // (never just because a swap is needed at all) — confirming re-sends the exact same + // call with `confirmed: true` so the route skips the check the second time. + if (ok && payload?.requiresConfirmation) { + const names = payload.affectedSessions.map((s) => s.name || s.id).join(', '); + const proceed = confirm( + `${names} ${payload.affectedSessions.length === 1 ? 'is' : 'are'} currently using ` + + `${payload.currentlyLoadedModel} on this endpoint. Switching to ${modelId} will unload it ` + + `for ${payload.affectedSessions.length === 1 ? 'that session' : 'those sessions'} too. Continue?` + ); + if (!proceed) { + this.showToast('Kept the native backend — model switch cancelled', 'info'); + return; + } + ({ ok, data, res } = await this._applyCustomModelToSession(sessionId, endpointId, modelId, true)); + payload = data?.success !== false ? data?.data : undefined; + } + + if (!ok || !data || data.success === false) { const detail = data?.error ? `: ${data.error}` : res ? ` (HTTP ${res.status})` : ' (request failed)'; this.showToast(`Session started on the native backend — could not apply the custom endpoint${detail}`, 'error'); return; } this.showToast(`Pointed at ${endpointId} — restarting the session...`, 'info'); + + // The apply above already succeeded — the session IS pointed at the endpoint — but + // llama-swap itself may still be unloading the old model and loading this one, which + // can take well over a minute. Without this, a prompt sent during that window either + // hangs silently or (the bug this whole feature exists to fix) gets answered by + // whatever was loaded a moment ago, reading as "it's still using the wrong model." + if (payload?.modelSwapInProgress) { + void this._watchLlamaSwapLoading(endpointId, modelId); + } + }, + + /** POST /api/sessions/:id/custom-model, returning {ok, data, res} rather than throwing — + * see runCustomModelEntry's own comment for why this goes through `_api()` (raw fetch) + * rather than `_apiJson()`: a failure's `error` detail must survive to the caller. */ + async _applyCustomModelToSession(sessionId, endpointId, modelId, confirmed) { + const res = await this._api(`/api/sessions/${sessionId}/custom-model`, { + method: 'POST', + body: confirmed ? { endpointId, modelId, confirmed } : { endpointId, modelId }, + }); + const data = res ? await res.json().catch(() => null) : null; + return { ok: !!res, data, res }; + }, + + /** + * Polls llama-swap's own `/running` (via the read-only running-status route) until + * `modelId` reports `state: 'ready'`, showing a sticky toast the whole time so a slow + * unload/reload (measured well over a minute for a large model) reads as "loading", + * never as silence or a wrong answer from whatever was loaded before. Bounded at 2 + * minutes; still not ready by then gets a toast saying so rather than polling forever. + * + * `pollIntervalMs`/`maxWaitMs` exist to let a test drive this in milliseconds instead of + * minutes — real callers never pass them, which is what keeps the defaults live here + * rather than only in a test fixture. + */ + async _watchLlamaSwapLoading(endpointId, modelId, pollIntervalMs = 3000, maxWaitMs = 120000) { + const toast = this.showToast(`Loading ${modelId} on ${endpointId}… this can take a while`, 'info', { + duration: 0, + }); + const deadline = Date.now() + maxWaitMs; + while (Date.now() < deadline) { + await new Promise((resolve) => setTimeout(resolve, pollIntervalMs)); + const status = await this._apiJson(`/api/model-endpoints/${encodeURIComponent(endpointId)}/running-status`); + if (!status) continue; // transient failure — keep waiting rather than giving up early + if (!status.isLlamaSwap) { + // Endpoint changed under us, or wasn't llama-swap after all — nothing more to + // watch for, and not a failure worth a toast of its own. + toast?.dismiss(); + return; + } + if (status.running.some((r) => r.model === modelId && r.state === 'ready')) { + toast?.dismiss(); + this.showToast(`${modelId} is ready`, 'success', { duration: 2500 }); + return; + } + } + toast?.dismiss(); + this.showToast(`Still waiting for ${modelId} to finish loading on ${endpointId} — check the llama-swap server`, 'warning'); }, /** diff --git a/src/web/routes/custom-model-routes.ts b/src/web/routes/custom-model-routes.ts index eabd191e..c75e40a4 100644 --- a/src/web/routes/custom-model-routes.ts +++ b/src/web/routes/custom-model-routes.ts @@ -177,6 +177,57 @@ type RedactedHost = ReturnType; * each do their own `discoverModels()` + error handling around one shared * "how to apply a successful result" step. */ +const RUNNING_TIMEOUT_MS = 5000; + +export interface LlamaSwapRunningModel { + model: string; + state: string; +} + +export interface LlamaSwapStatus { + /** + * Feature-detected via `GET /running`: true only when the server answered with + * llama-swap's own shape (`{ running: [...] }`). Plain llama.cpp (and any other + * OpenAI-compatible server) has no such endpoint and always runs the single model + * it was started with, so there is no "current model" to conflict with — every + * caller must treat `isLlamaSwap: false` as "nothing to check", never as an error. + */ + isLlamaSwap: boolean; + running: LlamaSwapRunningModel[]; +} + +/** + * Distinguishes llama-swap from a plain llama.cpp/OpenAI-compatible server, and reports + * what llama-swap currently has loaded — llama.cpp only ever runs one GGUF at a time, and + * llama-swap unloads/reloads it on demand when a request asks for a different one, which + * can take anywhere from a few seconds to over a minute. Read-only: this never triggers a + * swap itself (unlike `/props?model=`, `/running` takes no `model` parameter to route by). + * Best-effort like `discoverModels()`'s siblings: any failure (unreachable, non-2xx, + * unexpected shape) reads as "not llama-swap", never thrown. + */ +export async function getLlamaSwapStatus( + host: Pick +): Promise { + try { + const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/running`), { + headers: authHeaders(host), + signal: AbortSignal.timeout(RUNNING_TIMEOUT_MS), + }); + if (!res.ok) return { isLlamaSwap: false, running: [] }; + const body = (await res.json()) as { running?: unknown }; + if (!Array.isArray(body.running)) return { isLlamaSwap: false, running: [] }; + const running = body.running + .filter( + (r): r is { model: string; state?: unknown } => + !!r && typeof r === 'object' && typeof (r as { model?: unknown }).model === 'string' + ) + .map((r) => ({ model: r.model, state: typeof r.state === 'string' ? r.state : 'unknown' })); + return { isLlamaSwap: true, running }; + } catch { + return { isLlamaSwap: false, running: [] }; + } +} + function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult): CustomModelHost { const { models, contextLengths } = result; const defaultModelId = host.defaultModelId && models.includes(host.defaultModelId) ? host.defaultModelId : undefined; @@ -303,4 +354,18 @@ export function registerCustomModelRoutes(app: FastifyInstance): void { } } ); + + // Read-only, no admin gate: any session owner who can already point their own session + // at this endpoint (POST .../custom-model, ungated by design — see session-routes.ts) + // can equally ask what it currently has loaded, before or while that apply is pending. + app.get('/api/model-endpoints/:id/running-status', async (req): Promise> => { + const { id } = req.params as { id: string }; + const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR); + const host = hosts.find((item) => item.id === id); + if (!host) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found'); + if (isBlockedWebviewUrl(host.baseUrl)) { + return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed'); + } + return { success: true, data: await getLlamaSwapStatus(host) }; + }); } diff --git a/src/web/routes/session-routes.ts b/src/web/routes/session-routes.ts index 44b47b7c..f927dff9 100644 --- a/src/web/routes/session-routes.ts +++ b/src/web/routes/session-routes.ts @@ -55,6 +55,7 @@ import { } from '../schemas.js'; import { readCustomModelHosts } from '../../custom-model-hosts.js'; import { applyCustomModelInjection, removeConfigDir } from '../../custom-model-injection-apply.js'; +import { getLlamaSwapStatus } from './custom-model-routes.js'; import { matchesPattern } from '../../config/cli-registry/patterns.js'; import { ownerLayoutKey } from '../../tab-layout-persistence.js'; import { TabLayoutValidationError } from '../../tab-layout.js'; @@ -1209,6 +1210,32 @@ export function registerSessionRoutes( return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found'); } + // llama.cpp runs exactly one model at a time; llama-swap unloads and reloads it on + // demand, which can take anywhere from a few seconds to over a minute — long enough + // that a session mid-swap looks indistinguishable from one that never left the native + // backend. Feature-detected via llama-swap's own `GET /running` (a plain llama.cpp + // server has no such endpoint and reads as `isLlamaSwap: false` — nothing to check). + const swapStatus = await getLlamaSwapStatus(endpoint); + const currentlyLoaded = swapStatus.running.find((r) => r.state === 'ready')?.model ?? swapStatus.running[0]?.model; + const swapNeeded = swapStatus.isLlamaSwap && !!currentlyLoaded && currentlyLoaded !== body.modelId; + + // Only ask when switching would actually take the model away from another session + // that is currently using it — never just because a swap is needed at all. `confirmed` + // (set by the caller after showing that warning once) skips asking again. + if (swapNeeded && !body.confirmed) { + const affectedSessions = [...ctx.sessions.values()] + .filter( + (s) => + s.id !== session.id && + s.customModel?.endpointId === endpoint.id && + s.customModel?.modelId === currentlyLoaded + ) + .map((s) => ({ id: s.id, name: s.name })); + if (affectedSessions.length > 0) { + return { requiresConfirmation: true, currentlyLoadedModel: currentlyLoaded, affectedSessions }; + } + } + // A CLI whose config alone cannot select the model also gets its `model` launch param // forced (pi/omp `custom/`, grok's block name). The argv engine DROPS a token that // fails its pattern rather than quoting it, which would silently launch the CLI on its @@ -1250,7 +1277,7 @@ export function registerSessionRoutes( const restarted = await session.restartCli(); persistAndBroadcastSession(ctx, session); - return { customModel: session.customModel, restarted }; + return { customModel: session.customModel, restarted, modelSwapInProgress: swapNeeded }; }); // ========== Delete Session ========== diff --git a/src/web/schemas.ts b/src/web/schemas.ts index 6407e054..119d2836 100644 --- a/src/web/schemas.ts +++ b/src/web/schemas.ts @@ -1934,6 +1934,10 @@ export const CustomModelSelectionSchema = z.union([ z.object({ endpointId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid endpoint id'), modelId: z.string().min(1).max(200), + // Set once the caller has already shown the "this will unload for session(s) + // X" warning (see session-routes.ts's llama-swap conflict check) and the user chose to + // proceed anyway — skips that check on this call instead of asking again. + confirmed: z.boolean().optional(), }), z.object({ clear: z.literal(true) }), ]); diff --git a/test/custom-model-run-menu-ui.test.ts b/test/custom-model-run-menu-ui.test.ts index ea995940..2597d1a3 100644 --- a/test/custom-model-run-menu-ui.test.ts +++ b/test/custom-model-run-menu-ui.test.ts @@ -426,3 +426,168 @@ describe('Custom Model Endpoint Profiles: applying a picked entry', () => { expect(app._runMode).toBe('opencode'); }); }); + +describe('Custom Model Endpoint Profiles: llama-swap model-swap confirmation and loading state', () => { + function launchHarness(applyResponses: Array>) { + const { win, app } = bootApp({}); + app.activeSessionId = 'old-session'; + app.run = async () => { + app.activeSessionId = 'new-session'; + }; + const applyBodies: unknown[] = []; + let call = 0; + app._api = async (path: string, opts?: { body?: unknown }) => { + if (path.endsWith('/custom-model')) { + applyBodies.push(opts?.body); + const data = applyResponses[Math.min(call, applyResponses.length - 1)]; + call += 1; + return { ok: true, status: 200, json: async () => ({ success: true, data }) }; + } + throw new Error(`unexpected _api call: ${path}`); + }; + return { win, app, applyBodies }; + } + + it('confirming the native window.confirm() re-sends the apply with confirmed:true', async () => { + const { win, app, applyBodies } = launchHarness([ + { + requiresConfirmation: true, + currentlyLoadedModel: 'llama3', + affectedSessions: [{ id: 's2', name: 'w2-otherbox' }], + }, + { customModel: { endpointId: 'llama-box' }, restarted: true, modelSwapInProgress: true }, + ]); + let confirmMessage: string | undefined; + win.confirm = ((msg: string) => { + confirmMessage = msg; + return true; + }) as typeof win.confirm; + app._watchLlamaSwapLoading = async () => {}; // not under test here + + await app.runCustomModelEntry('claude', 'llama-box', 'qwen3'); + + expect(confirmMessage).toContain('w2-otherbox'); + expect(confirmMessage).toContain('llama3'); + expect(confirmMessage).toContain('qwen3'); + expect(applyBodies).toEqual([ + { endpointId: 'llama-box', modelId: 'qwen3' }, + { endpointId: 'llama-box', modelId: 'qwen3', confirmed: true }, + ]); + }); + + it('cancelling window.confirm() keeps the native backend and never re-sends the apply', async () => { + const { win, app, applyBodies } = launchHarness([ + { requiresConfirmation: true, currentlyLoadedModel: 'llama3', affectedSessions: [{ id: 's2', name: 'w2' }] }, + ]); + win.confirm = (() => false) as typeof win.confirm; + let toastMessage: string | undefined; + app.showToast = (msg: string) => { + toastMessage = msg; + }; + + await app.runCustomModelEntry('claude', 'llama-box', 'qwen3'); + + expect(applyBodies).toHaveLength(1); // no second (confirmed) call + expect(toastMessage).toMatch(/cancelled/i); + }); + + it('a successful apply with modelSwapInProgress kicks off the loading watcher', async () => { + const { app } = launchHarness([ + { customModel: { endpointId: 'llama-box' }, restarted: true, modelSwapInProgress: true }, + ]); + let watched: unknown[] | null = null; + app._watchLlamaSwapLoading = async (...args: unknown[]) => { + watched = args; + }; + + await app.runCustomModelEntry('claude', 'llama-box', 'qwen3'); + + expect(watched).toEqual(['llama-box', 'qwen3']); + }); + + it('a successful apply with no swap needed never starts the loading watcher', async () => { + const { app } = launchHarness([ + { customModel: { endpointId: 'llama-box' }, restarted: true, modelSwapInProgress: false }, + ]); + let watchCalled = false; + app._watchLlamaSwapLoading = async () => { + watchCalled = true; + }; + + await app.runCustomModelEntry('claude', 'llama-box', 'qwen3'); + + expect(watchCalled).toBe(false); + }); +}); + +describe('Custom Model Endpoint Profiles: _watchLlamaSwapLoading polling', () => { + // Driven with millisecond intervals (the function's own pollIntervalMs/maxWaitMs + // params — real callers never pass them) rather than fake timers: this code runs + // inside the JSDOM window's own realm (bootApp's `runScripts: "dangerously"` eval), + // whose setTimeout is NOT the one vi.useFakeTimers() patches, so advancing fake + // timers here would advance nothing and either hang or silently no-op. + + it('dismisses the loading toast as soon as the target model reports ready', async () => { + const { app } = bootApp({}); + const toastCalls: Array<{ message: string; type: string }> = []; + const dismissed: string[] = []; + app.showToast = (message: string, type: string) => { + toastCalls.push({ message, type }); + return { dismiss: () => dismissed.push(message), setMessage: () => {} }; + }; + app._apiJson = async () => ({ isLlamaSwap: true, running: [{ model: 'qwen3', state: 'ready' }] }); + + await app._watchLlamaSwapLoading('llama-box', 'qwen3', 5, 200); + + expect(toastCalls[0].message).toMatch(/loading qwen3/i); + expect(dismissed).toContain(toastCalls[0].message); + expect(toastCalls.at(-1)?.message).toMatch(/ready/i); + }); + + it('gives up after the bounded wait and warns instead of polling forever', async () => { + const { app } = bootApp({}); + const toastCalls: string[] = []; + app.showToast = (message: string) => { + toastCalls.push(message); + return { dismiss: () => {}, setMessage: () => {} }; + }; + app._apiJson = async () => ({ isLlamaSwap: true, running: [{ model: 'something-else', state: 'ready' }] }); + + await app._watchLlamaSwapLoading('llama-box', 'qwen3', 5, 30); + + expect(toastCalls.at(-1)).toMatch(/still waiting/i); + }); + + it('stops polling (without a warning) once the endpoint no longer reads as llama-swap', async () => { + const { app } = bootApp({}); + const toastCalls: string[] = []; + app.showToast = (message: string) => { + toastCalls.push(message); + return { dismiss: () => {}, setMessage: () => {} }; + }; + app._apiJson = async () => ({ isLlamaSwap: false, running: [] }); + + await app._watchLlamaSwapLoading('llama-box', 'qwen3', 5, 200); + + expect(toastCalls).toHaveLength(1); // only the initial "Loading..." toast, no follow-up warning + }); + + it('keeps waiting through a transient status-fetch failure instead of giving up early', async () => { + const { app } = bootApp({}); + const toastCalls: string[] = []; + app.showToast = (message: string) => { + toastCalls.push(message); + return { dismiss: () => {}, setMessage: () => {} }; + }; + let call = 0; + app._apiJson = async () => { + call += 1; + if (call === 1) return null; // transient failure + return { isLlamaSwap: true, running: [{ model: 'qwen3', state: 'ready' }] }; + }; + + await app._watchLlamaSwapLoading('llama-box', 'qwen3', 5, 200); + + expect(toastCalls.at(-1)).toMatch(/ready/i); + }); +}); diff --git a/test/routes/session-custom-model.test.ts b/test/routes/session-custom-model.test.ts index 8797e8ac..272aa7d3 100644 --- a/test/routes/session-custom-model.test.ts +++ b/test/routes/session-custom-model.test.ts @@ -3,13 +3,28 @@ * chunk 5 — applying/clearing a session's custom model endpoint + CLI restart). * Port: N/A (app.inject, no real port needed) */ -import { describe, it, expect, beforeEach } from 'vitest'; +import { describe, it, expect, beforeEach, vi } from 'vitest'; import { registerSessionRoutes } from '../../src/web/routes/session-routes.js'; import { createRouteTestHarness } from './_route-test-utils.js'; +import { createMockSession } from '../mocks/index.js'; import { getDataDir } from '../../src/config/instance.js'; import { writeCustomModelHosts, type CustomModelHost } from '../../src/custom-model-hosts.js'; import { existsSync, readFileSync, statSync } from 'node:fs'; import { join } from 'node:path'; +import { webviewFetch } from '../../src/web/webview-egress.js'; + +// Every apply now also checks llama-swap's `GET /running` (session-routes.ts) before +// applying — without this mock every test in this file would make a REAL network request +// to the fake 192.168.1.50 endpoint below and wait out its 5s timeout. Defaults to a plain +// 404 (reads as "not llama-swap", exercising none of the new conflict-check tests below), +// overridden per-test where the llama-swap behavior itself is what's under test. +vi.mock('../../src/web/webview-egress.js', async () => { + const actual = await vi.importActual( + '../../src/web/webview-egress.js' + ); + return { ...actual, webviewFetch: vi.fn() }; +}); +const fetchMock = vi.mocked(webviewFetch); const CLAUDE_ENDPOINT: CustomModelHost = { id: 'ep1', @@ -26,6 +41,8 @@ async function setup() { describe('POST /api/sessions/:id/custom-model', () => { beforeEach(async () => { await writeCustomModelHosts(getDataDir(), []); + fetchMock.mockReset(); + fetchMock.mockResolvedValue(new Response('not found', { status: 404 })); }); it('applies an endpoint/model to a claude-mode session and restarts the CLI', async () => { @@ -211,6 +228,129 @@ describe('POST /api/sessions/:id/custom-model', () => { expect(existsSync(dir)).toBe(false); }); + describe('llama-swap conflict check (llama.cpp runs one model at a time)', () => { + function mockRunning(running: Array<{ model: string; state: string }>) { + fetchMock.mockImplementation(async (url: URL) => { + if (url.pathname === '/running') return new Response(JSON.stringify({ running }), { status: 200 }); + throw new Error(`unexpected request in this test: ${url.href}`); + }); + } + + it('applies straight away when the requested model is already loaded', async () => { + const { app, ctx } = await setup(); + ctx.sessions.get('test-session-1')!.mode = 'claude'; + mockRunning([{ model: 'qwen3', state: 'ready' }]); + + const res = await app.inject({ + method: 'POST', + url: '/api/sessions/test-session-1/custom-model', + payload: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + expect(res.json().success).not.toBe(false); + expect(res.json().modelSwapInProgress).toBe(false); + expect(ctx.sessions.get('test-session-1')!.setCustomModel).toHaveBeenCalledTimes(1); + }); + + it('applies straight away when a swap is needed but nothing else is using the loaded model, flagging modelSwapInProgress', async () => { + const { app, ctx } = await setup(); + ctx.sessions.get('test-session-1')!.mode = 'claude'; + mockRunning([{ model: 'llama3', state: 'ready' }]); + + const res = await app.inject({ + method: 'POST', + url: '/api/sessions/test-session-1/custom-model', + payload: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + expect(res.json().success).not.toBe(false); + expect(res.json().modelSwapInProgress).toBe(true); + expect(ctx.sessions.get('test-session-1')!.setCustomModel).toHaveBeenCalledTimes(1); + }); + + it('asks for confirmation instead of applying when another session is actively using the currently loaded model', async () => { + const { app, ctx } = await setup(); + const session = ctx.sessions.get('test-session-1')!; + session.mode = 'claude'; + const other = createMockSession('other-session'); + other.name = 'w2-otherbox'; + other.customModel = { endpointId: 'ep1', modelId: 'llama3' }; + ctx.sessions.set('other-session', other); + mockRunning([{ model: 'llama3', state: 'ready' }]); + + const res = await app.inject({ + method: 'POST', + url: '/api/sessions/test-session-1/custom-model', + payload: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + const body = res.json(); + expect(body.success).not.toBe(false); + expect(body.requiresConfirmation).toBe(true); + expect(body.currentlyLoadedModel).toBe('llama3'); + expect(body.affectedSessions).toEqual([{ id: 'other-session', name: 'w2-otherbox' }]); + // Nothing actually applied yet — this call only asked, it did not switch. + expect(session.setCustomModel).not.toHaveBeenCalled(); + expect(session.restartCli).not.toHaveBeenCalled(); + }); + + it('applies once confirmed, skipping the conflict check the second time', async () => { + const { app, ctx } = await setup(); + const session = ctx.sessions.get('test-session-1')!; + session.mode = 'claude'; + const other = createMockSession('other-session'); + other.customModel = { endpointId: 'ep1', modelId: 'llama3' }; + ctx.sessions.set('other-session', other); + mockRunning([{ model: 'llama3', state: 'ready' }]); + + const res = await app.inject({ + method: 'POST', + url: '/api/sessions/test-session-1/custom-model', + payload: { endpointId: 'ep1', modelId: 'qwen3', confirmed: true }, + }); + + const body = res.json(); + expect(body.requiresConfirmation).toBeUndefined(); + expect(body.modelSwapInProgress).toBe(true); + expect(session.setCustomModel).toHaveBeenCalledTimes(1); + expect(session.restartCli).toHaveBeenCalledTimes(1); + }); + + it('a session pointed at the SAME endpoint but a DIFFERENT (not-currently-loaded) model is not treated as affected', async () => { + const { app, ctx } = await setup(); + const session = ctx.sessions.get('test-session-1')!; + session.mode = 'claude'; + const other = createMockSession('other-session'); + other.customModel = { endpointId: 'ep1', modelId: 'some-other-model' }; // not the loaded one + ctx.sessions.set('other-session', other); + mockRunning([{ model: 'llama3', state: 'ready' }]); + + const res = await app.inject({ + method: 'POST', + url: '/api/sessions/test-session-1/custom-model', + payload: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + expect(res.json().requiresConfirmation).toBeUndefined(); + expect(session.setCustomModel).toHaveBeenCalledTimes(1); + }); + + it('not llama-swap (plain llama.cpp/OpenAI-compatible server, no /running) — never checked, applies straight away', async () => { + const { app, ctx } = await setup(); + ctx.sessions.get('test-session-1')!.mode = 'claude'; + fetchMock.mockResolvedValue(new Response('not found', { status: 404 })); + + const res = await app.inject({ + method: 'POST', + url: '/api/sessions/test-session-1/custom-model', + payload: { endpointId: 'ep1', modelId: 'qwen3' }, + }); + + expect(res.json().modelSwapInProgress).toBe(false); + expect(res.json().requiresConfirmation).toBeUndefined(); + }); + }); + it('refuses to touch a busy session', async () => { const { app, ctx } = await setup(); const session = ctx.sessions.get('test-session-1')!;