mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-03 22:19:42 +02:00
feat(custom-model): detect llama-swap model conflicts before switching
Root-caused the user's earlier confusion ('the terminal says opus even though
something is waiting for llama to load'): llama.cpp runs exactly one model at
a time, and llama-swap unloads/reloads it on demand - a swap can take
anywhere from a few seconds to well over a minute, during which a session
looks indistinguishable from one still on the native backend.
1. Feature-detects llama-swap (vs. plain llama.cpp/any OpenAI-compatible
server) via its own GET /running, which plain llama.cpp has no concept of
at all. New GET /api/model-endpoints/:id/running-status route exposes this
read-only, for the frontend's polling loop below.
2. Before applying a selection, POST /api/sessions/:id/custom-model now checks
what llama-swap currently has loaded. If it differs from the requested
model AND another live session's own customModel selection is actively
using that loaded model, the apply is refused with a
{requiresConfirmation, currentlyLoadedModel, affectedSessions} payload
instead of silently switching. A "confirmed: true" field on the retry
skips the check. Switching with nothing else affected proceeds
immediately, no confirmation asked, only ever when there is something to
warn about.
3. The frontend (runCustomModelEntry) shows a native confirm() naming the
affected session(s) and the model they'd lose, matching this codebase's
existing convention for this class of decision (delete case, kill
session, etc.) rather than a new modal. On a successful apply the response
also carries modelSwapInProgress; when true, a new _watchLlamaSwapLoading
poll shows a sticky "Loading <model>..." toast via the new running-status
route until llama-swap reports the target model ready (bounded at 2
minutes), so a prompt sent mid-swap reads as "loading", never as silence
or an answer from whatever was loaded a moment before.
Checks are read-only against llama-swap's own /running - never /props, which
takes a ?model= and can itself trigger a load as a side effect of asking.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
25f22b9839
commit
bcebc81fcd
@@ -426,3 +426,168 @@ describe('Custom Model Endpoint Profiles: applying a picked entry', () => {
|
||||
expect(app._runMode).toBe('opencode');
|
||||
});
|
||||
});
|
||||
|
||||
describe('Custom Model Endpoint Profiles: llama-swap model-swap confirmation and loading state', () => {
|
||||
function launchHarness(applyResponses: Array<Record<string, unknown>>) {
|
||||
const { win, app } = bootApp({});
|
||||
app.activeSessionId = 'old-session';
|
||||
app.run = async () => {
|
||||
app.activeSessionId = 'new-session';
|
||||
};
|
||||
const applyBodies: unknown[] = [];
|
||||
let call = 0;
|
||||
app._api = async (path: string, opts?: { body?: unknown }) => {
|
||||
if (path.endsWith('/custom-model')) {
|
||||
applyBodies.push(opts?.body);
|
||||
const data = applyResponses[Math.min(call, applyResponses.length - 1)];
|
||||
call += 1;
|
||||
return { ok: true, status: 200, json: async () => ({ success: true, data }) };
|
||||
}
|
||||
throw new Error(`unexpected _api call: ${path}`);
|
||||
};
|
||||
return { win, app, applyBodies };
|
||||
}
|
||||
|
||||
it('confirming the native window.confirm() re-sends the apply with confirmed:true', async () => {
|
||||
const { win, app, applyBodies } = launchHarness([
|
||||
{
|
||||
requiresConfirmation: true,
|
||||
currentlyLoadedModel: 'llama3',
|
||||
affectedSessions: [{ id: 's2', name: 'w2-otherbox' }],
|
||||
},
|
||||
{ customModel: { endpointId: 'llama-box' }, restarted: true, modelSwapInProgress: true },
|
||||
]);
|
||||
let confirmMessage: string | undefined;
|
||||
win.confirm = ((msg: string) => {
|
||||
confirmMessage = msg;
|
||||
return true;
|
||||
}) as typeof win.confirm;
|
||||
app._watchLlamaSwapLoading = async () => {}; // not under test here
|
||||
|
||||
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
|
||||
|
||||
expect(confirmMessage).toContain('w2-otherbox');
|
||||
expect(confirmMessage).toContain('llama3');
|
||||
expect(confirmMessage).toContain('qwen3');
|
||||
expect(applyBodies).toEqual([
|
||||
{ endpointId: 'llama-box', modelId: 'qwen3' },
|
||||
{ endpointId: 'llama-box', modelId: 'qwen3', confirmed: true },
|
||||
]);
|
||||
});
|
||||
|
||||
it('cancelling window.confirm() keeps the native backend and never re-sends the apply', async () => {
|
||||
const { win, app, applyBodies } = launchHarness([
|
||||
{ requiresConfirmation: true, currentlyLoadedModel: 'llama3', affectedSessions: [{ id: 's2', name: 'w2' }] },
|
||||
]);
|
||||
win.confirm = (() => false) as typeof win.confirm;
|
||||
let toastMessage: string | undefined;
|
||||
app.showToast = (msg: string) => {
|
||||
toastMessage = msg;
|
||||
};
|
||||
|
||||
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
|
||||
|
||||
expect(applyBodies).toHaveLength(1); // no second (confirmed) call
|
||||
expect(toastMessage).toMatch(/cancelled/i);
|
||||
});
|
||||
|
||||
it('a successful apply with modelSwapInProgress kicks off the loading watcher', async () => {
|
||||
const { app } = launchHarness([
|
||||
{ customModel: { endpointId: 'llama-box' }, restarted: true, modelSwapInProgress: true },
|
||||
]);
|
||||
let watched: unknown[] | null = null;
|
||||
app._watchLlamaSwapLoading = async (...args: unknown[]) => {
|
||||
watched = args;
|
||||
};
|
||||
|
||||
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
|
||||
|
||||
expect(watched).toEqual(['llama-box', 'qwen3']);
|
||||
});
|
||||
|
||||
it('a successful apply with no swap needed never starts the loading watcher', async () => {
|
||||
const { app } = launchHarness([
|
||||
{ customModel: { endpointId: 'llama-box' }, restarted: true, modelSwapInProgress: false },
|
||||
]);
|
||||
let watchCalled = false;
|
||||
app._watchLlamaSwapLoading = async () => {
|
||||
watchCalled = true;
|
||||
};
|
||||
|
||||
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
|
||||
|
||||
expect(watchCalled).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Custom Model Endpoint Profiles: _watchLlamaSwapLoading polling', () => {
|
||||
// Driven with millisecond intervals (the function's own pollIntervalMs/maxWaitMs
|
||||
// params — real callers never pass them) rather than fake timers: this code runs
|
||||
// inside the JSDOM window's own realm (bootApp's `runScripts: "dangerously"` eval),
|
||||
// whose setTimeout is NOT the one vi.useFakeTimers() patches, so advancing fake
|
||||
// timers here would advance nothing and either hang or silently no-op.
|
||||
|
||||
it('dismisses the loading toast as soon as the target model reports ready', async () => {
|
||||
const { app } = bootApp({});
|
||||
const toastCalls: Array<{ message: string; type: string }> = [];
|
||||
const dismissed: string[] = [];
|
||||
app.showToast = (message: string, type: string) => {
|
||||
toastCalls.push({ message, type });
|
||||
return { dismiss: () => dismissed.push(message), setMessage: () => {} };
|
||||
};
|
||||
app._apiJson = async () => ({ isLlamaSwap: true, running: [{ model: 'qwen3', state: 'ready' }] });
|
||||
|
||||
await app._watchLlamaSwapLoading('llama-box', 'qwen3', 5, 200);
|
||||
|
||||
expect(toastCalls[0].message).toMatch(/loading qwen3/i);
|
||||
expect(dismissed).toContain(toastCalls[0].message);
|
||||
expect(toastCalls.at(-1)?.message).toMatch(/ready/i);
|
||||
});
|
||||
|
||||
it('gives up after the bounded wait and warns instead of polling forever', async () => {
|
||||
const { app } = bootApp({});
|
||||
const toastCalls: string[] = [];
|
||||
app.showToast = (message: string) => {
|
||||
toastCalls.push(message);
|
||||
return { dismiss: () => {}, setMessage: () => {} };
|
||||
};
|
||||
app._apiJson = async () => ({ isLlamaSwap: true, running: [{ model: 'something-else', state: 'ready' }] });
|
||||
|
||||
await app._watchLlamaSwapLoading('llama-box', 'qwen3', 5, 30);
|
||||
|
||||
expect(toastCalls.at(-1)).toMatch(/still waiting/i);
|
||||
});
|
||||
|
||||
it('stops polling (without a warning) once the endpoint no longer reads as llama-swap', async () => {
|
||||
const { app } = bootApp({});
|
||||
const toastCalls: string[] = [];
|
||||
app.showToast = (message: string) => {
|
||||
toastCalls.push(message);
|
||||
return { dismiss: () => {}, setMessage: () => {} };
|
||||
};
|
||||
app._apiJson = async () => ({ isLlamaSwap: false, running: [] });
|
||||
|
||||
await app._watchLlamaSwapLoading('llama-box', 'qwen3', 5, 200);
|
||||
|
||||
expect(toastCalls).toHaveLength(1); // only the initial "Loading..." toast, no follow-up warning
|
||||
});
|
||||
|
||||
it('keeps waiting through a transient status-fetch failure instead of giving up early', async () => {
|
||||
const { app } = bootApp({});
|
||||
const toastCalls: string[] = [];
|
||||
app.showToast = (message: string) => {
|
||||
toastCalls.push(message);
|
||||
return { dismiss: () => {}, setMessage: () => {} };
|
||||
};
|
||||
let call = 0;
|
||||
app._apiJson = async () => {
|
||||
call += 1;
|
||||
if (call === 1) return null; // transient failure
|
||||
return { isLlamaSwap: true, running: [{ model: 'qwen3', state: 'ready' }] };
|
||||
};
|
||||
|
||||
await app._watchLlamaSwapLoading('llama-box', 'qwen3', 5, 200);
|
||||
|
||||
expect(toastCalls.at(-1)).toMatch(/ready/i);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -3,13 +3,28 @@
|
||||
* chunk 5 — applying/clearing a session's custom model endpoint + CLI restart).
|
||||
* Port: N/A (app.inject, no real port needed)
|
||||
*/
|
||||
import { describe, it, expect, beforeEach } from 'vitest';
|
||||
import { describe, it, expect, beforeEach, vi } from 'vitest';
|
||||
import { registerSessionRoutes } from '../../src/web/routes/session-routes.js';
|
||||
import { createRouteTestHarness } from './_route-test-utils.js';
|
||||
import { createMockSession } from '../mocks/index.js';
|
||||
import { getDataDir } from '../../src/config/instance.js';
|
||||
import { writeCustomModelHosts, type CustomModelHost } from '../../src/custom-model-hosts.js';
|
||||
import { existsSync, readFileSync, statSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
import { webviewFetch } from '../../src/web/webview-egress.js';
|
||||
|
||||
// Every apply now also checks llama-swap's `GET /running` (session-routes.ts) before
|
||||
// applying — without this mock every test in this file would make a REAL network request
|
||||
// to the fake 192.168.1.50 endpoint below and wait out its 5s timeout. Defaults to a plain
|
||||
// 404 (reads as "not llama-swap", exercising none of the new conflict-check tests below),
|
||||
// overridden per-test where the llama-swap behavior itself is what's under test.
|
||||
vi.mock('../../src/web/webview-egress.js', async () => {
|
||||
const actual = await vi.importActual<typeof import('../../src/web/webview-egress.js')>(
|
||||
'../../src/web/webview-egress.js'
|
||||
);
|
||||
return { ...actual, webviewFetch: vi.fn() };
|
||||
});
|
||||
const fetchMock = vi.mocked(webviewFetch);
|
||||
|
||||
const CLAUDE_ENDPOINT: CustomModelHost = {
|
||||
id: 'ep1',
|
||||
@@ -26,6 +41,8 @@ async function setup() {
|
||||
describe('POST /api/sessions/:id/custom-model', () => {
|
||||
beforeEach(async () => {
|
||||
await writeCustomModelHosts(getDataDir(), []);
|
||||
fetchMock.mockReset();
|
||||
fetchMock.mockResolvedValue(new Response('not found', { status: 404 }));
|
||||
});
|
||||
|
||||
it('applies an endpoint/model to a claude-mode session and restarts the CLI', async () => {
|
||||
@@ -211,6 +228,129 @@ describe('POST /api/sessions/:id/custom-model', () => {
|
||||
expect(existsSync(dir)).toBe(false);
|
||||
});
|
||||
|
||||
describe('llama-swap conflict check (llama.cpp runs one model at a time)', () => {
|
||||
function mockRunning(running: Array<{ model: string; state: string }>) {
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/running') return new Response(JSON.stringify({ running }), { status: 200 });
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
}
|
||||
|
||||
it('applies straight away when the requested model is already loaded', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
ctx.sessions.get('test-session-1')!.mode = 'claude';
|
||||
mockRunning([{ model: 'qwen3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().success).not.toBe(false);
|
||||
expect(res.json().modelSwapInProgress).toBe(false);
|
||||
expect(ctx.sessions.get('test-session-1')!.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('applies straight away when a swap is needed but nothing else is using the loaded model, flagging modelSwapInProgress', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
ctx.sessions.get('test-session-1')!.mode = 'claude';
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().success).not.toBe(false);
|
||||
expect(res.json().modelSwapInProgress).toBe(true);
|
||||
expect(ctx.sessions.get('test-session-1')!.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('asks for confirmation instead of applying when another session is actively using the currently loaded model', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
const other = createMockSession('other-session');
|
||||
other.name = 'w2-otherbox';
|
||||
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
|
||||
ctx.sessions.set('other-session', other);
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.success).not.toBe(false);
|
||||
expect(body.requiresConfirmation).toBe(true);
|
||||
expect(body.currentlyLoadedModel).toBe('llama3');
|
||||
expect(body.affectedSessions).toEqual([{ id: 'other-session', name: 'w2-otherbox' }]);
|
||||
// Nothing actually applied yet — this call only asked, it did not switch.
|
||||
expect(session.setCustomModel).not.toHaveBeenCalled();
|
||||
expect(session.restartCli).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('applies once confirmed, skipping the conflict check the second time', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
const other = createMockSession('other-session');
|
||||
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
|
||||
ctx.sessions.set('other-session', other);
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3', confirmed: true },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.requiresConfirmation).toBeUndefined();
|
||||
expect(body.modelSwapInProgress).toBe(true);
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
expect(session.restartCli).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('a session pointed at the SAME endpoint but a DIFFERENT (not-currently-loaded) model is not treated as affected', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
const other = createMockSession('other-session');
|
||||
other.customModel = { endpointId: 'ep1', modelId: 'some-other-model' }; // not the loaded one
|
||||
ctx.sessions.set('other-session', other);
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresConfirmation).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('not llama-swap (plain llama.cpp/OpenAI-compatible server, no /running) — never checked, applies straight away', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
ctx.sessions.get('test-session-1')!.mode = 'claude';
|
||||
fetchMock.mockResolvedValue(new Response('not found', { status: 404 }));
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().modelSwapInProgress).toBe(false);
|
||||
expect(res.json().requiresConfirmation).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
it('refuses to touch a busy session', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
|
||||
Reference in New Issue
Block a user