fix(custom-model): actually trigger the llama-swap load, not just watch for it

Root cause of "it doesn't look like llama-swap is actually switching the
model" (confirmed live: no load_model line in llama-swap's own logs after
applying a selection). llama-swap has no "switch model" admin endpoint - the
ONLY thing that starts a swap is a real inference request naming the model.
Every previous fix (the conflict check, the loading banner) assumed a swap
would start on its own; nothing ever actually asked llama-swap to load
anything until the launched CLI's first real prompt did, which could be
much later than "applying the selection" implied.

Adds triggerLlamaSwapLoad() (custom-model-routes.ts): sends the smallest
real request that will start a load - POST <baseUrl>/v1/chat/completions,
max_tokens: 1, one throwaway message - fire-and-forget (never awaited by
the caller; the frontend's own running-status polling is what actually
confirms readiness). Wired into both apply paths (the dedicated restart
route and the one-shot quick-start route), fired whenever the target model
isn't already the one loaded and ready - a broader condition than the
existing swapNeeded (which only gates the "this will evict another
session's model" confirmation ask and deliberately stays narrow to that).
modelSwapInProgress in both routes' responses now reflects this same
broader condition too, so the frontend's loading banner actually correlates
with a real in-flight load rather than only firing when something else
happened to be loaded already.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
Devvyn
2026-09-16 15:53:34 +08:00
co-authored by Claude Sonnet 5
parent 01b32ee6cd
commit 0929694012
6 changed files with 246 additions and 2 deletions
@@ -274,4 +274,56 @@ describe('POST /api/quick-start: customModel (one-shot custom-model launch)', ()
expect(res.json().requiresConfirmation).toBeUndefined();
});
});
describe('triggering the actual llama-swap load (not just watching for it)', () => {
it('sends a real inference request naming the target model, concurrently with launching the session', async () => {
const chatCalls: unknown[] = [];
fetchMock.mockImplementation(async (url: URL, init?: { body?: unknown }) => {
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 });
}
if (url.pathname === '/v1/chat/completions') {
chatCalls.push(JSON.parse(init!.body as string));
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
const res = await quickStart({
caseName: 'cm-trigger',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
await new Promise((resolve) => setTimeout(resolve, 0)); // let the fire-and-forget trigger settle
expect(res.statusCode).toBe(200);
expect(res.json().modelSwapInProgress).toBe(true);
expect(chatCalls).toHaveLength(1);
expect(chatCalls[0]).toMatchObject({ model: 'qwen3', max_tokens: 1 });
});
it('never sends a load-trigger request when the target model is already loaded and ready', async () => {
let chatCalled = false;
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'qwen3', state: 'ready' }] }), { status: 200 });
}
if (url.pathname === '/v1/chat/completions') {
chatCalled = true;
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
const res = await quickStart({
caseName: 'cm-no-trigger',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
await new Promise((resolve) => setTimeout(resolve, 0));
expect(res.json().modelSwapInProgress).toBe(false);
expect(chatCalled).toBe(false);
});
});
});
+83
View File
@@ -351,6 +351,89 @@ describe('POST /api/sessions/:id/custom-model', () => {
});
});
describe('triggering the actual llama-swap load (not just watching for it)', () => {
it('sends a real inference request naming the target model when it is not already loaded and ready', async () => {
const { app, ctx } = await setup();
ctx.sessions.get('test-session-1')!.mode = 'claude';
const chatCalls: unknown[] = [];
fetchMock.mockImplementation(async (url: URL, init?: { body?: unknown }) => {
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 });
}
if (url.pathname === '/v1/chat/completions') {
chatCalls.push(JSON.parse(init!.body as string));
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
await new Promise((resolve) => setTimeout(resolve, 0)); // let the fire-and-forget trigger settle
expect(chatCalls).toHaveLength(1);
expect(chatCalls[0]).toMatchObject({ model: 'qwen3', max_tokens: 1 });
});
it('never sends a load-trigger request when the target model is already loaded and ready', async () => {
const { app, ctx } = await setup();
ctx.sessions.get('test-session-1')!.mode = 'claude';
let chatCalled = false;
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'qwen3', state: 'ready' }] }), { status: 200 });
}
if (url.pathname === '/v1/chat/completions') {
chatCalled = true;
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
await new Promise((resolve) => setTimeout(resolve, 0));
expect(chatCalled).toBe(false);
});
it('never sends a load-trigger request while confirmation is still pending', async () => {
const { app, ctx } = await setup();
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const other = createMockSession('other-session');
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
ctx.sessions.set('other-session', other);
let chatCalled = false;
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 });
}
if (url.pathname === '/v1/chat/completions') {
chatCalled = true;
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
await new Promise((resolve) => setTimeout(resolve, 0));
expect(res.json().requiresConfirmation).toBe(true);
expect(chatCalled).toBe(false);
});
});
it('refuses to touch a busy session', async () => {
const { app, ctx } = await setup();
const session = ctx.sessions.get('test-session-1')!;