fix(custom-model): actually trigger the llama-swap load, not just watch for it

Root cause of "it doesn't look like llama-swap is actually switching the
model" (confirmed live: no load_model line in llama-swap's own logs after
applying a selection). llama-swap has no "switch model" admin endpoint - the
ONLY thing that starts a swap is a real inference request naming the model.
Every previous fix (the conflict check, the loading banner) assumed a swap
would start on its own; nothing ever actually asked llama-swap to load
anything until the launched CLI's first real prompt did, which could be
much later than "applying the selection" implied.

Adds triggerLlamaSwapLoad() (custom-model-routes.ts): sends the smallest
real request that will start a load - POST <baseUrl>/v1/chat/completions,
max_tokens: 1, one throwaway message - fire-and-forget (never awaited by
the caller; the frontend's own running-status polling is what actually
confirms readiness). Wired into both apply paths (the dedicated restart
route and the one-shot quick-start route), fired whenever the target model
isn't already the one loaded and ready - a broader condition than the
existing swapNeeded (which only gates the "this will evict another
session's model" confirmation ask and deliberately stays narrow to that).
modelSwapInProgress in both routes' responses now reflects this same
broader condition too, so the frontend's loading banner actually correlates
with a real in-flight load rather than only firing when something else
happened to be loaded already.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
Devvyn
2026-09-16 15:53:34 +08:00
co-authored by Claude Sonnet 5
parent 01b32ee6cd
commit 0929694012
6 changed files with 246 additions and 2 deletions
+38
View File
@@ -228,6 +228,44 @@ export async function getLlamaSwapStatus(
}
}
/**
* Actually kicks off llama-swap's lazy model load, rather than waiting for the launched
* CLI's own first prompt to do it. llama-swap has no separate "switch model" admin
* endpoint — the ONLY thing that starts a swap is a real inference request naming the
* model (confirmed live: applying a selection alone never appeared in the llama-swap
* server's own logs; nothing had actually asked it to load anything). This sends the
* smallest real request that will — `max_tokens: 1`, one throwaway user message — to
* `${baseUrl}/v1/chat/completions`, the OpenAI-compatible endpoint every supported
* harness already points at.
*
* Deliberately fire-and-forget: the caller (the apply/create routes) returns to the
* client immediately, and the frontend's own polling (`GET .../running-status`) is what
* actually confirms readiness — this call's response is never read, just its side
* effect. No abort/timeout of its own either: a real load can take well over a minute for
* a large model, and this is a normal long-running Node process, so there is nothing to
* clean up by cutting it short. Errors are swallowed for the same reason `discoverModels`'s
* siblings swallow theirs — one endpoint's hiccup here is a nice-to-have that failed, not
* something worth surfacing as a request failure four layers up.
*/
export function triggerLlamaSwapLoad(
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>,
modelId: string
): void {
const url = new URL(`${host.baseUrl.replace(/\/+$/, '')}/v1/chat/completions`);
webviewFetch(url, {
method: 'POST',
headers: { ...authHeaders(host), 'content-type': 'application/json' },
body: JSON.stringify({
model: modelId,
messages: [{ role: 'user', content: 'Hi' }],
max_tokens: 1,
stream: false,
}),
}).catch(() => {
// best-effort — see the doc comment above
});
}
function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult): CustomModelHost {
const { models, contextLengths } = result;
const defaultModelId = host.defaultModelId && models.includes(host.defaultModelId) ? host.defaultModelId : undefined;
+34 -2
View File
@@ -55,7 +55,7 @@ import {
} from '../schemas.js';
import { readCustomModelHosts } from '../../custom-model-hosts.js';
import { applyCustomModelInjection, removeConfigDir } from '../../custom-model-injection-apply.js';
import { getLlamaSwapStatus } from './custom-model-routes.js';
import { getLlamaSwapStatus, triggerLlamaSwapLoad } from './custom-model-routes.js';
import { matchesPattern } from '../../config/cli-registry/patterns.js';
import { ownerLayoutKey } from '../../tab-layout-persistence.js';
import { TabLayoutValidationError } from '../../tab-layout.js';
@@ -1217,7 +1217,16 @@ export function registerSessionRoutes(
// server has no such endpoint and reads as `isLlamaSwap: false` — nothing to check).
const swapStatus = await getLlamaSwapStatus(endpoint);
const currentlyLoaded = swapStatus.running.find((r) => r.state === 'ready')?.model ?? swapStatus.running[0]?.model;
// Distinct from targetReady below: this is ONLY about whether proceeding would evict a
// model another session is actively using — true even if nothing is loaded at all yet
// would be wrong here (nothing to evict), so this stays narrowly "a DIFFERENT model is
// currently ready".
const swapNeeded = swapStatus.isLlamaSwap && !!currentlyLoaded && currentlyLoaded !== body.modelId;
// Whether the TARGET model itself is already the one loaded and ready — false whether
// nothing is loaded yet, a different model is loaded, or this one is loaded but still
// mid-load. Drives both the actual load trigger below and modelSwapInProgress in the
// response; deliberately broader than swapNeeded, which only gates the confirmation ask.
const targetReady = swapStatus.running.some((r) => r.model === body.modelId && r.state === 'ready');
// Only ask when switching would actually take the model away from another session
// that is currently using it — never just because a swap is needed at all. `confirmed`
@@ -1275,9 +1284,17 @@ export function registerSessionRoutes(
removeConfigDir(previousConfigDir);
}
// Actually kick off llama-swap's load now, rather than waiting on the restarted CLI's
// own first prompt to do it — confirmed live that applying a selection alone never
// reached the llama-swap server at all (nothing in its own logs), since llama-swap has
// no "switch model" admin call, only a real inference request naming the model.
if (swapStatus.isLlamaSwap && !targetReady) {
triggerLlamaSwapLoad(endpoint, body.modelId);
}
const restarted = await session.restartCli();
persistAndBroadcastSession(ctx, session);
return { customModel: session.customModel, restarted, modelSwapInProgress: swapNeeded };
return { customModel: session.customModel, restarted, modelSwapInProgress: swapStatus.isLlamaSwap && !targetReady };
});
// ========== Delete Session ==========
@@ -3533,6 +3550,7 @@ export function registerSessionRoutes(
let qsCustomModelEnvOverrides = qsGatedEnvOverrides;
let qsCustomModelLaunchModel: string | undefined;
let qsCustomModelSessionId: string | undefined;
let qsCustomModelSwapInProgress = false;
let qsCustomModelBookkeeping:
| {
endpointId: string;
@@ -3562,6 +3580,11 @@ export function registerSessionRoutes(
const cmCurrentlyLoaded =
cmSwapStatus.running.find((r) => r.state === 'ready')?.model ?? cmSwapStatus.running[0]?.model;
const cmSwapNeeded = cmSwapStatus.isLlamaSwap && !!cmCurrentlyLoaded && cmCurrentlyLoaded !== customModel.modelId;
// Broader than cmSwapNeeded (which only gates the confirmation ask above): true
// whenever the TARGET model isn't already loaded and ready, including when nothing
// is loaded at all yet. Drives the actual load trigger below.
const cmTargetReady = cmSwapStatus.running.some((r) => r.model === customModel.modelId && r.state === 'ready');
qsCustomModelSwapInProgress = cmSwapStatus.isLlamaSwap && !cmTargetReady;
if (cmSwapNeeded && !customModel.confirmed) {
const cmAffectedSessions = [...ctx.sessions.values()]
.filter((s) => s.customModel?.endpointId === cmEndpoint.id && s.customModel?.modelId === cmCurrentlyLoaded)
@@ -3614,6 +3637,14 @@ export function registerSessionRoutes(
configDir: cmApplied.configDir,
launchModel: cmApplied.launchModel,
};
// Actually kick off llama-swap's load now — see the dedicated apply route's own
// comment on triggerLlamaSwapLoad for why this can't just wait on the launched CLI's
// first prompt. Fired here, before the session is even created, so the load starts
// concurrently with Claude/Codex/etc. booting rather than after.
if (qsCustomModelSwapInProgress) {
triggerLlamaSwapLoad(cmEndpoint, customModel.modelId);
}
}
const session = new Session({
@@ -3761,6 +3792,7 @@ export function registerSessionRoutes(
sessionId: session.id,
casePath: resolvedCasePath,
caseName,
...(customModel ? { modelSwapInProgress: qsCustomModelSwapInProgress } : {}),
};
} catch (err) {
// Clean up session on error to prevent orphaned resources