mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-06 07:29:42 +02:00
feat(custom-model): detect and notify when a session's model gets swapped out later
The llama-swap conflict check on the apply/create routes only ever runs at THAT session's own launch/apply moment, and cannot see a swap caused by a DIFFERENT session's later, ordinary use. Confirmed live: a second Codex session picking a different model launched with no warning at all — nothing conflicted at that exact instant — yet it silently evicted the first session's model regardless (llama.cpp runs one model at a time). Reproduced and root-caused via direct API calls against a live test-picker instance rather than guessing. - detectCustomModelSwapDisplacements() (custom-model-routes.ts): groups live sessions with a customModel by endpointId, checks each group's endpoint via GET /running once, and flags a session whose own modelId is no longer in the running list. Read-only, best-effort per endpoint like refreshAllCustomModelHosts's sibling sweep. - Notifies once per displacement via a caller-owned de-dupe Set: a session id is added when displaced, removed once its own model is loaded/ready again, so a later genuinely-new displacement can notify again. - New periodic sweep in server.ts (CUSTOM_MODEL_SWAP_CHECK_INTERVAL_MS, 20s — much shorter than the 5-minute model-list refresh, since this is time-sensitive) broadcasts a new custom-model:swapped-out SSE event per displacement. De-dupe Set cleared per-session on session cleanup to avoid an unbounded leak. - Frontend: global toast (not tied to the displaced session's tab, since the point is warning before the user types into it) naming the session, its previous model, and what's currently loaded. Chose the "detect after the fact" scope (vs. checking before every message send, which would add a round-trip to every turn on every custom-model session) per explicit user decision after being presented the trade-off. 9 new tests for the detection logic (flag/clear/re-flag cycle, unreachable/deleted endpoints, non-llama-swap servers, multiple sessions on one endpoint). SSE registry bumped 158->159, parity test passing. Typecheck/lint/frontend-syntax clean; full suite shows no new regressions (9 more passing than baseline, matching the new tests; same pre-existing Windows-environment failures). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
470f75b08c
commit
5ddc028a2f
@@ -428,6 +428,91 @@ export async function refreshAllCustomModelHosts(): Promise<void> {
|
||||
}
|
||||
}
|
||||
|
||||
/** The subset of `Session` this sweep needs — kept minimal so a test can pass a plain object. */
|
||||
export interface CustomModelSessionLike {
|
||||
id: string;
|
||||
name: string;
|
||||
customModel?: { endpointId: string; modelId: string; label?: string };
|
||||
}
|
||||
|
||||
/** One session whose model was just found evicted, ready to broadcast as `CustomModelSwappedOut`. */
|
||||
export interface CustomModelSwapDisplacement {
|
||||
sessionId: string;
|
||||
sessionName: string;
|
||||
endpointId: string;
|
||||
previousModel: string;
|
||||
currentlyLoadedModel: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Detects when a live session's own custom-model selection is no longer the model
|
||||
* llama-swap actually has loaded — evicted by ANOTHER session's activity on the same
|
||||
* endpoint, since llama.cpp/llama-swap runs one model at a time (the apply/create routes'
|
||||
* own swap-conflict check only ever runs at THAT session's own launch/apply moment, so it
|
||||
* cannot catch a later eviction triggered by a different session's normal use — confirmed
|
||||
* live: a session created while nothing else had a live conflict at that instant can still
|
||||
* get silently displaced afterward). Read-only, and best-effort per endpoint exactly like
|
||||
* `refreshAllCustomModelHosts`'s sibling sweep — one endpoint's hiccup here never blocks
|
||||
* checking the others.
|
||||
*
|
||||
* `notifiedSessionIds` is the caller's own de-dupe state (`server.ts` keeps one `Set` across
|
||||
* sweeps), mutated in place: a session id is added once displaced and removed again once its
|
||||
* own model is loaded and ready — so a LATER, genuinely new displacement can notify again
|
||||
* rather than the session staying silently un-notified forever after the first one.
|
||||
*/
|
||||
export async function detectCustomModelSwapDisplacements(
|
||||
sessions: Iterable<CustomModelSessionLike>,
|
||||
notifiedSessionIds: Set<string>
|
||||
): Promise<CustomModelSwapDisplacement[]> {
|
||||
const byEndpoint = new Map<string, CustomModelSessionLike[]>();
|
||||
for (const session of sessions) {
|
||||
if (!session.customModel) continue;
|
||||
const group = byEndpoint.get(session.customModel.endpointId);
|
||||
if (group) group.push(session);
|
||||
else byEndpoint.set(session.customModel.endpointId, [session]);
|
||||
}
|
||||
if (byEndpoint.size === 0) return [];
|
||||
|
||||
const hosts = await readCustomModelHosts(getDataDir());
|
||||
const displacements: CustomModelSwapDisplacement[] = [];
|
||||
|
||||
for (const [endpointId, group] of byEndpoint) {
|
||||
const host = hosts.find((h) => h.id === endpointId);
|
||||
if (!host) continue; // endpoint deleted since these sessions were created — nothing to check
|
||||
let status: LlamaSwapStatus;
|
||||
try {
|
||||
status = await getLlamaSwapStatus(host);
|
||||
} catch {
|
||||
continue; // unreachable this cycle — try again next tick, not fatal to the sweep
|
||||
}
|
||||
// Not llama-swap (feature-detected) or nothing loaded at all: nothing has been evicted,
|
||||
// by construction — a plain llama.cpp/OpenAI-compatible server only ever runs the one
|
||||
// model it was started with, so there is no "current model" to conflict with.
|
||||
if (!status.isLlamaSwap || status.running.length === 0) continue;
|
||||
const currentlyLoaded = status.running.find((r) => r.state === 'ready')?.model ?? status.running[0]?.model;
|
||||
if (!currentlyLoaded) continue;
|
||||
|
||||
for (const session of group) {
|
||||
const modelId = session.customModel!.modelId;
|
||||
const stillLoaded = status.running.some((r) => r.model === modelId);
|
||||
if (stillLoaded) {
|
||||
notifiedSessionIds.delete(session.id); // back to normal — a future eviction can notify again
|
||||
continue;
|
||||
}
|
||||
if (notifiedSessionIds.has(session.id)) continue; // already told them once for this displacement
|
||||
notifiedSessionIds.add(session.id);
|
||||
displacements.push({
|
||||
sessionId: session.id,
|
||||
sessionName: session.name,
|
||||
endpointId,
|
||||
previousModel: modelId,
|
||||
currentlyLoadedModel: currentlyLoaded,
|
||||
});
|
||||
}
|
||||
}
|
||||
return displacements;
|
||||
}
|
||||
|
||||
export function registerCustomModelRoutes(app: FastifyInstance): void {
|
||||
app.get('/api/model-endpoints', async (req): Promise<RedactedHost[]> => {
|
||||
if (isMultiUserMode() && !isAdmin(req)) return [];
|
||||
|
||||
@@ -27,4 +27,10 @@ export { registerWsRoutes } from './ws-routes.js';
|
||||
export { registerVoiceRoutes } from './voice-routes.js';
|
||||
export { registerWebviewRoutes, tryWebviewRefererFallback } from './webview-routes.js';
|
||||
export { registerTabLayoutRoutes } from './tab-layout-routes.js';
|
||||
export { registerCustomModelRoutes, refreshAllCustomModelHosts } from './custom-model-routes.js';
|
||||
export {
|
||||
registerCustomModelRoutes,
|
||||
refreshAllCustomModelHosts,
|
||||
detectCustomModelSwapDisplacements,
|
||||
type CustomModelSessionLike,
|
||||
type CustomModelSwapDisplacement,
|
||||
} from './custom-model-routes.js';
|
||||
|
||||
Reference in New Issue
Block a user