fix(custom-model): poll llama-swap readiness every 1s, check immediately, extend the cap

Reported: the "Loading..." banner stayed up past 2 minutes even though
llama-swap itself had already finished loading the model. Three fixes:

1. pollIntervalMs default 3000ms -> 1000ms (as asked).
2. The loop now checks readiness IMMEDIATELY on entry rather than sleeping
   a full interval first - a model that's already ready (a fast load, or a
   re-apply onto one already loaded) shouldn't sit on "Loading..." at all.
3. maxWaitMs default 120000ms (2 min) -> 300000ms (5 min): a large (20GB+)
   model reading from disk can genuinely take longer than 2 minutes, which
   would have looked identical to the reported symptom - "still stuck past
   the point it should have resolved" - except it would have actually
   flipped to a "still waiting" warning toast at the 2-minute mark rather
   than staying on "Loading" indefinitely, so this alone doesn't explain
   what was reported, but is a real, separate improvement worth making.

Also fixes a real, separate bug this surfaced while reasoning through the
report: _showCenterStatus's banner is ONE shared, reused DOM node. A second
call to _watchLlamaSwapLoading (e.g. switching models again before the
first switch's loop had finished) would take over that shared banner, but
the FIRST loop was still running and would eventually dismiss or overwrite
it once ITS OWN deadline or readiness check resolved - clobbering whatever
the second, current loop had put there. A generation counter
(_watchLlamaSwapGeneration) now lets each call recognise when it no longer
owns the banner and stop touching it silently, rather than only the last
call to actually start ever safely reading or writing it.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
Devvyn
2026-09-16 19:44:37 +08:00
co-authored by Claude Sonnet 5
parent 0929694012
commit 0af233c96c
2 changed files with 82 additions and 9 deletions
+58
View File
@@ -676,6 +676,64 @@ describe('Custom Model Endpoint Profiles: _watchLlamaSwapLoading polling', () =>
expect(toastCalls.at(-1)).toMatch(/ready/i);
});
it('checks immediately rather than waiting a full interval before the first check', async () => {
// A model that is already ready by the time this runs (a fast load, or a re-apply
// onto one that was already loaded) shouldn't sit on "Loading..." for a whole
// pollIntervalMs before saying so.
const { app } = bootApp({});
app._showCenterStatus = () => ({ dismiss: () => {}, setMessage: () => {} });
let calls = 0;
app._apiJson = async () => {
calls += 1;
return { isLlamaSwap: true, running: [{ model: 'qwen3', state: 'ready' }] };
};
// A huge interval that would time the test out if the function actually waited for
// it before the first check.
await app._watchLlamaSwapLoading('llama-box', 'qwen3', 60000, 300000);
expect(calls).toBe(1);
});
it('a newer call takes over the shared banner — the older one neither dismisses nor overwrites it', async () => {
const { app } = bootApp({});
const bannerCalls: string[] = [];
const dismissCalls: string[] = [];
app._showCenterStatus = (message: string) => {
bannerCalls.push(message);
return { dismiss: () => dismissCalls.push(message), setMessage: () => {} };
};
app.showToast = () => {};
// The FIRST call never sees its target model ready, so it would otherwise run all
// the way to its own timeout and dismiss/warn — but a second call starts first.
let firstResolveApiJson: (() => void) | undefined;
const firstNeverReady = new Promise<void>((resolve) => {
firstResolveApiJson = resolve;
});
app._apiJson = async () => {
await firstNeverReady; // block the first loop's very first check indefinitely
return { isLlamaSwap: true, running: [] };
};
const firstCall = app._watchLlamaSwapLoading('llama-box', 'model-a', 5, 50);
// Second call, for a DIFFERENT model, starts while the first is still blocked on its
// very first status check — claims the banner as the newer generation.
app._apiJson = async () => ({ isLlamaSwap: true, running: [{ model: 'model-b', state: 'ready' }] });
await app._watchLlamaSwapLoading('llama-box', 'model-b', 5, 200);
// Now let the first call's blocked check resolve and run to completion.
firstResolveApiJson?.();
await firstCall;
expect(bannerCalls).toEqual([
'Loading model-a on llama-box… this can take a while',
'Loading model-b on llama-box… this can take a while',
]);
// Only the CURRENT (second) call's own dismiss ever ran — the stale first call's
// late resolution recognised it no longer owns the banner and touched nothing.
expect(dismissCalls).toEqual([bannerCalls[1]]);
});
});
describe('Custom Model Endpoint Profiles: _confirmModelSwap (in-app modal, replaces a native confirm() popup)', () => {