mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-05 15:09:42 +02:00
fix(custom-model): poll llama-swap readiness every 1s, check immediately, extend the cap
Reported: the "Loading..." banner stayed up past 2 minutes even though llama-swap itself had already finished loading the model. Three fixes: 1. pollIntervalMs default 3000ms -> 1000ms (as asked). 2. The loop now checks readiness IMMEDIATELY on entry rather than sleeping a full interval first - a model that's already ready (a fast load, or a re-apply onto one already loaded) shouldn't sit on "Loading..." at all. 3. maxWaitMs default 120000ms (2 min) -> 300000ms (5 min): a large (20GB+) model reading from disk can genuinely take longer than 2 minutes, which would have looked identical to the reported symptom - "still stuck past the point it should have resolved" - except it would have actually flipped to a "still waiting" warning toast at the 2-minute mark rather than staying on "Loading" indefinitely, so this alone doesn't explain what was reported, but is a real, separate improvement worth making. Also fixes a real, separate bug this surfaced while reasoning through the report: _showCenterStatus's banner is ONE shared, reused DOM node. A second call to _watchLlamaSwapLoading (e.g. switching models again before the first switch's loop had finished) would take over that shared banner, but the FIRST loop was still running and would eventually dismiss or overwrite it once ITS OWN deadline or readiness check resolved - clobbering whatever the second, current loop had put there. A generation counter (_watchLlamaSwapGeneration) now lets each call recognise when it no longer owns the banner and stop touching it silently, rather than only the last call to actually start ever safely reading or writing it. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
0929694012
commit
0af233c96c
@@ -676,6 +676,64 @@ describe('Custom Model Endpoint Profiles: _watchLlamaSwapLoading polling', () =>
|
||||
|
||||
expect(toastCalls.at(-1)).toMatch(/ready/i);
|
||||
});
|
||||
|
||||
it('checks immediately rather than waiting a full interval before the first check', async () => {
|
||||
// A model that is already ready by the time this runs (a fast load, or a re-apply
|
||||
// onto one that was already loaded) shouldn't sit on "Loading..." for a whole
|
||||
// pollIntervalMs before saying so.
|
||||
const { app } = bootApp({});
|
||||
app._showCenterStatus = () => ({ dismiss: () => {}, setMessage: () => {} });
|
||||
let calls = 0;
|
||||
app._apiJson = async () => {
|
||||
calls += 1;
|
||||
return { isLlamaSwap: true, running: [{ model: 'qwen3', state: 'ready' }] };
|
||||
};
|
||||
|
||||
// A huge interval that would time the test out if the function actually waited for
|
||||
// it before the first check.
|
||||
await app._watchLlamaSwapLoading('llama-box', 'qwen3', 60000, 300000);
|
||||
|
||||
expect(calls).toBe(1);
|
||||
});
|
||||
|
||||
it('a newer call takes over the shared banner — the older one neither dismisses nor overwrites it', async () => {
|
||||
const { app } = bootApp({});
|
||||
const bannerCalls: string[] = [];
|
||||
const dismissCalls: string[] = [];
|
||||
app._showCenterStatus = (message: string) => {
|
||||
bannerCalls.push(message);
|
||||
return { dismiss: () => dismissCalls.push(message), setMessage: () => {} };
|
||||
};
|
||||
app.showToast = () => {};
|
||||
// The FIRST call never sees its target model ready, so it would otherwise run all
|
||||
// the way to its own timeout and dismiss/warn — but a second call starts first.
|
||||
let firstResolveApiJson: (() => void) | undefined;
|
||||
const firstNeverReady = new Promise<void>((resolve) => {
|
||||
firstResolveApiJson = resolve;
|
||||
});
|
||||
app._apiJson = async () => {
|
||||
await firstNeverReady; // block the first loop's very first check indefinitely
|
||||
return { isLlamaSwap: true, running: [] };
|
||||
};
|
||||
const firstCall = app._watchLlamaSwapLoading('llama-box', 'model-a', 5, 50);
|
||||
|
||||
// Second call, for a DIFFERENT model, starts while the first is still blocked on its
|
||||
// very first status check — claims the banner as the newer generation.
|
||||
app._apiJson = async () => ({ isLlamaSwap: true, running: [{ model: 'model-b', state: 'ready' }] });
|
||||
await app._watchLlamaSwapLoading('llama-box', 'model-b', 5, 200);
|
||||
|
||||
// Now let the first call's blocked check resolve and run to completion.
|
||||
firstResolveApiJson?.();
|
||||
await firstCall;
|
||||
|
||||
expect(bannerCalls).toEqual([
|
||||
'Loading model-a on llama-box… this can take a while',
|
||||
'Loading model-b on llama-box… this can take a while',
|
||||
]);
|
||||
// Only the CURRENT (second) call's own dismiss ever ran — the stale first call's
|
||||
// late resolution recognised it no longer owns the banner and touched nothing.
|
||||
expect(dismissCalls).toEqual([bannerCalls[1]]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Custom Model Endpoint Profiles: _confirmModelSwap (in-app modal, replaces a native confirm() popup)', () => {
|
||||
|
||||
Reference in New Issue
Block a user