From 55dae31530d3af69770489348b996e19f50146ed Mon Sep 17 00:00:00 2001 From: Devvyn <22340871+opticon454@users.noreply.github.com> Date: Wed, 16 Sep 2026 19:58:46 +0800 Subject: [PATCH] feat(custom-model): estimate model load time from its discovered size Discovery now also parses a GB figure out of an auto-discovered model's own description (llama-swap writes "Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp"), stored per model as modelSizesGB - unlike context length this needs no /props probe (the figure is right there in /v1/models) so it is populated for every model regardless of loaded state. A hand-configured profile's own description has no such figure and correctly gets no entry. The loading banner (_watchLlamaSwapLoading) now looks this up and, when known, shows it plus a rough estimate from a small size->time matrix (_estimateModelLoad/_MODEL_LOAD_TIME_MATRIX, session-ui.js) - "Loading qwen3.8-27b-ud-q4_k_xl (16.4 GB, typically ~1-3 min) on llama-swap... this can take a while" - and uses that same estimate's own bracket to scale the banner's default give-up timeout for a very large model, instead of a flat 5 minutes for everything. Explicitly labelled as an UNMEASURED, typical-hardware estimate in every relevant comment - this is not benchmarked against any real endpoint's actual storage/GPU, just a reasonable expectation-setter. A model with no discoverable size (a hand-configured profile) gets no size/estimate shown at all, matching the "never a guess" convention modelContextLengths already established. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG --- docs/custom-model-endpoints.md | 14 ++ src/custom-model-hosts.ts | 11 ++ src/web/public/session-ui.js | 66 +++++++- src/web/routes/custom-model-routes.ts | 49 +++++- src/web/schemas.ts | 2 + .../custom-model-endpoint-rediscovery.test.ts | 74 +++++++++ test/custom-model-run-menu-ui.test.ts | 154 ++++++++++++++---- 7 files changed, 329 insertions(+), 41 deletions(-) diff --git a/docs/custom-model-endpoints.md b/docs/custom-model-endpoints.md index 25045d5f..9d7997b5 100644 --- a/docs/custom-model-endpoints.md +++ b/docs/custom-model-endpoints.md @@ -83,6 +83,20 @@ session" below) so a CLI that would otherwise assume a large default context window for an unrecognized model id stops silently overflowing a much smaller real one. +**File size is discovered too, when the server states one.** llama-swap +writes a GB figure into an auto-discovered model's own `description` +(`"Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp"`), parsed +into `modelSizesGB` — unlike context length, this needs no `/props` probe +(the figure is right there in the `/v1/models` response) and so is populated +for every model regardless of loaded state. A hand-configured profile's own +description has no such figure and correctly gets no entry, never a guess. +Used only to label the Run-menu picker's "loading model" banner with a +rough, UNMEASURED expected-time estimate (`_estimateModelLoad()` in +session-ui.js, based on typical local NVMe/SSD throughput — not benchmarked +against any real endpoint's actual hardware/storage) and to scale that same +banner's own give-up timeout for a very large model; never anything a +server-side check relies on. + `defaultModelId` names which discovered model the picker pre-marks for that endpoint — the settings panel's Edit form exposes it as a select populated from the endpoint's own discovered `models`, and the route refuses a value diff --git a/src/custom-model-hosts.ts b/src/custom-model-hosts.ts index cea7e7e5..61ef75f7 100644 --- a/src/custom-model-hosts.ts +++ b/src/custom-model-hosts.ts @@ -61,6 +61,17 @@ export interface CustomModelHost { * has no entry for simply gets no context-length env override applied — never a guess. */ modelContextLengths?: Record; + /** + * Discovered file size (GB) per model id, keyed by the same strings as `models`. + * Populated during discovery by parsing llama-swap's own `description` field for an + * auto-discovered model ("Auto-discovered 16.35 GB - parameters auto-fitted by + * llama.cpp") — a hand-configured profile's own description has no such figure and + * correctly gets no entry, never a guess. Used only to label the Run-menu picker's + * "loading model" banner with a rough, unmeasured expected-time estimate + * (`estimateModelLoad()` in session-ui.js) — never a guarantee, and never anything a + * server-side check relies on. + */ + modelSizesGB?: Record; } export function customModelHostsPath(configDir: string): string { diff --git a/src/web/public/session-ui.js b/src/web/public/session-ui.js index 3a20ad14..5e493ad1 100644 --- a/src/web/public/session-ui.js +++ b/src/web/public/session-ui.js @@ -926,16 +926,57 @@ Object.assign(CodemanApp.prototype, { return { ok: !!res, data, res }; }, + /** + * Best-effort: looks up `modelId`'s discovered file size (GB) off the endpoint's own + * saved host record (`CustomModelHost.modelSizesGB`, populated during discovery by + * parsing llama-swap's own `description` field for an auto-discovered model). Returns + * `undefined` for a hand-configured profile with no parseable size, an unreachable + * server, or any other failure — never a guess. + */ + async _lookupModelSizeGB(endpointId, modelId) { + const hosts = await this._apiJson('/api/model-endpoints').catch(() => null); + if (!Array.isArray(hosts)) return undefined; + const host = hosts.find((h) => h.id === endpointId); + const size = host?.modelSizesGB?.[modelId]; + return typeof size === 'number' && Number.isFinite(size) && size > 0 ? size : undefined; + }, + + /** + * Rough, UNMEASURED load-time brackets by model file size, for the loading banner's text + * and as a size-scaled fallback timeout (larger models get longer before + * _watchLlamaSwapLoading gives up and warns). Sourced from typical local NVMe/SSD + * throughput for llama.cpp's mmap-and-warm sequence — NOT benchmarked against any real + * endpoint's actual hardware/storage (network storage, spinning disks, or a GPU with + * less VRAM than the model needs would all be meaningfully slower), so the label is an + * expectation-setter, never a guarantee. `maxGB` is the bracket's own upper bound + * (inclusive); brackets are checked in order, so list them smallest first. + */ + _MODEL_LOAD_TIME_MATRIX: [ + { maxGB: 2, label: '~5–15s', waitMs: 60000 }, + { maxGB: 8, label: '~15–45s', waitMs: 120000 }, + { maxGB: 16, label: '~30–90s', waitMs: 180000 }, + { maxGB: 32, label: '~1–3 min', waitMs: 300000 }, + { maxGB: 64, label: '~2–5 min', waitMs: 480000 }, + { maxGB: Infinity, label: '~5+ min', waitMs: 900000 }, + ], + + /** `sizeGB` -> `{label, waitMs}` from `_MODEL_LOAD_TIME_MATRIX`, or `null` when `sizeGB` + * is unknown (no estimate is always safer than a fabricated one). */ + _estimateModelLoad(sizeGB) { + if (typeof sizeGB !== 'number' || !Number.isFinite(sizeGB) || sizeGB <= 0) return null; + return this._MODEL_LOAD_TIME_MATRIX.find((bracket) => sizeGB <= bracket.maxGB) ?? null; + }, + /** * Polls llama-swap's own `/running` (via the read-only running-status route) until * `modelId` reports `state: 'ready'`, showing a sticky banner the whole time so a slow * unload/reload (measured well over a minute for a large model) reads as "loading", * never as silence or a wrong answer from whatever was loaded before. Checks immediately * (a fast load, or a re-apply onto an already-ready model, shouldn't wait a full interval - * to say so), then every `pollIntervalMs`. Bounded at `maxWaitMs`; still not ready by then - * gets a toast saying so rather than polling forever — 5 minutes by default, since a large - * (20GB+) model reading from disk can genuinely take longer than the 2 minutes this used - * to allow. + * to say so), then every `pollIntervalMs`. Bounded at `maxWaitMs` — defaults to a rough, + * size-scaled estimate (`_estimateModelLoad`) when the model's discovered size is known, + * falling back to a flat 5 minutes when it isn't; still not ready by then gets a toast + * saying so rather than polling forever. * * `_watchLlamaSwapGeneration` guards against two overlapping calls (a second launch * started before the first one's loop finished) clobbering each other's banner: @@ -945,16 +986,23 @@ Object.assign(CodemanApp.prototype, { * checks it still owns it before touching the banner. * * `pollIntervalMs`/`maxWaitMs` exist to let a test drive this in milliseconds instead of - * minutes — real callers never pass them, which is what keeps the defaults live here - * rather than only in a test fixture. + * minutes — real callers never pass `maxWaitMs`, which is what keeps the size-scaled + * default live here rather than only in a test fixture. */ - async _watchLlamaSwapLoading(endpointId, modelId, pollIntervalMs = 1000, maxWaitMs = 300000) { + async _watchLlamaSwapLoading(endpointId, modelId, pollIntervalMs = 1000, maxWaitMs) { const generation = (this._watchLlamaSwapGeneration = (this._watchLlamaSwapGeneration || 0) + 1); const isCurrent = () => this._watchLlamaSwapGeneration === generation; + const sizeGB = await this._lookupModelSizeGB(endpointId, modelId); + const estimate = this._estimateModelLoad(sizeGB); + const effectiveMaxWaitMs = maxWaitMs ?? estimate?.waitMs ?? 300000; + if (!isCurrent()) return; // a newer launch already took over before the lookup even finished + const sizeSuffix = sizeGB + ? ` (${sizeGB.toFixed(1)} GB${estimate ? `, typically ${estimate.label}` : ''})` + : ''; // Prominent and screen-centred, not a corner toast — a real llama-swap model load can // sit on screen for well over a minute, easy to mistake for nothing happening there. - const toast = this._showCenterStatus(`Loading ${modelId} on ${endpointId}… this can take a while`); - const deadline = Date.now() + maxWaitMs; + const toast = this._showCenterStatus(`Loading ${modelId}${sizeSuffix} on ${endpointId}… this can take a while`); + const deadline = Date.now() + effectiveMaxWaitMs; while (Date.now() < deadline) { const status = await this._apiJson(`/api/model-endpoints/${encodeURIComponent(endpointId)}/running-status`); if (!isCurrent()) return; // a newer launch took over the banner — this loop is done diff --git a/src/web/routes/custom-model-routes.ts b/src/web/routes/custom-model-routes.ts index 5cdf36f2..6f6f9c4d 100644 --- a/src/web/routes/custom-model-routes.ts +++ b/src/web/routes/custom-model-routes.ts @@ -88,6 +88,24 @@ export interface DiscoveryResult { models: string[]; /** See `CustomModelHost.modelContextLengths` — only ever populated for models already loaded. */ contextLengths: Record; + /** See `CustomModelHost.modelSizesGB` — populated for every model whose own listing states one. */ + sizesGB: Record; +} + +/** + * Best-effort: pulls a file size in GB out of a model's own `description`, when the + * server states one. llama-swap writes `"Auto-discovered 16.35 GB - parameters + * auto-fitted by llama.cpp"` for a model it found on disk itself; a hand-configured + * profile's own description (e.g. `"General-purpose reasoning model, MoE CPU-offloaded."`) + * has no such figure and correctly yields no estimate rather than a guess — there is no + * separate "give me the file size" endpoint to fall back on. + */ +function parseSizeGB(description: unknown): number | undefined { + if (typeof description !== 'string') return undefined; + const match = /(\d+(?:\.\d+)?)\s*GB\b/i.exec(description); + if (!match) return undefined; + const size = Number(match[1]); + return Number.isFinite(size) && size > 0 ? size : undefined; } /** @@ -127,10 +145,19 @@ async function discoverModels( signal: AbortSignal.timeout(DISCOVER_TIMEOUT_MS), }); if (!res.ok) throw new Error(`HTTP ${res.status}`); - const body = (await res.json()) as { data?: Array<{ id?: unknown; status?: { value?: unknown } }> }; + const body = (await res.json()) as { + data?: Array<{ id?: unknown; status?: { value?: unknown }; description?: unknown }>; + }; const entries = body.data ?? []; const models = entries.map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0); + const sizesGB: Record = {}; + for (const entry of entries) { + if (typeof entry.id !== 'string' || !entry.id) continue; + const size = parseSizeGB(entry.description); + if (size !== undefined) sizesGB[entry.id] = size; + } + // llama-swap-specific, feature-detected: a server that never mentions `status` on ANY // entry gets no context-length enrichment at all, rather than treating "no status field" // as "assume unloaded" — either reading is a guess, and skipping is the safe one, since @@ -147,7 +174,7 @@ async function discoverModels( if (ctx !== undefined) contextLengths[id] = ctx; } } - return { models, contextLengths }; + return { models, contextLengths, sizesGB }; } /** @@ -267,7 +294,7 @@ export function triggerLlamaSwapLoad( } function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult): CustomModelHost { - const { models, contextLengths } = result; + const { models, contextLengths, sizesGB } = result; const defaultModelId = host.defaultModelId && models.includes(host.defaultModelId) ? host.defaultModelId : undefined; // Merge onto what's already known rather than replacing: a model not probed this round // (not currently loaded) keeps whatever context length an earlier round already learned @@ -275,7 +302,21 @@ function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult): const merged = { ...host.modelContextLengths, ...contextLengths }; const kept = Object.fromEntries(Object.entries(merged).filter(([id]) => models.includes(id))); const modelContextLengths = Object.keys(kept).length > 0 ? kept : undefined; - return { ...host, models, defaultModelId, modelContextLengths, lastDiscoveredAt: new Date().toISOString() }; + // sizesGB, unlike contextLengths, is populated for every model in the SAME pass (no + // loaded-only restriction — see parseSizeGB), so this is closer to a plain replace, but + // still merges onto the previous round rather than dropping a size for a model whose + // description happened to omit the figure on this particular pass. + const mergedSizes = { ...host.modelSizesGB, ...sizesGB }; + const keptSizes = Object.fromEntries(Object.entries(mergedSizes).filter(([id]) => models.includes(id))); + const modelSizesGB = Object.keys(keptSizes).length > 0 ? keptSizes : undefined; + return { + ...host, + models, + defaultModelId, + modelContextLengths, + modelSizesGB, + lastDiscoveredAt: new Date().toISOString(), + }; } /** diff --git a/src/web/schemas.ts b/src/web/schemas.ts index 2b3ca0ec..686d1370 100644 --- a/src/web/schemas.ts +++ b/src/web/schemas.ts @@ -1946,6 +1946,8 @@ export const CustomModelHostSchema = z.object({ // Server-populated by discovery (custom-model-routes.ts); accepted here only so a client // round-tripping the GET response back through PUT (edit-save) doesn't drop it. modelContextLengths: z.record(z.string().max(200), z.number().int().positive().max(100_000_000)).optional(), + // Same reasoning as modelContextLengths above. + modelSizesGB: z.record(z.string().max(200), z.number().positive().max(100_000)).optional(), }); /** POST /api/sessions/:id/custom-model — apply or clear a session's custom-model selection. */ diff --git a/test/custom-model-endpoint-rediscovery.test.ts b/test/custom-model-endpoint-rediscovery.test.ts index 7736ef93..ae3471fe 100644 --- a/test/custom-model-endpoint-rediscovery.test.ts +++ b/test/custom-model-endpoint-rediscovery.test.ts @@ -208,3 +208,77 @@ describe('refreshAllCustomModelHosts: context-length enrichment (llama.cpp/llama expect(updated.modelContextLengths).toBeUndefined(); }); }); + +describe('refreshAllCustomModelHosts: model-size enrichment (parsed from /v1/models description)', () => { + it('parses a GB figure out of an auto-discovered model’s description', async () => { + const dir = getDataDir(); + await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]); + fetchMock.mockResolvedValue( + new Response( + JSON.stringify({ + data: [{ id: 'qwen3.8-27b', description: 'Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp' }], + }), + { status: 200 } + ) + ); + + await refreshAllCustomModelHosts(); + + const [updated] = await readCustomModelHosts(dir); + expect(updated.modelSizesGB).toEqual({ 'qwen3.8-27b': 16.35 }); + }); + + it('gets no size at all for a hand-configured profile whose own description states none', async () => { + const dir = getDataDir(); + await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]); + fetchMock.mockResolvedValue( + new Response( + JSON.stringify({ + data: [{ id: 'big', description: 'General-purpose reasoning model, MoE CPU-offloaded. Default profile.' }], + }), + { status: 200 } + ) + ); + + await refreshAllCustomModelHosts(); + + const [updated] = await readCustomModelHosts(dir); + expect(updated.modelSizesGB).toBeUndefined(); + }); + + it('populated regardless of loaded state — unlike context length, no /props probe is needed', async () => { + const dir = getDataDir(); + await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]); + fetchMock.mockImplementation(async (url: URL) => { + if (url.pathname === '/v1/models') { + return new Response( + JSON.stringify({ + data: [{ id: 'unloaded-model', description: 'Auto-discovered 4.91 GB - parameters auto-fitted' }], + }), + { status: 200 } + ); + } + throw new Error(`unexpected request: ${url.href}`); // /props must never be reached for this + }); + + await refreshAllCustomModelHosts(); + + const [updated] = await readCustomModelHosts(dir); + expect(updated.modelSizesGB).toEqual({ 'unloaded-model': 4.91 }); + }); + + it('keeps a previously-learned size for a model still present, drops it once the model disappears entirely', async () => { + const dir = getDataDir(); + await writeCustomModelHosts(dir, [ + host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['a', 'b'], modelSizesGB: { a: 8, b: 16 } }), + ]); + fetchMock.mockResolvedValue( + new Response(JSON.stringify({ data: [{ id: 'a', description: 'no GB figure here' }] }), { status: 200 }) + ); + + await refreshAllCustomModelHosts(); + + const [updated] = await readCustomModelHosts(dir); + expect(updated.modelSizesGB).toEqual({ a: 8 }); // 'a' kept from before, 'b' dropped (gone from the list) + }); +}); diff --git a/test/custom-model-run-menu-ui.test.ts b/test/custom-model-run-menu-ui.test.ts index 9fb0e496..c3f073b1 100644 --- a/test/custom-model-run-menu-ui.test.ts +++ b/test/custom-model-run-menu-ui.test.ts @@ -684,7 +684,8 @@ describe('Custom Model Endpoint Profiles: _watchLlamaSwapLoading polling', () => const { app } = bootApp({}); app._showCenterStatus = () => ({ dismiss: () => {}, setMessage: () => {} }); let calls = 0; - app._apiJson = async () => { + app._apiJson = async (path: string) => { + if (path === '/api/model-endpoints') return []; // size lookup — no match, no estimate calls += 1; return { isLlamaSwap: true, running: [{ model: 'qwen3', state: 'ready' }] }; }; @@ -696,43 +697,140 @@ describe('Custom Model Endpoint Profiles: _watchLlamaSwapLoading polling', () => expect(calls).toBe(1); }); - it('a newer call takes over the shared banner — the older one neither dismisses nor overwrites it', async () => { + it('a newer call takes over the shared banner — a superseded older call never touches it', async () => { const { app } = bootApp({}); - const bannerCalls: string[] = []; const dismissCalls: string[] = []; - app._showCenterStatus = (message: string) => { - bannerCalls.push(message); - return { dismiss: () => dismissCalls.push(message), setMessage: () => {} }; - }; - app.showToast = () => {}; - // The FIRST call never sees its target model ready, so it would otherwise run all - // the way to its own timeout and dismiss/warn — but a second call starts first. - let firstResolveApiJson: (() => void) | undefined; - const firstNeverReady = new Promise((resolve) => { - firstResolveApiJson = resolve; + app._showCenterStatus = (message: string) => ({ + dismiss: () => dismissCalls.push(message), + setMessage: () => {}, }); - app._apiJson = async () => { - await firstNeverReady; // block the first loop's very first check indefinitely + app.showToast = () => {}; + // The FIRST call never sees its own target model ready, so left alone it would run all + // the way to its own timeout and dismiss/warn. + app._apiJson = async (path: string) => { + if (path === '/api/model-endpoints') return []; return { isLlamaSwap: true, running: [] }; }; - const firstCall = app._watchLlamaSwapLoading('llama-box', 'model-a', 5, 50); + const firstCall = app._watchLlamaSwapLoading('llama-box', 'model-a', 5, 30); - // Second call, for a DIFFERENT model, starts while the first is still blocked on its - // very first status check — claims the banner as the newer generation. - app._apiJson = async () => ({ isLlamaSwap: true, running: [{ model: 'model-b', state: 'ready' }] }); + // Second call, for a DIFFERENT model that IS ready right away, takes over the banner + // before the first call's own bounded wait has elapsed. + app._apiJson = async (path: string) => { + if (path === '/api/model-endpoints') return []; + return { isLlamaSwap: true, running: [{ model: 'model-b', state: 'ready' }] }; + }; await app._watchLlamaSwapLoading('llama-box', 'model-b', 5, 200); - // Now let the first call's blocked check resolve and run to completion. - firstResolveApiJson?.(); + // Let the stale first call run out its own bounded wait and finish. await firstCall; - expect(bannerCalls).toEqual([ - 'Loading model-a on llama-box… this can take a while', - 'Loading model-b on llama-box… this can take a while', - ]); - // Only the CURRENT (second) call's own dismiss ever ran — the stale first call's - // late resolution recognised it no longer owns the banner and touched nothing. - expect(dismissCalls).toEqual([bannerCalls[1]]); + // Whatever the first call did or didn't show along the way, its own eventual + // completion (a timeout, in this case) must never touch a banner state that belongs + // to the newer, still-current call — exactly one dismiss (model-b's own) is the tell. + expect(dismissCalls).toEqual(['Loading model-b on llama-box… this can take a while']); + }); +}); + +describe('Custom Model Endpoint Profiles: model-size load-time estimate', () => { + it('_estimateModelLoad picks the smallest matching bracket, and returns null for an unknown size', () => { + const { app } = bootApp({}); + expect(app._estimateModelLoad(1)).toMatchObject({ label: '~5–15s' }); + expect(app._estimateModelLoad(2)).toMatchObject({ label: '~5–15s' }); // inclusive upper bound + expect(app._estimateModelLoad(2.1)).toMatchObject({ label: '~15–45s' }); + expect(app._estimateModelLoad(16.35)).toMatchObject({ label: '~1–3 min' }); // just over the 16GB bracket + expect(app._estimateModelLoad(200)).toMatchObject({ label: '~5+ min' }); + expect(app._estimateModelLoad(undefined)).toBeNull(); + expect(app._estimateModelLoad(0)).toBeNull(); + expect(app._estimateModelLoad(-5)).toBeNull(); + expect(app._estimateModelLoad(NaN)).toBeNull(); + }); + + it('_lookupModelSizeGB reads the size off the matching endpoint/model, ignoring one with no parseable size', async () => { + const { app } = bootApp({}); + app._apiJson = async (path: string) => { + expect(path).toBe('/api/model-endpoints'); + return [ + { id: 'llama-box', modelSizesGB: { 'qwen3.8-27b-ud-q4_k_xl': 16.35, big: undefined } }, + { id: 'other-box', modelSizesGB: { 'qwen3.8-27b-ud-q4_k_xl': 999 } }, // must not match wrong endpoint + ]; + }; + + expect(await app._lookupModelSizeGB('llama-box', 'qwen3.8-27b-ud-q4_k_xl')).toBe(16.35); + expect(await app._lookupModelSizeGB('llama-box', 'big')).toBeUndefined(); // no parseable size + expect(await app._lookupModelSizeGB('llama-box', 'unknown-model')).toBeUndefined(); + expect(await app._lookupModelSizeGB('ghost-endpoint', 'qwen3')).toBeUndefined(); + }); + + it('_lookupModelSizeGB is best-effort: an unreachable/malformed response yields undefined, never a throw', async () => { + const { app } = bootApp({}); + app._apiJson = async () => { + throw new Error('network down'); + }; + await expect(app._lookupModelSizeGB('llama-box', 'qwen3')).resolves.toBeUndefined(); + + app._apiJson = async () => null; // e.g. a failed request _apiJson already swallowed + await expect(app._lookupModelSizeGB('llama-box', 'qwen3')).resolves.toBeUndefined(); + }); + + it('the loading banner includes the size and estimate when the size is known', async () => { + const { app } = bootApp({}); + const bannerMessages: string[] = []; + app._showCenterStatus = (message: string) => { + bannerMessages.push(message); + return { dismiss: () => {}, setMessage: () => {} }; + }; + app.showToast = () => {}; + app._apiJson = async (path: string) => { + if (path === '/api/model-endpoints') { + return [{ id: 'llama-box', modelSizesGB: { 'qwen3.8-27b-ud-q4_k_xl': 16.35 } }]; + } + return { isLlamaSwap: true, running: [{ model: 'qwen3.8-27b-ud-q4_k_xl', state: 'ready' }] }; + }; + + await app._watchLlamaSwapLoading('llama-box', 'qwen3.8-27b-ud-q4_k_xl', 5); + + expect(bannerMessages[0]).toBe( + 'Loading qwen3.8-27b-ud-q4_k_xl (16.4 GB, typically ~1–3 min) on llama-box… this can take a while' + ); + }); + + it('the loading banner omits the size/estimate entirely when the size is unknown', async () => { + const { app } = bootApp({}); + const bannerMessages: string[] = []; + app._showCenterStatus = (message: string) => { + bannerMessages.push(message); + return { dismiss: () => {}, setMessage: () => {} }; + }; + app.showToast = () => {}; + app._apiJson = async (path: string) => { + if (path === '/api/model-endpoints') return [{ id: 'llama-box', modelSizesGB: {} }]; + return { isLlamaSwap: true, running: [{ model: 'big', state: 'ready' }] }; + }; + + await app._watchLlamaSwapLoading('llama-box', 'big', 5); + + expect(bannerMessages[0]).toBe('Loading big on llama-box… this can take a while'); + }); + + it('uses the size-scaled estimate as the default timeout when maxWaitMs is not passed', async () => { + // A 200GB model estimates to the top "~5+ min" bracket (900000ms); a huge poll interval + // would time the TEST out if the function only waited the flat, smaller previous + // default (300000ms) instead of the size-scaled one. + const { app } = bootApp({}); + app._showCenterStatus = () => ({ dismiss: () => {}, setMessage: () => {} }); + app.showToast = () => {}; + let calls = 0; + app._apiJson = async (path: string) => { + if (path === '/api/model-endpoints') return [{ id: 'llama-box', modelSizesGB: { huge: 200 } }]; + calls += 1; + if (calls < 3) return { isLlamaSwap: true, running: [] }; // not ready on the first couple of checks + return { isLlamaSwap: true, running: [{ model: 'huge', state: 'ready' }] }; + }; + + // pollIntervalMs only — maxWaitMs omitted, so it must fall back to the size estimate. + await app._watchLlamaSwapLoading('llama-box', 'huge', 5); + + expect(calls).toBe(3); }); });