feat(custom-model): estimate model load time from its discovered size

Discovery now also parses a GB figure out of an auto-discovered model's own
description (llama-swap writes "Auto-discovered 16.35 GB - parameters
auto-fitted by llama.cpp"), stored per model as modelSizesGB - unlike
context length this needs no /props probe (the figure is right there in
/v1/models) so it is populated for every model regardless of loaded state.
A hand-configured profile's own description has no such figure and
correctly gets no entry.

The loading banner (_watchLlamaSwapLoading) now looks this up and, when
known, shows it plus a rough estimate from a small size->time matrix
(_estimateModelLoad/_MODEL_LOAD_TIME_MATRIX, session-ui.js) -
"Loading qwen3.8-27b-ud-q4_k_xl (16.4 GB, typically ~1-3 min) on
llama-swap... this can take a while" - and uses that same estimate's own
bracket to scale the banner's default give-up timeout for a very large
model, instead of a flat 5 minutes for everything. Explicitly labelled as
an UNMEASURED, typical-hardware estimate in every relevant comment - this
is not benchmarked against any real endpoint's actual storage/GPU, just a
reasonable expectation-setter. A model with no discoverable size (a
hand-configured profile) gets no size/estimate shown at all, matching the
"never a guess" convention modelContextLengths already established.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
Devvyn
2026-09-16 19:58:46 +08:00
co-authored by Claude Sonnet 5
parent 0af233c96c
commit 55dae31530
7 changed files with 329 additions and 41 deletions
@@ -208,3 +208,77 @@ describe('refreshAllCustomModelHosts: context-length enrichment (llama.cpp/llama
expect(updated.modelContextLengths).toBeUndefined();
});
});
describe('refreshAllCustomModelHosts: model-size enrichment (parsed from /v1/models description)', () => {
it('parses a GB figure out of an auto-discovered model’s description', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockResolvedValue(
new Response(
JSON.stringify({
data: [{ id: 'qwen3.8-27b', description: 'Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp' }],
}),
{ status: 200 }
)
);
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelSizesGB).toEqual({ 'qwen3.8-27b': 16.35 });
});
it('gets no size at all for a hand-configured profile whose own description states none', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockResolvedValue(
new Response(
JSON.stringify({
data: [{ id: 'big', description: 'General-purpose reasoning model, MoE CPU-offloaded. Default profile.' }],
}),
{ status: 200 }
)
);
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelSizesGB).toBeUndefined();
});
it('populated regardless of loaded state — unlike context length, no /props probe is needed', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/v1/models') {
return new Response(
JSON.stringify({
data: [{ id: 'unloaded-model', description: 'Auto-discovered 4.91 GB - parameters auto-fitted' }],
}),
{ status: 200 }
);
}
throw new Error(`unexpected request: ${url.href}`); // /props must never be reached for this
});
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelSizesGB).toEqual({ 'unloaded-model': 4.91 });
});
it('keeps a previously-learned size for a model still present, drops it once the model disappears entirely', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [
host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['a', 'b'], modelSizesGB: { a: 8, b: 16 } }),
]);
fetchMock.mockResolvedValue(
new Response(JSON.stringify({ data: [{ id: 'a', description: 'no GB figure here' }] }), { status: 200 })
);
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelSizesGB).toEqual({ a: 8 }); // 'a' kept from before, 'b' dropped (gone from the list)
});
});