mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-04 14:39:42 +02:00
feat(custom-model): estimate model load time from its discovered size
Discovery now also parses a GB figure out of an auto-discovered model's own description (llama-swap writes "Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp"), stored per model as modelSizesGB - unlike context length this needs no /props probe (the figure is right there in /v1/models) so it is populated for every model regardless of loaded state. A hand-configured profile's own description has no such figure and correctly gets no entry. The loading banner (_watchLlamaSwapLoading) now looks this up and, when known, shows it plus a rough estimate from a small size->time matrix (_estimateModelLoad/_MODEL_LOAD_TIME_MATRIX, session-ui.js) - "Loading qwen3.8-27b-ud-q4_k_xl (16.4 GB, typically ~1-3 min) on llama-swap... this can take a while" - and uses that same estimate's own bracket to scale the banner's default give-up timeout for a very large model, instead of a flat 5 minutes for everything. Explicitly labelled as an UNMEASURED, typical-hardware estimate in every relevant comment - this is not benchmarked against any real endpoint's actual storage/GPU, just a reasonable expectation-setter. A model with no discoverable size (a hand-configured profile) gets no size/estimate shown at all, matching the "never a guess" convention modelContextLengths already established. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
0af233c96c
commit
55dae31530
@@ -61,6 +61,17 @@ export interface CustomModelHost {
|
||||
* has no entry for simply gets no context-length env override applied — never a guess.
|
||||
*/
|
||||
modelContextLengths?: Record<string, number>;
|
||||
/**
|
||||
* Discovered file size (GB) per model id, keyed by the same strings as `models`.
|
||||
* Populated during discovery by parsing llama-swap's own `description` field for an
|
||||
* auto-discovered model ("Auto-discovered 16.35 GB - parameters auto-fitted by
|
||||
* llama.cpp") — a hand-configured profile's own description has no such figure and
|
||||
* correctly gets no entry, never a guess. Used only to label the Run-menu picker's
|
||||
* "loading model" banner with a rough, unmeasured expected-time estimate
|
||||
* (`estimateModelLoad()` in session-ui.js) — never a guarantee, and never anything a
|
||||
* server-side check relies on.
|
||||
*/
|
||||
modelSizesGB?: Record<string, number>;
|
||||
}
|
||||
|
||||
export function customModelHostsPath(configDir: string): string {
|
||||
|
||||
@@ -926,16 +926,57 @@ Object.assign(CodemanApp.prototype, {
|
||||
return { ok: !!res, data, res };
|
||||
},
|
||||
|
||||
/**
|
||||
* Best-effort: looks up `modelId`'s discovered file size (GB) off the endpoint's own
|
||||
* saved host record (`CustomModelHost.modelSizesGB`, populated during discovery by
|
||||
* parsing llama-swap's own `description` field for an auto-discovered model). Returns
|
||||
* `undefined` for a hand-configured profile with no parseable size, an unreachable
|
||||
* server, or any other failure — never a guess.
|
||||
*/
|
||||
async _lookupModelSizeGB(endpointId, modelId) {
|
||||
const hosts = await this._apiJson('/api/model-endpoints').catch(() => null);
|
||||
if (!Array.isArray(hosts)) return undefined;
|
||||
const host = hosts.find((h) => h.id === endpointId);
|
||||
const size = host?.modelSizesGB?.[modelId];
|
||||
return typeof size === 'number' && Number.isFinite(size) && size > 0 ? size : undefined;
|
||||
},
|
||||
|
||||
/**
|
||||
* Rough, UNMEASURED load-time brackets by model file size, for the loading banner's text
|
||||
* and as a size-scaled fallback timeout (larger models get longer before
|
||||
* _watchLlamaSwapLoading gives up and warns). Sourced from typical local NVMe/SSD
|
||||
* throughput for llama.cpp's mmap-and-warm sequence — NOT benchmarked against any real
|
||||
* endpoint's actual hardware/storage (network storage, spinning disks, or a GPU with
|
||||
* less VRAM than the model needs would all be meaningfully slower), so the label is an
|
||||
* expectation-setter, never a guarantee. `maxGB` is the bracket's own upper bound
|
||||
* (inclusive); brackets are checked in order, so list them smallest first.
|
||||
*/
|
||||
_MODEL_LOAD_TIME_MATRIX: [
|
||||
{ maxGB: 2, label: '~5–15s', waitMs: 60000 },
|
||||
{ maxGB: 8, label: '~15–45s', waitMs: 120000 },
|
||||
{ maxGB: 16, label: '~30–90s', waitMs: 180000 },
|
||||
{ maxGB: 32, label: '~1–3 min', waitMs: 300000 },
|
||||
{ maxGB: 64, label: '~2–5 min', waitMs: 480000 },
|
||||
{ maxGB: Infinity, label: '~5+ min', waitMs: 900000 },
|
||||
],
|
||||
|
||||
/** `sizeGB` -> `{label, waitMs}` from `_MODEL_LOAD_TIME_MATRIX`, or `null` when `sizeGB`
|
||||
* is unknown (no estimate is always safer than a fabricated one). */
|
||||
_estimateModelLoad(sizeGB) {
|
||||
if (typeof sizeGB !== 'number' || !Number.isFinite(sizeGB) || sizeGB <= 0) return null;
|
||||
return this._MODEL_LOAD_TIME_MATRIX.find((bracket) => sizeGB <= bracket.maxGB) ?? null;
|
||||
},
|
||||
|
||||
/**
|
||||
* Polls llama-swap's own `/running` (via the read-only running-status route) until
|
||||
* `modelId` reports `state: 'ready'`, showing a sticky banner the whole time so a slow
|
||||
* unload/reload (measured well over a minute for a large model) reads as "loading",
|
||||
* never as silence or a wrong answer from whatever was loaded before. Checks immediately
|
||||
* (a fast load, or a re-apply onto an already-ready model, shouldn't wait a full interval
|
||||
* to say so), then every `pollIntervalMs`. Bounded at `maxWaitMs`; still not ready by then
|
||||
* gets a toast saying so rather than polling forever — 5 minutes by default, since a large
|
||||
* (20GB+) model reading from disk can genuinely take longer than the 2 minutes this used
|
||||
* to allow.
|
||||
* to say so), then every `pollIntervalMs`. Bounded at `maxWaitMs` — defaults to a rough,
|
||||
* size-scaled estimate (`_estimateModelLoad`) when the model's discovered size is known,
|
||||
* falling back to a flat 5 minutes when it isn't; still not ready by then gets a toast
|
||||
* saying so rather than polling forever.
|
||||
*
|
||||
* `_watchLlamaSwapGeneration` guards against two overlapping calls (a second launch
|
||||
* started before the first one's loop finished) clobbering each other's banner:
|
||||
@@ -945,16 +986,23 @@ Object.assign(CodemanApp.prototype, {
|
||||
* checks it still owns it before touching the banner.
|
||||
*
|
||||
* `pollIntervalMs`/`maxWaitMs` exist to let a test drive this in milliseconds instead of
|
||||
* minutes — real callers never pass them, which is what keeps the defaults live here
|
||||
* rather than only in a test fixture.
|
||||
* minutes — real callers never pass `maxWaitMs`, which is what keeps the size-scaled
|
||||
* default live here rather than only in a test fixture.
|
||||
*/
|
||||
async _watchLlamaSwapLoading(endpointId, modelId, pollIntervalMs = 1000, maxWaitMs = 300000) {
|
||||
async _watchLlamaSwapLoading(endpointId, modelId, pollIntervalMs = 1000, maxWaitMs) {
|
||||
const generation = (this._watchLlamaSwapGeneration = (this._watchLlamaSwapGeneration || 0) + 1);
|
||||
const isCurrent = () => this._watchLlamaSwapGeneration === generation;
|
||||
const sizeGB = await this._lookupModelSizeGB(endpointId, modelId);
|
||||
const estimate = this._estimateModelLoad(sizeGB);
|
||||
const effectiveMaxWaitMs = maxWaitMs ?? estimate?.waitMs ?? 300000;
|
||||
if (!isCurrent()) return; // a newer launch already took over before the lookup even finished
|
||||
const sizeSuffix = sizeGB
|
||||
? ` (${sizeGB.toFixed(1)} GB${estimate ? `, typically ${estimate.label}` : ''})`
|
||||
: '';
|
||||
// Prominent and screen-centred, not a corner toast — a real llama-swap model load can
|
||||
// sit on screen for well over a minute, easy to mistake for nothing happening there.
|
||||
const toast = this._showCenterStatus(`Loading ${modelId} on ${endpointId}… this can take a while`);
|
||||
const deadline = Date.now() + maxWaitMs;
|
||||
const toast = this._showCenterStatus(`Loading ${modelId}${sizeSuffix} on ${endpointId}… this can take a while`);
|
||||
const deadline = Date.now() + effectiveMaxWaitMs;
|
||||
while (Date.now() < deadline) {
|
||||
const status = await this._apiJson(`/api/model-endpoints/${encodeURIComponent(endpointId)}/running-status`);
|
||||
if (!isCurrent()) return; // a newer launch took over the banner — this loop is done
|
||||
|
||||
@@ -88,6 +88,24 @@ export interface DiscoveryResult {
|
||||
models: string[];
|
||||
/** See `CustomModelHost.modelContextLengths` — only ever populated for models already loaded. */
|
||||
contextLengths: Record<string, number>;
|
||||
/** See `CustomModelHost.modelSizesGB` — populated for every model whose own listing states one. */
|
||||
sizesGB: Record<string, number>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort: pulls a file size in GB out of a model's own `description`, when the
|
||||
* server states one. llama-swap writes `"Auto-discovered 16.35 GB - parameters
|
||||
* auto-fitted by llama.cpp"` for a model it found on disk itself; a hand-configured
|
||||
* profile's own description (e.g. `"General-purpose reasoning model, MoE CPU-offloaded."`)
|
||||
* has no such figure and correctly yields no estimate rather than a guess — there is no
|
||||
* separate "give me the file size" endpoint to fall back on.
|
||||
*/
|
||||
function parseSizeGB(description: unknown): number | undefined {
|
||||
if (typeof description !== 'string') return undefined;
|
||||
const match = /(\d+(?:\.\d+)?)\s*GB\b/i.exec(description);
|
||||
if (!match) return undefined;
|
||||
const size = Number(match[1]);
|
||||
return Number.isFinite(size) && size > 0 ? size : undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -127,10 +145,19 @@ async function discoverModels(
|
||||
signal: AbortSignal.timeout(DISCOVER_TIMEOUT_MS),
|
||||
});
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
||||
const body = (await res.json()) as { data?: Array<{ id?: unknown; status?: { value?: unknown } }> };
|
||||
const body = (await res.json()) as {
|
||||
data?: Array<{ id?: unknown; status?: { value?: unknown }; description?: unknown }>;
|
||||
};
|
||||
const entries = body.data ?? [];
|
||||
const models = entries.map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
|
||||
|
||||
const sizesGB: Record<string, number> = {};
|
||||
for (const entry of entries) {
|
||||
if (typeof entry.id !== 'string' || !entry.id) continue;
|
||||
const size = parseSizeGB(entry.description);
|
||||
if (size !== undefined) sizesGB[entry.id] = size;
|
||||
}
|
||||
|
||||
// llama-swap-specific, feature-detected: a server that never mentions `status` on ANY
|
||||
// entry gets no context-length enrichment at all, rather than treating "no status field"
|
||||
// as "assume unloaded" — either reading is a guess, and skipping is the safe one, since
|
||||
@@ -147,7 +174,7 @@ async function discoverModels(
|
||||
if (ctx !== undefined) contextLengths[id] = ctx;
|
||||
}
|
||||
}
|
||||
return { models, contextLengths };
|
||||
return { models, contextLengths, sizesGB };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -267,7 +294,7 @@ export function triggerLlamaSwapLoad(
|
||||
}
|
||||
|
||||
function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult): CustomModelHost {
|
||||
const { models, contextLengths } = result;
|
||||
const { models, contextLengths, sizesGB } = result;
|
||||
const defaultModelId = host.defaultModelId && models.includes(host.defaultModelId) ? host.defaultModelId : undefined;
|
||||
// Merge onto what's already known rather than replacing: a model not probed this round
|
||||
// (not currently loaded) keeps whatever context length an earlier round already learned
|
||||
@@ -275,7 +302,21 @@ function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult):
|
||||
const merged = { ...host.modelContextLengths, ...contextLengths };
|
||||
const kept = Object.fromEntries(Object.entries(merged).filter(([id]) => models.includes(id)));
|
||||
const modelContextLengths = Object.keys(kept).length > 0 ? kept : undefined;
|
||||
return { ...host, models, defaultModelId, modelContextLengths, lastDiscoveredAt: new Date().toISOString() };
|
||||
// sizesGB, unlike contextLengths, is populated for every model in the SAME pass (no
|
||||
// loaded-only restriction — see parseSizeGB), so this is closer to a plain replace, but
|
||||
// still merges onto the previous round rather than dropping a size for a model whose
|
||||
// description happened to omit the figure on this particular pass.
|
||||
const mergedSizes = { ...host.modelSizesGB, ...sizesGB };
|
||||
const keptSizes = Object.fromEntries(Object.entries(mergedSizes).filter(([id]) => models.includes(id)));
|
||||
const modelSizesGB = Object.keys(keptSizes).length > 0 ? keptSizes : undefined;
|
||||
return {
|
||||
...host,
|
||||
models,
|
||||
defaultModelId,
|
||||
modelContextLengths,
|
||||
modelSizesGB,
|
||||
lastDiscoveredAt: new Date().toISOString(),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -1946,6 +1946,8 @@ export const CustomModelHostSchema = z.object({
|
||||
// Server-populated by discovery (custom-model-routes.ts); accepted here only so a client
|
||||
// round-tripping the GET response back through PUT (edit-save) doesn't drop it.
|
||||
modelContextLengths: z.record(z.string().max(200), z.number().int().positive().max(100_000_000)).optional(),
|
||||
// Same reasoning as modelContextLengths above.
|
||||
modelSizesGB: z.record(z.string().max(200), z.number().positive().max(100_000)).optional(),
|
||||
});
|
||||
|
||||
/** POST /api/sessions/:id/custom-model — apply or clear a session's custom-model selection. */
|
||||
|
||||
Reference in New Issue
Block a user