From 993710263d826bfcb7ba9b52a4c9c4052fbee325 Mon Sep 17 00:00:00 2001 From: Devvyn <22340871+opticon454@users.noreply.github.com> Date: Wed, 16 Sep 2026 20:47:12 +0800 Subject: [PATCH] fix(custom-model): stop trusting /props's n_ctx, parse the real context size from /running's cmd Root cause of the context-overflow regression reported live: "API Error: 400 request (36437 tokens) exceeds the available context size (16384 tokens)". Discovery had stored modelContextLengths.qwen3.8-27b-ud-q4_k_xl = 154112, so CLAUDE_CODE_MAX_CONTEXT_TOKENS told Claude Code it had a huge window and it never compacted - but the real llama-swap server was launched with --fit-ctx 16384 (confirmed against /running's own cmd field) and refused the request right at that real limit. /props?model='s n_ctx (the field discovery read) is confirmed live to be unreliable for a --fit-ctx-launched backend: it reported 154112 for the same model /running says was launched with --fit-ctx 16384 - appears to report the model's theoretical/trained maximum context, not the runtime- configured one. discoverModels() now parses the REAL configured size straight out of llama-swap's own launch command instead (parseCtxFromCmd(), reading /running's cmd field - --fit-ctx first, then the plain llama.cpp -c/ --ctx-size a hand-written command might use), and only falls back to the old /props probe when cmd states no recognizable flag at all. One /running call now covers every loaded model's context length in a single request, same as it already did for the swap-conflict check and the load trigger. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG --- docs/custom-model-endpoints.md | 19 ++++- src/web/routes/custom-model-routes.ts | 54 ++++++++++-- .../custom-model-endpoint-rediscovery.test.ts | 84 +++++++++++++++++++ 3 files changed, 149 insertions(+), 8 deletions(-) diff --git a/docs/custom-model-endpoints.md b/docs/custom-model-endpoints.md index b608fa2e..e3db6016 100644 --- a/docs/custom-model-endpoints.md +++ b/docs/custom-model-endpoints.md @@ -67,9 +67,8 @@ Endpoint management is admin-only in multi-user mode, same as remote/docker hosts — these are machine-level infra, not per-user settings. **Context length is discovered too, opportunistically and safely.** The plain -`GET /v1/models` response has no context-window field, but llama.cpp's -llama-swap-proxied `GET /props?model=` does (`n_ctx`). Discovery only ever -calls it for a model llama-swap's own response already reports +`GET /v1/models` response has no context-window field. Discovery only ever +looks for one for a model llama-swap's own response already reports `status.value === "loaded"` for — never for an unloaded one, because llama-swap treats `?model=` as a routing hint and asking about a model that isn't loaded risks triggering an actual (slow, GPU-swapping) load as a side @@ -83,6 +82,20 @@ session" below) so a CLI that would otherwise assume a large default context window for an unrecognized model id stops silently overflowing a much smaller real one. +**Where that number actually comes from matters, and got this wrong once +already.** The first cut read it from llama.cpp's own +`GET /props?model=` (`n_ctx`) — plausible, and it worked in testing, but +confirmed live to be actively WRONG for a `--fit-ctx`-launched llama-swap +backend: `/props` reported `n_ctx: 154112` for a model llama-swap itself had +launched with `--fit-ctx 16384`, and the real server then refused a request +right at that real 16384-token limit — `/props`'s `n_ctx` appears to report +the model's theoretical/trained maximum there, not the runtime-configured +one. Discovery now parses the REAL configured size straight out of +llama-swap's own launch command instead (`GET /running`'s `cmd` field — +`--fit-ctx ` first, then the plain llama.cpp `-c`/`--ctx-size` a +hand-written command might use), and only falls back to the `/props` probe +when `cmd` states no recognizable flag at all. + **File size is discovered too, when the server states one.** llama-swap writes a GB figure into an auto-discovered model's own `description` (`"Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp"`), parsed diff --git a/src/web/routes/custom-model-routes.ts b/src/web/routes/custom-model-routes.ts index 6f6f9c4d..252a4f12 100644 --- a/src/web/routes/custom-model-routes.ts +++ b/src/web/routes/custom-model-routes.ts @@ -117,6 +117,15 @@ function parseSizeGB(description: unknown): number | undefined { * actual (slow, GPU-swapping) load as a side effect of what should be read-only discovery. * Any failure (unreachable, non-2xx, missing/malformed field) is swallowed — one model's * context length is a nice-to-have, never worth failing the whole discovery pass over. + * + * ⚠️ FALLBACK ONLY — confirmed live to be actively WRONG for a `--fit-ctx`-launched llama- + * swap backend: `/props`'s `n_ctx` read 154112 for a model llama-swap itself had launched + * with `--fit-ctx 16384` (visible in `/running`'s own `cmd`), and the real server then + * refused a request at the real 16384-token limit — `n_ctx` here appears to report the + * model's theoretical/trained maximum, not the runtime-configured one. `parseCtxFromCmd` + * (below), which reads the actual launch flag `/running` reports, is the primary source; + * this is only used when that parse comes up empty (no recognized flag in `cmd`, or `cmd` + * itself unavailable). */ async function fetchContextLength( host: Pick, @@ -136,6 +145,25 @@ async function fetchContextLength( } } +/** + * Parses the REAL configured context size out of llama-swap's own launch command for a + * model (`/running`'s `cmd` field, e.g. `"llama-server -m ... --fit-ctx 16384 ..."`) — + * the primary source for `modelContextLengths`, preferred over `/props`'s `n_ctx` (see + * `fetchContextLength`'s own doc comment for why that field is unreliable here). Checks + * `--fit-ctx` first (llama-swap's own auto-fit flag), then the plain llama.cpp + * `-c`/`--ctx-size`/`--ctx_size` flags a hand-written launch command might use instead. + * Returns `undefined` when `cmd` has none of these — not every launch command needs to + * state one explicitly (llama.cpp has its own default), and guessing one would be worse + * than the "no override applied" the caller already treats an unknown length as. + */ +function parseCtxFromCmd(cmd: unknown): number | undefined { + if (typeof cmd !== 'string') return undefined; + const match = /--fit-ctx\s+(\d+)/.exec(cmd) ?? /(?:^|\s)(?:-c|--ctx-size|--ctx_size)\s+(\d+)/.exec(cmd); + if (!match) return undefined; + const value = Number(match[1]); + return Number.isFinite(value) && value > 0 ? value : undefined; +} + async function discoverModels( host: Pick ): Promise { @@ -169,9 +197,18 @@ async function discoverModels( .filter((m) => m.status && typeof m.status === 'object' && (m.status as { value?: unknown }).value === 'loaded') .map((m) => m.id) .filter((id): id is string => typeof id === 'string' && id.length > 0); - for (const id of loadedIds) { - const ctx = await fetchContextLength(host, id, headers); - if (ctx !== undefined) contextLengths[id] = ctx; + if (loadedIds.length > 0) { + // Primary source: the REAL launch command (see parseCtxFromCmd's own doc comment + // for why /props's n_ctx cannot be trusted here). One /running call covers every + // loaded model, so this never costs more requests than the old /props-only path did + // when the cmd parse succeeds, and exactly one extra when it has to fall back. + const swapStatus = await getLlamaSwapStatus(host); + const cmdById = new Map(swapStatus.running.map((r) => [r.model, r.cmd])); + for (const id of loadedIds) { + const fromCmd = parseCtxFromCmd(cmdById.get(id)); + const ctx = fromCmd ?? (await fetchContextLength(host, id, headers)); + if (ctx !== undefined) contextLengths[id] = ctx; + } } } return { models, contextLengths, sizesGB }; @@ -209,6 +246,9 @@ const RUNNING_TIMEOUT_MS = 5000; export interface LlamaSwapRunningModel { model: string; state: string; + /** The actual launch command llama-swap started this backend with, when it says one — + * see `parseCtxFromCmd`, which reads the real configured context size out of this. */ + cmd?: string; } export interface LlamaSwapStatus { @@ -245,10 +285,14 @@ export async function getLlamaSwapStatus( if (!Array.isArray(body.running)) return { isLlamaSwap: false, running: [] }; const running = body.running .filter( - (r): r is { model: string; state?: unknown } => + (r): r is { model: string; state?: unknown; cmd?: unknown } => !!r && typeof r === 'object' && typeof (r as { model?: unknown }).model === 'string' ) - .map((r) => ({ model: r.model, state: typeof r.state === 'string' ? r.state : 'unknown' })); + .map((r) => ({ + model: r.model, + state: typeof r.state === 'string' ? r.state : 'unknown', + cmd: typeof r.cmd === 'string' ? r.cmd : undefined, + })); return { isLlamaSwap: true, running }; } catch { return { isLlamaSwap: false, running: [] }; diff --git a/test/custom-model-endpoint-rediscovery.test.ts b/test/custom-model-endpoint-rediscovery.test.ts index ae3471fe..b0dedf7c 100644 --- a/test/custom-model-endpoint-rediscovery.test.ts +++ b/test/custom-model-endpoint-rediscovery.test.ts @@ -207,6 +207,90 @@ describe('refreshAllCustomModelHosts: context-length enrichment (llama.cpp/llama const [updated] = await readCustomModelHosts(dir); expect(updated.modelContextLengths).toBeUndefined(); }); + + it('prefers the REAL configured context size parsed from /running’s launch command over /props’s unreliable n_ctx', async () => { + // Confirmed live: llama-swap launched a model with --fit-ctx 16384 (the real, working + // limit — the actual server then refused a request over it), but /props reported + // n_ctx: 154112 for the same model, well over what it would really accept. /props must + // never be reached at all once the /running command parse already answered it. + const dir = getDataDir(); + await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]); + fetchMock.mockImplementation(async (url: URL) => { + if (url.pathname === '/v1/models') { + return new Response(JSON.stringify({ data: [{ id: 'qwen3.8-27b', status: { value: 'loaded' } }] }), { + status: 200, + }); + } + if (url.pathname === '/running') { + return new Response( + JSON.stringify({ + running: [ + { + model: 'qwen3.8-27b', + state: 'ready', + cmd: 'llama-server -m /models/Qwen3.8-27B.gguf --flash-attn on --jinja --fit-ctx 16384 --host 0.0.0.0 --port 5840', + }, + ], + }), + { status: 200 } + ); + } + if (url.pathname === '/props') throw new Error('must never be reached — the cmd parse already answered it'); + throw new Error(`unexpected request: ${url.href}`); + }); + + await refreshAllCustomModelHosts(); + + const [updated] = await readCustomModelHosts(dir); + expect(updated.modelContextLengths).toEqual({ 'qwen3.8-27b': 16384 }); + }); + + it('falls back to /props when /running has no cmd, or the cmd states no recognizable context flag', async () => { + const dir = getDataDir(); + await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]); + fetchMock.mockImplementation(async (url: URL) => { + if (url.pathname === '/v1/models') { + return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 }); + } + if (url.pathname === '/running') { + return new Response( + JSON.stringify({ running: [{ model: 'a', state: 'ready', cmd: 'llama-server -m /models/a.gguf' }] }), + { status: 200 } + ); + } + if (url.pathname === '/props') return new Response(JSON.stringify({ n_ctx: 8192 }), { status: 200 }); + throw new Error(`unexpected request: ${url.href}`); + }); + + await refreshAllCustomModelHosts(); + + const [updated] = await readCustomModelHosts(dir); + expect(updated.modelContextLengths).toEqual({ a: 8192 }); + }); + + it('also recognizes a plain -c/--ctx-size flag, not just llama-swap’s own --fit-ctx', async () => { + const dir = getDataDir(); + await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]); + fetchMock.mockImplementation(async (url: URL) => { + if (url.pathname === '/v1/models') { + return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 }); + } + if (url.pathname === '/running') { + return new Response( + JSON.stringify({ + running: [{ model: 'a', state: 'ready', cmd: 'llama-server -m /models/a.gguf --ctx-size 8192' }], + }), + { status: 200 } + ); + } + throw new Error(`unexpected request: ${url.href}`); // /props must never be reached + }); + + await refreshAllCustomModelHosts(); + + const [updated] = await readCustomModelHosts(dir); + expect(updated.modelContextLengths).toEqual({ a: 8192 }); + }); }); describe('refreshAllCustomModelHosts: model-size enrichment (parsed from /v1/models description)', () => {