mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-05 23:19:43 +02:00
fix(custom-model): stop trusting /props's n_ctx, parse the real context size from /running's cmd
Root cause of the context-overflow regression reported live: "API Error: 400 request (36437 tokens) exceeds the available context size (16384 tokens)". Discovery had stored modelContextLengths.qwen3.8-27b-ud-q4_k_xl = 154112, so CLAUDE_CODE_MAX_CONTEXT_TOKENS told Claude Code it had a huge window and it never compacted - but the real llama-swap server was launched with --fit-ctx 16384 (confirmed against /running's own cmd field) and refused the request right at that real limit. /props?model=<id>'s n_ctx (the field discovery read) is confirmed live to be unreliable for a --fit-ctx-launched backend: it reported 154112 for the same model /running says was launched with --fit-ctx 16384 - appears to report the model's theoretical/trained maximum context, not the runtime- configured one. discoverModels() now parses the REAL configured size straight out of llama-swap's own launch command instead (parseCtxFromCmd(), reading /running's cmd field - --fit-ctx first, then the plain llama.cpp -c/ --ctx-size a hand-written command might use), and only falls back to the old /props probe when cmd states no recognizable flag at all. One /running call now covers every loaded model's context length in a single request, same as it already did for the swap-conflict check and the load trigger. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
7bbe408e44
commit
993710263d
@@ -207,6 +207,90 @@ describe('refreshAllCustomModelHosts: context-length enrichment (llama.cpp/llama
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toBeUndefined();
|
||||
});
|
||||
|
||||
it('prefers the REAL configured context size parsed from /running’s launch command over /props’s unreliable n_ctx', async () => {
|
||||
// Confirmed live: llama-swap launched a model with --fit-ctx 16384 (the real, working
|
||||
// limit — the actual server then refused a request over it), but /props reported
|
||||
// n_ctx: 154112 for the same model, well over what it would really accept. /props must
|
||||
// never be reached at all once the /running command parse already answered it.
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'qwen3.8-27b', status: { value: 'loaded' } }] }), {
|
||||
status: 200,
|
||||
});
|
||||
}
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
running: [
|
||||
{
|
||||
model: 'qwen3.8-27b',
|
||||
state: 'ready',
|
||||
cmd: 'llama-server -m /models/Qwen3.8-27B.gguf --flash-attn on --jinja --fit-ctx 16384 --host 0.0.0.0 --port 5840',
|
||||
},
|
||||
],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
if (url.pathname === '/props') throw new Error('must never be reached — the cmd parse already answered it');
|
||||
throw new Error(`unexpected request: ${url.href}`);
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ 'qwen3.8-27b': 16384 });
|
||||
});
|
||||
|
||||
it('falls back to /props when /running has no cmd, or the cmd states no recognizable context flag', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(
|
||||
JSON.stringify({ running: [{ model: 'a', state: 'ready', cmd: 'llama-server -m /models/a.gguf' }] }),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
if (url.pathname === '/props') return new Response(JSON.stringify({ n_ctx: 8192 }), { status: 200 });
|
||||
throw new Error(`unexpected request: ${url.href}`);
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ a: 8192 });
|
||||
});
|
||||
|
||||
it('also recognizes a plain -c/--ctx-size flag, not just llama-swap’s own --fit-ctx', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
running: [{ model: 'a', state: 'ready', cmd: 'llama-server -m /models/a.gguf --ctx-size 8192' }],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
throw new Error(`unexpected request: ${url.href}`); // /props must never be reached
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ a: 8192 });
|
||||
});
|
||||
});
|
||||
|
||||
describe('refreshAllCustomModelHosts: model-size enrichment (parsed from /v1/models description)', () => {
|
||||
|
||||
Reference in New Issue
Block a user