mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-07 16:09:43 +02:00
feat(custom-model): warn before launching Claude on a model too small for its own overhead
Claude Code's own fixed per-turn overhead (system prompt + tool schemas,
~36.4K tokens measured live) can exceed a small local model's entire real
context before any conversation history exists to compact — confirmed
live twice as an in:0 out:0 failure on the very first message sent.
CLAUDE_CODE_MAX_CONTEXT_TOKENS cannot fix this: it only governs when
history gets compacted, and there is none on message one.
- exceedsSafeContextFloor() (custom-model-routes.ts): true when a CLI's
registry entry declares contextLengthVar (currently only claude) and
the model's discovered context is below CLAUDE_MIN_SAFE_CONTEXT_TOKENS
(40000). A no-op for every other CLI by construction.
- Both apply routes (POST /api/sessions/:id/custom-model and the
quick-start customModel path) check this before the swap-conflict
check and before launching/restarting anything, returning
{requiresContextWarning, modelId, contextLength, minSafeContextTokens}
— skipped when confirmed:true.
- Frontend: #customModelContextWarningModal + _confirmContextWarning/
_resolveContextWarningConfirm (session-ui.js), wired into both
_quickStartWithCustomModelConfirm and _runCustomModelEntryViaRestart
(the path Claude actually uses) ahead of the swap-confirmation check.
Explains the fix in-modal: give the model an explicit larger -c/
--ctx-size in llama-swap instead of relying on --fit-ctx, which
optimizes for the biggest model that fits rather than the biggest
context.
Tests added for the route-level warning/confirm/skip cases and the
frontend modal + launch-flow wiring. Docs updated (custom-model-
endpoints.md, wiki/Custom-Model-Endpoints.md) and the PR's running
changeset extended.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
993710263d
commit
b45a96358e
@@ -275,6 +275,59 @@ describe('POST /api/quick-start: customModel (one-shot custom-model launch)', ()
|
||||
});
|
||||
});
|
||||
|
||||
describe("context-window floor warning (this CLI's own overhead can exceed a small model's real context)", () => {
|
||||
const SMALL_CTX_ENDPOINT: CustomModelHost = {
|
||||
id: 'ep-small',
|
||||
label: 'tiny box',
|
||||
baseUrl: 'http://192.168.1.51:8080',
|
||||
apiKey: 'k',
|
||||
modelContextLengths: { 'qwen3.8-27b-ud-q4_k_xl': 16384 },
|
||||
};
|
||||
|
||||
it('warns instead of launching when the discovered context is below the safe floor', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-small-ctx',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.requiresContextWarning).toBe(true);
|
||||
expect(body.modelId).toBe('qwen3.8-27b-ud-q4_k_xl');
|
||||
expect(body.contextLength).toBe(16384);
|
||||
expect(body.minSafeContextTokens).toBe(40000);
|
||||
// Nothing was actually created.
|
||||
expect(ctx.sessions.size).toBe(1);
|
||||
});
|
||||
|
||||
it('launches once confirmed, skipping the context check', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-small-ctx-confirmed',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl', confirmed: true },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
expect(ctx.sessions.size).toBe(2);
|
||||
});
|
||||
|
||||
it('does not warn when nothing about context was discovered', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-no-ctx-data',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('triggering the actual llama-swap load (not just watching for it)', () => {
|
||||
it('sends a real inference request naming the target model, concurrently with launching the session', async () => {
|
||||
const chatCalls: unknown[] = [];
|
||||
|
||||
@@ -351,6 +351,113 @@ describe('POST /api/sessions/:id/custom-model', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("context-window floor warning (this CLI's own overhead can exceed a small model's real context)", () => {
|
||||
const SMALL_CTX_ENDPOINT: CustomModelHost = {
|
||||
id: 'ep-small',
|
||||
label: 'tiny box',
|
||||
baseUrl: 'http://192.168.1.51:8080',
|
||||
apiKey: 'k',
|
||||
modelContextLengths: { 'qwen3.8-27b-ud-q4_k_xl': 16384 },
|
||||
};
|
||||
|
||||
it('warns instead of applying when the discovered context is below the safe floor', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.success).not.toBe(false);
|
||||
expect(body.requiresContextWarning).toBe(true);
|
||||
expect(body.modelId).toBe('qwen3.8-27b-ud-q4_k_xl');
|
||||
expect(body.contextLength).toBe(16384);
|
||||
expect(body.minSafeContextTokens).toBe(40000);
|
||||
// Nothing actually applied yet — this call only warned, it did not switch.
|
||||
expect(session.setCustomModel).not.toHaveBeenCalled();
|
||||
expect(session.restartCli).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('applies once confirmed, skipping the context check the second time', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl', confirmed: true },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.requiresContextWarning).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
expect(session.restartCli).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not warn when the discovered context is comfortably above the floor', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const roomyEndpoint: CustomModelHost = {
|
||||
id: 'ep-roomy',
|
||||
label: 'roomy box',
|
||||
baseUrl: 'http://192.168.1.52:8080',
|
||||
apiKey: 'k',
|
||||
modelContextLengths: { qwen3: 65536 },
|
||||
};
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, roomyEndpoint]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-roomy', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not warn when the context length was never discovered (nothing to compare)', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not warn for a CLI whose registry entry declares no contextLengthVar (opencode)', async () => {
|
||||
// opencode's customModelInjection kind is configContentEnv, not env+contextLengthVar,
|
||||
// so exceedsSafeContextFloor is false by construction regardless of context size.
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'opencode';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('triggering the actual llama-swap load (not just watching for it)', () => {
|
||||
it('sends a real inference request naming the target model when it is not already loaded and ready', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
|
||||
Reference in New Issue
Block a user