feat(custom-model): warn before launching Claude on a model too small for its own overhead

Claude Code's own fixed per-turn overhead (system prompt + tool schemas,
~36.4K tokens measured live) can exceed a small local model's entire real
context before any conversation history exists to compact — confirmed
live twice as an in:0 out:0 failure on the very first message sent.
CLAUDE_CODE_MAX_CONTEXT_TOKENS cannot fix this: it only governs when
history gets compacted, and there is none on message one.

- exceedsSafeContextFloor() (custom-model-routes.ts): true when a CLI's
  registry entry declares contextLengthVar (currently only claude) and
  the model's discovered context is below CLAUDE_MIN_SAFE_CONTEXT_TOKENS
  (40000). A no-op for every other CLI by construction.
- Both apply routes (POST /api/sessions/:id/custom-model and the
  quick-start customModel path) check this before the swap-conflict
  check and before launching/restarting anything, returning
  {requiresContextWarning, modelId, contextLength, minSafeContextTokens}
  — skipped when confirmed:true.
- Frontend: #customModelContextWarningModal + _confirmContextWarning/
  _resolveContextWarningConfirm (session-ui.js), wired into both
  _quickStartWithCustomModelConfirm and _runCustomModelEntryViaRestart
  (the path Claude actually uses) ahead of the swap-confirmation check.
  Explains the fix in-modal: give the model an explicit larger -c/
  --ctx-size in llama-swap instead of relying on --fit-ctx, which
  optimizes for the biggest model that fits rather than the biggest
  context.

Tests added for the route-level warning/confirm/skip cases and the
frontend modal + launch-flow wiring. Docs updated (custom-model-
endpoints.md, wiki/Custom-Model-Endpoints.md) and the PR's running
changeset extended.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
Devvyn
2026-09-17 07:36:57 +08:00
co-authored by Claude Sonnet 5
parent 993710263d
commit b45a96358e
10 changed files with 500 additions and 20 deletions
@@ -275,6 +275,59 @@ describe('POST /api/quick-start: customModel (one-shot custom-model launch)', ()
});
});
describe("context-window floor warning (this CLI's own overhead can exceed a small model's real context)", () => {
const SMALL_CTX_ENDPOINT: CustomModelHost = {
id: 'ep-small',
label: 'tiny box',
baseUrl: 'http://192.168.1.51:8080',
apiKey: 'k',
modelContextLengths: { 'qwen3.8-27b-ud-q4_k_xl': 16384 },
};
it('warns instead of launching when the discovered context is below the safe floor', async () => {
await writeCustomModelHosts(getDataDir(), [ENDPOINT, SMALL_CTX_ENDPOINT]);
const res = await quickStart({
caseName: 'cm-small-ctx',
mode: 'claude',
customModel: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
});
const body = res.json();
expect(body.requiresContextWarning).toBe(true);
expect(body.modelId).toBe('qwen3.8-27b-ud-q4_k_xl');
expect(body.contextLength).toBe(16384);
expect(body.minSafeContextTokens).toBe(40000);
// Nothing was actually created.
expect(ctx.sessions.size).toBe(1);
});
it('launches once confirmed, skipping the context check', async () => {
await writeCustomModelHosts(getDataDir(), [ENDPOINT, SMALL_CTX_ENDPOINT]);
const res = await quickStart({
caseName: 'cm-small-ctx-confirmed',
mode: 'claude',
customModel: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl', confirmed: true },
});
expect(res.statusCode).toBe(200);
expect(res.json().requiresContextWarning).toBeUndefined();
expect(ctx.sessions.size).toBe(2);
});
it('does not warn when nothing about context was discovered', async () => {
const res = await quickStart({
caseName: 'cm-no-ctx-data',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.statusCode).toBe(200);
expect(res.json().requiresContextWarning).toBeUndefined();
});
});
describe('triggering the actual llama-swap load (not just watching for it)', () => {
it('sends a real inference request naming the target model, concurrently with launching the session', async () => {
const chatCalls: unknown[] = [];
+107
View File
@@ -351,6 +351,113 @@ describe('POST /api/sessions/:id/custom-model', () => {
});
});
describe("context-window floor warning (this CLI's own overhead can exceed a small model's real context)", () => {
const SMALL_CTX_ENDPOINT: CustomModelHost = {
id: 'ep-small',
label: 'tiny box',
baseUrl: 'http://192.168.1.51:8080',
apiKey: 'k',
modelContextLengths: { 'qwen3.8-27b-ud-q4_k_xl': 16384 },
};
it('warns instead of applying when the discovered context is below the safe floor', async () => {
const { app, ctx } = await setup();
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
});
const body = res.json();
expect(body.success).not.toBe(false);
expect(body.requiresContextWarning).toBe(true);
expect(body.modelId).toBe('qwen3.8-27b-ud-q4_k_xl');
expect(body.contextLength).toBe(16384);
expect(body.minSafeContextTokens).toBe(40000);
// Nothing actually applied yet — this call only warned, it did not switch.
expect(session.setCustomModel).not.toHaveBeenCalled();
expect(session.restartCli).not.toHaveBeenCalled();
});
it('applies once confirmed, skipping the context check the second time', async () => {
const { app, ctx } = await setup();
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl', confirmed: true },
});
const body = res.json();
expect(body.requiresContextWarning).toBeUndefined();
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
expect(session.restartCli).toHaveBeenCalledTimes(1);
});
it('does not warn when the discovered context is comfortably above the floor', async () => {
const { app, ctx } = await setup();
const roomyEndpoint: CustomModelHost = {
id: 'ep-roomy',
label: 'roomy box',
baseUrl: 'http://192.168.1.52:8080',
apiKey: 'k',
modelContextLengths: { qwen3: 65536 },
};
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, roomyEndpoint]);
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep-roomy', modelId: 'qwen3' },
});
expect(res.json().requiresContextWarning).toBeUndefined();
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
});
it('does not warn when the context length was never discovered (nothing to compare)', async () => {
const { app, ctx } = await setup();
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT]);
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.json().requiresContextWarning).toBeUndefined();
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
});
it('does not warn for a CLI whose registry entry declares no contextLengthVar (opencode)', async () => {
// opencode's customModelInjection kind is configContentEnv, not env+contextLengthVar,
// so exceedsSafeContextFloor is false by construction regardless of context size.
const { app, ctx } = await setup();
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'opencode';
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
});
expect(res.json().requiresContextWarning).toBeUndefined();
});
});
describe('triggering the actual llama-swap load (not just watching for it)', () => {
it('sends a real inference request naming the target model when it is not already loaded and ready', async () => {
const { app, ctx } = await setup();