mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-05 06:59:42 +02:00
feat(custom-model): warn before launching Claude on a model too small for its own overhead
Claude Code's own fixed per-turn overhead (system prompt + tool schemas,
~36.4K tokens measured live) can exceed a small local model's entire real
context before any conversation history exists to compact — confirmed
live twice as an in:0 out:0 failure on the very first message sent.
CLAUDE_CODE_MAX_CONTEXT_TOKENS cannot fix this: it only governs when
history gets compacted, and there is none on message one.
- exceedsSafeContextFloor() (custom-model-routes.ts): true when a CLI's
registry entry declares contextLengthVar (currently only claude) and
the model's discovered context is below CLAUDE_MIN_SAFE_CONTEXT_TOKENS
(40000). A no-op for every other CLI by construction.
- Both apply routes (POST /api/sessions/:id/custom-model and the
quick-start customModel path) check this before the swap-conflict
check and before launching/restarting anything, returning
{requiresContextWarning, modelId, contextLength, minSafeContextTokens}
— skipped when confirmed:true.
- Frontend: #customModelContextWarningModal + _confirmContextWarning/
_resolveContextWarningConfirm (session-ui.js), wired into both
_quickStartWithCustomModelConfirm and _runCustomModelEntryViaRestart
(the path Claude actually uses) ahead of the swap-confirmation check.
Explains the fix in-modal: give the model an explicit larger -c/
--ctx-size in llama-swap instead of relying on --fit-ctx, which
optimizes for the biggest model that fits rather than the biggest
context.
Tests added for the route-level warning/confirm/skip cases and the
frontend modal + launch-flow wiring. Docs updated (custom-model-
endpoints.md, wiki/Custom-Model-Endpoints.md) and the PR's running
changeset extended.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
993710263d
commit
b45a96358e
@@ -57,6 +57,9 @@ function bootApp(
|
||||
<div class="modal" id="customModelSwapConfirmModal">
|
||||
<p id="customModelSwapConfirmMessage"></p>
|
||||
</div>
|
||||
<div class="modal" id="customModelContextWarningModal">
|
||||
<p id="customModelContextWarningMessage"></p>
|
||||
</div>
|
||||
</body>`,
|
||||
{ url: 'http://localhost/', runScripts: 'dangerously' }
|
||||
);
|
||||
@@ -859,6 +862,64 @@ describe('Custom Model Endpoint Profiles: model-size load-time estimate', () =>
|
||||
});
|
||||
});
|
||||
|
||||
describe("Custom Model Endpoint Profiles: requiresContextWarning (this CLI's own overhead can exceed a small model's real context)", () => {
|
||||
function launchHarness(applyResponses: Array<Record<string, unknown>>) {
|
||||
const { win, app } = bootApp({});
|
||||
app.activeSessionId = 'old-session';
|
||||
app.run = async () => {
|
||||
app.activeSessionId = 'new-session';
|
||||
};
|
||||
const applyBodies: unknown[] = [];
|
||||
let call = 0;
|
||||
app._api = async (path: string, opts?: { body?: unknown }) => {
|
||||
if (path.endsWith('/custom-model')) {
|
||||
applyBodies.push(opts?.body);
|
||||
const data = applyResponses[Math.min(call, applyResponses.length - 1)];
|
||||
call += 1;
|
||||
return { ok: true, status: 200, json: async () => ({ success: true, data }) };
|
||||
}
|
||||
throw new Error(`unexpected _api call: ${path}`);
|
||||
};
|
||||
return { win, app, applyBodies };
|
||||
}
|
||||
|
||||
it('confirming the in-app context-warning modal re-sends the apply with confirmed:true', async () => {
|
||||
const { app, applyBodies } = launchHarness([
|
||||
{ requiresContextWarning: true, modelId: 'qwen3', contextLength: 16384, minSafeContextTokens: 40000 },
|
||||
{ customModel: { endpointId: 'llama-box' }, restarted: true, modelSwapInProgress: false },
|
||||
]);
|
||||
let confirmArgs: unknown[] | undefined;
|
||||
app._confirmContextWarning = async (...args: unknown[]) => {
|
||||
confirmArgs = args;
|
||||
return true;
|
||||
};
|
||||
|
||||
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
|
||||
|
||||
expect(confirmArgs).toEqual(['qwen3', 16384, 40000]);
|
||||
expect(applyBodies).toEqual([
|
||||
{ endpointId: 'llama-box', modelId: 'qwen3' },
|
||||
{ endpointId: 'llama-box', modelId: 'qwen3', confirmed: true },
|
||||
]);
|
||||
});
|
||||
|
||||
it('declining the in-app context-warning modal keeps the native backend and never re-sends the apply', async () => {
|
||||
const { app, applyBodies } = launchHarness([
|
||||
{ requiresContextWarning: true, modelId: 'qwen3', contextLength: 16384, minSafeContextTokens: 40000 },
|
||||
]);
|
||||
app._confirmContextWarning = async () => false;
|
||||
let toastMessage: string | undefined;
|
||||
app.showToast = (msg: string) => {
|
||||
toastMessage = msg;
|
||||
};
|
||||
|
||||
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
|
||||
|
||||
expect(applyBodies).toHaveLength(1); // no second (confirmed) call
|
||||
expect(toastMessage).toMatch(/context window too small/i);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Custom Model Endpoint Profiles: _confirmModelSwap (in-app modal, replaces a native confirm() popup)', () => {
|
||||
it('shows the message, activates the modal, and resolves true when "Switch anyway" is clicked', async () => {
|
||||
const { win, app } = bootApp({});
|
||||
@@ -884,3 +945,41 @@ describe('Custom Model Endpoint Profiles: _confirmModelSwap (in-app modal, repla
|
||||
expect(win.document.getElementById('customModelSwapConfirmModal')!.classList.contains('active')).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Custom Model Endpoint Profiles: _confirmContextWarning (in-app modal, native backend never restarted while it is up)', () => {
|
||||
it('shows a message naming the model, the discovered context and the safe floor, activates the modal, and resolves true on "Launch anyway"', async () => {
|
||||
const { win, app } = bootApp({});
|
||||
const promise = app._confirmContextWarning('qwen3.8-27b-ud-q4_k_xl', 16384, 40000);
|
||||
|
||||
const modal = win.document.getElementById('customModelContextWarningModal')!;
|
||||
expect(modal.classList.contains('active')).toBe(true);
|
||||
const message = win.document.getElementById('customModelContextWarningMessage')!.textContent!;
|
||||
expect(message).toContain('qwen3.8-27b-ud-q4_k_xl');
|
||||
expect(message).toContain('16,384');
|
||||
expect(message).toContain('40,000');
|
||||
expect(message).toMatch(/llama-swap/i);
|
||||
expect(message).toMatch(/fit-ctx/i);
|
||||
|
||||
app._resolveContextWarningConfirm(true);
|
||||
|
||||
expect(await promise).toBe(true);
|
||||
expect(modal.classList.contains('active')).toBe(false);
|
||||
});
|
||||
|
||||
it('resolves false when Cancel is clicked', async () => {
|
||||
const { win, app } = bootApp({});
|
||||
const promise = app._confirmContextWarning('qwen3', 16384, 40000);
|
||||
app._resolveContextWarningConfirm(false);
|
||||
expect(await promise).toBe(false);
|
||||
expect(win.document.getElementById('customModelContextWarningModal')!.classList.contains('active')).toBe(false);
|
||||
});
|
||||
|
||||
it('describes an unknown context length without printing a bogus number', async () => {
|
||||
const { win, app } = bootApp({});
|
||||
void app._confirmContextWarning('qwen3', undefined, 40000);
|
||||
const message = win.document.getElementById('customModelContextWarningMessage')!.textContent!;
|
||||
expect(message).not.toMatch(/undefined/);
|
||||
expect(message).toMatch(/unknown/i);
|
||||
app._resolveContextWarningConfirm(false);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -275,6 +275,59 @@ describe('POST /api/quick-start: customModel (one-shot custom-model launch)', ()
|
||||
});
|
||||
});
|
||||
|
||||
describe("context-window floor warning (this CLI's own overhead can exceed a small model's real context)", () => {
|
||||
const SMALL_CTX_ENDPOINT: CustomModelHost = {
|
||||
id: 'ep-small',
|
||||
label: 'tiny box',
|
||||
baseUrl: 'http://192.168.1.51:8080',
|
||||
apiKey: 'k',
|
||||
modelContextLengths: { 'qwen3.8-27b-ud-q4_k_xl': 16384 },
|
||||
};
|
||||
|
||||
it('warns instead of launching when the discovered context is below the safe floor', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-small-ctx',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.requiresContextWarning).toBe(true);
|
||||
expect(body.modelId).toBe('qwen3.8-27b-ud-q4_k_xl');
|
||||
expect(body.contextLength).toBe(16384);
|
||||
expect(body.minSafeContextTokens).toBe(40000);
|
||||
// Nothing was actually created.
|
||||
expect(ctx.sessions.size).toBe(1);
|
||||
});
|
||||
|
||||
it('launches once confirmed, skipping the context check', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-small-ctx-confirmed',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl', confirmed: true },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
expect(ctx.sessions.size).toBe(2);
|
||||
});
|
||||
|
||||
it('does not warn when nothing about context was discovered', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-no-ctx-data',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('triggering the actual llama-swap load (not just watching for it)', () => {
|
||||
it('sends a real inference request naming the target model, concurrently with launching the session', async () => {
|
||||
const chatCalls: unknown[] = [];
|
||||
|
||||
@@ -351,6 +351,113 @@ describe('POST /api/sessions/:id/custom-model', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("context-window floor warning (this CLI's own overhead can exceed a small model's real context)", () => {
|
||||
const SMALL_CTX_ENDPOINT: CustomModelHost = {
|
||||
id: 'ep-small',
|
||||
label: 'tiny box',
|
||||
baseUrl: 'http://192.168.1.51:8080',
|
||||
apiKey: 'k',
|
||||
modelContextLengths: { 'qwen3.8-27b-ud-q4_k_xl': 16384 },
|
||||
};
|
||||
|
||||
it('warns instead of applying when the discovered context is below the safe floor', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.success).not.toBe(false);
|
||||
expect(body.requiresContextWarning).toBe(true);
|
||||
expect(body.modelId).toBe('qwen3.8-27b-ud-q4_k_xl');
|
||||
expect(body.contextLength).toBe(16384);
|
||||
expect(body.minSafeContextTokens).toBe(40000);
|
||||
// Nothing actually applied yet — this call only warned, it did not switch.
|
||||
expect(session.setCustomModel).not.toHaveBeenCalled();
|
||||
expect(session.restartCli).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('applies once confirmed, skipping the context check the second time', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl', confirmed: true },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.requiresContextWarning).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
expect(session.restartCli).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not warn when the discovered context is comfortably above the floor', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const roomyEndpoint: CustomModelHost = {
|
||||
id: 'ep-roomy',
|
||||
label: 'roomy box',
|
||||
baseUrl: 'http://192.168.1.52:8080',
|
||||
apiKey: 'k',
|
||||
modelContextLengths: { qwen3: 65536 },
|
||||
};
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, roomyEndpoint]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-roomy', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not warn when the context length was never discovered (nothing to compare)', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not warn for a CLI whose registry entry declares no contextLengthVar (opencode)', async () => {
|
||||
// opencode's customModelInjection kind is configContentEnv, not env+contextLengthVar,
|
||||
// so exceedsSafeContextFloor is false by construction regardless of context size.
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'opencode';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('triggering the actual llama-swap load (not just watching for it)', () => {
|
||||
it('sends a real inference request naming the target model when it is not already loaded and ready', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
|
||||
Reference in New Issue
Block a user