mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-04 14:39:42 +02:00
feat(custom-model): warn before launching Claude on a model too small for its own overhead
Claude Code's own fixed per-turn overhead (system prompt + tool schemas,
~36.4K tokens measured live) can exceed a small local model's entire real
context before any conversation history exists to compact — confirmed
live twice as an in:0 out:0 failure on the very first message sent.
CLAUDE_CODE_MAX_CONTEXT_TOKENS cannot fix this: it only governs when
history gets compacted, and there is none on message one.
- exceedsSafeContextFloor() (custom-model-routes.ts): true when a CLI's
registry entry declares contextLengthVar (currently only claude) and
the model's discovered context is below CLAUDE_MIN_SAFE_CONTEXT_TOKENS
(40000). A no-op for every other CLI by construction.
- Both apply routes (POST /api/sessions/:id/custom-model and the
quick-start customModel path) check this before the swap-conflict
check and before launching/restarting anything, returning
{requiresContextWarning, modelId, contextLength, minSafeContextTokens}
— skipped when confirmed:true.
- Frontend: #customModelContextWarningModal + _confirmContextWarning/
_resolveContextWarningConfirm (session-ui.js), wired into both
_quickStartWithCustomModelConfirm and _runCustomModelEntryViaRestart
(the path Claude actually uses) ahead of the swap-confirmation check.
Explains the fix in-modal: give the model an explicit larger -c/
--ctx-size in llama-swap instead of relying on --fit-ctx, which
optimizes for the biggest model that fits rather than the biggest
context.
Tests added for the route-level warning/confirm/skip cases and the
frontend modal + launch-flow wiring. Docs updated (custom-model-
endpoints.md, wiki/Custom-Model-Endpoints.md) and the PR's running
changeset extended.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
993710263d
commit
b45a96358e
@@ -948,6 +948,29 @@
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Custom Model Endpoint Profiles: context-window-too-small warning
|
||||
(docs/custom-model-endpoints-plan.md) — shown before launching a CLI
|
||||
whose own fixed system-prompt/tool-schema overhead exceeds the
|
||||
model's real discovered context, which guarantees a first-message
|
||||
failure regardless of CLAUDE_CODE_MAX_CONTEXT_TOKENS. See
|
||||
_confirmContextWarning() in session-ui.js. -->
|
||||
<div class="modal" id="customModelContextWarningModal">
|
||||
<div class="modal-backdrop" onclick="app._resolveContextWarningConfirm(false)"></div>
|
||||
<div class="modal-content modal-sm">
|
||||
<div class="modal-header">
|
||||
<h3>Context window too small</h3>
|
||||
<button class="modal-close" onclick="app._resolveContextWarningConfirm(false)" aria-label="Cancel">×</button>
|
||||
</div>
|
||||
<div class="modal-body">
|
||||
<p class="form-hint" id="customModelContextWarningMessage" style="white-space: pre-wrap;"></p>
|
||||
</div>
|
||||
<div class="modal-footer">
|
||||
<button class="btn-toolbar" onclick="app._resolveContextWarningConfirm(false)">Cancel</button>
|
||||
<button class="btn-toolbar btn-primary" onclick="app._resolveContextWarningConfirm(true)">Launch anyway</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Cron Jobs Modal -->
|
||||
<div class="modal" id="cronModal">
|
||||
<div class="modal-backdrop" onclick="app.closeCron()"></div>
|
||||
|
||||
@@ -693,6 +693,45 @@ Object.assign(CodemanApp.prototype, {
|
||||
resolve?.(proceed);
|
||||
},
|
||||
|
||||
/**
|
||||
* In-app warning shown when the apply route reports `requiresContextWarning`: this
|
||||
* model's real discovered context is smaller than the CLI's own fixed system-prompt/
|
||||
* tool-schema overhead, which guarantees the very first message fails outright — no
|
||||
* `CLAUDE_CODE_MAX_CONTEXT_TOKENS` value fixes that, since there is no conversation
|
||||
* history yet for compaction to trim. Same promise-based pattern as
|
||||
* `_confirmModelSwap`; `_resolveContextWarningConfirm` settles it.
|
||||
*/
|
||||
_confirmContextWarning(modelId, contextLength, minSafeContextTokens) {
|
||||
const modal = document.getElementById('customModelContextWarningModal');
|
||||
const messageEl = document.getElementById('customModelContextWarningMessage');
|
||||
if (messageEl) {
|
||||
const known = typeof contextLength === 'number';
|
||||
messageEl.textContent =
|
||||
`${modelId} is configured with ` +
|
||||
(known ? `only ${contextLength.toLocaleString()} tokens of` : 'an unknown (too small)') +
|
||||
` context, but this CLI needs roughly ${minSafeContextTokens.toLocaleString()}+ tokens just for its own ` +
|
||||
`system prompt and tools — before any conversation history. Its very first message will fail outright, ` +
|
||||
`no matter what context size Codeman tells it to expect.\n\n` +
|
||||
`To fix this, reconfigure llama-swap to give this model (or a smaller one) an explicit larger context ` +
|
||||
`instead of relying on auto-fit (--fit-ctx), which optimizes for the biggest MODEL that fits, not the ` +
|
||||
`biggest CONTEXT — e.g. add "-c 65536" (or as large a --ctx-size as your hardware holds) to its llama-swap ` +
|
||||
`config entry. A smaller model at a much larger explicit context often fits in the same VRAM a bigger ` +
|
||||
`model's auto-fit context gets shrunk to make room for.`;
|
||||
}
|
||||
modal?.classList.add('active');
|
||||
return new Promise((resolve) => {
|
||||
this._resolveContextWarningConfirmPromise = resolve;
|
||||
});
|
||||
},
|
||||
|
||||
/** Called by the modal's Cancel/Launch-anyway buttons and its backdrop click. */
|
||||
_resolveContextWarningConfirm(proceed) {
|
||||
document.getElementById('customModelContextWarningModal')?.classList.remove('active');
|
||||
const resolve = this._resolveContextWarningConfirmPromise;
|
||||
this._resolveContextWarningConfirmPromise = null;
|
||||
resolve?.(proceed);
|
||||
},
|
||||
|
||||
/** A model row in the picker modal was clicked: close it and launch with that choice. */
|
||||
chooseCustomModelAndRun(modelId) {
|
||||
const pending = this._pendingCustomModelPick;
|
||||
@@ -789,6 +828,15 @@ Object.assign(CodemanApp.prototype, {
|
||||
return res.json();
|
||||
};
|
||||
let data = await post(bodyObj);
|
||||
if (data?.data?.requiresContextWarning) {
|
||||
const { modelId, contextLength, minSafeContextTokens } = data.data;
|
||||
const proceed = await this._confirmContextWarning(modelId, contextLength, minSafeContextTokens);
|
||||
if (!proceed) {
|
||||
this._lastCustomModelLaunchResult = undefined;
|
||||
return { success: false, error: 'Launch cancelled — context window too small' };
|
||||
}
|
||||
data = await post({ ...bodyObj, customModel: { ...bodyObj.customModel, confirmed: true } });
|
||||
}
|
||||
if (data?.data?.requiresConfirmation) {
|
||||
const { currentlyLoadedModel, affectedSessions } = data.data;
|
||||
const names = affectedSessions.map((s) => s.name || s.id).join(', ');
|
||||
@@ -871,6 +919,26 @@ Object.assign(CodemanApp.prototype, {
|
||||
// once `data.success !== false`.
|
||||
let payload = data?.success !== false ? data?.data : undefined;
|
||||
|
||||
// This CLI's own fixed overhead (system prompt + tool schemas) may exceed the
|
||||
// model's real discovered context outright — no context-length declaration can
|
||||
// fix that, since compaction only trims conversation history and there is none
|
||||
// on message 1. Warn and let the user decide whether to launch anyway, same
|
||||
// confirmed:true re-send pattern as the swap check below.
|
||||
if (ok && payload?.requiresContextWarning) {
|
||||
const proceed = await this._confirmContextWarning(
|
||||
payload.modelId,
|
||||
payload.contextLength,
|
||||
payload.minSafeContextTokens
|
||||
);
|
||||
if (!proceed) {
|
||||
switchingToast?.dismiss();
|
||||
this.showToast('Kept the native backend — context window too small', 'info');
|
||||
return;
|
||||
}
|
||||
({ ok, data, res } = await this._applyCustomModelToSession(sessionId, endpointId, modelId, true));
|
||||
payload = data?.success !== false ? data?.data : undefined;
|
||||
}
|
||||
|
||||
// llama-swap runs one model at a time: switching would unload it out from under
|
||||
// another session actively using it. The route only asks when that's actually true
|
||||
// (never just because a swap is needed at all) — confirming re-sends the exact same
|
||||
|
||||
@@ -23,11 +23,46 @@ import { isBlockedWebviewUrl } from '../webview-egress-policy.js';
|
||||
import { egressBlockedReason, webviewFetch } from '../webview-egress.js';
|
||||
import { CustomModelHostSchema } from '../schemas.js';
|
||||
import { readCustomModelHosts, writeCustomModelHosts, type CustomModelHost } from '../../custom-model-hosts.js';
|
||||
import type { CliEntry } from '../../config/cli-registry/types.js';
|
||||
|
||||
const CODEMAN_CONFIG_DIR = getDataDir();
|
||||
const DISCOVER_TIMEOUT_MS = 8000;
|
||||
const PROPS_TIMEOUT_MS = 5000;
|
||||
|
||||
/**
|
||||
* Claude Code's own system prompt + tool schemas cost roughly this many tokens on EVERY
|
||||
* request, before a single character of conversation history — confirmed live, twice, on
|
||||
* requests reporting `in:0 out:0` (the very first exchange) failing at ~36.4K tokens. No
|
||||
* `CLAUDE_CODE_MAX_CONTEXT_TOKENS` value fixes this: that setting only changes when Claude
|
||||
* Code decides to COMPACT conversation history, and there is no history yet on the first
|
||||
* message for it to trim. A model whose real context is below this floor will refuse
|
||||
* Claude Code's very first message outright, unconditionally.
|
||||
*
|
||||
* Set well above the ~36.4K actually measured — CLAUDE.md size, active MCP servers, and
|
||||
* enabled skills all add to a project's real baseline, so the observed figure is a floor
|
||||
* for THAT one workspace, not a ceiling for every one. Erring conservative here means a
|
||||
* borderline-safe model still gets warned about (the user can launch anyway), rather than
|
||||
* this floor missing a genuinely-too-small one because a smaller test project happened to
|
||||
* fit.
|
||||
*/
|
||||
export const CLAUDE_MIN_SAFE_CONTEXT_TOKENS = 40000;
|
||||
|
||||
/**
|
||||
* True when applying this model to this CLI is heading for a guaranteed first-message
|
||||
* failure per `CLAUDE_MIN_SAFE_CONTEXT_TOKENS` above. Gated on `contextLengthVar` (today,
|
||||
* only claude's registry entry declares one) rather than a hardcoded mode check: a CLI
|
||||
* with a small enough baseline of its own to never trip this would have no reason to
|
||||
* declare the field in the first place, so the check simply never applies to it.
|
||||
*/
|
||||
export function exceedsSafeContextFloor(
|
||||
entry: Pick<CliEntry, 'capabilities'>,
|
||||
contextLength: number | undefined
|
||||
): boolean {
|
||||
const cap = entry.capabilities.customModelInjection;
|
||||
if (cap.kind !== 'env' || !cap.contextLengthVar) return false;
|
||||
return typeof contextLength === 'number' && contextLength < CLAUDE_MIN_SAFE_CONTEXT_TOKENS;
|
||||
}
|
||||
|
||||
function adminOnly(req: FastifyRequest, reply: { code: (n: number) => unknown }): ApiResponse<never> | null {
|
||||
if (!isMultiUserMode() || isAdmin(req)) return null;
|
||||
reply.code(403);
|
||||
|
||||
@@ -55,7 +55,12 @@ import {
|
||||
} from '../schemas.js';
|
||||
import { readCustomModelHosts } from '../../custom-model-hosts.js';
|
||||
import { applyCustomModelInjection, removeConfigDir } from '../../custom-model-injection-apply.js';
|
||||
import { getLlamaSwapStatus, triggerLlamaSwapLoad } from './custom-model-routes.js';
|
||||
import {
|
||||
getLlamaSwapStatus,
|
||||
triggerLlamaSwapLoad,
|
||||
exceedsSafeContextFloor,
|
||||
CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
|
||||
} from './custom-model-routes.js';
|
||||
import { matchesPattern } from '../../config/cli-registry/patterns.js';
|
||||
import { ownerLayoutKey } from '../../tab-layout-persistence.js';
|
||||
import { TabLayoutValidationError } from '../../tab-layout.js';
|
||||
@@ -1209,6 +1214,23 @@ export function registerSessionRoutes(
|
||||
if (!endpoint) {
|
||||
return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
}
|
||||
const contextLength = endpoint.modelContextLengths?.[body.modelId];
|
||||
|
||||
// Some CLIs (today: only claude) carry enough of their own fixed system-prompt/tool-
|
||||
// schema overhead that a small enough real context guarantees a first-message failure
|
||||
// no matter what CLAUDE_CODE_MAX_CONTEXT_TOKENS says — confirmed live at ~36.4K tokens
|
||||
// against a model configured with a real 16384-token context. Warn before committing
|
||||
// to a restart that's certain to fail, rather than letting the user discover it via a
|
||||
// cryptic 400 from the CLI itself. `confirmed` (already used for the swap-conflict
|
||||
// warning below) skips this too — the user has already said "launch anyway" once.
|
||||
if (!body.confirmed && exceedsSafeContextFloor(entry, contextLength)) {
|
||||
return {
|
||||
requiresContextWarning: true,
|
||||
modelId: body.modelId,
|
||||
contextLength,
|
||||
minSafeContextTokens: CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
|
||||
};
|
||||
}
|
||||
|
||||
// llama.cpp runs exactly one model at a time; llama-swap unloads and reloads it on
|
||||
// demand, which can take anywhere from a few seconds to over a minute — long enough
|
||||
@@ -1250,7 +1272,6 @@ export function registerSessionRoutes(
|
||||
// fails its pattern rather than quoting it, which would silently launch the CLI on its
|
||||
// own default provider again, so refuse an id the pattern cannot carry up front.
|
||||
const modelSpec = entry.launch.params.model;
|
||||
const contextLength = endpoint.modelContextLengths?.[body.modelId];
|
||||
const applied = applyCustomModelInjection(entry, endpoint, body.modelId, session.id, contextLength);
|
||||
if (!applied) {
|
||||
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${session.mode} has no known custom-model mechanism`);
|
||||
@@ -3570,6 +3591,20 @@ export function registerSessionRoutes(
|
||||
const cmHosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
const cmEndpoint = cmHosts.find((h) => h.id === customModel.endpointId);
|
||||
if (!cmEndpoint) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
const cmContextLength = cmEndpoint.modelContextLengths?.[customModel.modelId];
|
||||
|
||||
// See the dedicated route's own comment for the full reasoning: some CLIs' own fixed
|
||||
// overhead can exceed a small enough real context on the very first message,
|
||||
// regardless of contextLengthVar. Warn before creating a session that's certain to
|
||||
// fail immediately.
|
||||
if (!customModel.confirmed && exceedsSafeContextFloor(cmEntry, cmContextLength)) {
|
||||
return {
|
||||
requiresContextWarning: true,
|
||||
modelId: customModel.modelId,
|
||||
contextLength: cmContextLength,
|
||||
minSafeContextTokens: CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
|
||||
};
|
||||
}
|
||||
|
||||
// See the dedicated route's own comment for the full reasoning: llama.cpp runs one
|
||||
// model at a time, llama-swap swaps on demand, and switching away from what another
|
||||
@@ -3603,7 +3638,6 @@ export function registerSessionRoutes(
|
||||
// with, not a placeholder: `new Session({ id: ... })` accepts an explicit id for
|
||||
// exactly this reason.
|
||||
qsCustomModelSessionId = randomUUID();
|
||||
const cmContextLength = cmEndpoint.modelContextLengths?.[customModel.modelId];
|
||||
const cmApplied = applyCustomModelInjection(
|
||||
cmEntry,
|
||||
cmEndpoint,
|
||||
|
||||
Reference in New Issue
Block a user