mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-03 05:59:43 +02:00
fix(custom-model): isolate Claude config dir and inject real context length
Addresses two live-validation findings on the Run-menu custom-model picker: 1. Both claude.ai and ANTHROPIC_API_KEY set warning. Claude Code still coexists an OAuth login with an injected ANTHROPIC_API_KEY in the same config directory and warns about it (confirmed cosmetic - the API key wins for actual requests, verified via a real session's own API Usage Billing line). A custom-model claude session now gets an isolated CLAUDE_CONFIG_DIR (registry-declared via a new configDirVar field, empty, no files written into it) so there is nothing to conflict with. projects is symlinked (junction on Windows) back into the real config dir so the response viewer, subagent windows and Read My Mind keep working for that session, best-effort. 2. Context-window overflow. Claude Code assumes a large default context window for a model id it doesn't recognise and never compacts, so a custom endpoint's real, much smaller context (verified live: a 400 exceeding a 16384-token llama-swap model with a stock ~33.7K-token system prompt) silently overflows. Discovery now also learns each model's real context length from llama.cpp/llama-swap's GET /props?model=<id> (n_ctx), but ONLY for a model llama-swap's own /v1/models response already marks status.value === 'loaded' - never an unloaded one, since llama-swap treats ?model= as a routing hint and probing an unloaded model risks triggering an actual, slow, GPU-swapping load as a side effect of read-only discovery. A server with no status field at all gets no enrichment rather than a guess; a model not probed this round keeps its previously-learned value until it disappears from the list entirely. Stored per model (CustomModelHost.modelContextLengths) and applied via a new contextLengthVar registry field, set to CLAUDE_CODE_MAX_CONTEXT_TOKENS for claude. Both new fields live on the existing env-kind customModelInjection capability shape, declared only on claude's registry entry - every other CLI's injection is unaffected (pinned by test). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
5c25a52f95
commit
0e8b1981af
@@ -26,6 +26,7 @@ import { readCustomModelHosts, writeCustomModelHosts, type CustomModelHost } fro
|
||||
|
||||
const CODEMAN_CONFIG_DIR = getDataDir();
|
||||
const DISCOVER_TIMEOUT_MS = 8000;
|
||||
const PROPS_TIMEOUT_MS = 5000;
|
||||
|
||||
function adminOnly(req: FastifyRequest, reply: { code: (n: number) => unknown }): ApiResponse<never> | null {
|
||||
if (!isMultiUserMode() || isAdmin(req)) return null;
|
||||
@@ -72,7 +73,7 @@ function applyStoredApiKey(incoming: CustomModelHost, existing: CustomModelHost)
|
||||
return incoming.apiKey ? incoming : { ...incoming, apiKey: existing.apiKey };
|
||||
}
|
||||
|
||||
async function discoverModels(host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>): Promise<string[]> {
|
||||
function authHeaders(host: Pick<CustomModelHost, 'apiKey' | 'authStyle'>): Record<string, string> {
|
||||
const headers: Record<string, string> = {};
|
||||
const apiKey = host.apiKey?.trim();
|
||||
// Exactly ONE header, never both — see custom-model-hosts.ts's CustomModelAuthStyle
|
||||
@@ -80,14 +81,73 @@ async function discoverModels(host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' |
|
||||
const style = host.authStyle ?? 'bearer';
|
||||
if (apiKey && style === 'bearer') headers.Authorization = `Bearer ${apiKey}`;
|
||||
if (apiKey && style === 'api-key') headers['api-key'] = apiKey;
|
||||
return headers;
|
||||
}
|
||||
|
||||
export interface DiscoveryResult {
|
||||
models: string[];
|
||||
/** See `CustomModelHost.modelContextLengths` — only ever populated for models already loaded. */
|
||||
contextLengths: Record<string, number>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort: fetches `GET /props?model=<id>` (llama.cpp-native, llama-swap-proxied) for
|
||||
* ONE already-loaded model and pulls its real `n_ctx` out. Never called for a model that
|
||||
* isn't already loaded — see the caller and `CustomModelHost.modelContextLengths` for why
|
||||
* that's a hard safety requirement, not just a nicety: llama-swap treats this endpoint's
|
||||
* `?model=` as a routing hint, and asking it about an unloaded model risks triggering an
|
||||
* actual (slow, GPU-swapping) load as a side effect of what should be read-only discovery.
|
||||
* Any failure (unreachable, non-2xx, missing/malformed field) is swallowed — one model's
|
||||
* context length is a nice-to-have, never worth failing the whole discovery pass over.
|
||||
*/
|
||||
async function fetchContextLength(
|
||||
host: Pick<CustomModelHost, 'baseUrl'>,
|
||||
modelId: string,
|
||||
headers: Record<string, string>
|
||||
): Promise<number | undefined> {
|
||||
try {
|
||||
const url = new URL(`${host.baseUrl.replace(/\/+$/, '')}/props`);
|
||||
url.searchParams.set('model', modelId);
|
||||
const res = await webviewFetch(url, { headers, signal: AbortSignal.timeout(PROPS_TIMEOUT_MS) });
|
||||
if (!res.ok) return undefined;
|
||||
const body = (await res.json()) as { n_ctx?: unknown; default_generation_settings?: { n_ctx?: unknown } };
|
||||
const nCtx = body.n_ctx ?? body.default_generation_settings?.n_ctx;
|
||||
return typeof nCtx === 'number' && Number.isFinite(nCtx) && nCtx > 0 ? nCtx : undefined;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
async function discoverModels(
|
||||
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>
|
||||
): Promise<DiscoveryResult> {
|
||||
const headers = authHeaders(host);
|
||||
const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/v1/models`), {
|
||||
headers,
|
||||
signal: AbortSignal.timeout(DISCOVER_TIMEOUT_MS),
|
||||
});
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
||||
const body = (await res.json()) as { data?: Array<{ id?: unknown }> };
|
||||
return (body.data ?? []).map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
|
||||
const body = (await res.json()) as { data?: Array<{ id?: unknown; status?: { value?: unknown } }> };
|
||||
const entries = body.data ?? [];
|
||||
const models = entries.map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
|
||||
|
||||
// llama-swap-specific, feature-detected: a server that never mentions `status` on ANY
|
||||
// entry gets no context-length enrichment at all, rather than treating "no status field"
|
||||
// as "assume unloaded" — either reading is a guess, and skipping is the safe one, since
|
||||
// fetchContextLength must only ever run against a model this server itself calls loaded.
|
||||
const hasStatusField = entries.some((m) => m && typeof m === 'object' && 'status' in m);
|
||||
const contextLengths: Record<string, number> = {};
|
||||
if (hasStatusField) {
|
||||
const loadedIds = entries
|
||||
.filter((m) => m.status && typeof m.status === 'object' && (m.status as { value?: unknown }).value === 'loaded')
|
||||
.map((m) => m.id)
|
||||
.filter((id): id is string => typeof id === 'string' && id.length > 0);
|
||||
for (const id of loadedIds) {
|
||||
const ctx = await fetchContextLength(host, id, headers);
|
||||
if (ctx !== undefined) contextLengths[id] = ctx;
|
||||
}
|
||||
}
|
||||
return { models, contextLengths };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -117,9 +177,16 @@ type RedactedHost = ReturnType<typeof redactApiKey>;
|
||||
* each do their own `discoverModels()` + error handling around one shared
|
||||
* "how to apply a successful result" step.
|
||||
*/
|
||||
function applyDiscoveredModels(host: CustomModelHost, models: string[]): CustomModelHost {
|
||||
function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult): CustomModelHost {
|
||||
const { models, contextLengths } = result;
|
||||
const defaultModelId = host.defaultModelId && models.includes(host.defaultModelId) ? host.defaultModelId : undefined;
|
||||
return { ...host, models, defaultModelId, lastDiscoveredAt: new Date().toISOString() };
|
||||
// Merge onto what's already known rather than replacing: a model not probed this round
|
||||
// (not currently loaded) keeps whatever context length an earlier round already learned
|
||||
// for it, and one no longer in the fresh list is dropped, same reasoning as defaultModelId.
|
||||
const merged = { ...host.modelContextLengths, ...contextLengths };
|
||||
const kept = Object.fromEntries(Object.entries(merged).filter(([id]) => models.includes(id)));
|
||||
const modelContextLengths = Object.keys(kept).length > 0 ? kept : undefined;
|
||||
return { ...host, models, defaultModelId, modelContextLengths, lastDiscoveredAt: new Date().toISOString() };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -135,9 +202,9 @@ export async function refreshAllCustomModelHosts(): Promise<void> {
|
||||
const hosts = await readCustomModelHosts(dataDir);
|
||||
for (const host of hosts) {
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) continue;
|
||||
let models: string[];
|
||||
let result: DiscoveryResult;
|
||||
try {
|
||||
models = await discoverModels(host);
|
||||
result = await discoverModels(host);
|
||||
} catch {
|
||||
continue; // unreachable this cycle — try again next tick, not fatal to the sweep
|
||||
}
|
||||
@@ -147,7 +214,7 @@ export async function refreshAllCustomModelHosts(): Promise<void> {
|
||||
const current = await readCustomModelHosts(dataDir);
|
||||
const index = current.findIndex((item) => item.id === host.id);
|
||||
if (index === -1) continue; // deleted mid-sweep
|
||||
current[index] = applyDiscoveredModels(current[index], models);
|
||||
current[index] = applyDiscoveredModels(current[index], result);
|
||||
await writeCustomModelHosts(dataDir, current);
|
||||
}
|
||||
}
|
||||
@@ -222,11 +289,11 @@ export function registerCustomModelRoutes(app: FastifyInstance): void {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
try {
|
||||
const models = await discoverModels(host);
|
||||
const result = await discoverModels(host);
|
||||
const next = [...hosts];
|
||||
next[index] = applyDiscoveredModels(host, models);
|
||||
next[index] = applyDiscoveredModels(host, result);
|
||||
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
|
||||
return { success: true, data: { models } };
|
||||
return { success: true, data: { models: result.models } };
|
||||
} catch (err) {
|
||||
const blocked = egressBlockedReason(err);
|
||||
return createErrorResponse(
|
||||
|
||||
@@ -1214,7 +1214,8 @@ export function registerSessionRoutes(
|
||||
// fails its pattern rather than quoting it, which would silently launch the CLI on its
|
||||
// own default provider again, so refuse an id the pattern cannot carry up front.
|
||||
const modelSpec = entry.launch.params.model;
|
||||
const applied = applyCustomModelInjection(entry, endpoint, body.modelId, session.id);
|
||||
const contextLength = endpoint.modelContextLengths?.[body.modelId];
|
||||
const applied = applyCustomModelInjection(entry, endpoint, body.modelId, session.id, contextLength);
|
||||
if (!applied) {
|
||||
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${session.mode} has no known custom-model mechanism`);
|
||||
}
|
||||
|
||||
@@ -1924,6 +1924,9 @@ export const CustomModelHostSchema = z.object({
|
||||
// host to apply it, so the check belongs there once, not duplicated into a refine
|
||||
// that would run on every unrelated field edit too).
|
||||
defaultModelId: z.string().max(200).optional(),
|
||||
// Server-populated by discovery (custom-model-routes.ts); accepted here only so a client
|
||||
// round-tripping the GET response back through PUT (edit-save) doesn't drop it.
|
||||
modelContextLengths: z.record(z.string().max(200), z.number().int().positive().max(100_000_000)).optional(),
|
||||
});
|
||||
|
||||
/** POST /api/sessions/:id/custom-model — apply or clear a session's custom-model selection. */
|
||||
|
||||
Reference in New Issue
Block a user