Merge pull request #430 from opticon454/custom-model-run-menu

This commit is contained in:
Codeman maintainer
2026-09-19 12:25:11 +02:00
43 changed files with 7304 additions and 144 deletions
+692 -17
View File
@@ -16,16 +16,52 @@
import type { FastifyInstance, FastifyRequest } from 'fastify';
import { ApiErrorCode, createErrorResponse, type ApiResponse } from '../../types.js';
import { isAdmin, parseBody } from '../route-helpers.js';
import { isAdmin, parseBody, readJsonConfig, SETTINGS_PATH } from '../route-helpers.js';
import { isMultiUserMode } from '../../config/multiuser.js';
import { getDataDir } from '../../config/instance.js';
import { isBlockedWebviewUrl } from '../webview-egress-policy.js';
import { egressBlockedReason, webviewFetch } from '../webview-egress.js';
import { CustomModelHostSchema } from '../schemas.js';
import { readCustomModelHosts, writeCustomModelHosts, type CustomModelHost } from '../../custom-model-hosts.js';
import type { CliEntry } from '../../config/cli-registry/types.js';
const CODEMAN_CONFIG_DIR = getDataDir();
const DISCOVER_TIMEOUT_MS = 8000;
const PROPS_TIMEOUT_MS = 5000;
/**
* Claude Code's own system prompt + tool schemas cost roughly this many tokens on EVERY
* request, before a single character of conversation history — confirmed live, twice, on
* requests reporting `in:0 out:0` (the very first exchange) failing at ~36.4K tokens. No
* `CLAUDE_CODE_MAX_CONTEXT_TOKENS` value fixes this: that setting only changes when Claude
* Code decides to COMPACT conversation history, and there is no history yet on the first
* message for it to trim. A model whose real context is below this floor will refuse
* Claude Code's very first message outright, unconditionally.
*
* Set well above the ~36.4K actually measured — CLAUDE.md size, active MCP servers, and
* enabled skills all add to a project's real baseline, so the observed figure is a floor
* for THAT one workspace, not a ceiling for every one. Erring conservative here means a
* borderline-safe model still gets warned about (the user can launch anyway), rather than
* this floor missing a genuinely-too-small one because a smaller test project happened to
* fit.
*/
export const CLAUDE_MIN_SAFE_CONTEXT_TOKENS = 40000;
/**
* True when applying this model to this CLI is heading for a guaranteed first-message
* failure per `CLAUDE_MIN_SAFE_CONTEXT_TOKENS` above. Gated on `contextLengthVar` (today,
* only claude's registry entry declares one) rather than a hardcoded mode check: a CLI
* with a small enough baseline of its own to never trip this would have no reason to
* declare the field in the first place, so the check simply never applies to it.
*/
export function exceedsSafeContextFloor(
entry: Pick<CliEntry, 'capabilities'>,
contextLength: number | undefined
): boolean {
const cap = entry.capabilities.customModelInjection;
if (cap.kind !== 'env' || !cap.contextLengthVar) return false;
return typeof contextLength === 'number' && contextLength < CLAUDE_MIN_SAFE_CONTEXT_TOKENS;
}
function adminOnly(req: FastifyRequest, reply: { code: (n: number) => unknown }): ApiResponse<never> | null {
if (!isMultiUserMode() || isAdmin(req)) return null;
@@ -33,7 +69,58 @@ function adminOnly(req: FastifyRequest, reply: { code: (n: number) => unknown })
return createErrorResponse(ApiErrorCode.FORBIDDEN, 'Admin only in multi-user mode');
}
async function discoverModels(host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>): Promise<string[]> {
/**
* `defaultModelId` names the model the Run-menu picker applies for this endpoint with
* no further choice, so it must actually be one of the discovered `models` — a schema
* `.refine()` can't see across the two fields the way this can, and would also run on
* every unrelated field edit rather than only when either of these two changes.
*/
function invalidDefaultModel(host: Pick<CustomModelHost, 'defaultModelId' | 'models'>): ApiResponse<never> | null {
if (host.defaultModelId === undefined) return null;
if ((host.models ?? []).includes(host.defaultModelId)) return null;
return createErrorResponse(
ApiErrorCode.INVALID_INPUT,
'defaultModelId must be one of the endpoint’s discovered models'
);
}
/**
* Never hand the stored credential back to the browser, on GET, POST or PUT
* alike — the file is written 0600 precisely because it holds one. `apiKeySet`
* is what lets the editor say "unchanged if left blank" without the client
* ever holding the real value: `applyStoredApiKey()` below is the other half,
* treating an absent key on PUT as "keep the stored one" rather than clearing
* it, which is what makes never returning it survivable for the edit flow.
*/
function redactApiKey(host: CustomModelHost): Omit<CustomModelHost, 'apiKey'> & { apiKeySet: boolean } {
const { apiKey, ...rest } = host;
return { ...rest, apiKeySet: !!apiKey };
}
/**
* A PUT body with no `apiKey` (or a blank one) means "leave it alone", never
* "clear it": the editor never receives the real value to resend deliberately
* unchanged (see redactApiKey), so the only way it can tell the two apart is
* by omission. There is deliberately no way to CLEAR a key back to unset this
* way — a pre-existing limitation, not something this changes.
*/
function applyStoredApiKey(incoming: CustomModelHost, existing: CustomModelHost): CustomModelHost {
return incoming.apiKey ? incoming : { ...incoming, apiKey: existing.apiKey };
}
/**
* `modelContextLengths`/`modelSizesGB` are server-populated by discovery, never
* user-entered, and PUT replaces the whole record — so merge them back in from the
* stored host rather than trust whatever the editor's body carried (or omitted).
* The editor only ever sends `models`/`lastDiscoveredAt` verbatim from its cached
* copy; requiring it to also round-trip these two is exactly the kind of thing a
* future caller forgets, same class of bug `applyStoredApiKey` exists to prevent.
*/
function applyDiscoveredFields(incoming: CustomModelHost, existing: CustomModelHost): CustomModelHost {
return { ...incoming, modelContextLengths: existing.modelContextLengths, modelSizesGB: existing.modelSizesGB };
}
function authHeaders(host: Pick<CustomModelHost, 'apiKey' | 'authStyle'>): Record<string, string> {
const headers: Record<string, string> = {};
const apiKey = host.apiKey?.trim();
// Exactly ONE header, never both — see custom-model-hosts.ts's CustomModelAuthStyle
@@ -41,14 +128,137 @@ async function discoverModels(host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' |
const style = host.authStyle ?? 'bearer';
if (apiKey && style === 'bearer') headers.Authorization = `Bearer ${apiKey}`;
if (apiKey && style === 'api-key') headers['api-key'] = apiKey;
return headers;
}
export interface DiscoveryResult {
models: string[];
/** See `CustomModelHost.modelContextLengths` — only ever populated for models already loaded. */
contextLengths: Record<string, number>;
/** See `CustomModelHost.modelSizesGB` — populated for every model whose own listing states one. */
sizesGB: Record<string, number>;
}
/**
* Best-effort: pulls a file size in GB out of a model's own `description`, when the
* server states one. llama-swap writes `"Auto-discovered 16.35 GB - parameters
* auto-fitted by llama.cpp"` for a model it found on disk itself; a hand-configured
* profile's own description (e.g. `"General-purpose reasoning model, MoE CPU-offloaded."`)
* has no such figure and correctly yields no estimate rather than a guess — there is no
* separate "give me the file size" endpoint to fall back on.
*/
function parseSizeGB(description: unknown): number | undefined {
if (typeof description !== 'string') return undefined;
const match = /(\d+(?:\.\d+)?)\s*GB\b/i.exec(description);
if (!match) return undefined;
const size = Number(match[1]);
return Number.isFinite(size) && size > 0 ? size : undefined;
}
/**
* Best-effort: fetches `GET /props?model=<id>` (llama.cpp-native, llama-swap-proxied) for
* ONE already-loaded model and pulls its real `n_ctx` out. Never called for a model that
* isn't already loaded — see the caller and `CustomModelHost.modelContextLengths` for why
* that's a hard safety requirement, not just a nicety: llama-swap treats this endpoint's
* `?model=` as a routing hint, and asking it about an unloaded model risks triggering an
* actual (slow, GPU-swapping) load as a side effect of what should be read-only discovery.
* Any failure (unreachable, non-2xx, missing/malformed field) is swallowed — one model's
* context length is a nice-to-have, never worth failing the whole discovery pass over.
*
* ⚠️ FALLBACK ONLY — confirmed live to be actively WRONG for a `--fit-ctx`-launched llama-
* swap backend: `/props`'s `n_ctx` read 154112 for a model llama-swap itself had launched
* with `--fit-ctx 16384` (visible in `/running`'s own `cmd`), and the real server then
* refused a request at the real 16384-token limit — `n_ctx` here appears to report the
* model's theoretical/trained maximum, not the runtime-configured one. `parseCtxFromCmd`
* (below), which reads the actual launch flag `/running` reports, is the primary source;
* this is only used when that parse comes up empty (no recognized flag in `cmd`, or `cmd`
* itself unavailable).
*/
async function fetchContextLength(
host: Pick<CustomModelHost, 'baseUrl'>,
modelId: string,
headers: Record<string, string>
): Promise<number | undefined> {
try {
const url = new URL(`${host.baseUrl.replace(/\/+$/, '')}/props`);
url.searchParams.set('model', modelId);
const res = await webviewFetch(url, { headers, signal: AbortSignal.timeout(PROPS_TIMEOUT_MS) });
if (!res.ok) return undefined;
const body = (await res.json()) as { n_ctx?: unknown; default_generation_settings?: { n_ctx?: unknown } };
const nCtx = body.n_ctx ?? body.default_generation_settings?.n_ctx;
return typeof nCtx === 'number' && Number.isFinite(nCtx) && nCtx > 0 ? nCtx : undefined;
} catch {
return undefined;
}
}
/**
* Parses the REAL configured context size out of llama-swap's own launch command for a
* model (`/running`'s `cmd` field, e.g. `"llama-server -m ... --fit-ctx 16384 ..."`) —
* the primary source for `modelContextLengths`, preferred over `/props`'s `n_ctx` (see
* `fetchContextLength`'s own doc comment for why that field is unreliable here). Checks
* `--fit-ctx` first (llama-swap's own auto-fit flag), then the plain llama.cpp
* `-c`/`--ctx-size`/`--ctx_size` flags a hand-written launch command might use instead.
* Returns `undefined` when `cmd` has none of these — not every launch command needs to
* state one explicitly (llama.cpp has its own default), and guessing one would be worse
* than the "no override applied" the caller already treats an unknown length as.
*/
function parseCtxFromCmd(cmd: unknown): number | undefined {
if (typeof cmd !== 'string') return undefined;
const match = /--fit-ctx\s+(\d+)/.exec(cmd) ?? /(?:^|\s)(?:-c|--ctx-size|--ctx_size)\s+(\d+)/.exec(cmd);
if (!match) return undefined;
const value = Number(match[1]);
return Number.isFinite(value) && value > 0 ? value : undefined;
}
async function discoverModels(
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>
): Promise<DiscoveryResult> {
const headers = authHeaders(host);
const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/v1/models`), {
headers,
signal: AbortSignal.timeout(DISCOVER_TIMEOUT_MS),
});
if (!res.ok) throw new Error(`HTTP ${res.status}`);
const body = (await res.json()) as { data?: Array<{ id?: unknown }> };
return (body.data ?? []).map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
const body = (await res.json()) as {
data?: Array<{ id?: unknown; status?: { value?: unknown }; description?: unknown }>;
};
const entries = body.data ?? [];
const models = entries.map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
const sizesGB: Record<string, number> = {};
for (const entry of entries) {
if (typeof entry.id !== 'string' || !entry.id) continue;
const size = parseSizeGB(entry.description);
if (size !== undefined) sizesGB[entry.id] = size;
}
// llama-swap-specific, feature-detected: a server that never mentions `status` on ANY
// entry gets no context-length enrichment at all, rather than treating "no status field"
// as "assume unloaded" — either reading is a guess, and skipping is the safe one, since
// fetchContextLength must only ever run against a model this server itself calls loaded.
const hasStatusField = entries.some((m) => m && typeof m === 'object' && 'status' in m);
const contextLengths: Record<string, number> = {};
if (hasStatusField) {
const loadedIds = entries
.filter((m) => m.status && typeof m.status === 'object' && (m.status as { value?: unknown }).value === 'loaded')
.map((m) => m.id)
.filter((id): id is string => typeof id === 'string' && id.length > 0);
if (loadedIds.length > 0) {
// Primary source: the REAL launch command (see parseCtxFromCmd's own doc comment
// for why /props's n_ctx cannot be trusted here). One /running call covers every
// loaded model, so this never costs more requests than the old /props-only path did
// when the cmd parse succeeds, and exactly one extra when it has to fall back.
const swapStatus = await getLlamaSwapStatus(host);
const cmdById = new Map(swapStatus.running.map((r) => [r.model, r.cmd]));
for (const id of loadedIds) {
const fromCmd = parseCtxFromCmd(cmdById.get(id));
const ctx = fromCmd ?? (await fetchContextLength(host, id, headers));
if (ctx !== undefined) contextLengths[id] = ctx;
}
}
}
return { models, contextLengths, sizesGB };
}
/**
@@ -66,41 +276,472 @@ function describeFetchError(err: unknown): string {
return message;
}
export function registerCustomModelRoutes(app: FastifyInstance): void {
app.get('/api/model-endpoints', async (req) =>
isMultiUserMode() && !isAdmin(req) ? [] : readCustomModelHosts(CODEMAN_CONFIG_DIR)
);
type RedactedHost = ReturnType<typeof redactApiKey>;
app.post('/api/model-endpoints', async (req, reply): Promise<ApiResponse<{ host: CustomModelHost }>> => {
/**
* Merges a fresh `GET /v1/models` result into a host record: stamps
* `lastDiscoveredAt`, and drops `defaultModelId` if it no longer appears in
* the fresh list (it would otherwise leave the Run-menu picker applying a
* model id the endpoint just told us it doesn't serve). Pure — no IO, so the
* manual route (which reports a fetch failure's *reason* to the caller) and
* the periodic sweep below (which only cares whether it can move on) can
* each do their own `discoverModels()` + error handling around one shared
* "how to apply a successful result" step.
*/
const RUNNING_TIMEOUT_MS = 5000;
export interface LlamaSwapRunningModel {
model: string;
state: string;
/** The actual launch command llama-swap started this backend with, when it says one —
* see `parseCtxFromCmd`, which reads the real configured context size out of this. */
cmd?: string;
}
export interface LlamaSwapStatus {
/**
* Feature-detected via `GET /running`: true only when the server answered with
* llama-swap's own shape (`{ running: [...] }`). Plain llama.cpp (and any other
* OpenAI-compatible server) has no such endpoint and always runs the single model
* it was started with, so there is no "current model" to conflict with — every
* caller must treat `isLlamaSwap: false` as "nothing to check", never as an error.
*/
isLlamaSwap: boolean;
running: LlamaSwapRunningModel[];
}
/**
* Distinguishes llama-swap from a plain llama.cpp/OpenAI-compatible server, and reports
* what llama-swap currently has loaded — llama.cpp only ever runs one GGUF at a time, and
* llama-swap unloads/reloads it on demand when a request asks for a different one, which
* can take anywhere from a few seconds to over a minute. Read-only: this never triggers a
* swap itself (unlike `/props?model=`, `/running` takes no `model` parameter to route by).
* Best-effort like `discoverModels()`'s siblings: any failure (unreachable, non-2xx,
* unexpected shape) reads as "not llama-swap", never thrown.
*/
export async function getLlamaSwapStatus(
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>
): Promise<LlamaSwapStatus> {
try {
const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/running`), {
headers: authHeaders(host),
signal: AbortSignal.timeout(RUNNING_TIMEOUT_MS),
});
if (!res.ok) return { isLlamaSwap: false, running: [] };
const body = (await res.json()) as { running?: unknown };
if (!Array.isArray(body.running)) return { isLlamaSwap: false, running: [] };
const running = body.running
.filter(
(r): r is { model: string; state?: unknown; cmd?: unknown } =>
!!r && typeof r === 'object' && typeof (r as { model?: unknown }).model === 'string'
)
.map((r) => ({
model: r.model,
state: typeof r.state === 'string' ? r.state : 'unknown',
cmd: typeof r.cmd === 'string' ? r.cmd : undefined,
}));
return { isLlamaSwap: true, running };
} catch {
return { isLlamaSwap: false, running: [] };
}
}
interface LlamaSwapLogTail {
latestLine?: string;
lastAccessedAt: number;
controller: AbortController;
}
/** One open `/api/events` tail per endpoint, keyed by host id — see `getLatestLlamaSwapLogLine`. */
const llamaSwapLogTails = new Map<string, LlamaSwapLogTail>();
/** A tail nothing has asked about in this long is closed by the next `pruneIdleLlamaSwapLogTails` sweep. */
const LOG_TAIL_IDLE_MS = 30_000;
/**
* Parses one `data: {...}` payload from llama-swap's `GET /api/events` SSE stream and
* returns the backend (never llama-swap's own proxy) log text it carries, or `undefined`
* for anything else (a different event `type`, a malformed frame, a proxy-sourced one).
*
* The real shape, confirmed live against a real llama-swap deployment — NOT documented
* anywhere the plan doc's original research found, and genuinely surprising the first
* time around: `GET /logs` (the endpoint that name suggests, and this feature's own
* first cut was built against) turns out to carry ONLY llama-swap's own proxy
* request-access log — it never once showed a single backend line even seconds after a
* real, confirmed model swap. The backend llama-server process's actual stdout
* (`load_model: ...`, `llama_server: model loaded`) only ever showed up in `/api/events`,
* as `{"type":"logData","data":"<JSON-string>"}` whose OWN `data` field parses to a
* second object, `{"data": "<newline-joined log text>", "source": "proxy" | "upstream"}`
* — `source` is the exact, explicit distinguisher (`upstream` = the backend process,
* `proxy` = llama-swap's own line), not a guessed regex against the text itself.
*/
function parseBackendLogDataEvent(dataLine: string): string | undefined {
let outer: unknown;
try {
outer = JSON.parse(dataLine);
} catch {
return undefined;
}
if (
!outer ||
typeof outer !== 'object' ||
(outer as { type?: unknown }).type !== 'logData' ||
typeof (outer as { data?: unknown }).data !== 'string'
) {
return undefined;
}
let inner: unknown;
try {
inner = JSON.parse((outer as { data: string }).data);
} catch {
return undefined;
}
if (
!inner ||
typeof inner !== 'object' ||
(inner as { source?: unknown }).source !== 'upstream' ||
typeof (inner as { data?: unknown }).data !== 'string'
) {
return undefined;
}
return (inner as { data: string }).data;
}
/**
* Reads `GET /api/events` forever (until `entry.controller` aborts it), updating
* `entry.latestLine` with the most recent BACKEND log line seen (see
* `parseBackendLogDataEvent`). Fire-and-forget: the caller never awaits this — it runs
* for the tail's whole lifetime in the background, and `getLatestLlamaSwapLogLine` just
* reads whatever `entry.latestLine` currently holds. SSE frames are separated by a blank
* line (`\n\n`), buffered the same way `/running`'s NDJSON-shaped siblings buffer partial
* chunks — a frame split across two `reader.read()` calls must not be parsed early.
*/
async function pumpLlamaSwapLogTail(
host: Pick<CustomModelHost, 'id' | 'baseUrl' | 'apiKey' | 'authStyle'>,
entry: LlamaSwapLogTail
): Promise<void> {
try {
const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/api/events`), {
headers: authHeaders(host),
signal: entry.controller.signal,
});
if (!res.ok || !res.body) return;
const reader = res.body.getReader();
const decoder = new TextDecoder();
let buffer = '';
for (;;) {
const { done, value } = await reader.read();
if (done) break;
buffer += decoder.decode(value, { stream: true });
const frames = buffer.split('\n\n');
buffer = frames.pop() ?? '';
for (const frame of frames) {
const dataLine = frame.split('\n').find((l) => l.startsWith('data:'));
if (!dataLine) continue;
const backendText = parseBackendLogDataEvent(dataLine.slice('data:'.length));
if (!backendText) continue;
const lines = backendText.split('\n').filter((l) => l.trim());
if (lines.length > 0) entry.latestLine = lines[lines.length - 1]!.trim();
}
}
} catch {
// connection dropped / aborted / endpoint unreachable — a future access starts fresh
} finally {
// Delete by IDENTITY, not just by key: an aborted pump can finish after a NEWER
// entry was already created for the same endpoint id (e.g. abort-then-immediately-
// re-request), and deleting unconditionally would remove that newer entry and orphan
// its connection — nothing would ever prune it, since pruneIdleLlamaSwapLogTails only
// walks entries still present in the map.
if (llamaSwapLogTails.get(host.id) === entry) {
llamaSwapLogTails.delete(host.id);
}
}
}
/**
* Real-time "what is llama.cpp actually doing right now" for the loading banner
* (docs/custom-model-endpoints-plan.md): llama-swap's `GET /api/events` SSE stream
* carries the backend llama-server process's own stdout — `load_model: loading model
* '<path>'`, `load_model: initializing, n_slots = N, n_ctx_slot = N`, `llama_server:
* model loaded`, etc — tagged `source: "upstream"`, distinct from llama-swap's own
* `source: "proxy"` request-access lines (see `parseBackendLogDataEvent`). Confirmed
* live against a real llama-swap deployment, including through an actual forced model
* swap end-to-end.
*
* Held OPEN per endpoint rather than re-opened on every 1s poll — confirmed live to stay
* open indefinitely (read past 220KB over 8 seconds with no `done`), unlike `/logs`
* (see `parseBackendLogDataEvent`'s doc comment), so reconnecting each poll would be
* pure waste. One connection is reused across every session currently watching a load on
* that endpoint; since llama.cpp/llama-swap only ever runs one model at a time, a line
* seen while a load is in flight is safe to attribute to that load (a deployment that
* could load several models concurrently would need a per-model tag this format doesn't
* provide).
*
* Lazily started on first access and idle-closed rather than left open forever — see
* `pruneIdleLlamaSwapLogTails`.
*/
export function getLatestLlamaSwapLogLine(
host: Pick<CustomModelHost, 'id' | 'baseUrl' | 'apiKey' | 'authStyle'>
): string | undefined {
let entry = llamaSwapLogTails.get(host.id);
if (!entry) {
entry = { lastAccessedAt: Date.now(), controller: new AbortController() };
llamaSwapLogTails.set(host.id, entry);
void pumpLlamaSwapLogTail(host, entry);
}
entry.lastAccessedAt = Date.now();
return entry.latestLine;
}
/**
* Closes any log tail nothing has called `getLatestLlamaSwapLogLine` about in
* `LOG_TAIL_IDLE_MS` — a stream nobody is polling is an open connection with nothing to
* show for it. Called from the same periodic sweep as `detectCustomModelSwapDisplacements`
* in server.ts, not its own timer.
*/
export function pruneIdleLlamaSwapLogTails(now = Date.now()): void {
for (const [id, entry] of llamaSwapLogTails) {
if (now - entry.lastAccessedAt > LOG_TAIL_IDLE_MS) {
entry.controller.abort();
llamaSwapLogTails.delete(id);
}
}
}
/**
* Actually kicks off llama-swap's lazy model load, rather than waiting for the launched
* CLI's own first prompt to do it. llama-swap has no separate "switch model" admin
* endpoint — the ONLY thing that starts a swap is a real inference request naming the
* model (confirmed live: applying a selection alone never appeared in the llama-swap
* server's own logs; nothing had actually asked it to load anything). This sends the
* smallest real request that will — `max_tokens: 1`, one throwaway user message — to
* `${baseUrl}/v1/chat/completions`, the OpenAI-compatible endpoint every supported
* harness already points at.
*
* Deliberately fire-and-forget: the caller (the apply/create routes) returns to the
* client immediately, and the frontend's own polling (`GET .../running-status`) is what
* actually confirms readiness — this call's response is never read, just its side
* effect. No abort/timeout of its own either: a real load can take well over a minute for
* a large model, and this is a normal long-running Node process, so there is nothing to
* clean up by cutting it short. Errors are swallowed for the same reason `discoverModels`'s
* siblings swallow theirs — one endpoint's hiccup here is a nice-to-have that failed, not
* something worth surfacing as a request failure four layers up.
*/
export function triggerLlamaSwapLoad(
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>,
modelId: string
): void {
const url = new URL(`${host.baseUrl.replace(/\/+$/, '')}/v1/chat/completions`);
webviewFetch(url, {
method: 'POST',
headers: { ...authHeaders(host), 'content-type': 'application/json' },
body: JSON.stringify({
model: modelId,
messages: [{ role: 'user', content: 'Hi' }],
max_tokens: 1,
stream: false,
}),
}).catch(() => {
// best-effort — see the doc comment above
});
}
function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult): CustomModelHost {
const { models, contextLengths, sizesGB } = result;
const defaultModelId = host.defaultModelId && models.includes(host.defaultModelId) ? host.defaultModelId : undefined;
// Merge onto what's already known rather than replacing: a model not probed this round
// (not currently loaded) keeps whatever context length an earlier round already learned
// for it, and one no longer in the fresh list is dropped, same reasoning as defaultModelId.
const merged = { ...host.modelContextLengths, ...contextLengths };
const kept = Object.fromEntries(Object.entries(merged).filter(([id]) => models.includes(id)));
const modelContextLengths = Object.keys(kept).length > 0 ? kept : undefined;
// sizesGB, unlike contextLengths, is populated for every model in the SAME pass (no
// loaded-only restriction — see parseSizeGB), so this is closer to a plain replace, but
// still merges onto the previous round rather than dropping a size for a model whose
// description happened to omit the figure on this particular pass.
const mergedSizes = { ...host.modelSizesGB, ...sizesGB };
const keptSizes = Object.fromEntries(Object.entries(mergedSizes).filter(([id]) => models.includes(id)));
const modelSizesGB = Object.keys(keptSizes).length > 0 ? keptSizes : undefined;
return {
...host,
models,
defaultModelId,
modelContextLengths,
modelSizesGB,
lastDiscoveredAt: new Date().toISOString(),
};
}
/**
* `customModelEndpointsEnabled` defaults OFF (unlike `showPlanUsageLimits`'s
* absent-means-on in `readPlanUsageTelemetryEnabled`), so mirror the frontend's
* own gate (`session-ui.js`'s `!settings.customModelEndpointsEnabled`) rather
* than that reader's default. Exists so the periodic re-discovery sweep in
* server.ts can skip entirely while the feature is off, instead of polling
* every saved endpoint forever regardless of the setting.
*/
export async function readCustomModelEndpointsEnabled(): Promise<boolean> {
const settings = await readJsonConfig<Record<string, unknown>>(SETTINGS_PATH, 'settings.json', {});
return settings.customModelEndpointsEnabled === true;
}
/**
* Re-discovers every saved endpoint's models, best-effort. One endpoint being
* unreachable (powered off, wrong network) must not stop the others from
* refreshing, and a read-modify-write per host (rather than one batch write
* at the end) means a crash or restart mid-sweep loses at most the endpoints
* not yet reached, never a write already applied. Exported so both the
* periodic timer (server.ts) and a test can drive it directly.
*/
export async function refreshAllCustomModelHosts(): Promise<void> {
const dataDir = getDataDir();
const hosts = await readCustomModelHosts(dataDir);
for (const host of hosts) {
if (isBlockedWebviewUrl(host.baseUrl)) continue;
let result: DiscoveryResult;
try {
result = await discoverModels(host);
} catch {
continue; // unreachable this cycle — try again next tick, not fatal to the sweep
}
// Re-read + splice by id rather than reusing the array captured above: an
// admin editing or deleting an endpoint via the API mid-sweep must win,
// not be silently overwritten by a refresh that started before their change.
const current = await readCustomModelHosts(dataDir);
const index = current.findIndex((item) => item.id === host.id);
if (index === -1) continue; // deleted mid-sweep
current[index] = applyDiscoveredModels(current[index], result);
await writeCustomModelHosts(dataDir, current);
}
}
/** The subset of `Session` this sweep needs — kept minimal so a test can pass a plain object. */
export interface CustomModelSessionLike {
id: string;
name: string;
customModel?: { endpointId: string; modelId: string; label?: string };
}
/** One session whose model was just found evicted, ready to broadcast as `CustomModelSwappedOut`. */
export interface CustomModelSwapDisplacement {
sessionId: string;
sessionName: string;
endpointId: string;
previousModel: string;
currentlyLoadedModel: string;
}
/**
* Detects when a live session's own custom-model selection is no longer the model
* llama-swap actually has loaded — evicted by ANOTHER session's activity on the same
* endpoint, since llama.cpp/llama-swap runs one model at a time (the apply/create routes'
* own swap-conflict check only ever runs at THAT session's own launch/apply moment, so it
* cannot catch a later eviction triggered by a different session's normal use — confirmed
* live: a session created while nothing else had a live conflict at that instant can still
* get silently displaced afterward). Read-only, and best-effort per endpoint exactly like
* `refreshAllCustomModelHosts`'s sibling sweep — one endpoint's hiccup here never blocks
* checking the others.
*
* `notifiedSessionIds` is the caller's own de-dupe state (`server.ts` keeps one `Set` across
* sweeps), mutated in place: a session id is added once displaced and removed again once its
* own model is loaded and ready — so a LATER, genuinely new displacement can notify again
* rather than the session staying silently un-notified forever after the first one.
*/
export async function detectCustomModelSwapDisplacements(
sessions: Iterable<CustomModelSessionLike>,
notifiedSessionIds: Set<string>
): Promise<CustomModelSwapDisplacement[]> {
const byEndpoint = new Map<string, CustomModelSessionLike[]>();
for (const session of sessions) {
if (!session.customModel) continue;
const group = byEndpoint.get(session.customModel.endpointId);
if (group) group.push(session);
else byEndpoint.set(session.customModel.endpointId, [session]);
}
if (byEndpoint.size === 0) return [];
const hosts = await readCustomModelHosts(getDataDir());
const displacements: CustomModelSwapDisplacement[] = [];
for (const [endpointId, group] of byEndpoint) {
const host = hosts.find((h) => h.id === endpointId);
if (!host) continue; // endpoint deleted since these sessions were created — nothing to check
let status: LlamaSwapStatus;
try {
status = await getLlamaSwapStatus(host);
} catch {
continue; // unreachable this cycle — try again next tick, not fatal to the sweep
}
// Not llama-swap (feature-detected) or nothing loaded at all: nothing has been evicted,
// by construction — a plain llama.cpp/OpenAI-compatible server only ever runs the one
// model it was started with, so there is no "current model" to conflict with.
if (!status.isLlamaSwap || status.running.length === 0) continue;
const currentlyLoaded = status.running.find((r) => r.state === 'ready')?.model ?? status.running[0]?.model;
if (!currentlyLoaded) continue;
for (const session of group) {
const modelId = session.customModel!.modelId;
const stillLoaded = status.running.some((r) => r.model === modelId);
if (stillLoaded) {
notifiedSessionIds.delete(session.id); // back to normal — a future eviction can notify again
continue;
}
if (notifiedSessionIds.has(session.id)) continue; // already told them once for this displacement
notifiedSessionIds.add(session.id);
displacements.push({
sessionId: session.id,
sessionName: session.name,
endpointId,
previousModel: modelId,
currentlyLoadedModel: currentlyLoaded,
});
}
}
return displacements;
}
export function registerCustomModelRoutes(app: FastifyInstance): void {
app.get('/api/model-endpoints', async (req): Promise<RedactedHost[]> => {
if (isMultiUserMode() && !isAdmin(req)) return [];
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
return hosts.map(redactApiKey);
});
app.post('/api/model-endpoints', async (req, reply): Promise<ApiResponse<{ host: RedactedHost }>> => {
const denied = adminOnly(req, reply);
if (denied) return denied;
const host = parseBody(CustomModelHostSchema, req.body);
if (isBlockedWebviewUrl(host.baseUrl)) {
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
}
const badDefault = invalidDefaultModel(host);
if (badDefault) return badDefault;
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
if (hosts.some((item) => item.id === host.id)) {
return createErrorResponse(ApiErrorCode.ALREADY_EXISTS, 'Model endpoint already exists');
}
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, [...hosts, host]);
return { success: true, data: { host } };
return { success: true, data: { host: redactApiKey(host) } };
});
app.put('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ host: CustomModelHost }>> => {
app.put('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ host: RedactedHost }>> => {
const denied = adminOnly(req, reply);
if (denied) return denied;
const { id } = req.params as { id: string };
const host = parseBody(CustomModelHostSchema, { ...(req.body as object), id });
if (isBlockedWebviewUrl(host.baseUrl)) {
const incoming = parseBody(CustomModelHostSchema, { ...(req.body as object), id });
if (isBlockedWebviewUrl(incoming.baseUrl)) {
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
}
const badDefault = invalidDefaultModel(incoming);
if (badDefault) return badDefault;
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
const index = hosts.findIndex((item) => item.id === id);
if (index === -1) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
const host = applyDiscoveredFields(applyStoredApiKey(incoming, hosts[index]), hosts[index]);
const next = [...hosts];
next[index] = host;
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
return { success: true, data: { host } };
return { success: true, data: { host: redactApiKey(host) } };
});
app.delete('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ id: string }>> => {
@@ -129,11 +770,11 @@ export function registerCustomModelRoutes(app: FastifyInstance): void {
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
}
try {
const models = await discoverModels(host);
const result = await discoverModels(host);
const next = [...hosts];
next[index] = { ...host, models, lastDiscoveredAt: new Date().toISOString() };
next[index] = applyDiscoveredModels(host, result);
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
return { success: true, data: { models } };
return { success: true, data: { models: result.models } };
} catch (err) {
const blocked = egressBlockedReason(err);
return createErrorResponse(
@@ -143,4 +784,38 @@ export function registerCustomModelRoutes(app: FastifyInstance): void {
}
}
);
// Read-only, no admin gate: any session owner who can already point their own session
// at this endpoint (POST .../custom-model, ungated by design — see session-routes.ts)
// can equally ask what it currently has loaded, before or while that apply is pending.
app.get(
'/api/model-endpoints/:id/running-status',
async (
req
): Promise<
ApiResponse<{
isLlamaSwap: boolean;
running: Array<Pick<LlamaSwapRunningModel, 'model' | 'state'>>;
logLine?: string;
}>
> => {
const { id } = req.params as { id: string };
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
const host = hosts.find((item) => item.id === id);
if (!host) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
if (isBlockedWebviewUrl(host.baseUrl)) {
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
}
const status = await getLlamaSwapStatus(host);
// Only worth tailing /logs once llama-swap is actually confirmed — a plain
// llama.cpp/OpenAI-compatible server has no such endpoint at all.
const logLine = status.isLlamaSwap ? getLatestLlamaSwapLogLine(host) : undefined;
// `cmd` (the literal llama-server launch line, which can carry model paths and
// --api-key) exists only so parseCtxFromCmd() can read it server-side during
// discovery — this un-gated, polled-every-second route has no reason to hand it
// to the browser, which only ever reads `model`/`state`.
const running = status.running.map(({ model, state }) => ({ model, state }));
return { success: true, data: { isLlamaSwap: status.isLlamaSwap, running, logLine } };
}
);
}
+9 -1
View File
@@ -28,4 +28,12 @@ export { registerWsRoutes } from './ws-routes.js';
export { registerVoiceRoutes } from './voice-routes.js';
export { registerWebviewRoutes, tryWebviewRefererFallback } from './webview-routes.js';
export { registerTabLayoutRoutes } from './tab-layout-routes.js';
export { registerCustomModelRoutes } from './custom-model-routes.js';
export {
registerCustomModelRoutes,
refreshAllCustomModelHosts,
readCustomModelEndpointsEnabled,
detectCustomModelSwapDisplacements,
pruneIdleLlamaSwapLogTails,
type CustomModelSessionLike,
type CustomModelSwapDisplacement,
} from './custom-model-routes.js';
+262 -11
View File
@@ -11,7 +11,7 @@ import { homedir } from 'node:os';
import { existsSync, statSync, mkdirSync, writeFileSync } from 'node:fs';
import { execFile } from 'node:child_process';
import fs from 'node:fs/promises';
import { randomBytes } from 'node:crypto';
import { randomBytes, randomUUID } from 'node:crypto';
import { performance } from 'node:perf_hooks';
import {
ApiErrorCode,
@@ -57,6 +57,12 @@ import {
} from '../schemas.js';
import { readCustomModelHosts } from '../../custom-model-hosts.js';
import { applyCustomModelInjection, removeConfigDir } from '../../custom-model-injection-apply.js';
import {
getLlamaSwapStatus,
triggerLlamaSwapLoad,
exceedsSafeContextFloor,
CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
} from './custom-model-routes.js';
import { matchesPattern } from '../../config/cli-registry/patterns.js';
import { ownerLayoutKey } from '../../tab-layout-persistence.js';
import { TabLayoutValidationError } from '../../tab-layout.js';
@@ -1232,13 +1238,78 @@ export function registerSessionRoutes(
if (!endpoint) {
return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
}
const contextLength = endpoint.modelContextLengths?.[body.modelId];
// Some CLIs (today: only claude) carry enough of their own fixed system-prompt/tool-
// schema overhead that a small enough real context guarantees a first-message failure
// no matter what CLAUDE_CODE_MAX_CONTEXT_TOKENS says — confirmed live at ~36.4K tokens
// against a model configured with a real 16384-token context. Warn before committing
// to a restart that's certain to fail, rather than letting the user discover it via a
// cryptic 400 from the CLI itself. `confirmed` (already used for the swap-conflict
// warning below) skips this too — the user has already said "launch anyway" once.
if (!body.confirmed && exceedsSafeContextFloor(entry, contextLength)) {
return {
requiresContextWarning: true,
modelId: body.modelId,
contextLength,
minSafeContextTokens: CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
};
}
// llama.cpp runs exactly one model at a time; llama-swap unloads and reloads it on
// demand, which can take anywhere from a few seconds to over a minute — long enough
// that a session mid-swap looks indistinguishable from one that never left the native
// backend. Feature-detected via llama-swap's own `GET /running` (a plain llama.cpp
// server has no such endpoint and reads as `isLlamaSwap: false` — nothing to check).
const swapStatus = await getLlamaSwapStatus(endpoint);
const currentlyLoaded = swapStatus.running.find((r) => r.state === 'ready')?.model ?? swapStatus.running[0]?.model;
// Distinct from targetReady below: this is ONLY about whether proceeding would evict a
// model another session is actively using — true even if nothing is loaded at all yet
// would be wrong here (nothing to evict), so this stays narrowly "a DIFFERENT model is
// currently ready".
const swapNeeded = swapStatus.isLlamaSwap && !!currentlyLoaded && currentlyLoaded !== body.modelId;
// Whether the TARGET model itself is already the one loaded and ready — false whether
// nothing is loaded yet, a different model is loaded, or this one is loaded but still
// mid-load. Drives both the actual load trigger below and modelSwapInProgress in the
// response; deliberately broader than swapNeeded, which only gates the confirmation ask.
const targetReady = swapStatus.running.some((r) => r.model === body.modelId && r.state === 'ready');
// Only ask when switching would actually take the model away from another session
// that is currently using it — never just because a swap is needed at all. `confirmed`
// (set by the caller after showing that warning once) skips asking again.
if (swapNeeded && !body.confirmed) {
const conflicting = [...ctx.sessions.values()].filter(
(s) =>
s.id !== session.id && s.customModel?.endpointId === endpoint.id && s.customModel?.modelId === currentlyLoaded
);
if (conflicting.length > 0) {
// Applying a custom model is ungated for any session owner, so in multi-user
// mode a non-admin pointing their own session at a shared endpoint must not
// learn another user's session names in the confirm dialog — with
// autoNameSessions on, those names are that user's own prompts. The swap is
// still blocked pending confirmation regardless of ownership (a foreign
// session is just as real a disruption); only which ones get NAMED is scoped.
const requestUser = getAuthUser(req);
const affectedSessions = conflicting
.filter((s) => canAccessOwned(requestUser, s.owner))
.map((s) => ({ id: s.id, name: s.name }));
return { requiresConfirmation: true, currentlyLoadedModel: currentlyLoaded, affectedSessions };
}
}
// A CLI whose config alone cannot select the model also gets its `model` launch param
// forced (pi/omp `custom/<id>`, grok's block name). The argv engine DROPS a token that
// fails its pattern rather than quoting it, which would silently launch the CLI on its
// own default provider again, so refuse an id the pattern cannot carry up front.
const modelSpec = entry.launch.params.model;
const applied = applyCustomModelInjection(entry, endpoint, body.modelId, session.id);
const applied = applyCustomModelInjection(
entry,
endpoint,
body.modelId,
session.id,
contextLength,
session.workingDir
);
if (!applied) {
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${session.mode} has no known custom-model mechanism`);
}
@@ -1271,9 +1342,17 @@ export function registerSessionRoutes(
removeConfigDir(previousConfigDir);
}
// Actually kick off llama-swap's load now, rather than waiting on the restarted CLI's
// own first prompt to do it — confirmed live that applying a selection alone never
// reached the llama-swap server at all (nothing in its own logs), since llama-swap has
// no "switch model" admin call, only a real inference request naming the model.
if (swapStatus.isLlamaSwap && !targetReady) {
triggerLlamaSwapLoad(endpoint, body.modelId);
}
const restarted = await session.restartCli();
persistAndBroadcastSession(ctx, session);
return { customModel: session.customModel, restarted };
return { customModel: session.customModel, restarted, modelSwapInProgress: swapStatus.isLlamaSwap && !targetReady };
});
// ========== Delete Session ==========
@@ -3252,6 +3331,7 @@ export function registerSessionRoutes(
effort,
parentSessionId,
agentOrigin,
customModel,
} = parseBody(QuickStartSchema, req.body);
// Resolved ONCE here: the same value labels a case directory this request creates
@@ -3304,11 +3384,12 @@ export function registerSessionRoutes(
grokConfig ||
deepSeekConfig ||
ompConfig ||
openCodeConfig
openCodeConfig ||
customModel
) {
return createErrorResponse(
ApiErrorCode.INVALID_INPUT,
'envOverrides, effort, modelOverride, and per-CLI config are not supported for remote cases (they do not cross ssh). Configure the remote command via the host command override instead.'
'envOverrides, effort, modelOverride, per-CLI config, and custom model endpoints are not supported for remote cases (they do not cross ssh). Configure the remote command via the host command override instead.'
);
}
@@ -3372,11 +3453,12 @@ export function registerSessionRoutes(
grokConfig ||
deepSeekConfig ||
ompConfig ||
openCodeConfig
openCodeConfig ||
customModel
) {
return createErrorResponse(
ApiErrorCode.INVALID_INPUT,
'envOverrides, effort, and per-CLI config are not supported for docker cases (they do not cross into the container). Configure the container via the docker host command override instead.'
'envOverrides, effort, per-CLI config, and custom model endpoints are not supported for docker cases (they do not cross into the container). Configure the container via the docker host command override instead.'
);
}
@@ -3664,7 +3746,148 @@ export function registerSessionRoutes(
);
const qsTerminalHistoryConfig = await ctx.getTerminalHistoryConfig();
const qsGatedEnvOverrides = await clampEnvOverridesForOwner(owner, envOverrides);
const session = new Session({
const qsResolvedOmpConfig = resolveOmpConfigForCreate(mode, resolvedCasePath, ompConfig);
// Custom Model Endpoint Profiles, applied AT CREATE TIME (docs/custom-model-endpoints-plan.md)
// rather than via the dedicated restart-in-place route (POST /api/sessions/:id/custom-
// model, still what an ALREADY-RUNNING session uses to switch later): computing the
// injection before the process exists and launching directly on it avoids the visible
// native-boot-then-restart the restart-after-launch design otherwise shows on every
// custom-model run — most jarring on a CLI like Codex whose TUI fully reinitializes.
// Mirrors the dedicated route's own checks (llama-swap conflict, unsupported CLI,
// unknown endpoint, a model id the CLI's argv pattern can't carry) rather than trusting
// a lighter version of them, since this is the same server-side authority reached a
// different way, not a separate, less-checked path.
let qsCustomModelEnvOverrides = qsGatedEnvOverrides;
// Only the INJECTED keys (never the caller's envOverrides merged in) — this is what
// setCustomModel() bookkeeping must be given below. The Session constructor already
// applies qsCustomModelEnvOverrides (the full merged set) directly; re-merging that
// full set into setCustomModel() would put CLAUDE_CODE_EFFORT_LEVEL back after the
// constructor stripped it (see setCustomModel()'s own doc comment in session.ts).
let qsCustomModelAppliedEnvOverrides: Record<string, string> | undefined;
let qsCustomModelLaunchModel: string | undefined;
let qsCustomModelSessionId: string | undefined;
let qsCustomModelSwapInProgress = false;
let qsCustomModelBookkeeping:
| {
endpointId: string;
modelId: string;
label?: string;
envKeys: string[];
configDir?: string;
launchModel?: string;
}
| undefined;
if (customModel) {
const cmEntry = getCli(mode);
if (!cmEntry) return createErrorResponse(ApiErrorCode.INVALID_INPUT, `No CLI registry entry for mode ${mode}`);
if (cmEntry.capabilities.customModelInjection.kind === 'unsupported') {
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${mode} has no known custom-model mechanism`);
}
const cmHosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
const cmEndpoint = cmHosts.find((h) => h.id === customModel.endpointId);
if (!cmEndpoint) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
const cmContextLength = cmEndpoint.modelContextLengths?.[customModel.modelId];
// See the dedicated route's own comment for the full reasoning: some CLIs' own fixed
// overhead can exceed a small enough real context on the very first message,
// regardless of contextLengthVar. Warn before creating a session that's certain to
// fail immediately.
if (!customModel.confirmed && exceedsSafeContextFloor(cmEntry, cmContextLength)) {
return {
requiresContextWarning: true,
modelId: customModel.modelId,
contextLength: cmContextLength,
minSafeContextTokens: CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
};
}
// See the dedicated route's own comment for the full reasoning: llama.cpp runs one
// model at a time, llama-swap swaps on demand, and switching away from what another
// live session is actively using deserves a warning, not a silent switch. There is no
// "self" to exclude from the affected-sessions scan here — this session doesn't exist
// yet.
const cmSwapStatus = await getLlamaSwapStatus(cmEndpoint);
const cmCurrentlyLoaded =
cmSwapStatus.running.find((r) => r.state === 'ready')?.model ?? cmSwapStatus.running[0]?.model;
const cmSwapNeeded = cmSwapStatus.isLlamaSwap && !!cmCurrentlyLoaded && cmCurrentlyLoaded !== customModel.modelId;
// Broader than cmSwapNeeded (which only gates the confirmation ask above): true
// whenever the TARGET model isn't already loaded and ready, including when nothing
// is loaded at all yet. Drives the actual load trigger below.
const cmTargetReady = cmSwapStatus.running.some((r) => r.model === customModel.modelId && r.state === 'ready');
qsCustomModelSwapInProgress = cmSwapStatus.isLlamaSwap && !cmTargetReady;
if (cmSwapNeeded && !customModel.confirmed) {
const cmConflicting = [...ctx.sessions.values()].filter(
(s) => s.customModel?.endpointId === cmEndpoint.id && s.customModel?.modelId === cmCurrentlyLoaded
);
if (cmConflicting.length > 0) {
// Same reasoning as the dedicated /custom-model route above: the swap is
// still blocked pending confirmation regardless of ownership, but a
// non-admin caller only learns the names of sessions they can access.
const cmRequestUser = getAuthUser(req);
const cmAffectedSessions = cmConflicting
.filter((s) => canAccessOwned(cmRequestUser, s.owner))
.map((s) => ({ id: s.id, name: s.name }));
return {
requiresConfirmation: true,
currentlyLoadedModel: cmCurrentlyLoaded,
affectedSessions: cmAffectedSessions,
};
}
}
// Minted ourselves (rather than left to Session's own default) so the injection
// below — and any configDir it writes — can target the REAL id the session launches
// with, not a placeholder: `new Session({ id: ... })` accepts an explicit id for
// exactly this reason.
qsCustomModelSessionId = randomUUID();
const cmApplied = applyCustomModelInjection(
cmEntry,
cmEndpoint,
customModel.modelId,
qsCustomModelSessionId,
cmContextLength,
resolvedCasePath
);
if (!cmApplied) {
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${mode} has no known custom-model mechanism`);
}
const cmModelSpec = cmEntry.launch.params.model;
if (
cmApplied.launchModel !== undefined &&
cmModelSpec?.type === 'token' &&
!matchesPattern(cmModelSpec.pattern, cmApplied.launchModel)
) {
removeConfigDir(cmApplied.configDir);
return createErrorResponse(
ApiErrorCode.INVALID_INPUT,
`Model id ${JSON.stringify(customModel.modelId)} cannot be passed to ${mode} on its command line`
);
}
qsCustomModelEnvOverrides = { ...qsGatedEnvOverrides, ...cmApplied.envOverrides };
qsCustomModelAppliedEnvOverrides = cmApplied.envOverrides;
qsCustomModelLaunchModel = cmApplied.launchModel;
qsCustomModelBookkeeping = {
endpointId: cmEndpoint.id,
modelId: customModel.modelId,
label: cmEndpoint.label,
envKeys: cmApplied.envKeys,
configDir: cmApplied.configDir,
launchModel: cmApplied.launchModel,
};
// Actually kick off llama-swap's load now — see the dedicated apply route's own
// comment on triggerLlamaSwapLoad for why this can't just wait on the launched CLI's
// first prompt. Fired here, before the session is even created, so the load starts
// concurrently with Claude/Codex/etc. booting rather than after.
if (qsCustomModelSwapInProgress) {
triggerLlamaSwapLoad(cmEndpoint, customModel.modelId);
}
}
const qsSessionOptions: ConstructorParameters<typeof Session>[0] = {
id: qsCustomModelSessionId,
workingDir: resolvedCasePath,
name: sessionName ? sessionName.slice(0, MAX_SESSION_NAME_LENGTH) : '',
mux: ctx.mux,
@@ -3682,15 +3905,42 @@ export function registerSessionRoutes(
piConfig: mode === 'pi' ? qsGatedPiConfig : undefined,
grokConfig: mode === 'grok' ? qsGatedGrokConfig : undefined,
deepSeekConfig: mode === 'deepseek' ? qsGatedDeepSeekConfig : undefined,
ompConfig: resolveOmpConfigForCreate(mode, resolvedCasePath, ompConfig),
envOverrides: qsGatedEnvOverrides,
ompConfig: qsResolvedOmpConfig,
envOverrides: qsCustomModelEnvOverrides,
effort,
remote,
docker,
resumeSessionId: dockerResumeId,
tmuxHistoryLimit: qsTerminalHistoryConfig.tmuxHistoryLimit,
parentSessionId: qsParentSessionId,
});
};
// Force the custom-model selection's launchModel (pi/omp `custom/<id>`, grok's
// `[model.<name>]` block name) onto whichever config field the registry says the
// CLI's `model` launch param lives in — mirrors Session._withCustomModelLaunchModel,
// which the restart-in-place path already uses, rather than a hardcoded per-CLI
// branch here that a CLI landing its injection recipe later would silently miss.
if (qsCustomModelLaunchModel !== undefined) {
const qsCustomModelField = getCli(mode)?.launch.legacyConfigField;
if (qsCustomModelField) {
const qsSessionOptionsBag = qsSessionOptions as unknown as Record<string, unknown>;
qsSessionOptionsBag[qsCustomModelField] = {
...((qsSessionOptionsBag[qsCustomModelField] as Record<string, unknown>) ?? {}),
model: qsCustomModelLaunchModel,
};
} else {
qsSessionOptions.model = qsCustomModelLaunchModel;
}
}
const session = new Session(qsSessionOptions);
// Records the selection for session.customModel/getCustomModelForPersist() and future
// clear/switch calls — the actual env vars and launch-model config are already part of
// the launch above (constructor envOverrides, piConfig/grokConfig/ompConfig.model), so
// this is bookkeeping only, never a restart: setCustomModel() is synchronous state, no
// tmux IO of its own (see its own doc comment in session.ts).
if (qsCustomModelBookkeeping) {
session.setCustomModel(qsCustomModelBookkeeping, qsCustomModelAppliedEnvOverrides);
}
// Auto-detect completion phrase from CLAUDE.md BEFORE broadcasting
// so the initial state already has the phrase configured (only if globally enabled)
@@ -3786,6 +4036,7 @@ export function registerSessionRoutes(
sessionId: session.id,
casePath: resolvedCasePath,
caseName,
...(customModel ? { modelSwapInProgress: qsCustomModelSwapInProgress } : {}),
};
} catch (err) {
// Clean up session on error to prevent orphaned resources