Merge pull request #430 from opticon454/custom-model-run-menu

This commit is contained in:
Codeman maintainer
2026-09-19 12:25:11 +02:00
43 changed files with 7304 additions and 144 deletions
+19
View File
@@ -1720,6 +1720,25 @@ class CodemanApp {
console.error('[SSE] docker container recreated:', err);
}
});
// Custom Model Endpoint Profiles: a session's own model got evicted on llama-swap by
// another session's activity, detected AFTER the fact by a periodic server sweep (there
// is no push notification from llama-swap itself) — see detectCustomModelSwapDisplacements
// in custom-model-routes.ts. Global toast rather than a per-tab indicator: the displaced
// session need not be the one currently open, and the whole point is telling the user
// BEFORE they type into it expecting the model they picked.
addListener(SSE_EVENTS.CUSTOM_MODEL_SWAPPED_OUT, (e) => {
try {
const d = e.data ? JSON.parse(e.data) : {};
this.showToast(
`${d.sessionName || d.sessionId}'s model (${d.previousModel}) was swapped out on llama-swap by another ` +
`session — currently loaded: ${d.currentlyLoadedModel}. Sending a message there will reload it.`,
'warning',
{ duration: 0 }
);
} catch (err) {
console.error('[SSE] custom model swapped out:', err);
}
});
// Multi-user admin: live-refresh whichever admin views (panel/Users tab) are open.
addListener(SSE_EVENTS.ADMIN_USERS_CHANGED, () => {
window.codemanAdmin?.onUsersChanged?.();
+3
View File
@@ -1165,6 +1165,9 @@ const SSE_EVENTS = {
APPROVAL_UPDATED: 'approval:updated',
APPROVAL_RESOLVED: 'approval:resolved',
// Custom Model Endpoint Profiles
CUSTOM_MODEL_SWAPPED_OUT: 'custom-model:swapped-out',
// Subagents (Claude Code background agents)
SUBAGENT_DISCOVERED: 'subagent:discovered',
SUBAGENT_UPDATED: 'subagent:updated',
+28
View File
@@ -286,6 +286,34 @@
'Prompt sent': '提示已发送',
'Inserted, press Enter in the terminal to send': '已插入,在终端中按 Enter 发送',
'Could not reach the session': '无法连接到会话',
'Custom model endpoints': '自定义模型端点',
'Point a harness at your own OpenAI-compatible server (llama.cpp, vLLM, DGX Spark, Azure AI Foundry, OpenRouter) instead of its native cloud backend. When on, the Run menu offers an extra entry per harness that supports it, per saved endpoint.':
'让工具指向您自己的兼容 OpenAI 服务器(llama.cpp、vLLM、DGX Spark、Azure AI Foundry、OpenRouter),而非其原生云端后端。开启后,"运行"菜单会为每个支持此功能的工具、每个已保存的端点新增一个条目。',
'Enable custom model endpoints': '启用自定义模型端点',
'Adds a per-endpoint entry to the Run menu for every harness that can redirect to one.':
'为每个可重定向到端点的工具,在"运行"菜单中添加对应条目。',
'No endpoints yet. Add one below to point a harness at a local or cloud OpenAI-compatible server.':
'暂无端点。请在下方添加一个,以便将工具指向本地或云端的兼容 OpenAI 服务器。',
Discover: '发现模型',
'+ Add endpoint': '+ 添加端点',
'Add endpoint': '添加端点',
Id: 'ID',
'Short, stable — used in URLs, never shown to the CLI.': '简短且固定 — 用于 URL,不会展示给 CLI。',
Label: '标签',
'Base URL': '基础 URL',
'API key': 'API 密钥',
'Optional. Left blank on edit keeps the existing key.': '可选。编辑时留空将保留现有密钥。',
'Auth header': '认证请求头',
'Never send both — some servers hang indefinitely.': '切勿同时发送两者 — 部分服务器会因此无限期挂起。',
'Authorization: Bearer (default)': 'Authorization: Bearer(默认)',
'api-key header (Azure)': 'api-key 请求头(Azure)',
'Default model': '默认模型',
'What the Run-menu picker applies for this endpoint. Discover models first.':
'运行菜单选择器会为此端点应用该模型。请先发现可用模型。',
'Custom Endpoints': '自定义端点',
'Choose a model': '选择模型',
'That endpoint no longer exists': '该端点已不存在',
'No models discovered for this endpoint yet': '此端点尚未发现任何模型',
'Subagent Options': '子智能体选项',
'Enable Tracking': '启用跟踪',
'Active Tab Only': '仅活动标签页',
+123
View File
@@ -680,6 +680,14 @@
<button class="run-mode-option" data-mode="omp" onclick="app.setRunMode('omp')">
<span class="run-mode-dot omp"></span>OMP
</button>
<!-- Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md): one
generated entry per (harness, saved endpoint) pair, e.g. "Claude Code
(llama.cpp)". Built entirely by _refreshCustomModelRunOptions() — hidden
when the feature is off or no endpoint has a usable default model, never
a fixed per-harness duplicate in this markup. -->
<div class="run-mode-sep" id="runModeCustomModelSep" style="display: none;"></div>
<div class="run-mode-header" id="runModeCustomModelHeader" style="display: none;">Custom Endpoints</div>
<div class="run-mode-custom-models" id="runModeCustomModels"></div>
<div class="run-mode-sep"></div>
<button class="run-mode-option" data-mode="shell" onclick="app.setRunMode('shell')">
<span class="run-mode-dot shell"></span>Terminal / Shell
@@ -932,6 +940,67 @@
</div>
</div>
<!-- Custom Model Endpoint Profiles: "which model" picker (docs/custom-model-endpoints-plan.md).
Shown only when the chosen endpoint has more than one discovered model — see
selectCustomModelEntry() in session-ui.js, which skips straight to launch otherwise. -->
<div class="modal" id="customModelPickModal">
<div class="modal-backdrop" onclick="app.closeCustomModelPickModal()"></div>
<div class="modal-content modal-sm">
<div class="modal-header">
<h3 id="customModelPickTitle">Choose a model</h3>
<button class="modal-close" onclick="app.closeCustomModelPickModal()" aria-label="Close model picker">&times;</button>
</div>
<div class="modal-body">
<p class="form-hint" id="customModelPickHint"></p>
<div id="customModelPickList" class="run-mode-custom-models"></div>
</div>
</div>
</div>
<!-- Custom Model Endpoint Profiles: llama-swap model-swap confirmation
(docs/custom-model-endpoints-plan.md) — replaces a native confirm()
popup, shown when switching would unload a model another live
session is actively using. See _confirmModelSwap() in session-ui.js. -->
<div class="modal" id="customModelSwapConfirmModal">
<div class="modal-backdrop" onclick="app._resolveModelSwapConfirm(false)"></div>
<div class="modal-content modal-sm">
<div class="modal-header">
<h3>Switch models?</h3>
<button class="modal-close" onclick="app._resolveModelSwapConfirm(false)" aria-label="Cancel">&times;</button>
</div>
<div class="modal-body">
<p class="form-hint" id="customModelSwapConfirmMessage"></p>
</div>
<div class="modal-footer">
<button class="btn-toolbar" onclick="app._resolveModelSwapConfirm(false)">Cancel</button>
<button class="btn-toolbar btn-primary" onclick="app._resolveModelSwapConfirm(true)">Switch anyway</button>
</div>
</div>
</div>
<!-- Custom Model Endpoint Profiles: context-window-too-small warning
(docs/custom-model-endpoints-plan.md) — shown before launching a CLI
whose own fixed system-prompt/tool-schema overhead exceeds the
model's real discovered context, which guarantees a first-message
failure regardless of CLAUDE_CODE_MAX_CONTEXT_TOKENS. See
_confirmContextWarning() in session-ui.js. -->
<div class="modal" id="customModelContextWarningModal">
<div class="modal-backdrop" onclick="app._resolveContextWarningConfirm(false)"></div>
<div class="modal-content modal-sm">
<div class="modal-header">
<h3>Context window too small</h3>
<button class="modal-close" onclick="app._resolveContextWarningConfirm(false)" aria-label="Cancel">&times;</button>
</div>
<div class="modal-body">
<p class="form-hint" id="customModelContextWarningMessage" style="white-space: pre-wrap;"></p>
</div>
<div class="modal-footer">
<button class="btn-toolbar" onclick="app._resolveContextWarningConfirm(false)">Cancel</button>
<button class="btn-toolbar btn-primary" onclick="app._resolveContextWarningConfirm(true)">Launch anyway</button>
</div>
</div>
</div>
<!-- Cron Jobs Modal -->
<div class="modal" id="cronModal">
<div class="modal-backdrop" onclick="app.closeCron()"></div>
@@ -2228,6 +2297,60 @@
</div>
</div>
</div>
<div class="set-group" id="customModelEndpointsGroup">
<div class="set-group-head"><h4>Custom model endpoints</h4><span class="set-scope">synced</span></div>
<p class="set-group-hint">Point a harness at your own OpenAI-compatible server (llama.cpp, vLLM, DGX Spark, Azure AI Foundry, OpenRouter) instead of its native cloud backend. When on, the Run menu offers an extra entry per harness that supports it, per saved endpoint.</p>
<div class="set-group-body">
<div class="set-row" data-search="custom model endpoint llama.cpp local llm run menu picker">
<div class="set-row-text">
<span class="set-row-label">Enable custom model endpoints</span>
<span class="set-row-desc">Adds a per-endpoint entry to the Run menu for every harness that can redirect to one.</span>
</div>
<label class="switch switch-sm"><input type="checkbox" id="appSettingsCustomModelEndpoints" onchange="app.applyCustomModelEndpointsVisibility()"><span class="slider"></span></label>
</div>
<!-- Gated on the toggle above (applyCustomModelEndpointsVisibility): with the
feature off, a list of endpoints that do nothing is worse than nothing. -->
<div id="customModelEndpointsBody" style="display:none">
<div id="customModelHostsList" class="set-group-body" data-search="endpoints"></div>
<button type="button" class="btn-toolbar btn-sm" id="customModelHostAddBtn" onclick="app.openCustomModelHostEditor()">+ Add endpoint</button>
<div id="customModelHostEditor" class="set-inline-form" style="display:none">
<h5 id="customModelHostEditorTitle">Add endpoint</h5>
<div class="set-row has-field">
<div class="set-row-text"><span class="set-row-label">Id</span><span class="set-row-desc">Short, stable — used in URLs, never shown to the CLI.</span></div>
<input type="text" id="customModelHostId" class="set-input" placeholder="llama-cpp-local">
</div>
<div class="set-row has-field">
<div class="set-row-text"><span class="set-row-label">Label</span></div>
<input type="text" id="customModelHostLabel" class="set-input" placeholder="llama.cpp (local)">
</div>
<div class="set-row has-field">
<div class="set-row-text"><span class="set-row-label">Base URL</span></div>
<input type="text" id="customModelHostBaseUrl" class="set-input" placeholder="http://192.168.1.50:8080">
</div>
<div class="set-row has-field">
<div class="set-row-text"><span class="set-row-label">API key</span><span class="set-row-desc">Optional. Left blank on edit keeps the existing key.</span></div>
<input type="password" id="customModelHostApiKey" class="set-input" autocomplete="new-password">
</div>
<div class="set-row has-field">
<div class="set-row-text"><span class="set-row-label">Auth header</span><span class="set-row-desc">Never send both — some servers hang indefinitely.</span></div>
<select id="customModelHostAuthStyle" class="set-select">
<option value="bearer">Authorization: Bearer (default)</option>
<option value="api-key">api-key header (Azure)</option>
</select>
</div>
<div class="set-row has-field">
<div class="set-row-text"><span class="set-row-label">Default model</span><span class="set-row-desc">What the Run-menu picker applies for this endpoint. Discover models first.</span></div>
<select id="customModelHostDefaultModel" class="set-select" disabled></select>
</div>
<div class="set-row-actions">
<button type="button" class="btn-toolbar btn-sm" onclick="app.saveCustomModelHostFromEditor()">Save</button>
<button type="button" class="btn-toolbar btn-sm" onclick="app.closeCustomModelHostEditor()">Cancel</button>
</div>
</div>
</div>
</div>
</div>
</section>
<!-- ══ Agents &amp; CLIs ═════════════════════════════════════════ -->
+125 -4
View File
@@ -5497,12 +5497,25 @@ Object.assign(CodemanApp.prototype, {
return this.showToast(message, type);
},
/**
* `duration` defaults to 3000ms for every toast type. A message worth
* reading rather than glancing at (e.g. "Session started on the native
* backend — could not apply the custom endpoint: <the actual reason>")
* passes an explicit `opts.duration: 0` at its own call site instead of
* widening the default: this used to default every `error` toast to
* sticky, and with no cap on `.toast-container` and no eviction, a
* repeatedly failing path (a flapping SSE reconnect, a poll loop) stacked
* sticky toasts off the bottom of the viewport where they could not be
* read or dismissed. Every toast still gets an explicit close button
* regardless of duration.
*/
showToast(message, type = 'info', opts = {}) {
const { duration = 3000, action } = opts;
const toast = document.createElement('div');
toast.className = `toast toast-${type}`;
const msgSpan = document.createElement('span');
msgSpan.className = 'toast-message';
msgSpan.textContent = message;
toast.appendChild(msgSpan);
@@ -5514,6 +5527,20 @@ Object.assign(CodemanApp.prototype, {
toast.appendChild(btn);
}
let dismissTimer = null;
const dismiss = () => {
if (dismissTimer) clearTimeout(dismissTimer);
toast.classList.remove('show');
setTimeout(() => toast.remove(), 200);
};
const closeBtn = document.createElement('button');
closeBtn.className = 'toast-close';
closeBtn.textContent = '×';
closeBtn.setAttribute('aria-label', 'Dismiss');
closeBtn.onclick = (e) => { e.stopPropagation(); dismiss(); };
toast.appendChild(closeBtn);
// Cache toast container reference
if (!this._toastContainer) {
this._toastContainer = document.querySelector('.toast-container');
@@ -5527,10 +5554,104 @@ Object.assign(CodemanApp.prototype, {
requestAnimationFrame(() => toast.classList.add('show'));
setTimeout(() => {
toast.classList.remove('show');
setTimeout(() => toast.remove(), 200);
}, duration);
if (duration > 0) {
dismissTimer = setTimeout(dismiss, duration);
}
// Most callers ignore this — a handle exists for a long-running toast a caller needs
// to update or dismiss itself once its own condition resolves (e.g. a "loading model"
// toast a poll loop dismisses once the model reports ready).
return { dismiss, setMessage: (text) => { msgSpan.textContent = text; } };
},
/**
* A prominent, screen-centred status banner — for the small set of messages that are
* genuinely worth interrupting the eye for rather than living in the corner with every
* other toast (currently: a custom-model session's "switching backends" and "loading
* model" states, both of which can sit on screen for well over a minute and are easy to
* mistake for nothing happening). Non-blocking (`pointer-events: none` on the wrapper,
* restored only on the card) — an info banner is never a gate the user has to dismiss to
* keep working. Only one is ever shown at a time (the DOM node is created once and
* reused), which matches every current caller: each hands off to the next rather than
* stacking.
*
* `opts.type` — `'info'` (default, spinner, no close button — a caller ends it itself via
* `dismiss()`) or `'error'` (no spinner — nothing is in progress once this shows — with a
* close button, since a sticky error the user cannot dismiss would just sit there). The
* DOM is rebuilt fresh each call rather than patched, since which children exist differs
* by type; `setMessage` still only ever touches the text node afterwards.
*
* `opts.onCancel` — when given (any type, but in practice only 'info': an 'error' banner
* already has its own close button), renders a "Cancel" button that calls it on click.
* The callback owns everything that follows (dismissing the banner, stopping whatever
* loop this was showing progress for, closing a session it was for) — this helper only
* renders the button and wires the click, the same "caller decides what cancel means"
* split as `_confirmModelSwap`'s promise-resolving buttons.
*/
_showCenterStatus(message, opts = {}) {
const { type = 'info', onCancel } = opts;
let el = document.getElementById('customModelCenterStatus');
if (!el) {
el = document.createElement('div');
el.id = 'customModelCenterStatus';
document.body.appendChild(el);
}
// A pending hide from a PREVIOUS dismiss() (e.g. switchingToast.dismiss() right
// before this same-origin call reopens the banner within its 200ms fade) must
// never fire against the node this call is about to show — clear it before
// reusing the shared DOM node, or the old timer hides the fresh banner ~200ms in.
if (el._hideTimer) {
clearTimeout(el._hideTimer);
el._hideTimer = null;
}
el.className = `center-status-banner center-status-${type}`;
el.innerHTML = '';
const dismiss = () => {
el.classList.remove('show');
el._hideTimer = setTimeout(() => {
el.hidden = true;
el._hideTimer = null;
}, 200);
};
if (type !== 'error') {
const spinner = document.createElement('span');
spinner.className = 'center-status-spinner';
spinner.setAttribute('aria-hidden', 'true');
el.appendChild(spinner);
}
const text = document.createElement('span');
text.className = 'center-status-text';
text.textContent = message;
el.appendChild(text);
if (type === 'error') {
const closeBtn = document.createElement('button');
closeBtn.className = 'center-status-close';
closeBtn.textContent = '×';
closeBtn.setAttribute('aria-label', 'Dismiss');
closeBtn.onclick = (e) => {
e.stopPropagation();
dismiss();
};
el.appendChild(closeBtn);
} else if (onCancel) {
const cancelBtn = document.createElement('button');
cancelBtn.className = 'center-status-cancel';
cancelBtn.textContent = 'Cancel';
cancelBtn.onclick = (e) => {
e.stopPropagation();
onCancel();
};
el.appendChild(cancelBtn);
}
el.hidden = false;
requestAnimationFrame(() => el.classList.add('show'));
return {
dismiss,
setMessage: (next) => {
const t = el.querySelector('.center-status-text');
if (t) t.textContent = next;
},
};
},
+604 -6
View File
@@ -461,6 +461,7 @@ Object.assign(CodemanApp.prototype, {
if (menu.classList.contains('active')) {
this._loadRunModeHistory();
this._refreshRunModeAvailability(menu);
this._refreshCustomModelRunOptions(menu);
const close = (ev) => {
if (!menu.contains(ev.target)) {
menu.classList.remove('active');
@@ -534,6 +535,587 @@ Object.assign(CodemanApp.prototype, {
if (dsWeb) dsWeb.style.display = avail.deepseekBinary ? 'flex' : 'none';
},
/**
* Generates the Run menu's Custom Model Endpoint entries
* (docs/custom-model-endpoints-plan.md): one button per (capable harness, saved
* endpoint) pair, e.g. "Claude Code (llama.cpp)". Hidden entirely when the
* feature is off, no endpoint has a usable default model, or the active case is
* remote/docker (the apply route refuses both — see session-routes.ts).
*
* `window.__codemanCustomModelClis` is server-injected at render time from the
* CLI registry's own `capabilities.customModelInjection` (never a hardcoded id
* list here), so a CLI gaining or losing the capability shows up with no
* frontend change.
*/
async _refreshCustomModelRunOptions(menu) {
const sep = menu.querySelector('#runModeCustomModelSep');
const header = menu.querySelector('#runModeCustomModelHeader');
const container = menu.querySelector('#runModeCustomModels');
if (!container) return;
const hide = () => {
if (sep) sep.style.display = 'none';
if (header) header.style.display = 'none';
container.innerHTML = '';
};
const settings = this.loadAppSettingsFromStorage();
// Matches _refreshRunModeAvailability's own gate: a stock entry for an
// uninstalled CLI is hidden, so a generated one must be too, or a box with
// no codex still offers "Codex (llama.cpp)" and fails at launch.
const capableClis = (window.__codemanCustomModelClis || []).filter((cli) => this.isCliAvailable(cli.id));
if (!settings.customModelEndpointsEnabled || capableClis.length === 0) return hide();
const caseName = document.getElementById('quickStartCase')?.value;
const activeCase = caseName ? (this.cases || []).find((c) => c.name === caseName) : null;
if (activeCase?.location === 'remote' || activeCase?.location === 'docker') return hide();
// GET /api/model-endpoints wraps its body in the { success, data } envelope
// like every other /api route (server.ts's preSerialization hook applies to
// arrays too) — _apiJson() unwraps it. A raw fetch().json() here would
// silently see the envelope object instead of the array and hide this
// section unconditionally.
const hosts = await this._apiJson('/api/model-endpoints');
if (!Array.isArray(hosts) || hosts.length === 0) return hide();
const rows = [];
for (const host of hosts) {
const models = host.models || [];
if (models.length === 0) continue; // nothing discovered yet — the settings panel explains why
const modelId = host.defaultModelId || models[0];
for (const cli of capableClis) {
// escapeHtml(JSON.stringify(...)) on EVERY arg, not just the untrusted
// one: JSON.stringify's own double quotes would otherwise terminate this
// double-quoted attribute at the first one, and everything after parses
// as raw tag content rather than a quoted string — which is what turns
// modelId (server-controlled, from the endpoint's own /v1/models reply,
// not this box's) into markup instead of inert data. Same idiom as
// deleteCase's onclick a few hundred lines down.
const args = [cli.id, host.id].map((v) => escapeHtml(JSON.stringify(v))).join(', ');
rows.push(`
<button class="run-mode-option" data-mode="${escapeHtml(cli.id)}" data-endpoint="${escapeHtml(host.id)}"
onclick="app.selectCustomModelEntry(${args})"
title="${escapeHtml(cli.label)} → ${escapeHtml(host.baseUrl)} (${escapeHtml(modelId)}${models.length > 1 ? `, +${models.length - 1} more` : ''})">
<span class="run-mode-dot ${escapeHtml(cli.id)}"></span>${escapeHtml(cli.label)} (${escapeHtml(host.label)})
</button>`);
}
}
if (rows.length === 0) return hide();
if (sep) sep.style.display = '';
if (header) header.style.display = '';
container.innerHTML = rows.join('');
},
/**
* Decides whether picking a Run-menu Custom Endpoint entry can launch
* straight away or needs to ask which model first. Re-fetches the endpoint
* rather than trusting anything cached from the menu render: the models
* list (or the default) could have changed — a re-discovery cycle running
* every 5 minutes in the background, or an edit in the settings panel —
* between opening the dropdown and clicking a row.
*/
async selectCustomModelEntry(mode, endpointId) {
document.getElementById('runModeMenu')?.classList.remove('active');
const hosts = await this._apiJson('/api/model-endpoints');
const host = (hosts || []).find((h) => h.id === endpointId);
if (!host) {
this.showToast('That endpoint no longer exists', 'error');
return;
}
const models = host.models || [];
if (models.length === 0) {
this.showToast('No models discovered for this endpoint yet', 'warning');
return;
}
// Exactly one model: nothing to choose, so asking would just be an extra
// click for the same answer every time. Two or more: always ask, even
// with a defaultModelId set — the point of asking is letting THIS launch
// differ from the default, not just confirming it.
if (models.length === 1) {
return this.runCustomModelEntry(mode, endpointId, models[0]);
}
this._openCustomModelPickModal(mode, host);
},
/** Renders the "which model" picker for a (harness, endpoint) pair with more than one discovered model. */
_openCustomModelPickModal(mode, host) {
const modal = document.getElementById('customModelPickModal');
const list = document.getElementById('customModelPickList');
if (!modal || !list) return;
this._pendingCustomModelPick = { mode, endpointId: host.id };
const cliLabel = (window.__codemanCustomModelClis || []).find((c) => c.id === mode)?.label || mode;
// A static title (translatable by i18n.js's exact-string walker) plus a
// dynamic hint carrying the specifics — same split webviewModalTitle uses,
// since the walker cannot i18n a string a variable is already spliced into.
document.getElementById('customModelPickTitle').textContent = 'Choose a model';
document.getElementById('customModelPickHint').textContent =
`${cliLabel} → ${host.label} — ${(host.models || []).length} models discovered.`;
list.innerHTML = (host.models || [])
.map((m) => {
const isDefault = m === host.defaultModelId;
const arg = escapeHtml(JSON.stringify(m));
return `
<button class="run-mode-option" onclick="app.chooseCustomModelAndRun(${arg})">
<span class="run-mode-dot ${escapeHtml(mode)}"></span>${escapeHtml(m)}${isDefault ? ' <span class="set-scope">Default</span>' : ''}
</button>`;
})
.join('');
modal.classList.add('active');
},
closeCustomModelPickModal() {
document.getElementById('customModelPickModal')?.classList.remove('active');
this._pendingCustomModelPick = null;
},
/**
* In-app replacement for a native `confirm()` popup, used specifically for the
* llama-swap "this will unload it for session X" warning (both launch paths below) —
* a browser-chrome dialog there looked out of place next to the rest of the app's own
* modals. Resolves true/false the same way `confirm()` would; `_resolveModelSwapConfirm`
* (the modal's own Cancel/Switch-anyway buttons, and its backdrop click) is what settles
* the returned promise.
*/
_confirmModelSwap(message) {
const modal = document.getElementById('customModelSwapConfirmModal');
const messageEl = document.getElementById('customModelSwapConfirmMessage');
if (messageEl) messageEl.textContent = message;
modal?.classList.add('active');
return new Promise((resolve) => {
this._resolveModelSwapConfirmPromise = resolve;
});
},
/** Called by the modal's Cancel/Switch-anyway buttons and its backdrop click. */
_resolveModelSwapConfirm(proceed) {
document.getElementById('customModelSwapConfirmModal')?.classList.remove('active');
const resolve = this._resolveModelSwapConfirmPromise;
this._resolveModelSwapConfirmPromise = null;
resolve?.(proceed);
},
/**
* In-app warning shown when the apply route reports `requiresContextWarning`: this
* model's real discovered context is smaller than the CLI's own fixed system-prompt/
* tool-schema overhead, which guarantees the very first message fails outright — no
* `CLAUDE_CODE_MAX_CONTEXT_TOKENS` value fixes that, since there is no conversation
* history yet for compaction to trim. Same promise-based pattern as
* `_confirmModelSwap`; `_resolveContextWarningConfirm` settles it.
*/
_confirmContextWarning(modelId, contextLength, minSafeContextTokens) {
const modal = document.getElementById('customModelContextWarningModal');
const messageEl = document.getElementById('customModelContextWarningMessage');
if (messageEl) {
const known = typeof contextLength === 'number';
messageEl.textContent =
`${modelId} is configured with ` +
(known ? `only ${contextLength.toLocaleString()} tokens of` : 'an unknown (too small)') +
` context, but this CLI needs roughly ${minSafeContextTokens.toLocaleString()}+ tokens just for its own ` +
`system prompt and tools — before any conversation history. Its very first message will fail outright, ` +
`no matter what context size Codeman tells it to expect.\n\n` +
`To fix this, reconfigure llama-swap to give this model (or a smaller one) an explicit larger context ` +
`instead of relying on auto-fit (--fit-ctx), which optimizes for the biggest MODEL that fits, not the ` +
`biggest CONTEXT — e.g. add "-c 65536" (or as large a --ctx-size as your hardware holds) to its llama-swap ` +
`config entry. A smaller model at a much larger explicit context often fits in the same VRAM a bigger ` +
`model's auto-fit context gets shrunk to make room for.`;
}
modal?.classList.add('active');
return new Promise((resolve) => {
this._resolveContextWarningConfirmPromise = resolve;
});
},
/** Called by the modal's Cancel/Launch-anyway buttons and its backdrop click. */
_resolveContextWarningConfirm(proceed) {
document.getElementById('customModelContextWarningModal')?.classList.remove('active');
const resolve = this._resolveContextWarningConfirmPromise;
this._resolveContextWarningConfirmPromise = null;
resolve?.(proceed);
},
/** A model row in the picker modal was clicked: close it and launch with that choice. */
chooseCustomModelAndRun(modelId) {
const pending = this._pendingCustomModelPick;
this.closeCustomModelPickModal();
if (!pending) return; // modal reopened/closed from elsewhere between render and click
void this.runCustomModelEntry(pending.mode, pending.endpointId, modelId);
},
/**
* Runs a session on `mode` and immediately applies `endpointId`/`modelId` to it
* via POST /api/sessions/:id/custom-model (see session-routes.ts) — the same
* restart-in-place apply path the (not-yet-built) endpoint-management surface
* would use for an already-running session. A custom-model run is a one-off
* "try this endpoint" action, not a sticky mode.
*
* Routes through run() itself, via a temporary `_runMode` swap, rather than a
* parallel dispatch table: that is what gives this the same in-flight lock
* every other Run click gets (CLAUDE.md, Run launch synchronization — the lock
* exists so a double click cannot create duplicate sessions with the same
* `w<n>-<case>` name, and it guards the OTHER direction too: without it, the
* main Run button could start a second concurrent launch while this one was
* still resolving), and it means a CLI whose customModelInjection recipe
* lands later needs no update here, only in run()'s own dispatch. The swap
* never persists — setRunMode() would sync it to the server as the user's new
* default, which a one-off endpoint run must not do — and is restored in
* `finally` even if run() throws.
*/
/**
* Dispatches to the ONE-SHOT launch path (below) for every custom-model-eligible CLI
* except claude, which still goes through the restart-after-native-boot path
* (`_runCustomModelEntryViaRestart`): claude's own `runClaude()` carries multi-tab
* launch and a docker-config-drift confirm/retry loop neither of the other seven
* functions has, and folding those into the one-shot flow is unstarted, separate work.
* The other seven (opencode/codex/gemini/pi/grok/deepseek/omp) are each a single,
* simple launch, so they get the one-shot path — the one visibly worth it, since a
* native-boot-then-restart is far more jarring on a CLI whose TUI fully reinitializes
* (Codex, confirmed live) than on claude's own `--resume`-based restart.
*/
async runCustomModelEntry(mode, endpointId, modelId) {
if (mode === 'claude') {
return this._runCustomModelEntryViaRestart(mode, endpointId, modelId);
}
return this._runCustomModelEntryOneShot(mode, endpointId, modelId);
},
/**
* Launches directly on the endpoint — no restart, so no visible relaunch. Stashes the
* pick on `_pendingCustomModelForLaunch` for the targeted run<Mode>() function to read
* and fold into its own /api/quick-start body (see `_quickStartWithCustomModelConfirm`);
* cleared in `finally` the same way `_runMode`'s temporary swap is, even if run() throws.
*/
async _runCustomModelEntryOneShot(mode, endpointId, modelId) {
document.getElementById('runModeMenu')?.classList.remove('active');
const previousRunMode = this._runMode;
const tabCountEl = document.getElementById('tabCount');
const prevTabCount = tabCountEl?.value;
this._runMode = mode;
this._pendingCustomModelForLaunch = { endpointId, modelId };
if (tabCountEl) tabCountEl.value = '1';
try {
await this.run();
} finally {
this._runMode = previousRunMode;
this._pendingCustomModelForLaunch = undefined;
if (tabCountEl && prevTabCount !== undefined) tabCountEl.value = prevTabCount;
}
// run() (via _quickStartWithCustomModelConfirm) reports its own launch error or
// cancellation via toast and leaves this unset — nothing more to do here then.
const result = this._lastCustomModelLaunchResult;
this._lastCustomModelLaunchResult = undefined;
if (result?.modelSwapInProgress) {
void this._watchLlamaSwapLoading(endpointId, modelId, result.sessionId);
}
},
/**
* POSTs a /api/quick-start body already carrying `customModel` (see the run<Mode>()
* call sites below), showing the same llama-swap "this will unload it for session X"
* warning the restart path's `_applyCustomModelToSession` shows when the route asks
* for confirmation, and retrying with `confirmed: true` on accept. Stashes the final
* response's payload on `_lastCustomModelLaunchResult` for
* `_runCustomModelEntryOneShot` to read `modelSwapInProgress` off afterward — run()'s
* eleven per-mode dispatch targets have no shared return-value contract of their own,
* so a side channel here is simpler than threading one through every one of them.
*/
async _quickStartWithCustomModelConfirm(bodyObj) {
const post = async (body) => {
const res = await fetch('/api/quick-start', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(body),
});
return res.json();
};
let data = await post(bodyObj);
if (data?.data?.requiresContextWarning) {
const { modelId, contextLength, minSafeContextTokens } = data.data;
const proceed = await this._confirmContextWarning(modelId, contextLength, minSafeContextTokens);
if (!proceed) {
this._lastCustomModelLaunchResult = undefined;
return { success: false, error: 'Launch cancelled — context window too small' };
}
data = await post({ ...bodyObj, customModel: { ...bodyObj.customModel, confirmed: true } });
}
if (data?.data?.requiresConfirmation) {
const { currentlyLoadedModel, affectedSessions } = data.data;
const names = affectedSessions.map((s) => s.name || s.id).join(', ');
const proceed = await this._confirmModelSwap(
`${names} ${affectedSessions.length === 1 ? 'is' : 'are'} currently using ` +
`${currentlyLoadedModel} on this endpoint. Switching will unload it for ` +
`${affectedSessions.length === 1 ? 'that session' : 'those sessions'} too. Continue?`
);
if (!proceed) {
this._lastCustomModelLaunchResult = undefined;
return { success: false, error: 'Model switch cancelled' };
}
data = await post({ ...bodyObj, customModel: { ...bodyObj.customModel, confirmed: true } });
}
this._lastCustomModelLaunchResult = data?.success !== false ? data?.data : undefined;
return data;
},
/** The restart-after-native-boot path — see `runCustomModelEntry`'s own comment for
* which CLIs still use this one. */
async _runCustomModelEntryViaRestart(mode, endpointId, modelId) {
document.getElementById('runModeMenu')?.classList.remove('active');
const previousRunMode = this._runMode;
const before = this.activeSessionId;
const tabCountEl = document.getElementById('tabCount');
const prevTabCount = tabCountEl?.value;
this._runMode = mode;
if (tabCountEl) tabCountEl.value = '1';
try {
await this.run();
} finally {
this._runMode = previousRunMode;
if (tabCountEl && prevTabCount !== undefined) tabCountEl.value = prevTabCount;
}
// run() reports its own errors via toast. Every run*() function handles its
// own failure internally and returns normally rather than throwing or
// leaving activeSessionId null, so a declined/failed launch (missing CLI, a
// caught exception, isBusy on the session the launch would have targeted)
// falls through to here with the PREVIOUSLY active session still active.
// Requiring the id to have actually changed — not just to be non-null — is
// what stops that case from silently re-pointing and restarting whatever
// session the user was already looking at.
const sessionId = this.activeSessionId;
if (!sessionId || sessionId === before) return;
// Claude just launched on the NATIVE backend and is about to be restarted onto
// the endpoint — without something saying so, that native boot (which can talk
// to Opus for a moment) reads as "the endpoint didn't apply" rather than "the
// switch hasn't happened yet". Prominent and screen-centred (not a corner toast)
// since this can sit on screen for a while; sticky until the apply below settles
// one way or the other, or hands off to _watchLlamaSwapLoading's own banner.
const switchingToast = this._showCenterStatus(`Claude started — switching to ${endpointId}…`);
// A freshly launched CLI reports its OWN startup as 'busy' (spinner, the
// workspace-trust check, whatever else it does before its first prompt) —
// measured landing well before this line reliably reaches it — and the
// apply route's isBusy() guard correctly refuses to restart a session
// mid-turn, "mid-turn" included, which this fresh boot looks exactly
// like from the outside. Give it a bounded chance to settle first rather
// than raising a false "Session is busy" on every single launch. Per the
// wait contract a timeout here is a normal 200, never an error — a
// session still busy after 20s just reaches the apply call below and
// gets the route's own honest, now-visible SESSION_BUSY error instead of
// this guessing about it.
await this._apiJson(`/api/sessions/${sessionId}/wait?until=idle&timeout=20000`);
// _apiJson() (used everywhere else in this file) unwraps a success body to
// its `data`, but on failure it swallows the response entirely and returns
// null — exactly the `error` text a caller needs to tell "the endpoint is
// unreachable" apart from "the CLI can't be redirected", "not one of the
// discovered models", or "this is a Docker/remote session". Go through the
// raw response here instead so a failure is diagnosable, not just present.
let { ok, data, res } = await this._applyCustomModelToSession(sessionId, endpointId, modelId);
// A success body comes back as {success:true, data:{...}} (server.ts's preSerialization
// envelope), but a route-level error is {success:false, error, errorCode} with no nested
// data — createErrorResponse() never wraps one. `payload` below is only ever meaningful
// once `data.success !== false`.
let payload = data?.success !== false ? data?.data : undefined;
// This CLI's own fixed overhead (system prompt + tool schemas) may exceed the
// model's real discovered context outright — no context-length declaration can
// fix that, since compaction only trims conversation history and there is none
// on message 1. Warn and let the user decide whether to launch anyway, same
// confirmed:true re-send pattern as the swap check below.
if (ok && payload?.requiresContextWarning) {
const proceed = await this._confirmContextWarning(
payload.modelId,
payload.contextLength,
payload.minSafeContextTokens
);
if (!proceed) {
switchingToast?.dismiss();
this.showToast('Kept the native backend — context window too small', 'info');
return;
}
({ ok, data, res } = await this._applyCustomModelToSession(sessionId, endpointId, modelId, true));
payload = data?.success !== false ? data?.data : undefined;
}
// llama-swap runs one model at a time: switching would unload it out from under
// another session actively using it. The route only asks when that's actually true
// (never just because a swap is needed at all) — confirming re-sends the exact same
// call with `confirmed: true` so the route skips the check the second time.
if (ok && payload?.requiresConfirmation) {
const names = payload.affectedSessions.map((s) => s.name || s.id).join(', ');
const proceed = await this._confirmModelSwap(
`${names} ${payload.affectedSessions.length === 1 ? 'is' : 'are'} currently using ` +
`${payload.currentlyLoadedModel} on this endpoint. Switching to ${modelId} will unload it ` +
`for ${payload.affectedSessions.length === 1 ? 'that session' : 'those sessions'} too. Continue?`
);
if (!proceed) {
switchingToast?.dismiss();
this.showToast('Kept the native backend — model switch cancelled', 'info');
return;
}
({ ok, data, res } = await this._applyCustomModelToSession(sessionId, endpointId, modelId, true));
payload = data?.success !== false ? data?.data : undefined;
}
if (!ok || !data || data.success === false) {
switchingToast?.dismiss();
const detail = data?.error ? `: ${data.error}` : res ? ` (HTTP ${res.status})` : ' (request failed)';
this.showToast(`Session started on the native backend — could not apply the custom endpoint${detail}`, 'error', {
duration: 0,
});
return;
}
// The apply above already succeeded — the session IS pointed at the endpoint — but
// llama-swap itself may still be unloading the old model and loading this one, which
// can take well over a minute. Without this, a prompt sent during that window either
// hangs silently or (the bug this whole feature exists to fix) gets answered by
// whatever was loaded a moment ago, reading as "it's still using the wrong model."
// Hand off to its own sticky toast rather than stacking a second one on top.
if (payload?.modelSwapInProgress) {
switchingToast?.dismiss();
void this._watchLlamaSwapLoading(endpointId, modelId, sessionId);
return;
}
switchingToast?.setMessage(`Pointed at ${endpointId} — restarting the session...`);
setTimeout(() => switchingToast?.dismiss(), 3000);
},
/** POST /api/sessions/:id/custom-model, returning {ok, data, res} rather than throwing —
* see runCustomModelEntry's own comment for why this goes through `_api()` (raw fetch)
* rather than `_apiJson()`: a failure's `error` detail must survive to the caller. */
async _applyCustomModelToSession(sessionId, endpointId, modelId, confirmed) {
const res = await this._api(`/api/sessions/${sessionId}/custom-model`, {
method: 'POST',
body: confirmed ? { endpointId, modelId, confirmed } : { endpointId, modelId },
});
const data = res ? await res.json().catch(() => null) : null;
return { ok: !!res, data, res };
},
/**
* Best-effort: looks up `modelId`'s discovered file size (GB) off the endpoint's own
* saved host record (`CustomModelHost.modelSizesGB`, populated during discovery by
* parsing llama-swap's own `description` field for an auto-discovered model). Returns
* `undefined` for a hand-configured profile with no parseable size, an unreachable
* server, or any other failure — never a guess.
*/
async _lookupModelSizeGB(endpointId, modelId) {
const hosts = await this._apiJson('/api/model-endpoints').catch(() => null);
if (!Array.isArray(hosts)) return undefined;
const host = hosts.find((h) => h.id === endpointId);
const size = host?.modelSizesGB?.[modelId];
return typeof size === 'number' && Number.isFinite(size) && size > 0 ? size : undefined;
},
/**
* Strips llama.cpp's own bootlog prefix (`<uptime> <I|W|E> <component> `, e.g.
* `0.31.428.568 I srv llama_server: model loaded`) for display, leaving just
* `llama_server: model loaded` — the raw line from the server is kept as-is
* (`GET .../running-status`'s `logLine` field), this trims it only for the loading
* banner's second line. Defensive: a line that doesn't match this shape (a different
* llama.cpp build, or llama-swap's own format changing) is shown verbatim rather than
* mangled or dropped.
*/
_formatLlamaLogLine(line) {
return typeof line === 'string' ? line.replace(/^[\d.]+\s+[IWE]\s+\S+\s+/, '') : line;
},
/**
* Polls llama-swap's own `/running` (via the read-only running-status route) until
* `modelId` reports `state: 'ready'`, showing a sticky banner the whole time so a slow
* unload/reload (measured well over a minute for a large model) reads as "loading,
* still working on it", never as silence or a wrong answer from whatever was loaded
* before. Checks immediately (a fast load, or a re-apply onto an already-ready model,
* shouldn't wait a full interval to say so), then every `pollIntervalMs`.
*
* Deliberately UNBOUNDED — no estimate, no countdown, no automatic give-up. An earlier
* version scaled a timeout off the model's discovered file size and auto-closed the
* session when it elapsed, but a real load's actual duration depends on hardware this
* feature has no way to know (VRAM, storage speed, what else is contending for the
* GPU), so any fixed number was a guess dressed up as a fact — the banner now says so
* outright instead of pretending to a precision it doesn't have, and a Cancel button on
* the banner itself (`_showCenterStatus`'s `onCancel`) is how the user ends it if it's
* taking too long, closing `sessionId` the same way the old timeout used to.
*
* `_watchLlamaSwapGeneration` guards against two overlapping calls (a second launch
* started before the first one's loop finished) clobbering each other's banner:
* `_showCenterStatus` reuses one shared DOM node, so an older loop's `dismiss()`/message
* update firing after a newer one has already taken over the banner would otherwise hide
* or overwrite the WRONG one, or close the WRONG session. Each call claims the counter
* as its own "generation" and checks it still owns it before touching either.
*
* `pollIntervalMs` exists to let a test drive this in milliseconds instead of seconds —
* real callers never pass it.
*/
async _watchLlamaSwapLoading(endpointId, modelId, sessionId, pollIntervalMs = 1000) {
const generation = (this._watchLlamaSwapGeneration = (this._watchLlamaSwapGeneration || 0) + 1);
const isCurrent = () => this._watchLlamaSwapGeneration === generation;
const sizeGB = await this._lookupModelSizeGB(endpointId, modelId);
if (!isCurrent()) return; // a newer launch already took over before the lookup even finished
const sizeSuffix = sizeGB ? ` (${sizeGB.toFixed(1)} GB)` : '';
const baseMessage =
`Loading ${modelId}${sizeSuffix} on ${endpointId} — this can take a while depending on ` +
`your hardware and the model size.`;
// Second line, when llama-swap's own event feed actually gives us one: the real
// backend llama-server process's own latest log line (load_model:/llama_server: ...,
// see getLatestLlamaSwapLogLine) — a bare "please wait" says nothing is broken, this
// says what's actually happening. Absent on the very first render (no poll has
// landed yet) and whenever the endpoint doesn't expose it at all — never fabricated,
// and never cleared back to blank once seen (stays on the last real thing llama.cpp
// said if a later poll comes back with nothing new).
const buildMessage = (logLine) => {
const line = this._formatLlamaLogLine(logLine);
return baseMessage + (line ? `\nllama.cpp: ${line}` : '');
};
let cancelled = false;
// Prominent and screen-centred, not a corner toast — a real llama-swap model load can
// sit on screen for well over a minute, easy to mistake for nothing happening there.
const toast = this._showCenterStatus(buildMessage(), {
onCancel: () => {
cancelled = true;
},
});
while (!cancelled) {
const status = await this._apiJson(`/api/model-endpoints/${encodeURIComponent(endpointId)}/running-status`);
if (!isCurrent()) return; // a newer launch took over the banner — this loop is done
if (cancelled) break;
if (!status) {
// transient failure — keep waiting rather than giving up early
} else if (!status.isLlamaSwap) {
// Endpoint changed under us, or wasn't llama-swap after all — nothing more to
// watch for, and not a failure worth a toast of its own.
toast?.dismiss();
return;
} else if (status.running.some((r) => r.model === modelId && r.state === 'ready')) {
toast?.dismiss();
this.showToast(`${modelId} is ready`, 'success', { duration: 2500 });
return;
}
if (!isCurrent() || cancelled) break;
toast?.setMessage(buildMessage(status?.logLine));
await new Promise((resolve) => setTimeout(resolve, pollIntervalMs));
}
if (!isCurrent()) return;
// Cancelled by the user, not a timeout — an ordinary info toast, not a scary error
// banner, since this was deliberate rather than something going wrong.
toast?.dismiss();
this.showToast(
`Cancelled loading ${modelId} on ${endpointId}` + (sessionId ? ' — the session has been closed.' : '.'),
'info'
);
if (sessionId) {
try {
await this.closeSession(sessionId);
} catch {
// closeSession already reports its own failure via toast — nothing more to do here
}
}
},
/**
* Start the DeepSeek Harness browser UI and open it as a Codeman web tab.
*
@@ -1279,15 +1861,24 @@ Object.assign(CodemanApp.prototype, {
async _launchQuickStartInstances(caseName, tabCount, label, buildBody, ownsLaunchTerminal) {
const startNumber = this._nextCaseSessionStartNumber(caseName);
let firstSessionId = null;
// A custom-model launch can ask up to two questions before it starts anything: the
// endpoint's context window is too small for this model, and loading it will unload
// the model another session is using. Both are decisions about the ENDPOINT, not
// about each session, and every instance in this batch targets the same one, so the
// answer is taken once and carried to the rest. Without this a 20-instance launch
// asks the same question 20 times, which is the interaction between the Instance
// count stepper and the custom-model picker that neither feature had on its own.
let customModelAnswered = false;
for (let i = 0; i < tabCount; i++) {
const sessionName = `w${startNumber + i}-${caseName}`;
const res = await fetch('/api/quick-start', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(buildBody(sessionName)),
});
const data = await res.json();
const body = buildBody(sessionName);
const data = await this._quickStartWithCustomModelConfirm(
customModelAnswered && body.customModel
? { ...body, customModel: { ...body.customModel, confirmed: true } }
: body
);
if (!data.success) throw new Error(data.error || `Failed to start ${label}`);
customModelAnswered = true;
await this._ensureCreatedSessionVisible(data.data.sessionId, data.data.session);
if (!firstSessionId) firstSessionId = data.data.sessionId;
}
@@ -1339,6 +1930,7 @@ Object.assign(CodemanApp.prototype, {
...(isRemote ? {} : {
openCodeConfig: { autoAllowTools: true },
...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}),
...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}),
}),
}),
ownsLaunchTerminal
@@ -1399,6 +1991,7 @@ Object.assign(CodemanApp.prototype, {
renderMode: 'hybrid',
},
...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}),
...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}),
}),
}),
ownsLaunchTerminal
@@ -1454,6 +2047,7 @@ Object.assign(CodemanApp.prototype, {
...(isRemote ? {} : {
geminiConfig: { approvalMode: 'yolo' },
...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}),
...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}),
}),
}),
ownsLaunchTerminal
@@ -1567,6 +2161,7 @@ Object.assign(CodemanApp.prototype, {
mode: 'pi',
sessionName,
...(isRemote || Object.keys(envOverrides).length === 0 ? {} : { envOverrides }),
...(!isRemote && this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}),
}),
ownsLaunchTerminal
);
@@ -1618,6 +2213,7 @@ Object.assign(CodemanApp.prototype, {
sessionName,
...(isRemote ? {} : {
...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}),
...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}),
}),
}),
ownsLaunchTerminal
@@ -1680,6 +2276,7 @@ Object.assign(CodemanApp.prototype, {
...(isRemote ? {} : {
grokConfig: { alwaysApprove: true },
...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}),
...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}),
}),
}),
ownsLaunchTerminal
@@ -1760,6 +2357,7 @@ Object.assign(CodemanApp.prototype, {
...(isRemote ? {} : {
deepSeekConfig: { permissionMode: 'danger-full-access' },
...(Object.keys(envOverrides).length > 0 ? { envOverrides } : {}),
...(this._pendingCustomModelForLaunch ? { customModel: this._pendingCustomModelForLaunch } : {}),
}),
}),
ownsLaunchTerminal
+226
View File
@@ -395,6 +395,13 @@ Object.assign(CodemanApp.prototype, {
document.getElementById('appSettingsShowUltracodeAgents').checked = settings.showUltracodeAgents ?? defaults.showUltracodeAgents ?? false;
// Approvals Inbox: synced, default OFF (opt-in; only an explicit true enables).
document.getElementById('appSettingsApprovalsInbox').checked = settings.approvalsInboxEnabled === true;
// Custom Model Endpoint Profiles: synced, default OFF. The toggle governs both
// the Run-menu picker's generated entries and this settings panel's visibility;
// the endpoint list itself is server state, loaded on demand below.
document.getElementById('appSettingsCustomModelEndpoints').checked = settings.customModelEndpointsEnabled === true;
// Assigning .checked above does not fire onchange, so the body's visibility
// (and its lazy load) needs an explicit sync on every open, not just a save.
this.applyCustomModelEndpointsVisibility();
// Read My Mind: synced, default OFF (opt-in; capture + prediction cost real tokens).
document.getElementById('appSettingsReadMyMind').checked = settings.readMyMindEnabled === true;
document.getElementById('appSettingsUltracodeFloatingWindows').checked =
@@ -509,6 +516,9 @@ Object.assign(CodemanApp.prototype, {
document.getElementById('appSettingsNiceValue').value = niceSettings.niceValue ?? 10;
// Model configuration (loaded from server)
this.loadModelConfigForSettings();
// Custom Model Endpoint Profiles' own load is gated on the toggle above (see
// applyCustomModelEndpointsVisibility) — unlike model config, this GET is
// pointless work with the feature off, so it is not fired unconditionally.
// Notification settings
const notifPrefs = this.notificationManager?.preferences || {};
document.getElementById('appSettingsNotifEnabled').checked = notifPrefs.enabled ?? true;
@@ -2106,6 +2116,7 @@ Object.assign(CodemanApp.prototype, {
showSubagents: document.getElementById('appSettingsShowSubagents').checked,
showUltracodeAgents: document.getElementById('appSettingsShowUltracodeAgents').checked,
approvalsInboxEnabled: document.getElementById('appSettingsApprovalsInbox').checked,
customModelEndpointsEnabled: document.getElementById('appSettingsCustomModelEndpoints').checked,
readMyMindEnabled: document.getElementById('appSettingsReadMyMind').checked,
ultracodeFloatingWindows: document.getElementById('appSettingsUltracodeFloatingWindows').checked,
showMultiMonitorButton: document.getElementById('appSettingsShowMultiMonitorButton').checked,
@@ -2487,6 +2498,209 @@ Object.assign(CodemanApp.prototype, {
}
},
// ═══════════════════════════════════════════════════════════════
// Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md)
//
// CRUD against /api/model-endpoints, rendered into the Models settings section.
// Deliberately its own load/save pair rather than folded into openAppSettings/
// saveAppSettings: these are server-side infra records (like remote/docker
// hosts), not a settings-payload field, so the app-settings-structure guard's
// by-id contract does not apply to them — only the `customModelEndpointsEnabled`
// toggle itself goes through that path.
// ═══════════════════════════════════════════════════════════════
/**
* Toggles the endpoint-management body's visibility to match the setting and,
* turning it on, lazily loads the endpoint list. Assigning `.checked` (as the
* settings load path does) fires no `change` event, so this must be called
* explicitly on open as well as wired to the checkbox's own onchange — a
* gate that only worked one of those two ways would show a stale "off"
* body right after opening, or a stale "on" one right after saving it off.
* With the feature off the body is a list of controls that do nothing, so it
* is hidden entirely rather than shown disabled.
*/
applyCustomModelEndpointsVisibility() {
const enabled = document.getElementById('appSettingsCustomModelEndpoints').checked;
const body = document.getElementById('customModelEndpointsBody');
if (body) body.style.display = enabled ? '' : 'none';
if (enabled) this.loadCustomModelEndpointsForSettings();
else this.closeCustomModelHostEditor();
this._applyCustomModelAdminGate();
},
/**
* Endpoint writes are admin-only in multi-user mode (custom-model-routes.ts),
* and GET already answers a non-admin with an empty list, which hides every
* per-row Edit/Discover/Delete button on its own. The "+ Add endpoint" button
* has no row to hide behind, so it needs its own gate — otherwise a non-admin
* can open the form, fill it in, and get a 403 toast on Save. Wired to the
* `codeman:me` event (admin-ui.js) as well as called from
* applyCustomModelEndpointsVisibility(), because `window.__codemanUser`'s
* real role can resolve AFTER settings have already been opened once.
*/
_applyCustomModelAdminGate() {
const addBtn = document.getElementById('customModelHostAddBtn');
if (!addBtn) return;
const me = window.__codemanUser || {};
const blocked = me.multiUser && me.role !== 'admin';
addBtn.style.display = blocked ? 'none' : '';
},
async loadCustomModelEndpointsForSettings() {
// GET /api/model-endpoints wraps its body in the { success, data } envelope
// like every other /api route (server.ts's preSerialization hook applies to
// arrays too) — _apiJson() unwraps it. A raw fetch().json() here would
// silently see the envelope object instead of the array and this panel
// would read as "No endpoints yet" forever, even with endpoints saved.
const hosts = await this._apiJson('/api/model-endpoints');
this._customModelHosts = Array.isArray(hosts) ? hosts : [];
this.renderCustomModelHostsList();
},
renderCustomModelHostsList() {
const list = document.getElementById('customModelHostsList');
if (!list) return;
const hosts = this._customModelHosts || [];
if (hosts.length === 0) {
list.innerHTML = '<p class="set-group-hint">No endpoints yet. Add one below to point a harness at a local or cloud OpenAI-compatible server.</p>';
return;
}
list.innerHTML = hosts
.map((h) => {
const modelCount = (h.models || []).length;
const modelSummary = modelCount === 0
? 'No models discovered yet'
: `${modelCount} model${modelCount === 1 ? '' : 's'}${h.defaultModelId ? ` · default: ${escapeHtml(h.defaultModelId)}` : ' · no default set'}`;
// escapeHtml(JSON.stringify(h.id)) — not JSON.stringify(h.id) alone —
// because JSON.stringify's own double quotes would otherwise terminate
// this double-quoted attribute at the first one, and everything after
// parses as raw tag content rather than the rest of the quoted string.
// Same idiom as deleteCase's onclick in session-ui.js. h.id is
// regex-constrained server-side (safe either way) but the pattern must
// match everywhere it is used, including where the argument is not.
const idArg = escapeHtml(JSON.stringify(h.id));
return `
<div class="set-row" data-endpoint-id="${escapeHtml(h.id)}">
<div class="set-row-text">
<span class="set-row-label">${escapeHtml(h.label)}</span>
<span class="set-row-desc">${escapeHtml(h.baseUrl)} — ${modelSummary}</span>
</div>
<div class="set-row-actions">
<button type="button" class="btn-toolbar btn-sm" onclick="app.discoverCustomModelHostModels(${idArg})">Discover</button>
<button type="button" class="btn-toolbar btn-sm" onclick="app.openCustomModelHostEditor(${idArg})">Edit</button>
<button type="button" class="btn-toolbar btn-danger btn-sm" onclick="app.deleteCustomModelHost(${idArg})">Delete</button>
</div>
</div>`;
})
.join('');
},
/** Opens the inline add/edit form. Pass no id to add a new endpoint. */
openCustomModelHostEditor(hostId) {
const host = hostId ? (this._customModelHosts || []).find((h) => h.id === hostId) : null;
this._editingCustomModelHostId = host ? host.id : null;
document.getElementById('customModelHostEditorTitle').textContent = host ? `Edit ${host.label}` : 'Add endpoint';
document.getElementById('customModelHostId').value = host?.id || '';
document.getElementById('customModelHostId').disabled = !!host; // id is immutable once created
document.getElementById('customModelHostLabel').value = host?.label || '';
document.getElementById('customModelHostBaseUrl').value = host?.baseUrl || '';
document.getElementById('customModelHostApiKey').value = ''; // the server never returns the real value (apiKeySet is a bool)
document.getElementById('customModelHostApiKey').placeholder = host?.apiKeySet ? '•••••••• (unchanged if left blank)' : '';
document.getElementById('customModelHostAuthStyle').value = host?.authStyle || 'bearer';
this._populateCustomModelDefaultSelect(host);
document.getElementById('customModelHostEditor').style.display = '';
},
closeCustomModelHostEditor() {
document.getElementById('customModelHostEditor').style.display = 'none';
this._editingCustomModelHostId = null;
},
_populateCustomModelDefaultSelect(host) {
const select = document.getElementById('customModelHostDefaultModel');
const models = host?.models || [];
select.innerHTML =
'<option value="">No default (picker uses the first discovered model)</option>' +
models.map((m) => `<option value="${escapeHtml(m)}">${escapeHtml(m)}</option>`).join('');
select.value = host?.defaultModelId || '';
select.disabled = models.length === 0;
},
async saveCustomModelHostFromEditor() {
const id = document.getElementById('customModelHostId').value.trim();
const label = document.getElementById('customModelHostLabel').value.trim();
const baseUrl = document.getElementById('customModelHostBaseUrl').value.trim();
const apiKeyInput = document.getElementById('customModelHostApiKey').value;
const authStyle = document.getElementById('customModelHostAuthStyle').value;
const defaultModelId = document.getElementById('customModelHostDefaultModel').value || undefined;
if (!id || !label || !baseUrl) {
this.showToast('Id, label and base URL are all required', 'warning');
return;
}
const editing = this._editingCustomModelHostId;
// PUT (server-side) treats an absent apiKey as "keep the stored one" — the
// browser never holds the real value to resend deliberately unchanged (see
// openCustomModelHostEditor and custom-model-routes.ts's applyStoredApiKey),
// so a blank field here means omitting the key entirely, not resending
// something we do not have. models/lastDiscoveredAt DO still need
// re-sending: PUT replaces the whole record, and this cached copy still
// carries both (only apiKey is redacted from what GET hands back).
const existing = editing ? (this._customModelHosts || []).find((h) => h.id === editing) : null;
const body = {
id,
label,
baseUrl,
authStyle,
defaultModelId,
apiKey: apiKeyInput || undefined,
models: existing?.models,
lastDiscoveredAt: existing?.lastDiscoveredAt,
};
try {
const res = await fetch(editing ? `/api/model-endpoints/${encodeURIComponent(editing)}` : '/api/model-endpoints', {
method: editing ? 'PUT' : 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(body),
});
const data = await res.json();
if (!data.success) {
this.showToast(data.error || 'Failed to save endpoint', 'error');
return;
}
this.showToast(editing ? 'Endpoint updated' : 'Endpoint added', 'success');
this.closeCustomModelHostEditor();
await this.loadCustomModelEndpointsForSettings();
} catch (err) {
this.showToast(`Failed to save endpoint: ${err.message}`, 'error');
}
},
async discoverCustomModelHostModels(hostId) {
this.showToast('Discovering models…', 'info');
try {
const res = await fetch(`/api/model-endpoints/${encodeURIComponent(hostId)}/discover-models`, { method: 'POST' });
const data = await res.json();
if (!data.success) {
this.showToast(data.error || 'Discovery failed', 'error');
return;
}
this.showToast(`Found ${data.data.models.length} model${data.data.models.length === 1 ? '' : 's'}`, 'success');
await this.loadCustomModelEndpointsForSettings();
} catch (err) {
this.showToast(`Discovery failed: ${err.message}`, 'error');
}
},
async deleteCustomModelHost(hostId) {
const host = (this._customModelHosts || []).find((h) => h.id === hostId);
if (!confirm(`Delete endpoint "${host?.label || hostId}"? Any session currently pointed at it keeps running until cleared.`)) return;
try {
await fetch(`/api/model-endpoints/${encodeURIComponent(hostId)}`, { method: 'DELETE' });
await this.loadCustomModelEndpointsForSettings();
} catch (err) {
this.showToast(`Failed to delete endpoint: ${err.message}`, 'error');
}
},
// ═══════════════════════════════════════════════════════════════
// Visibility Settings & Device-Specific Defaults
@@ -3543,3 +3757,15 @@ Object.assign(CodemanApp.prototype, {
this.subagentPanelVisible = false;
},
});
// window.__codemanUser's real role can resolve after settings have already been
// opened once (admin-ui.js fetches /api/me asynchronously and dispatches this on
// arrival), so the Custom Model Endpoints admin gate needs to be re-applied when
// it does, not just when the modal opens. Optional chaining on addEventListener
// itself: several frontend tests (run-mode-ui.test.ts) load this file into a vm
// context with a minimal fake `document` that has no event-target methods at
// all, and a module-level statement that throws there fails the whole file's
// evaluation, not just this feature.
document.addEventListener?.('codeman:me', () => {
window.app?._applyCustomModelAdminGate?.();
});
+237
View File
@@ -6807,6 +6807,67 @@ body.touch-device .terminal-container .xterm .xterm-helper-textarea {
min-height: 0;
}
/* Custom Model Endpoint Profiles' "which model" picker: same bounded-height +
scrollable-body shape as .modal-lg above, scoped by id rather than added to
.modal-sm itself (three other modals share that class for short, fixed
content and do not need a height cap). Without this the modal had no
max-height at all, so an endpoint with many discovered models grew the
dialog past the viewport with nothing to scroll — "the whole page" and
"the list is truncated" turned out to be one and the same bug. `min(70vh,
520px)` scales with the monitor (a phone gets 70% of its height, a 4K
display never gets a needlessly tall dialog) rather than a fixed value
that would be wrong at one end or the other. */
#customModelPickModal .modal-content {
max-height: min(70vh, 520px);
display: flex;
flex-direction: column;
}
#customModelPickModal .modal-body {
overflow-y: auto;
flex: 1;
min-height: 0;
}
/* Custom Model Endpoint Profiles: llama-swap model-swap confirmation — replaces a native
confirm() popup (docs/custom-model-endpoints-plan.md) so it looks and feels like the
rest of the app instead of a browser chrome dialog. Shares the context-window-too-small
modal's fixes below since both can appear mid-launch, in the same spot, for the same
reason — including this rule itself: there is no bare `.modal-footer` base style
anywhere in this file, and `.btn-toolbar` is `display: flex` (a block-level flex
container with no explicit `inline-flex`), so with no row layout of its own each
button took its own full-width line and the two stacked instead of sitting side by
side. Centred rather than flex-end per feedback — a two-button Cancel/confirm footer
reads better centred than pinned to one edge. */
#customModelSwapConfirmModal .modal-footer,
#customModelContextWarningModal .modal-footer {
display: flex;
justify-content: center;
gap: 0.5rem;
padding: 0.75rem 1rem;
border-top: 1px solid var(--border-color);
}
/* Both dialogs can appear while the centred llama-swap status banner (10001, see
.center-status-banner) is still on screen — right after "Claude started — switching
to llama-swap…" — and .modal's own z-index (1000) sat well under it, so the dialog
rendered fully hidden behind the banner (confirmed live, reported against the
context-window one but structurally identical for the swap-confirm modal too). */
#customModelSwapConfirmModal,
#customModelContextWarningModal {
z-index: 10010;
}
/* Both messages ARE the modal's whole explanatory content, not a one-line caption under
a form field, so .form-hint's 0.65rem caption size (right for what it was designed for)
read as illegibly small here, worst on the multi-sentence context-window explanation. */
#customModelSwapConfirmMessage,
#customModelContextWarningMessage {
font-size: 0.85rem;
line-height: 1.5;
color: var(--text);
}
/* Mobile Case Picker - Base Styles */
.mobile-case-picker-sheet {
@@ -8445,6 +8506,9 @@ kbd {
}
.toast {
display: flex;
align-items: center;
gap: 0.5rem;
background: var(--bg-card);
border: 1px solid var(--border);
border-radius: 6px;
@@ -8456,6 +8520,7 @@ kbd {
opacity: 0;
transition: all 0.2s ease;
pointer-events: auto;
max-width: 420px;
}
.toast.show {
@@ -8463,6 +8528,149 @@ kbd {
opacity: 1;
}
.toast-message {
flex: 1;
/* A sticky toast (showToast's opts.duration: 0) can carry a longer, specific
message — let it wrap instead of clipping. */
white-space: pre-wrap;
word-break: break-word;
}
/* Every toast gets one, sticky or not: a sticky toast with no way to close it
would just accumulate on screen across repeated failures. */
.toast-close {
flex-shrink: 0;
background: none;
border: none;
color: inherit;
opacity: 0.6;
font-size: 1.1rem;
line-height: 1;
padding: 0 0.15rem;
cursor: pointer;
}
.toast-close:hover {
opacity: 1;
}
/* Custom Model Endpoint Profiles: the "switching backends" / "loading model" states
(docs/custom-model-endpoints-plan.md) — a small set of messages prominent and
screen-centred rather than corner toasts, since they can sit on screen for well
over a minute (a real llama-swap model load) and are easy to mistake for nothing
happening. Non-blocking: `pointer-events: none` on the wrapper (no backdrop, no
click-catcher) with `auto` restored only on the card itself, purely so the text
inside remains selectable — there is nothing to click to dismiss it early. */
.center-status-banner {
position: fixed;
top: 50%;
left: 50%;
transform: translate(-50%, -50%) scale(0.96);
z-index: 10001;
display: flex;
align-items: center;
gap: 0.75rem;
background: var(--bg-card);
border: 1px solid var(--border);
border-radius: 10px;
padding: 1rem 1.5rem;
box-shadow: 0 8px 32px rgba(0, 0, 0, 0.4);
font-size: 0.95rem;
font-weight: 500;
color: var(--text);
max-width: min(90vw, 460px);
text-align: left;
opacity: 0;
pointer-events: none;
transition:
opacity 0.2s ease,
transform 0.2s ease;
}
.center-status-banner.show {
opacity: 1;
transform: translate(-50%, -50%) scale(1);
}
/* `hidden` has to be re-asserted over the `display: flex` above, or `dismiss()`
setting `el.hidden = true` does nothing (same trap as `.home-sessions[hidden]`
below): the card stays laid out at `opacity: 0` with its text/cancel/close
children still `pointer-events: auto`, an invisible click-blocker dead centre
over the terminal until the page reloads. */
.center-status-banner[hidden] {
display: none;
}
.center-status-spinner {
flex-shrink: 0;
width: 18px;
height: 18px;
border-radius: 50%;
border: 2px solid var(--border);
border-top-color: var(--accent, var(--text));
animation: center-status-spin 0.8s linear infinite;
}
@keyframes center-status-spin {
to {
transform: rotate(360deg);
}
}
.center-status-text {
flex: 1;
pointer-events: auto;
white-space: pre-wrap;
word-break: break-word;
}
/* Error variant: the load didn't finish in time — nothing is "in progress" anymore (no
spinner), and since this one doesn't dismiss itself, it needs a close button the user
can actually click, so pointer-events is restored here too (see the wrapper's own
comment on why that's `none` by default). */
.center-status-error {
border-color: rgba(239, 68, 68, 0.5);
}
.center-status-close {
flex-shrink: 0;
pointer-events: auto;
background: none;
border: none;
color: inherit;
opacity: 0.6;
font-size: 1.2rem;
line-height: 1;
padding: 0 0.15rem;
cursor: pointer;
}
.center-status-close:hover {
opacity: 1;
}
/* The Cancel button on an 'info' banner (e.g. the model-loading banner) — a real button
rather than the bare "×" close glyph above, since "Cancel" is an action with a
consequence (the caller's onCancel closes a session), not a plain dismiss. */
.center-status-cancel {
flex-shrink: 0;
pointer-events: auto;
background: none;
border: 1px solid var(--border);
border-radius: 6px;
color: inherit;
opacity: 0.75;
font-size: 0.8rem;
font-weight: 500;
padding: 0.25rem 0.6rem;
cursor: pointer;
}
.center-status-cancel:hover {
opacity: 1;
border-color: var(--text-muted, var(--border));
}
.toast-success { border-color: rgba(34, 197, 94, 0.4); }
.toast-error { border-color: rgba(239, 68, 68, 0.4); }
.toast-warning { border-color: rgba(234, 179, 8, 0.4); }
@@ -15138,6 +15346,12 @@ html[data-skin="daylight-blue"] .welcome-btn-tunnel.active:hover {
.run-mode-dot.web { background: #38bdf8; }
.run-mode-webviews { max-height: 180px; overflow-y: auto; }
/* Custom Model Endpoint Profiles' generated entries: `.run-mode-menu.active`'s
own `gap: 2px` only spaces its DIRECT children, and this container (like
`.run-mode-webviews` above) is one such child holding several buttons of
its own, so it needs the same gap repeated one level down or its rows sit
flush against each other. */
.run-mode-custom-models { display: flex; flex-direction: column; gap: 2px; }
/* A saved URL is a ROW: open on the left, edit + delete on the right, so a URL can
be changed or removed without first opening it as a tab. The side buttons stay
@@ -16302,6 +16516,29 @@ html[data-tab-orientation='vertical'] .home-sessions {
gap: 3px;
}
/* Custom Model Endpoint Profiles' inline add/edit form: a nested panel rather
than a modal, so it needs its own border to read as a distinct sub-section
inside .set-group-body's flat row stack. `--control-bg` rather than a
hardcoded black alpha — CLAUDE.md records that literal fill turning the
settings live preview into a grey slab on the light skins, and this panel
sits in the very same modal. */
:is(#appSettingsModal, #sessionOptionsModal, #createCaseModal) .set-inline-form {
display: flex;
flex-direction: column;
gap: 3px;
margin-top: 6px;
padding: 10px 12px;
border: 1px solid var(--border);
border-radius: 8px;
background: var(--control-bg);
}
:is(#appSettingsModal, #sessionOptionsModal, #createCaseModal) .set-inline-form h5 {
margin: 0 0 4px;
font-size: 0.72rem;
color: var(--text-muted);
}
/* ── rows ─────────────────────────────────────────────────────────────── */
:is(#appSettingsModal, #sessionOptionsModal, #createCaseModal) .set-row {
display: flex;
+692 -17
View File
@@ -16,16 +16,52 @@
import type { FastifyInstance, FastifyRequest } from 'fastify';
import { ApiErrorCode, createErrorResponse, type ApiResponse } from '../../types.js';
import { isAdmin, parseBody } from '../route-helpers.js';
import { isAdmin, parseBody, readJsonConfig, SETTINGS_PATH } from '../route-helpers.js';
import { isMultiUserMode } from '../../config/multiuser.js';
import { getDataDir } from '../../config/instance.js';
import { isBlockedWebviewUrl } from '../webview-egress-policy.js';
import { egressBlockedReason, webviewFetch } from '../webview-egress.js';
import { CustomModelHostSchema } from '../schemas.js';
import { readCustomModelHosts, writeCustomModelHosts, type CustomModelHost } from '../../custom-model-hosts.js';
import type { CliEntry } from '../../config/cli-registry/types.js';
const CODEMAN_CONFIG_DIR = getDataDir();
const DISCOVER_TIMEOUT_MS = 8000;
const PROPS_TIMEOUT_MS = 5000;
/**
* Claude Code's own system prompt + tool schemas cost roughly this many tokens on EVERY
* request, before a single character of conversation history — confirmed live, twice, on
* requests reporting `in:0 out:0` (the very first exchange) failing at ~36.4K tokens. No
* `CLAUDE_CODE_MAX_CONTEXT_TOKENS` value fixes this: that setting only changes when Claude
* Code decides to COMPACT conversation history, and there is no history yet on the first
* message for it to trim. A model whose real context is below this floor will refuse
* Claude Code's very first message outright, unconditionally.
*
* Set well above the ~36.4K actually measured — CLAUDE.md size, active MCP servers, and
* enabled skills all add to a project's real baseline, so the observed figure is a floor
* for THAT one workspace, not a ceiling for every one. Erring conservative here means a
* borderline-safe model still gets warned about (the user can launch anyway), rather than
* this floor missing a genuinely-too-small one because a smaller test project happened to
* fit.
*/
export const CLAUDE_MIN_SAFE_CONTEXT_TOKENS = 40000;
/**
* True when applying this model to this CLI is heading for a guaranteed first-message
* failure per `CLAUDE_MIN_SAFE_CONTEXT_TOKENS` above. Gated on `contextLengthVar` (today,
* only claude's registry entry declares one) rather than a hardcoded mode check: a CLI
* with a small enough baseline of its own to never trip this would have no reason to
* declare the field in the first place, so the check simply never applies to it.
*/
export function exceedsSafeContextFloor(
entry: Pick<CliEntry, 'capabilities'>,
contextLength: number | undefined
): boolean {
const cap = entry.capabilities.customModelInjection;
if (cap.kind !== 'env' || !cap.contextLengthVar) return false;
return typeof contextLength === 'number' && contextLength < CLAUDE_MIN_SAFE_CONTEXT_TOKENS;
}
function adminOnly(req: FastifyRequest, reply: { code: (n: number) => unknown }): ApiResponse<never> | null {
if (!isMultiUserMode() || isAdmin(req)) return null;
@@ -33,7 +69,58 @@ function adminOnly(req: FastifyRequest, reply: { code: (n: number) => unknown })
return createErrorResponse(ApiErrorCode.FORBIDDEN, 'Admin only in multi-user mode');
}
async function discoverModels(host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>): Promise<string[]> {
/**
* `defaultModelId` names the model the Run-menu picker applies for this endpoint with
* no further choice, so it must actually be one of the discovered `models` — a schema
* `.refine()` can't see across the two fields the way this can, and would also run on
* every unrelated field edit rather than only when either of these two changes.
*/
function invalidDefaultModel(host: Pick<CustomModelHost, 'defaultModelId' | 'models'>): ApiResponse<never> | null {
if (host.defaultModelId === undefined) return null;
if ((host.models ?? []).includes(host.defaultModelId)) return null;
return createErrorResponse(
ApiErrorCode.INVALID_INPUT,
'defaultModelId must be one of the endpoint’s discovered models'
);
}
/**
* Never hand the stored credential back to the browser, on GET, POST or PUT
* alike — the file is written 0600 precisely because it holds one. `apiKeySet`
* is what lets the editor say "unchanged if left blank" without the client
* ever holding the real value: `applyStoredApiKey()` below is the other half,
* treating an absent key on PUT as "keep the stored one" rather than clearing
* it, which is what makes never returning it survivable for the edit flow.
*/
function redactApiKey(host: CustomModelHost): Omit<CustomModelHost, 'apiKey'> & { apiKeySet: boolean } {
const { apiKey, ...rest } = host;
return { ...rest, apiKeySet: !!apiKey };
}
/**
* A PUT body with no `apiKey` (or a blank one) means "leave it alone", never
* "clear it": the editor never receives the real value to resend deliberately
* unchanged (see redactApiKey), so the only way it can tell the two apart is
* by omission. There is deliberately no way to CLEAR a key back to unset this
* way — a pre-existing limitation, not something this changes.
*/
function applyStoredApiKey(incoming: CustomModelHost, existing: CustomModelHost): CustomModelHost {
return incoming.apiKey ? incoming : { ...incoming, apiKey: existing.apiKey };
}
/**
* `modelContextLengths`/`modelSizesGB` are server-populated by discovery, never
* user-entered, and PUT replaces the whole record — so merge them back in from the
* stored host rather than trust whatever the editor's body carried (or omitted).
* The editor only ever sends `models`/`lastDiscoveredAt` verbatim from its cached
* copy; requiring it to also round-trip these two is exactly the kind of thing a
* future caller forgets, same class of bug `applyStoredApiKey` exists to prevent.
*/
function applyDiscoveredFields(incoming: CustomModelHost, existing: CustomModelHost): CustomModelHost {
return { ...incoming, modelContextLengths: existing.modelContextLengths, modelSizesGB: existing.modelSizesGB };
}
function authHeaders(host: Pick<CustomModelHost, 'apiKey' | 'authStyle'>): Record<string, string> {
const headers: Record<string, string> = {};
const apiKey = host.apiKey?.trim();
// Exactly ONE header, never both — see custom-model-hosts.ts's CustomModelAuthStyle
@@ -41,14 +128,137 @@ async function discoverModels(host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' |
const style = host.authStyle ?? 'bearer';
if (apiKey && style === 'bearer') headers.Authorization = `Bearer ${apiKey}`;
if (apiKey && style === 'api-key') headers['api-key'] = apiKey;
return headers;
}
export interface DiscoveryResult {
models: string[];
/** See `CustomModelHost.modelContextLengths` — only ever populated for models already loaded. */
contextLengths: Record<string, number>;
/** See `CustomModelHost.modelSizesGB` — populated for every model whose own listing states one. */
sizesGB: Record<string, number>;
}
/**
* Best-effort: pulls a file size in GB out of a model's own `description`, when the
* server states one. llama-swap writes `"Auto-discovered 16.35 GB - parameters
* auto-fitted by llama.cpp"` for a model it found on disk itself; a hand-configured
* profile's own description (e.g. `"General-purpose reasoning model, MoE CPU-offloaded."`)
* has no such figure and correctly yields no estimate rather than a guess — there is no
* separate "give me the file size" endpoint to fall back on.
*/
function parseSizeGB(description: unknown): number | undefined {
if (typeof description !== 'string') return undefined;
const match = /(\d+(?:\.\d+)?)\s*GB\b/i.exec(description);
if (!match) return undefined;
const size = Number(match[1]);
return Number.isFinite(size) && size > 0 ? size : undefined;
}
/**
* Best-effort: fetches `GET /props?model=<id>` (llama.cpp-native, llama-swap-proxied) for
* ONE already-loaded model and pulls its real `n_ctx` out. Never called for a model that
* isn't already loaded — see the caller and `CustomModelHost.modelContextLengths` for why
* that's a hard safety requirement, not just a nicety: llama-swap treats this endpoint's
* `?model=` as a routing hint, and asking it about an unloaded model risks triggering an
* actual (slow, GPU-swapping) load as a side effect of what should be read-only discovery.
* Any failure (unreachable, non-2xx, missing/malformed field) is swallowed — one model's
* context length is a nice-to-have, never worth failing the whole discovery pass over.
*
* ⚠️ FALLBACK ONLY — confirmed live to be actively WRONG for a `--fit-ctx`-launched llama-
* swap backend: `/props`'s `n_ctx` read 154112 for a model llama-swap itself had launched
* with `--fit-ctx 16384` (visible in `/running`'s own `cmd`), and the real server then
* refused a request at the real 16384-token limit — `n_ctx` here appears to report the
* model's theoretical/trained maximum, not the runtime-configured one. `parseCtxFromCmd`
* (below), which reads the actual launch flag `/running` reports, is the primary source;
* this is only used when that parse comes up empty (no recognized flag in `cmd`, or `cmd`
* itself unavailable).
*/
async function fetchContextLength(
host: Pick<CustomModelHost, 'baseUrl'>,
modelId: string,
headers: Record<string, string>
): Promise<number | undefined> {
try {
const url = new URL(`${host.baseUrl.replace(/\/+$/, '')}/props`);
url.searchParams.set('model', modelId);
const res = await webviewFetch(url, { headers, signal: AbortSignal.timeout(PROPS_TIMEOUT_MS) });
if (!res.ok) return undefined;
const body = (await res.json()) as { n_ctx?: unknown; default_generation_settings?: { n_ctx?: unknown } };
const nCtx = body.n_ctx ?? body.default_generation_settings?.n_ctx;
return typeof nCtx === 'number' && Number.isFinite(nCtx) && nCtx > 0 ? nCtx : undefined;
} catch {
return undefined;
}
}
/**
* Parses the REAL configured context size out of llama-swap's own launch command for a
* model (`/running`'s `cmd` field, e.g. `"llama-server -m ... --fit-ctx 16384 ..."`) —
* the primary source for `modelContextLengths`, preferred over `/props`'s `n_ctx` (see
* `fetchContextLength`'s own doc comment for why that field is unreliable here). Checks
* `--fit-ctx` first (llama-swap's own auto-fit flag), then the plain llama.cpp
* `-c`/`--ctx-size`/`--ctx_size` flags a hand-written launch command might use instead.
* Returns `undefined` when `cmd` has none of these — not every launch command needs to
* state one explicitly (llama.cpp has its own default), and guessing one would be worse
* than the "no override applied" the caller already treats an unknown length as.
*/
function parseCtxFromCmd(cmd: unknown): number | undefined {
if (typeof cmd !== 'string') return undefined;
const match = /--fit-ctx\s+(\d+)/.exec(cmd) ?? /(?:^|\s)(?:-c|--ctx-size|--ctx_size)\s+(\d+)/.exec(cmd);
if (!match) return undefined;
const value = Number(match[1]);
return Number.isFinite(value) && value > 0 ? value : undefined;
}
async function discoverModels(
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>
): Promise<DiscoveryResult> {
const headers = authHeaders(host);
const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/v1/models`), {
headers,
signal: AbortSignal.timeout(DISCOVER_TIMEOUT_MS),
});
if (!res.ok) throw new Error(`HTTP ${res.status}`);
const body = (await res.json()) as { data?: Array<{ id?: unknown }> };
return (body.data ?? []).map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
const body = (await res.json()) as {
data?: Array<{ id?: unknown; status?: { value?: unknown }; description?: unknown }>;
};
const entries = body.data ?? [];
const models = entries.map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
const sizesGB: Record<string, number> = {};
for (const entry of entries) {
if (typeof entry.id !== 'string' || !entry.id) continue;
const size = parseSizeGB(entry.description);
if (size !== undefined) sizesGB[entry.id] = size;
}
// llama-swap-specific, feature-detected: a server that never mentions `status` on ANY
// entry gets no context-length enrichment at all, rather than treating "no status field"
// as "assume unloaded" — either reading is a guess, and skipping is the safe one, since
// fetchContextLength must only ever run against a model this server itself calls loaded.
const hasStatusField = entries.some((m) => m && typeof m === 'object' && 'status' in m);
const contextLengths: Record<string, number> = {};
if (hasStatusField) {
const loadedIds = entries
.filter((m) => m.status && typeof m.status === 'object' && (m.status as { value?: unknown }).value === 'loaded')
.map((m) => m.id)
.filter((id): id is string => typeof id === 'string' && id.length > 0);
if (loadedIds.length > 0) {
// Primary source: the REAL launch command (see parseCtxFromCmd's own doc comment
// for why /props's n_ctx cannot be trusted here). One /running call covers every
// loaded model, so this never costs more requests than the old /props-only path did
// when the cmd parse succeeds, and exactly one extra when it has to fall back.
const swapStatus = await getLlamaSwapStatus(host);
const cmdById = new Map(swapStatus.running.map((r) => [r.model, r.cmd]));
for (const id of loadedIds) {
const fromCmd = parseCtxFromCmd(cmdById.get(id));
const ctx = fromCmd ?? (await fetchContextLength(host, id, headers));
if (ctx !== undefined) contextLengths[id] = ctx;
}
}
}
return { models, contextLengths, sizesGB };
}
/**
@@ -66,41 +276,472 @@ function describeFetchError(err: unknown): string {
return message;
}
export function registerCustomModelRoutes(app: FastifyInstance): void {
app.get('/api/model-endpoints', async (req) =>
isMultiUserMode() && !isAdmin(req) ? [] : readCustomModelHosts(CODEMAN_CONFIG_DIR)
);
type RedactedHost = ReturnType<typeof redactApiKey>;
app.post('/api/model-endpoints', async (req, reply): Promise<ApiResponse<{ host: CustomModelHost }>> => {
/**
* Merges a fresh `GET /v1/models` result into a host record: stamps
* `lastDiscoveredAt`, and drops `defaultModelId` if it no longer appears in
* the fresh list (it would otherwise leave the Run-menu picker applying a
* model id the endpoint just told us it doesn't serve). Pure — no IO, so the
* manual route (which reports a fetch failure's *reason* to the caller) and
* the periodic sweep below (which only cares whether it can move on) can
* each do their own `discoverModels()` + error handling around one shared
* "how to apply a successful result" step.
*/
const RUNNING_TIMEOUT_MS = 5000;
export interface LlamaSwapRunningModel {
model: string;
state: string;
/** The actual launch command llama-swap started this backend with, when it says one —
* see `parseCtxFromCmd`, which reads the real configured context size out of this. */
cmd?: string;
}
export interface LlamaSwapStatus {
/**
* Feature-detected via `GET /running`: true only when the server answered with
* llama-swap's own shape (`{ running: [...] }`). Plain llama.cpp (and any other
* OpenAI-compatible server) has no such endpoint and always runs the single model
* it was started with, so there is no "current model" to conflict with — every
* caller must treat `isLlamaSwap: false` as "nothing to check", never as an error.
*/
isLlamaSwap: boolean;
running: LlamaSwapRunningModel[];
}
/**
* Distinguishes llama-swap from a plain llama.cpp/OpenAI-compatible server, and reports
* what llama-swap currently has loaded — llama.cpp only ever runs one GGUF at a time, and
* llama-swap unloads/reloads it on demand when a request asks for a different one, which
* can take anywhere from a few seconds to over a minute. Read-only: this never triggers a
* swap itself (unlike `/props?model=`, `/running` takes no `model` parameter to route by).
* Best-effort like `discoverModels()`'s siblings: any failure (unreachable, non-2xx,
* unexpected shape) reads as "not llama-swap", never thrown.
*/
export async function getLlamaSwapStatus(
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>
): Promise<LlamaSwapStatus> {
try {
const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/running`), {
headers: authHeaders(host),
signal: AbortSignal.timeout(RUNNING_TIMEOUT_MS),
});
if (!res.ok) return { isLlamaSwap: false, running: [] };
const body = (await res.json()) as { running?: unknown };
if (!Array.isArray(body.running)) return { isLlamaSwap: false, running: [] };
const running = body.running
.filter(
(r): r is { model: string; state?: unknown; cmd?: unknown } =>
!!r && typeof r === 'object' && typeof (r as { model?: unknown }).model === 'string'
)
.map((r) => ({
model: r.model,
state: typeof r.state === 'string' ? r.state : 'unknown',
cmd: typeof r.cmd === 'string' ? r.cmd : undefined,
}));
return { isLlamaSwap: true, running };
} catch {
return { isLlamaSwap: false, running: [] };
}
}
interface LlamaSwapLogTail {
latestLine?: string;
lastAccessedAt: number;
controller: AbortController;
}
/** One open `/api/events` tail per endpoint, keyed by host id — see `getLatestLlamaSwapLogLine`. */
const llamaSwapLogTails = new Map<string, LlamaSwapLogTail>();
/** A tail nothing has asked about in this long is closed by the next `pruneIdleLlamaSwapLogTails` sweep. */
const LOG_TAIL_IDLE_MS = 30_000;
/**
* Parses one `data: {...}` payload from llama-swap's `GET /api/events` SSE stream and
* returns the backend (never llama-swap's own proxy) log text it carries, or `undefined`
* for anything else (a different event `type`, a malformed frame, a proxy-sourced one).
*
* The real shape, confirmed live against a real llama-swap deployment — NOT documented
* anywhere the plan doc's original research found, and genuinely surprising the first
* time around: `GET /logs` (the endpoint that name suggests, and this feature's own
* first cut was built against) turns out to carry ONLY llama-swap's own proxy
* request-access log — it never once showed a single backend line even seconds after a
* real, confirmed model swap. The backend llama-server process's actual stdout
* (`load_model: ...`, `llama_server: model loaded`) only ever showed up in `/api/events`,
* as `{"type":"logData","data":"<JSON-string>"}` whose OWN `data` field parses to a
* second object, `{"data": "<newline-joined log text>", "source": "proxy" | "upstream"}`
* — `source` is the exact, explicit distinguisher (`upstream` = the backend process,
* `proxy` = llama-swap's own line), not a guessed regex against the text itself.
*/
function parseBackendLogDataEvent(dataLine: string): string | undefined {
let outer: unknown;
try {
outer = JSON.parse(dataLine);
} catch {
return undefined;
}
if (
!outer ||
typeof outer !== 'object' ||
(outer as { type?: unknown }).type !== 'logData' ||
typeof (outer as { data?: unknown }).data !== 'string'
) {
return undefined;
}
let inner: unknown;
try {
inner = JSON.parse((outer as { data: string }).data);
} catch {
return undefined;
}
if (
!inner ||
typeof inner !== 'object' ||
(inner as { source?: unknown }).source !== 'upstream' ||
typeof (inner as { data?: unknown }).data !== 'string'
) {
return undefined;
}
return (inner as { data: string }).data;
}
/**
* Reads `GET /api/events` forever (until `entry.controller` aborts it), updating
* `entry.latestLine` with the most recent BACKEND log line seen (see
* `parseBackendLogDataEvent`). Fire-and-forget: the caller never awaits this — it runs
* for the tail's whole lifetime in the background, and `getLatestLlamaSwapLogLine` just
* reads whatever `entry.latestLine` currently holds. SSE frames are separated by a blank
* line (`\n\n`), buffered the same way `/running`'s NDJSON-shaped siblings buffer partial
* chunks — a frame split across two `reader.read()` calls must not be parsed early.
*/
async function pumpLlamaSwapLogTail(
host: Pick<CustomModelHost, 'id' | 'baseUrl' | 'apiKey' | 'authStyle'>,
entry: LlamaSwapLogTail
): Promise<void> {
try {
const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/api/events`), {
headers: authHeaders(host),
signal: entry.controller.signal,
});
if (!res.ok || !res.body) return;
const reader = res.body.getReader();
const decoder = new TextDecoder();
let buffer = '';
for (;;) {
const { done, value } = await reader.read();
if (done) break;
buffer += decoder.decode(value, { stream: true });
const frames = buffer.split('\n\n');
buffer = frames.pop() ?? '';
for (const frame of frames) {
const dataLine = frame.split('\n').find((l) => l.startsWith('data:'));
if (!dataLine) continue;
const backendText = parseBackendLogDataEvent(dataLine.slice('data:'.length));
if (!backendText) continue;
const lines = backendText.split('\n').filter((l) => l.trim());
if (lines.length > 0) entry.latestLine = lines[lines.length - 1]!.trim();
}
}
} catch {
// connection dropped / aborted / endpoint unreachable — a future access starts fresh
} finally {
// Delete by IDENTITY, not just by key: an aborted pump can finish after a NEWER
// entry was already created for the same endpoint id (e.g. abort-then-immediately-
// re-request), and deleting unconditionally would remove that newer entry and orphan
// its connection — nothing would ever prune it, since pruneIdleLlamaSwapLogTails only
// walks entries still present in the map.
if (llamaSwapLogTails.get(host.id) === entry) {
llamaSwapLogTails.delete(host.id);
}
}
}
/**
* Real-time "what is llama.cpp actually doing right now" for the loading banner
* (docs/custom-model-endpoints-plan.md): llama-swap's `GET /api/events` SSE stream
* carries the backend llama-server process's own stdout — `load_model: loading model
* '<path>'`, `load_model: initializing, n_slots = N, n_ctx_slot = N`, `llama_server:
* model loaded`, etc — tagged `source: "upstream"`, distinct from llama-swap's own
* `source: "proxy"` request-access lines (see `parseBackendLogDataEvent`). Confirmed
* live against a real llama-swap deployment, including through an actual forced model
* swap end-to-end.
*
* Held OPEN per endpoint rather than re-opened on every 1s poll — confirmed live to stay
* open indefinitely (read past 220KB over 8 seconds with no `done`), unlike `/logs`
* (see `parseBackendLogDataEvent`'s doc comment), so reconnecting each poll would be
* pure waste. One connection is reused across every session currently watching a load on
* that endpoint; since llama.cpp/llama-swap only ever runs one model at a time, a line
* seen while a load is in flight is safe to attribute to that load (a deployment that
* could load several models concurrently would need a per-model tag this format doesn't
* provide).
*
* Lazily started on first access and idle-closed rather than left open forever — see
* `pruneIdleLlamaSwapLogTails`.
*/
export function getLatestLlamaSwapLogLine(
host: Pick<CustomModelHost, 'id' | 'baseUrl' | 'apiKey' | 'authStyle'>
): string | undefined {
let entry = llamaSwapLogTails.get(host.id);
if (!entry) {
entry = { lastAccessedAt: Date.now(), controller: new AbortController() };
llamaSwapLogTails.set(host.id, entry);
void pumpLlamaSwapLogTail(host, entry);
}
entry.lastAccessedAt = Date.now();
return entry.latestLine;
}
/**
* Closes any log tail nothing has called `getLatestLlamaSwapLogLine` about in
* `LOG_TAIL_IDLE_MS` — a stream nobody is polling is an open connection with nothing to
* show for it. Called from the same periodic sweep as `detectCustomModelSwapDisplacements`
* in server.ts, not its own timer.
*/
export function pruneIdleLlamaSwapLogTails(now = Date.now()): void {
for (const [id, entry] of llamaSwapLogTails) {
if (now - entry.lastAccessedAt > LOG_TAIL_IDLE_MS) {
entry.controller.abort();
llamaSwapLogTails.delete(id);
}
}
}
/**
* Actually kicks off llama-swap's lazy model load, rather than waiting for the launched
* CLI's own first prompt to do it. llama-swap has no separate "switch model" admin
* endpoint — the ONLY thing that starts a swap is a real inference request naming the
* model (confirmed live: applying a selection alone never appeared in the llama-swap
* server's own logs; nothing had actually asked it to load anything). This sends the
* smallest real request that will — `max_tokens: 1`, one throwaway user message — to
* `${baseUrl}/v1/chat/completions`, the OpenAI-compatible endpoint every supported
* harness already points at.
*
* Deliberately fire-and-forget: the caller (the apply/create routes) returns to the
* client immediately, and the frontend's own polling (`GET .../running-status`) is what
* actually confirms readiness — this call's response is never read, just its side
* effect. No abort/timeout of its own either: a real load can take well over a minute for
* a large model, and this is a normal long-running Node process, so there is nothing to
* clean up by cutting it short. Errors are swallowed for the same reason `discoverModels`'s
* siblings swallow theirs — one endpoint's hiccup here is a nice-to-have that failed, not
* something worth surfacing as a request failure four layers up.
*/
export function triggerLlamaSwapLoad(
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>,
modelId: string
): void {
const url = new URL(`${host.baseUrl.replace(/\/+$/, '')}/v1/chat/completions`);
webviewFetch(url, {
method: 'POST',
headers: { ...authHeaders(host), 'content-type': 'application/json' },
body: JSON.stringify({
model: modelId,
messages: [{ role: 'user', content: 'Hi' }],
max_tokens: 1,
stream: false,
}),
}).catch(() => {
// best-effort — see the doc comment above
});
}
function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult): CustomModelHost {
const { models, contextLengths, sizesGB } = result;
const defaultModelId = host.defaultModelId && models.includes(host.defaultModelId) ? host.defaultModelId : undefined;
// Merge onto what's already known rather than replacing: a model not probed this round
// (not currently loaded) keeps whatever context length an earlier round already learned
// for it, and one no longer in the fresh list is dropped, same reasoning as defaultModelId.
const merged = { ...host.modelContextLengths, ...contextLengths };
const kept = Object.fromEntries(Object.entries(merged).filter(([id]) => models.includes(id)));
const modelContextLengths = Object.keys(kept).length > 0 ? kept : undefined;
// sizesGB, unlike contextLengths, is populated for every model in the SAME pass (no
// loaded-only restriction — see parseSizeGB), so this is closer to a plain replace, but
// still merges onto the previous round rather than dropping a size for a model whose
// description happened to omit the figure on this particular pass.
const mergedSizes = { ...host.modelSizesGB, ...sizesGB };
const keptSizes = Object.fromEntries(Object.entries(mergedSizes).filter(([id]) => models.includes(id)));
const modelSizesGB = Object.keys(keptSizes).length > 0 ? keptSizes : undefined;
return {
...host,
models,
defaultModelId,
modelContextLengths,
modelSizesGB,
lastDiscoveredAt: new Date().toISOString(),
};
}
/**
* `customModelEndpointsEnabled` defaults OFF (unlike `showPlanUsageLimits`'s
* absent-means-on in `readPlanUsageTelemetryEnabled`), so mirror the frontend's
* own gate (`session-ui.js`'s `!settings.customModelEndpointsEnabled`) rather
* than that reader's default. Exists so the periodic re-discovery sweep in
* server.ts can skip entirely while the feature is off, instead of polling
* every saved endpoint forever regardless of the setting.
*/
export async function readCustomModelEndpointsEnabled(): Promise<boolean> {
const settings = await readJsonConfig<Record<string, unknown>>(SETTINGS_PATH, 'settings.json', {});
return settings.customModelEndpointsEnabled === true;
}
/**
* Re-discovers every saved endpoint's models, best-effort. One endpoint being
* unreachable (powered off, wrong network) must not stop the others from
* refreshing, and a read-modify-write per host (rather than one batch write
* at the end) means a crash or restart mid-sweep loses at most the endpoints
* not yet reached, never a write already applied. Exported so both the
* periodic timer (server.ts) and a test can drive it directly.
*/
export async function refreshAllCustomModelHosts(): Promise<void> {
const dataDir = getDataDir();
const hosts = await readCustomModelHosts(dataDir);
for (const host of hosts) {
if (isBlockedWebviewUrl(host.baseUrl)) continue;
let result: DiscoveryResult;
try {
result = await discoverModels(host);
} catch {
continue; // unreachable this cycle — try again next tick, not fatal to the sweep
}
// Re-read + splice by id rather than reusing the array captured above: an
// admin editing or deleting an endpoint via the API mid-sweep must win,
// not be silently overwritten by a refresh that started before their change.
const current = await readCustomModelHosts(dataDir);
const index = current.findIndex((item) => item.id === host.id);
if (index === -1) continue; // deleted mid-sweep
current[index] = applyDiscoveredModels(current[index], result);
await writeCustomModelHosts(dataDir, current);
}
}
/** The subset of `Session` this sweep needs — kept minimal so a test can pass a plain object. */
export interface CustomModelSessionLike {
id: string;
name: string;
customModel?: { endpointId: string; modelId: string; label?: string };
}
/** One session whose model was just found evicted, ready to broadcast as `CustomModelSwappedOut`. */
export interface CustomModelSwapDisplacement {
sessionId: string;
sessionName: string;
endpointId: string;
previousModel: string;
currentlyLoadedModel: string;
}
/**
* Detects when a live session's own custom-model selection is no longer the model
* llama-swap actually has loaded — evicted by ANOTHER session's activity on the same
* endpoint, since llama.cpp/llama-swap runs one model at a time (the apply/create routes'
* own swap-conflict check only ever runs at THAT session's own launch/apply moment, so it
* cannot catch a later eviction triggered by a different session's normal use — confirmed
* live: a session created while nothing else had a live conflict at that instant can still
* get silently displaced afterward). Read-only, and best-effort per endpoint exactly like
* `refreshAllCustomModelHosts`'s sibling sweep — one endpoint's hiccup here never blocks
* checking the others.
*
* `notifiedSessionIds` is the caller's own de-dupe state (`server.ts` keeps one `Set` across
* sweeps), mutated in place: a session id is added once displaced and removed again once its
* own model is loaded and ready — so a LATER, genuinely new displacement can notify again
* rather than the session staying silently un-notified forever after the first one.
*/
export async function detectCustomModelSwapDisplacements(
sessions: Iterable<CustomModelSessionLike>,
notifiedSessionIds: Set<string>
): Promise<CustomModelSwapDisplacement[]> {
const byEndpoint = new Map<string, CustomModelSessionLike[]>();
for (const session of sessions) {
if (!session.customModel) continue;
const group = byEndpoint.get(session.customModel.endpointId);
if (group) group.push(session);
else byEndpoint.set(session.customModel.endpointId, [session]);
}
if (byEndpoint.size === 0) return [];
const hosts = await readCustomModelHosts(getDataDir());
const displacements: CustomModelSwapDisplacement[] = [];
for (const [endpointId, group] of byEndpoint) {
const host = hosts.find((h) => h.id === endpointId);
if (!host) continue; // endpoint deleted since these sessions were created — nothing to check
let status: LlamaSwapStatus;
try {
status = await getLlamaSwapStatus(host);
} catch {
continue; // unreachable this cycle — try again next tick, not fatal to the sweep
}
// Not llama-swap (feature-detected) or nothing loaded at all: nothing has been evicted,
// by construction — a plain llama.cpp/OpenAI-compatible server only ever runs the one
// model it was started with, so there is no "current model" to conflict with.
if (!status.isLlamaSwap || status.running.length === 0) continue;
const currentlyLoaded = status.running.find((r) => r.state === 'ready')?.model ?? status.running[0]?.model;
if (!currentlyLoaded) continue;
for (const session of group) {
const modelId = session.customModel!.modelId;
const stillLoaded = status.running.some((r) => r.model === modelId);
if (stillLoaded) {
notifiedSessionIds.delete(session.id); // back to normal — a future eviction can notify again
continue;
}
if (notifiedSessionIds.has(session.id)) continue; // already told them once for this displacement
notifiedSessionIds.add(session.id);
displacements.push({
sessionId: session.id,
sessionName: session.name,
endpointId,
previousModel: modelId,
currentlyLoadedModel: currentlyLoaded,
});
}
}
return displacements;
}
export function registerCustomModelRoutes(app: FastifyInstance): void {
app.get('/api/model-endpoints', async (req): Promise<RedactedHost[]> => {
if (isMultiUserMode() && !isAdmin(req)) return [];
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
return hosts.map(redactApiKey);
});
app.post('/api/model-endpoints', async (req, reply): Promise<ApiResponse<{ host: RedactedHost }>> => {
const denied = adminOnly(req, reply);
if (denied) return denied;
const host = parseBody(CustomModelHostSchema, req.body);
if (isBlockedWebviewUrl(host.baseUrl)) {
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
}
const badDefault = invalidDefaultModel(host);
if (badDefault) return badDefault;
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
if (hosts.some((item) => item.id === host.id)) {
return createErrorResponse(ApiErrorCode.ALREADY_EXISTS, 'Model endpoint already exists');
}
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, [...hosts, host]);
return { success: true, data: { host } };
return { success: true, data: { host: redactApiKey(host) } };
});
app.put('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ host: CustomModelHost }>> => {
app.put('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ host: RedactedHost }>> => {
const denied = adminOnly(req, reply);
if (denied) return denied;
const { id } = req.params as { id: string };
const host = parseBody(CustomModelHostSchema, { ...(req.body as object), id });
if (isBlockedWebviewUrl(host.baseUrl)) {
const incoming = parseBody(CustomModelHostSchema, { ...(req.body as object), id });
if (isBlockedWebviewUrl(incoming.baseUrl)) {
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
}
const badDefault = invalidDefaultModel(incoming);
if (badDefault) return badDefault;
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
const index = hosts.findIndex((item) => item.id === id);
if (index === -1) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
const host = applyDiscoveredFields(applyStoredApiKey(incoming, hosts[index]), hosts[index]);
const next = [...hosts];
next[index] = host;
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
return { success: true, data: { host } };
return { success: true, data: { host: redactApiKey(host) } };
});
app.delete('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ id: string }>> => {
@@ -129,11 +770,11 @@ export function registerCustomModelRoutes(app: FastifyInstance): void {
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
}
try {
const models = await discoverModels(host);
const result = await discoverModels(host);
const next = [...hosts];
next[index] = { ...host, models, lastDiscoveredAt: new Date().toISOString() };
next[index] = applyDiscoveredModels(host, result);
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
return { success: true, data: { models } };
return { success: true, data: { models: result.models } };
} catch (err) {
const blocked = egressBlockedReason(err);
return createErrorResponse(
@@ -143,4 +784,38 @@ export function registerCustomModelRoutes(app: FastifyInstance): void {
}
}
);
// Read-only, no admin gate: any session owner who can already point their own session
// at this endpoint (POST .../custom-model, ungated by design — see session-routes.ts)
// can equally ask what it currently has loaded, before or while that apply is pending.
app.get(
'/api/model-endpoints/:id/running-status',
async (
req
): Promise<
ApiResponse<{
isLlamaSwap: boolean;
running: Array<Pick<LlamaSwapRunningModel, 'model' | 'state'>>;
logLine?: string;
}>
> => {
const { id } = req.params as { id: string };
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
const host = hosts.find((item) => item.id === id);
if (!host) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
if (isBlockedWebviewUrl(host.baseUrl)) {
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
}
const status = await getLlamaSwapStatus(host);
// Only worth tailing /logs once llama-swap is actually confirmed — a plain
// llama.cpp/OpenAI-compatible server has no such endpoint at all.
const logLine = status.isLlamaSwap ? getLatestLlamaSwapLogLine(host) : undefined;
// `cmd` (the literal llama-server launch line, which can carry model paths and
// --api-key) exists only so parseCtxFromCmd() can read it server-side during
// discovery — this un-gated, polled-every-second route has no reason to hand it
// to the browser, which only ever reads `model`/`state`.
const running = status.running.map(({ model, state }) => ({ model, state }));
return { success: true, data: { isLlamaSwap: status.isLlamaSwap, running, logLine } };
}
);
}
+9 -1
View File
@@ -28,4 +28,12 @@ export { registerWsRoutes } from './ws-routes.js';
export { registerVoiceRoutes } from './voice-routes.js';
export { registerWebviewRoutes, tryWebviewRefererFallback } from './webview-routes.js';
export { registerTabLayoutRoutes } from './tab-layout-routes.js';
export { registerCustomModelRoutes } from './custom-model-routes.js';
export {
registerCustomModelRoutes,
refreshAllCustomModelHosts,
readCustomModelEndpointsEnabled,
detectCustomModelSwapDisplacements,
pruneIdleLlamaSwapLogTails,
type CustomModelSessionLike,
type CustomModelSwapDisplacement,
} from './custom-model-routes.js';
+262 -11
View File
@@ -11,7 +11,7 @@ import { homedir } from 'node:os';
import { existsSync, statSync, mkdirSync, writeFileSync } from 'node:fs';
import { execFile } from 'node:child_process';
import fs from 'node:fs/promises';
import { randomBytes } from 'node:crypto';
import { randomBytes, randomUUID } from 'node:crypto';
import { performance } from 'node:perf_hooks';
import {
ApiErrorCode,
@@ -57,6 +57,12 @@ import {
} from '../schemas.js';
import { readCustomModelHosts } from '../../custom-model-hosts.js';
import { applyCustomModelInjection, removeConfigDir } from '../../custom-model-injection-apply.js';
import {
getLlamaSwapStatus,
triggerLlamaSwapLoad,
exceedsSafeContextFloor,
CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
} from './custom-model-routes.js';
import { matchesPattern } from '../../config/cli-registry/patterns.js';
import { ownerLayoutKey } from '../../tab-layout-persistence.js';
import { TabLayoutValidationError } from '../../tab-layout.js';
@@ -1232,13 +1238,78 @@ export function registerSessionRoutes(
if (!endpoint) {
return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
}
const contextLength = endpoint.modelContextLengths?.[body.modelId];
// Some CLIs (today: only claude) carry enough of their own fixed system-prompt/tool-
// schema overhead that a small enough real context guarantees a first-message failure
// no matter what CLAUDE_CODE_MAX_CONTEXT_TOKENS says — confirmed live at ~36.4K tokens
// against a model configured with a real 16384-token context. Warn before committing
// to a restart that's certain to fail, rather than letting the user discover it via a
// cryptic 400 from the CLI itself. `confirmed` (already used for the swap-conflict
// warning below) skips this too — the user has already said "launch anyway" once.
if (!body.confirmed && exceedsSafeContextFloor(entry, contextLength)) {
return {
requiresContextWarning: true,
modelId: body.modelId,
contextLength,
minSafeContextTokens: CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
};
}
// llama.cpp runs exactly one model at a time; llama-swap unloads and reloads it on
// demand, which can take anywhere from a few seconds to over a minute — long enough
// that a session mid-swap looks indistinguishable from one that never left the native
// backend. Feature-detected via llama-swap's own `GET /running` (a plain llama.cpp
// server has no such endpoint and reads as `isLlamaSwap: false` — nothing to check).
const swapStatus = await getLlamaSwapStatus(endpoint);
const currentlyLoaded = swapStatus.running.find((r) => r.state === 'ready')?.model ?? swapStatus.running[0]?.model;
// Distinct from targetReady below: this is ONLY about whether proceeding would evict a
// model another session is actively using — true even if nothing is loaded at all yet
// would be wrong here (nothing to evict), so this stays narrowly "a DIFFERENT model is
// currently ready".
const swapNeeded = swapStatus.isLlamaSwap && !!currentlyLoaded && currentlyLoaded !== body.modelId;
// Whether the TARGET model itself is already the one loaded and ready — false whether
// nothing is loaded yet, a different model is loaded, or this one is loaded but still
// mid-load. Drives both the actual load trigger below and modelSwapInProgress in the
// response; deliberately broader than swapNeeded, which only gates the confirmation ask.
const targetReady = swapStatus.running.some((r) => r.model === body.modelId && r.state === 'ready');
// Only ask when switching would actually take the model away from another session
// that is currently using it — never just because a swap is needed at all. `confirmed`
// (set by the caller after showing that warning once) skips asking again.
if (swapNeeded && !body.confirmed) {
const conflicting = [...ctx.sessions.values()].filter(
(s) =>
s.id !== session.id && s.customModel?.endpointId === endpoint.id && s.customModel?.modelId === currentlyLoaded
);
if (conflicting.length > 0) {
// Applying a custom model is ungated for any session owner, so in multi-user
// mode a non-admin pointing their own session at a shared endpoint must not
// learn another user's session names in the confirm dialog — with
// autoNameSessions on, those names are that user's own prompts. The swap is
// still blocked pending confirmation regardless of ownership (a foreign
// session is just as real a disruption); only which ones get NAMED is scoped.
const requestUser = getAuthUser(req);
const affectedSessions = conflicting
.filter((s) => canAccessOwned(requestUser, s.owner))
.map((s) => ({ id: s.id, name: s.name }));
return { requiresConfirmation: true, currentlyLoadedModel: currentlyLoaded, affectedSessions };
}
}
// A CLI whose config alone cannot select the model also gets its `model` launch param
// forced (pi/omp `custom/<id>`, grok's block name). The argv engine DROPS a token that
// fails its pattern rather than quoting it, which would silently launch the CLI on its
// own default provider again, so refuse an id the pattern cannot carry up front.
const modelSpec = entry.launch.params.model;
const applied = applyCustomModelInjection(entry, endpoint, body.modelId, session.id);
const applied = applyCustomModelInjection(
entry,
endpoint,
body.modelId,
session.id,
contextLength,
session.workingDir
);
if (!applied) {
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${session.mode} has no known custom-model mechanism`);
}
@@ -1271,9 +1342,17 @@ export function registerSessionRoutes(
removeConfigDir(previousConfigDir);
}
// Actually kick off llama-swap's load now, rather than waiting on the restarted CLI's
// own first prompt to do it — confirmed live that applying a selection alone never
// reached the llama-swap server at all (nothing in its own logs), since llama-swap has
// no "switch model" admin call, only a real inference request naming the model.
if (swapStatus.isLlamaSwap && !targetReady) {
triggerLlamaSwapLoad(endpoint, body.modelId);
}
const restarted = await session.restartCli();
persistAndBroadcastSession(ctx, session);
return { customModel: session.customModel, restarted };
return { customModel: session.customModel, restarted, modelSwapInProgress: swapStatus.isLlamaSwap && !targetReady };
});
// ========== Delete Session ==========
@@ -3252,6 +3331,7 @@ export function registerSessionRoutes(
effort,
parentSessionId,
agentOrigin,
customModel,
} = parseBody(QuickStartSchema, req.body);
// Resolved ONCE here: the same value labels a case directory this request creates
@@ -3304,11 +3384,12 @@ export function registerSessionRoutes(
grokConfig ||
deepSeekConfig ||
ompConfig ||
openCodeConfig
openCodeConfig ||
customModel
) {
return createErrorResponse(
ApiErrorCode.INVALID_INPUT,
'envOverrides, effort, modelOverride, and per-CLI config are not supported for remote cases (they do not cross ssh). Configure the remote command via the host command override instead.'
'envOverrides, effort, modelOverride, per-CLI config, and custom model endpoints are not supported for remote cases (they do not cross ssh). Configure the remote command via the host command override instead.'
);
}
@@ -3372,11 +3453,12 @@ export function registerSessionRoutes(
grokConfig ||
deepSeekConfig ||
ompConfig ||
openCodeConfig
openCodeConfig ||
customModel
) {
return createErrorResponse(
ApiErrorCode.INVALID_INPUT,
'envOverrides, effort, and per-CLI config are not supported for docker cases (they do not cross into the container). Configure the container via the docker host command override instead.'
'envOverrides, effort, per-CLI config, and custom model endpoints are not supported for docker cases (they do not cross into the container). Configure the container via the docker host command override instead.'
);
}
@@ -3664,7 +3746,148 @@ export function registerSessionRoutes(
);
const qsTerminalHistoryConfig = await ctx.getTerminalHistoryConfig();
const qsGatedEnvOverrides = await clampEnvOverridesForOwner(owner, envOverrides);
const session = new Session({
const qsResolvedOmpConfig = resolveOmpConfigForCreate(mode, resolvedCasePath, ompConfig);
// Custom Model Endpoint Profiles, applied AT CREATE TIME (docs/custom-model-endpoints-plan.md)
// rather than via the dedicated restart-in-place route (POST /api/sessions/:id/custom-
// model, still what an ALREADY-RUNNING session uses to switch later): computing the
// injection before the process exists and launching directly on it avoids the visible
// native-boot-then-restart the restart-after-launch design otherwise shows on every
// custom-model run — most jarring on a CLI like Codex whose TUI fully reinitializes.
// Mirrors the dedicated route's own checks (llama-swap conflict, unsupported CLI,
// unknown endpoint, a model id the CLI's argv pattern can't carry) rather than trusting
// a lighter version of them, since this is the same server-side authority reached a
// different way, not a separate, less-checked path.
let qsCustomModelEnvOverrides = qsGatedEnvOverrides;
// Only the INJECTED keys (never the caller's envOverrides merged in) — this is what
// setCustomModel() bookkeeping must be given below. The Session constructor already
// applies qsCustomModelEnvOverrides (the full merged set) directly; re-merging that
// full set into setCustomModel() would put CLAUDE_CODE_EFFORT_LEVEL back after the
// constructor stripped it (see setCustomModel()'s own doc comment in session.ts).
let qsCustomModelAppliedEnvOverrides: Record<string, string> | undefined;
let qsCustomModelLaunchModel: string | undefined;
let qsCustomModelSessionId: string | undefined;
let qsCustomModelSwapInProgress = false;
let qsCustomModelBookkeeping:
| {
endpointId: string;
modelId: string;
label?: string;
envKeys: string[];
configDir?: string;
launchModel?: string;
}
| undefined;
if (customModel) {
const cmEntry = getCli(mode);
if (!cmEntry) return createErrorResponse(ApiErrorCode.INVALID_INPUT, `No CLI registry entry for mode ${mode}`);
if (cmEntry.capabilities.customModelInjection.kind === 'unsupported') {
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${mode} has no known custom-model mechanism`);
}
const cmHosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
const cmEndpoint = cmHosts.find((h) => h.id === customModel.endpointId);
if (!cmEndpoint) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
const cmContextLength = cmEndpoint.modelContextLengths?.[customModel.modelId];
// See the dedicated route's own comment for the full reasoning: some CLIs' own fixed
// overhead can exceed a small enough real context on the very first message,
// regardless of contextLengthVar. Warn before creating a session that's certain to
// fail immediately.
if (!customModel.confirmed && exceedsSafeContextFloor(cmEntry, cmContextLength)) {
return {
requiresContextWarning: true,
modelId: customModel.modelId,
contextLength: cmContextLength,
minSafeContextTokens: CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
};
}
// See the dedicated route's own comment for the full reasoning: llama.cpp runs one
// model at a time, llama-swap swaps on demand, and switching away from what another
// live session is actively using deserves a warning, not a silent switch. There is no
// "self" to exclude from the affected-sessions scan here — this session doesn't exist
// yet.
const cmSwapStatus = await getLlamaSwapStatus(cmEndpoint);
const cmCurrentlyLoaded =
cmSwapStatus.running.find((r) => r.state === 'ready')?.model ?? cmSwapStatus.running[0]?.model;
const cmSwapNeeded = cmSwapStatus.isLlamaSwap && !!cmCurrentlyLoaded && cmCurrentlyLoaded !== customModel.modelId;
// Broader than cmSwapNeeded (which only gates the confirmation ask above): true
// whenever the TARGET model isn't already loaded and ready, including when nothing
// is loaded at all yet. Drives the actual load trigger below.
const cmTargetReady = cmSwapStatus.running.some((r) => r.model === customModel.modelId && r.state === 'ready');
qsCustomModelSwapInProgress = cmSwapStatus.isLlamaSwap && !cmTargetReady;
if (cmSwapNeeded && !customModel.confirmed) {
const cmConflicting = [...ctx.sessions.values()].filter(
(s) => s.customModel?.endpointId === cmEndpoint.id && s.customModel?.modelId === cmCurrentlyLoaded
);
if (cmConflicting.length > 0) {
// Same reasoning as the dedicated /custom-model route above: the swap is
// still blocked pending confirmation regardless of ownership, but a
// non-admin caller only learns the names of sessions they can access.
const cmRequestUser = getAuthUser(req);
const cmAffectedSessions = cmConflicting
.filter((s) => canAccessOwned(cmRequestUser, s.owner))
.map((s) => ({ id: s.id, name: s.name }));
return {
requiresConfirmation: true,
currentlyLoadedModel: cmCurrentlyLoaded,
affectedSessions: cmAffectedSessions,
};
}
}
// Minted ourselves (rather than left to Session's own default) so the injection
// below — and any configDir it writes — can target the REAL id the session launches
// with, not a placeholder: `new Session({ id: ... })` accepts an explicit id for
// exactly this reason.
qsCustomModelSessionId = randomUUID();
const cmApplied = applyCustomModelInjection(
cmEntry,
cmEndpoint,
customModel.modelId,
qsCustomModelSessionId,
cmContextLength,
resolvedCasePath
);
if (!cmApplied) {
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${mode} has no known custom-model mechanism`);
}
const cmModelSpec = cmEntry.launch.params.model;
if (
cmApplied.launchModel !== undefined &&
cmModelSpec?.type === 'token' &&
!matchesPattern(cmModelSpec.pattern, cmApplied.launchModel)
) {
removeConfigDir(cmApplied.configDir);
return createErrorResponse(
ApiErrorCode.INVALID_INPUT,
`Model id ${JSON.stringify(customModel.modelId)} cannot be passed to ${mode} on its command line`
);
}
qsCustomModelEnvOverrides = { ...qsGatedEnvOverrides, ...cmApplied.envOverrides };
qsCustomModelAppliedEnvOverrides = cmApplied.envOverrides;
qsCustomModelLaunchModel = cmApplied.launchModel;
qsCustomModelBookkeeping = {
endpointId: cmEndpoint.id,
modelId: customModel.modelId,
label: cmEndpoint.label,
envKeys: cmApplied.envKeys,
configDir: cmApplied.configDir,
launchModel: cmApplied.launchModel,
};
// Actually kick off llama-swap's load now — see the dedicated apply route's own
// comment on triggerLlamaSwapLoad for why this can't just wait on the launched CLI's
// first prompt. Fired here, before the session is even created, so the load starts
// concurrently with Claude/Codex/etc. booting rather than after.
if (qsCustomModelSwapInProgress) {
triggerLlamaSwapLoad(cmEndpoint, customModel.modelId);
}
}
const qsSessionOptions: ConstructorParameters<typeof Session>[0] = {
id: qsCustomModelSessionId,
workingDir: resolvedCasePath,
name: sessionName ? sessionName.slice(0, MAX_SESSION_NAME_LENGTH) : '',
mux: ctx.mux,
@@ -3682,15 +3905,42 @@ export function registerSessionRoutes(
piConfig: mode === 'pi' ? qsGatedPiConfig : undefined,
grokConfig: mode === 'grok' ? qsGatedGrokConfig : undefined,
deepSeekConfig: mode === 'deepseek' ? qsGatedDeepSeekConfig : undefined,
ompConfig: resolveOmpConfigForCreate(mode, resolvedCasePath, ompConfig),
envOverrides: qsGatedEnvOverrides,
ompConfig: qsResolvedOmpConfig,
envOverrides: qsCustomModelEnvOverrides,
effort,
remote,
docker,
resumeSessionId: dockerResumeId,
tmuxHistoryLimit: qsTerminalHistoryConfig.tmuxHistoryLimit,
parentSessionId: qsParentSessionId,
});
};
// Force the custom-model selection's launchModel (pi/omp `custom/<id>`, grok's
// `[model.<name>]` block name) onto whichever config field the registry says the
// CLI's `model` launch param lives in — mirrors Session._withCustomModelLaunchModel,
// which the restart-in-place path already uses, rather than a hardcoded per-CLI
// branch here that a CLI landing its injection recipe later would silently miss.
if (qsCustomModelLaunchModel !== undefined) {
const qsCustomModelField = getCli(mode)?.launch.legacyConfigField;
if (qsCustomModelField) {
const qsSessionOptionsBag = qsSessionOptions as unknown as Record<string, unknown>;
qsSessionOptionsBag[qsCustomModelField] = {
...((qsSessionOptionsBag[qsCustomModelField] as Record<string, unknown>) ?? {}),
model: qsCustomModelLaunchModel,
};
} else {
qsSessionOptions.model = qsCustomModelLaunchModel;
}
}
const session = new Session(qsSessionOptions);
// Records the selection for session.customModel/getCustomModelForPersist() and future
// clear/switch calls — the actual env vars and launch-model config are already part of
// the launch above (constructor envOverrides, piConfig/grokConfig/ompConfig.model), so
// this is bookkeeping only, never a restart: setCustomModel() is synchronous state, no
// tmux IO of its own (see its own doc comment in session.ts).
if (qsCustomModelBookkeeping) {
session.setCustomModel(qsCustomModelBookkeeping, qsCustomModelAppliedEnvOverrides);
}
// Auto-detect completion phrase from CLAUDE.md BEFORE broadcasting
// so the initial state already has the phrase configured (only if globally enabled)
@@ -3786,6 +4036,7 @@ export function registerSessionRoutes(
sessionId: session.id,
casePath: resolvedCasePath,
caseName,
...(customModel ? { modelSwapInProgress: qsCustomModelSwapInProgress } : {}),
};
} catch (err) {
// Clean up session on error to prevent orphaned resources
+34
View File
@@ -1064,6 +1064,25 @@ export const QuickStartSchema = z.object({
* because it takes an existing `workingDir` and so never creates a directory to label.
*/
agentOrigin: z.string().max(64).optional(),
/**
* Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md): launches directly
* on this saved endpoint/model instead of the mode's native backend, computed server-side
* from the admin-configured endpoint store the same way `POST /api/sessions/:id/custom-
* model` does — never trusting raw env values from the client. One-shot, launch-time
* equivalent of that route: no restart, so no visible relaunch (that route's restart-in-
* place is still what an ALREADY-RUNNING session uses to switch later). Rejected for
* remote/docker cases, same reasoning as `envOverrides` above. `confirmed` mirrors that
* route's field: skips the llama-swap "this will unload it for another session" check on
* a deliberate retry.
*/
customModel: z
.object({
endpointId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid endpoint id'),
modelId: z.string().min(1).max(200),
confirmed: z.boolean().optional(),
})
.strict()
.optional(),
});
// ========== Hook Events ==========
@@ -1963,6 +1982,17 @@ export const CustomModelHostSchema = z.object({
authStyle: z.enum(['bearer', 'api-key']).optional(),
models: z.array(z.string().max(200)).max(200).optional(),
lastDiscoveredAt: z.string().max(64).optional(),
// The Run-menu picker's per-endpoint default; validated against `models` at the
// route layer (schema-level cross-field checks can't see the array narrowed the
// same way a `.refine()` closure could, and the route already re-reads the stored
// host to apply it, so the check belongs there once, not duplicated into a refine
// that would run on every unrelated field edit too).
defaultModelId: z.string().max(200).optional(),
// Server-populated by discovery (custom-model-routes.ts); accepted here only so a client
// round-tripping the GET response back through PUT (edit-save) doesn't drop it.
modelContextLengths: z.record(z.string().max(200), z.number().int().positive().max(100_000_000)).optional(),
// Same reasoning as modelContextLengths above.
modelSizesGB: z.record(z.string().max(200), z.number().positive().max(100_000)).optional(),
});
/** POST /api/sessions/:id/custom-model — apply or clear a session's custom-model selection. */
@@ -1970,6 +2000,10 @@ export const CustomModelSelectionSchema = z.union([
z.object({
endpointId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid endpoint id'),
modelId: z.string().min(1).max(200),
// Set once the caller has already shown the "this will unload <model> for session(s)
// X" warning (see session-routes.ts's llama-swap conflict check) and the user chose to
// proceed anyway — skips that check on this call instead of asking again.
confirmed: z.boolean().optional(),
}),
z.object({ clear: z.literal(true) }),
]);
+115 -2
View File
@@ -71,7 +71,7 @@ import {
import { imageWatcher } from '../image-watcher.js';
import { workflowRunWatcher, summarizeRun } from '../workflow-run-watcher.js';
import { attachmentRegistry, buildFileThumbnailRoute, registerExternalAttachment } from '../attachment-registry.js';
import { getCli } from '../config/cli-registry/registry.js';
import { getCli, enabledClis } from '../config/cli-registry/registry.js';
import { readCustomModelHosts } from '../custom-model-hosts.js';
import { applyCustomModelInjection, customModelConfigDir, removeConfigDir } from '../custom-model-injection-apply.js';
import type { CustomModelBookkeeping } from '../types/session.js';
@@ -195,6 +195,10 @@ import {
registerWebviewRoutes,
registerTabLayoutRoutes,
registerCustomModelRoutes,
refreshAllCustomModelHosts,
readCustomModelEndpointsEnabled,
detectCustomModelSwapDisplacements,
pruneIdleLlamaSwapLogTails,
tryWebviewRefererFallback,
} from './routes/index.js';
import { isLostWebviewFrameNavigation } from './webview-proxy.js';
@@ -207,11 +211,32 @@ const __dirname = dirname(fileURLToPath(import.meta.url));
// while capping growth of `sseClientsById` and blocking pathological inputs.
const SSE_CLIENT_ID_RE = /^[A-Za-z0-9_-]{8,64}$/;
const CODEX_USAGE_POLL_INTERVAL_MS = 5 * 60_000;
const CUSTOM_MODEL_REDISCOVER_INTERVAL_MS = 5 * 60_000;
// Much shorter than the model-LIST refresh above on purpose: this catches an actual
// eviction (a session's model no longer loaded, silently swapped out by another
// session's use), which the user wants to know about promptly, not once every 5
// minutes. Cheap either way — one /running GET per distinct endpoint with at least
// one live custom-model session, not per session.
const CUSTOM_MODEL_SWAP_CHECK_INTERVAL_MS = 20_000;
function escapeHtmlText(value: string): string {
return value.replaceAll('&', '&amp;').replaceAll('<', '&lt;').replaceAll('>', '&gt;');
}
/**
* Escapes a JSON string for safe embedding as the body of an inline `<script>`
* tag: `<` becomes the six-character sequence `<`, which both a JSON
* parser and a plain JS string literal decode back to `<` (both treat
* `\uXXXX` identically), but which can never itself form the two literal
* characters `<` `/` a browser's HTML tokenizer looks for to end the tag. A
* value containing a literal `</script>` would otherwise close the tag early
* and turn the rest of the document into inert script-body text. Exported so
* it unit-tests without constructing a WebServer (which needs a real tmux).
*/
export function escapeScriptJson(json: string): string {
return json.replace(/</g, '\\u003c');
}
import {
SESSIONS_LIST_CACHE_TTL,
SCHEDULED_CLEANUP_INTERVAL,
@@ -270,6 +295,8 @@ export class WebServer extends EventEmitter {
// Store session listener references for explicit cleanup (prevents memory leaks)
private sessionListenerRefs: Map<string, SessionListenerRefs> = new Map();
private scheduledRuns: Map<string, ScheduledRun> = new Map();
/** De-dupe state for the swap-displacement sweep — see detectCustomModelSwapDisplacements. */
private _customModelDisplacedNotified: Set<string> = new Set();
/** Cron service (assigned in setupRoutes). */
private cronService!: CronService;
/**
@@ -1224,7 +1251,13 @@ export class WebServer extends EventEmitter {
return undefined;
}
try {
return applyCustomModelInjection(entry, endpoint, saved.modelId, session.id)?.envOverrides;
return applyCustomModelInjection(
entry,
endpoint,
saved.modelId,
session.id,
endpoint.modelContextLengths?.[saved.modelId]
)?.envOverrides;
} catch (err) {
console.warn('[WebServer] Failed to rebuild custom-model env on recovery:', err);
return undefined;
@@ -1319,6 +1352,10 @@ export class WebServer extends EventEmitter {
session.ralphTracker.stopWatchingFixPlan();
}
// Custom Model Endpoint Profiles: drop this session's swap-displacement notify flag
// (see _checkCustomModelSwapDisplacements below) so it can't linger in that Set forever.
this._customModelDisplacedNotified.delete(sessionId);
// Kill all subagents spawned by this session (scoped to sessionId to avoid cross-session kills)
if (session && killMux) {
try {
@@ -1619,6 +1656,23 @@ export class WebServer extends EventEmitter {
'</head>',
`<script>window.__codemanCliAvailable=${JSON.stringify(available)};</script>\n</head>`
);
// Which run modes the Run-menu picker (docs/custom-model-endpoints-plan.md) may
// generate an entry for: read generically off the registry's `capabilities`
// (never an id list here) so a CLI whose customModelInjection lands later shows
// up in the picker with no frontend change, and one that ships `unsupported`
// (antigravity, and `shell`'s `kind !== 'agent'`) never does.
const customModelClis = enabledClis()
.filter((entry) => entry.kind === 'agent' && entry.capabilities.customModelInjection.kind !== 'unsupported')
.map((entry) => ({ id: entry.id, label: entry.label }));
// Unlike the boolean-only __codemanCliAvailable above, this payload carries
// `label`, a string a user's own clis.json can set (CliEntry.label, up to 60
// chars) — see escapeScriptJson's own doc comment for why that needs escaping
// and __codemanCliAvailable's booleans never did.
const customModelClisJson = escapeScriptJson(JSON.stringify(customModelClis));
html = html.replace(
'</head>',
`<script>window.__codemanCustomModelClis=${customModelClisJson};</script>\n</head>`
);
}
if (!soloSessionId && process.env.CODEMAN_GESTURE === '1') {
html = html.replace('</head>', `<script>window.__codemanGestureAvailable=true;</script>\n</head>`);
@@ -2350,6 +2404,7 @@ export class WebServer extends EventEmitter {
'team:',
'case:',
'remote:',
'custom-model:',
];
if (SESSION_PREFIXES.some((p) => event.startsWith(p))) {
const d = (data ?? {}) as { sessionId?: string; id?: string; session?: { id?: string }; username?: string };
@@ -2734,6 +2789,64 @@ export class WebServer extends EventEmitter {
});
}
// Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md): keeps
// each saved endpoint's discovered model list current with no manual
// "Discover" click, so a model added on the server side (or one that drops
// off) shows up in the Run-menu picker within one cycle. Best-effort per
// endpoint (refreshAllCustomModelHosts skips one that's unreachable rather
// than failing the sweep) and off in tests for the same reason the Codex
// poll above is — no real network to hit, no server instance to keep alive.
if (!this.testMode) {
this.cleanup.setInterval(
() => {
// Reads the setting fresh on every tick, same reasoning as
// readPlanUsageTelemetryEnabled() beside it: a live toggle takes effect
// on the very next cycle, not just at server boot, and turning the
// feature off actually stops the polling instead of only hiding the UI.
void readCustomModelEndpointsEnabled()
.then((enabled) => {
if (!enabled) return;
return refreshAllCustomModelHosts();
})
.catch((err) => {
console.error('[custom-model] periodic re-discovery failed:', getErrorMessage(err));
});
},
CUSTOM_MODEL_REDISCOVER_INTERVAL_MS,
{ description: 'custom model endpoint re-discovery' }
);
}
// Custom Model Endpoint Profiles: the swap-conflict check on the apply/create routes
// only ever runs at THAT session's own launch/apply moment — it cannot catch a LATER
// eviction triggered by a different session's normal use, since llama-swap has no push
// notification of its own and only swaps in response to a real inference request
// (confirmed live: a session created while nothing else conflicted at that instant can
// still get silently displaced afterward). This periodic sweep is what catches that
// case after the fact and tells the displaced session's user, rather than leaving them
// to discover it only when their next prompt behaves unexpectedly.
if (!this.testMode) {
this.cleanup.setInterval(
() => {
detectCustomModelSwapDisplacements(this.sessions.values(), this._customModelDisplacedNotified)
.then((displacements) => {
for (const displacement of displacements) {
this.broadcast(SseEvent.CustomModelSwappedOut, displacement);
}
})
.catch((err) => {
console.error('[custom-model] swap-displacement check failed:', getErrorMessage(err));
});
// Same cadence, unrelated concern: close any /logs tail (see
// getLatestLlamaSwapLogLine) nothing has polled in a while, so a loading banner
// that finished (or was abandoned) doesn't leave a connection open forever.
pruneIdleLlamaSwapLogTails();
},
CUSTOM_MODEL_SWAP_CHECK_INTERVAL_MS,
{ description: 'custom model swap-displacement check' }
);
}
// Start scheduled runs cleanup timer
this.cleanup.setInterval(
() => {
+18 -1
View File
@@ -5,7 +5,7 @@
* and referenced by the frontend (`SSE_EVENTS` in `constants.js`).
* Both files MUST be kept in sync.
*
* 160 event constants organized by category:
* 161 event constants organized by category:
* - **Core** (1): init
* - **Transport** (1): sse:heartbeat
* - **Session lifecycle** (23): created, updated, deleted, terminal, idle, working, ...
@@ -28,6 +28,7 @@
* - **Hooks** (10): idle_prompt, permission_prompt, elicitation_dialog, elicitation_complete, elicitation_response, stop, agent_working, teammate_idle, task_completed, prompt_submitted
* (agent_working is the odd one out: reported by the DeepSeek Harness status bridge, not by a Claude Code hook)
* - **Approvals** (3): pending, updated, resolved (cross-session Approvals Inbox)
* - **Custom Model Endpoint Profiles** (1): swapped-out (a session's model got evicted by another session on the same llama-swap endpoint)
* - **Orchestrator** (12): stateChanged, planProgress, planReady, phase*, verification, task*, completed, error
* - **Clipboard** (1): write
* - **Cases** (4): created, linked, deleted, order-changed
@@ -395,6 +396,19 @@ export const ApprovalUpdated = 'approval:updated' as const;
/** A pending approval left the inbox (answered, superseded, expired, ...). */
export const ApprovalResolved = 'approval:resolved' as const;
// ─── Custom Model Endpoint Profiles ──────────────────────────────────────────
/**
* A session's own custom-model selection is no longer the model llama-swap has loaded —
* ANOTHER session's activity on the same endpoint evicted it (llama.cpp/llama-swap runs
* one model at a time). Detected after the fact by a periodic sweep (`server.ts`), never
* at the moment of eviction itself, since llama-swap has no push notification of its own;
* this session's next prompt will trigger reloading its model, evicting whatever displaced
* it in turn. Fires at most once per displacement (cleared once the sweep sees the
* session's own model loaded again), so it can't spam on every sweep interval.
*/
export const CustomModelSwappedOut = 'custom-model:swapped-out' as const;
// ─── Orchestrator ────────────────────────────────────────────────────────────
/** Orchestrator state machine transitioned. */
@@ -651,6 +665,9 @@ export const SseEvent = {
ApprovalUpdated,
ApprovalResolved,
// Custom Model Endpoint Profiles
CustomModelSwappedOut,
// Orchestrator
OrchestratorStateChanged,
OrchestratorPlanProgress,