mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-03 05:59:43 +02:00
feat(custom-model): Custom Model Endpoint Profiles (local or cloud, all harnesses)
Point any Codeman-supported harness (Claude, opencode, Codex, Gemini, Pi, Grok, DeepSeek, OMP) at a custom OpenAI-compatible endpoint instead of its native cloud backend, for a given session. Covers local hardware (llama.cpp, Ollama, vLLM, DGX Spark, Strix Halo) and cloud (Azure AI Foundry, OpenRouter). Off by default (customModelEndpointsEnabled, synced, default OFF). - Registry: capabilities.customModelInjection per CLI entry (env / configContentEnv / configDir / unsupported kinds) - Pure injection builder (custom-model-injection.ts) turning an endpoint + model id into the real env vars / config content per CLI - Endpoint store + CRUD routes (custom-model-hosts.ts, custom-model-routes.ts), discovery via GET /v1/models, SSRF-guarded - Session integration: Session.setCustomModel()/restartCli() (POST /api/sessions/:id/custom-model), reusing the existing respawn-pane -k primitive to restart the CLI process with new env - Multi-user hardening: every new redirect-capable env var added to its CLI's privilegedEnvKeys, closing a pre-existing gap where several were already reachable via the generic envOverrides field's prefix allowlist - Standalone scripts/test-local-llm-harnesses.mjs: spawns real CLI binaries against a real endpoint outside the web UI, independent of tmux/sessions - Mock-server contract tests (test/fixtures/mock-openai-server.ts) replaying every CLI's injected values through a real HTTP shape Real end-to-end validation against a live llama-swap server (inside a codeman/agent:llm-test Docker image with all 9 CLI binaries) found and fixed three real bugs before they shipped: - Codex's config.toml schema was wrong ([model].default table instead of a top-level model string + [model_providers.custom]); fixing it then surfaced a genuine, documented protocol incompatibility (Codex only speaks the Responses API since Feb 2026, which llama.cpp/llama-swap don't implement) - Claude Code's async session-title-generation call validates ANTHROPIC_DEFAULT_HAIKU_MODEL against its own internal model list and hangs the whole -p invocation on an unrecognized name; documented for chunk 6, worked around in the standalone script only (--bare is NOT safe for a real interactive session, which needs hooks) - The discovery route's authStyle: 'both' option (send both Authorization and api-key headers) reliably hung a real server; removed the option entirely rather than just changing the default Status: draft. Chunk 6 (frontend toolbar/settings UI) not yet built — see PR.md and deployment_plan.md for the full chunk breakdown and confidence table. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017HqNWfmtBU2KN29SvSVWB3
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
a017e9a8e0
commit
41416566aa
@@ -310,6 +310,34 @@ const capabilitiesSchema = z
|
||||
privilegedEnvKeys: z.array(envName).max(8),
|
||||
gates: z.record(z.string(), z.object({ minVersion: z.string().max(20), failClosed: z.boolean() }).strict()),
|
||||
maxFrameBytes: z.number().int().positive().optional(),
|
||||
customModelInjection: z.discriminatedUnion('kind', [
|
||||
z
|
||||
.object({
|
||||
kind: z.literal('env'),
|
||||
baseUrlVar: envName,
|
||||
apiKeyVar: envName,
|
||||
// Empty is valid: deepseek's model routing is a profile-composition concern, not
|
||||
// an env var, so it declares baseUrl/apiKey injection with no model var at all.
|
||||
modelVars: z.array(envName).max(8),
|
||||
})
|
||||
.strict(),
|
||||
z
|
||||
.object({
|
||||
kind: z.literal('configContentEnv'),
|
||||
envVar: envName,
|
||||
template: z.literal('opencode-json'),
|
||||
})
|
||||
.strict(),
|
||||
z
|
||||
.object({
|
||||
kind: z.literal('configDir'),
|
||||
dirEnvVar: envName,
|
||||
fileName: z.string().min(1).max(80),
|
||||
template: z.enum(['codex-toml', 'pi-models-json', 'omp-models-yml']),
|
||||
})
|
||||
.strict(),
|
||||
z.object({ kind: z.literal('unsupported') }).strict(),
|
||||
]),
|
||||
})
|
||||
.strict();
|
||||
|
||||
|
||||
@@ -178,6 +178,12 @@ const CLAUDE: CliEntry = {
|
||||
unset: ['CLAUDECODE', 'COLORTERM'],
|
||||
tmuxSetenvKeys: [],
|
||||
dockerExecEnvNames: [],
|
||||
// Deliberately excludes ANTHROPIC_* (base URL / API key / default-model overrides):
|
||||
// custom-model-injection.ts's claude recipe uses those names, but they must reach a
|
||||
// session ONLY through the admin-configured, SSRF-guarded custom-model route, never
|
||||
// through a plain client-supplied envOverrides field. Widening this prefix would let
|
||||
// any session-create caller redirect a session's Anthropic traffic and credentials to
|
||||
// an arbitrary, unvalidated URL.
|
||||
allowedPrefixes: ['CLAUDE_CODE_'],
|
||||
allowedKeys: ['CLAUDE_CONFIG_DIR'],
|
||||
},
|
||||
@@ -209,8 +215,28 @@ const CLAUDE: CliEntry = {
|
||||
statusLineTelemetry: true,
|
||||
model: { source: 'claude-settings-file' },
|
||||
privilegedParams: [],
|
||||
privilegedEnvKeys: [],
|
||||
// ANTHROPIC_* is NOT in allowedPrefixes/allowedKeys above (deliberately — see the
|
||||
// allowedPrefixes comment nearby), so these are unreachable via plain envOverrides
|
||||
// today; listed here only so the dedicated custom-model route (deployment_plan.md
|
||||
// chunk 5) clamps them for a non-granted multi-user owner the same way every other
|
||||
// CLI's injection vars are clamped, the day that route widens who can set them.
|
||||
privilegedEnvKeys: [
|
||||
'ANTHROPIC_BASE_URL',
|
||||
'ANTHROPIC_API_KEY',
|
||||
'ANTHROPIC_DEFAULT_SONNET_MODEL',
|
||||
'ANTHROPIC_DEFAULT_HAIKU_MODEL',
|
||||
'ANTHROPIC_DEFAULT_OPUS_MODEL',
|
||||
],
|
||||
gates: { nameFlag: { minVersion: '2.1.224', failClosed: true } },
|
||||
// Custom Model Endpoint Profiles (deployment_plan.md) — verified by hand against a real
|
||||
// llama.cpp server. Claude reads these at process start only, so switching requires a
|
||||
// respawn, never a live hot-swap.
|
||||
customModelInjection: {
|
||||
kind: 'env',
|
||||
baseUrlVar: 'ANTHROPIC_BASE_URL',
|
||||
apiKeyVar: 'ANTHROPIC_API_KEY',
|
||||
modelVars: ['ANTHROPIC_DEFAULT_SONNET_MODEL', 'ANTHROPIC_DEFAULT_HAIKU_MODEL', 'ANTHROPIC_DEFAULT_OPUS_MODEL'],
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
// Mirrors the local default so the remote/in-container agent runs non-interactively
|
||||
@@ -276,6 +302,7 @@ const SHELL: CliEntry = {
|
||||
privilegedParams: [],
|
||||
privilegedEnvKeys: [],
|
||||
gates: {},
|
||||
customModelInjection: { kind: 'unsupported' }, // a raw shell has no "model" concept
|
||||
},
|
||||
overlays: {
|
||||
// No `remote` entry: defaultRemoteCommandForMode special-cases kind==='shell' directly
|
||||
@@ -355,6 +382,15 @@ const OPENCODE: CliEntry = {
|
||||
...agentDefaults(),
|
||||
altScreen: 'strip-mux-only',
|
||||
echo: { policy: 'buffer', anchor: { kind: 'cursor' }, predictProfile: undefined },
|
||||
// Verified by hand against a real llama.cpp server. Reuses the SAME env var opencode's
|
||||
// own `env.configContentVar` already declares — the builder in custom-model-injection.ts
|
||||
// must merge into whatever opencode config Codeman would otherwise send, not clobber it.
|
||||
customModelInjection: { kind: 'configContentEnv', envVar: 'OPENCODE_CONFIG_CONTENT', template: 'opencode-json' },
|
||||
// OPENCODE_CONFIG_CONTENT already matches the OPENCODE_ allowedPrefix above, so it was
|
||||
// ALREADY reachable via plain envOverrides before this feature existed — it replaces
|
||||
// opencode's whole config, provider api keys included, so a non-granted multi-user owner
|
||||
// sending it is a pre-existing credential-redirection gap, not one this feature opens.
|
||||
privilegedEnvKeys: ['OPENCODE_CONFIG_CONTENT'],
|
||||
},
|
||||
overlays: {
|
||||
credStore: { rel: '.config/opencode', seedWhole: true },
|
||||
@@ -444,6 +480,23 @@ const CODEX: CliEntry = {
|
||||
// `dangerouslyBypassApprovals` on the wire), so it is the one that would have caught a
|
||||
// regression; `schema.ts` now rejects a name that is not a declared param.
|
||||
privilegedParams: [{ param: 'bypassApprovals', clampTo: false }],
|
||||
// Verified by hand against a real llama.cpp server. Written to an isolated CODEX_HOME
|
||||
// so the user's real ~/.codex/config.toml is never touched.
|
||||
customModelInjection: {
|
||||
kind: 'configDir',
|
||||
dirEnvVar: 'CODEX_HOME',
|
||||
fileName: 'config.toml',
|
||||
template: 'codex-toml',
|
||||
},
|
||||
// CODEX_HOME already matches the CODEX_ allowedPrefix above, so it was ALREADY
|
||||
// reachable via plain envOverrides before this feature existed. It is arguably
|
||||
// MORE sensitive than a bare base-url var: a redirected CODEX_HOME points codex at a
|
||||
// config.toml a non-granted owner fully controls, which can restate sandbox/approval
|
||||
// policy INSIDE that file — a path the argv-level `bypassApprovals` clamp above
|
||||
// cannot see or stop.
|
||||
// CODEMAN_CUSTOM_MODEL_API_KEY: the credential config.toml's env_key references
|
||||
// (see custom-model-injection.ts) — same reasoning as CODEX_HOME above.
|
||||
privilegedEnvKeys: ['CODEX_HOME', 'CODEMAN_CUSTOM_MODEL_API_KEY'],
|
||||
},
|
||||
overlays: {
|
||||
credStore: {
|
||||
@@ -527,6 +580,20 @@ const GEMINI: CliEntry = {
|
||||
// MATERIALIZE a config (not just touch an already-sent one) or a non-granted owner who
|
||||
// sends no geminiConfig at all would still get yolo for free.
|
||||
privilegedParams: [{ param: 'approvalMode', clampTo: 'auto_edit', materializeWhenAbsent: true }],
|
||||
// Web-researched, unverified — needs a restart to pick up (CLI reads these at process
|
||||
// start). Confirm the exact model-override env var name against the installed
|
||||
// gemini-cli version before shipping.
|
||||
customModelInjection: {
|
||||
kind: 'env',
|
||||
baseUrlVar: 'GOOGLE_GEMINI_BASE_URL',
|
||||
apiKeyVar: 'GEMINI_API_KEY',
|
||||
modelVars: ['GEMINI_MODEL'],
|
||||
},
|
||||
// All three already match the GEMINI_/GOOGLE_ allowedPrefixes above, so they were
|
||||
// ALREADY reachable via plain envOverrides before this feature existed — a non-granted
|
||||
// multi-user owner redirecting a gemini session's endpoint/credentials is a
|
||||
// pre-existing gap this feature's analysis surfaced, not one it opens.
|
||||
privilegedEnvKeys: ['GOOGLE_GEMINI_BASE_URL', 'GEMINI_API_KEY', 'GEMINI_MODEL'],
|
||||
},
|
||||
overlays: {
|
||||
credStore: { rel: '.gemini', seedWhole: true }, // also covers antigravity — see its own entry
|
||||
@@ -592,6 +659,10 @@ const ANTIGRAVITY: CliEntry = {
|
||||
// Like codex: an ABSENT config already defaults safe (no bypass flag), so only a
|
||||
// SENT config needs the flag forced off — nothing is materialized.
|
||||
privilegedParams: [{ param: 'dangerouslySkipPermissions', clampTo: false }],
|
||||
// No known CLI/env/config mechanism — Antigravity's own docs describe a GUI-only
|
||||
// custom-endpoint setting and explicitly say it "cannot currently" become the core
|
||||
// reasoning model. Toolbar entry stays disabled for this mode.
|
||||
customModelInjection: { kind: 'unsupported' },
|
||||
},
|
||||
overlays: {
|
||||
// No credStore of its own: agy nests its whole state under ~/.gemini/antigravity-cli/,
|
||||
@@ -678,6 +749,20 @@ const PI: CliEntry = {
|
||||
// just answer "yes" to, so omitting --approve is not itself a clamp — MATERIALIZE
|
||||
// approveProjectTrust:false so buildPiCommand emits --no-approve outright.
|
||||
privilegedParams: [{ param: 'approveProjectTrust', clampTo: false, materializeWhenAbsent: true }],
|
||||
// Web-researched, unverified. pi's models.json hot-reloads, but this feature always
|
||||
// restarts the CLI on switch for consistency with the other 8 harnesses. Written to an
|
||||
// isolated PI_CONFIG_DIR so the user's real ~/.pi/agent/models.json is never touched.
|
||||
customModelInjection: {
|
||||
kind: 'configDir',
|
||||
dirEnvVar: 'PI_CONFIG_DIR',
|
||||
fileName: 'agent/models.json',
|
||||
template: 'pi-models-json',
|
||||
},
|
||||
// PI_CONFIG_DIR already matches the PI_ allowedPrefix above, so it was ALREADY
|
||||
// reachable via plain envOverrides before this feature existed — and pi executes
|
||||
// repo-local .pi/extensions TypeScript (see the External CLI modes note in CLAUDE.md),
|
||||
// so redirecting this dir is a code-execution surface, not just a config swap.
|
||||
privilegedEnvKeys: ['PI_CONFIG_DIR'],
|
||||
},
|
||||
overlays: {
|
||||
credStore: {
|
||||
@@ -773,6 +858,16 @@ const GROK: CliEntry = {
|
||||
// already its safe interactive ask-mode, so the multi-user clamp only needs to force an
|
||||
// EXPLICITLY-SENT bypass flag back off — nothing is materialized when config is absent.
|
||||
privilegedParams: [{ param: 'alwaysApprove', clampTo: false }],
|
||||
// Web-researched, unverified.
|
||||
customModelInjection: {
|
||||
kind: 'env',
|
||||
baseUrlVar: 'GROK_BASE_URL',
|
||||
apiKeyVar: 'XAI_API_KEY',
|
||||
modelVars: ['GROK_MODEL'],
|
||||
},
|
||||
// All three already match the GROK_/XAI_ allowedPrefixes above, so they were ALREADY
|
||||
// reachable via plain envOverrides before this feature existed.
|
||||
privilegedEnvKeys: ['GROK_BASE_URL', 'XAI_API_KEY', 'GROK_MODEL'],
|
||||
},
|
||||
overlays: {
|
||||
// ~/.grok also holds sessions/, memory/, downloads/ (the ~160MB binary), completions/,
|
||||
@@ -924,7 +1019,19 @@ const DEEPSEEK: CliEntry = {
|
||||
// The half no other CLI needs. `DSH_*` is an allowlisted envOverrides prefix and
|
||||
// applyEnvOverrides() runs LAST, so without this a non-granted owner could send
|
||||
// DSH_PERMISSION_MODE on the same request and land after the config clamp.
|
||||
privilegedEnvKeys: ['DSH_PERMISSION_MODE', 'DSH_HOME', 'DEEPSEEK_BASE_URL'],
|
||||
// DEEPSEEK_API_KEY added alongside DEEPSEEK_BASE_URL for the custom-model-injection.ts
|
||||
// recipe (deployment_plan.md) — the pair travels together, same reasoning as base URL.
|
||||
privilegedEnvKeys: ['DSH_PERMISSION_MODE', 'DSH_HOME', 'DEEPSEEK_BASE_URL', 'DEEPSEEK_API_KEY'],
|
||||
// Web-researched, unverified, partial: reuses the already-existing DEEPSEEK_BASE_URL/
|
||||
// DEEPSEEK_API_KEY keys above. No modelVars — dsh's model is a profile-composition
|
||||
// entry (see `model: { source: 'none' }` above), not an env var, so forcing a specific
|
||||
// model name may not fully work; verify against a real profile before shipping.
|
||||
customModelInjection: {
|
||||
kind: 'env',
|
||||
baseUrlVar: 'DEEPSEEK_BASE_URL',
|
||||
apiKeyVar: 'DEEPSEEK_API_KEY',
|
||||
modelVars: [],
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
// No credStore: dsh keeps everything under $DSH_HOME (default ~/.dsh), which is
|
||||
@@ -1026,7 +1133,20 @@ const OMP: CliEntry = {
|
||||
// Where omp resolves its auth from. No known concrete exfiltration path today (omp
|
||||
// forwards no operator-held key into a pane), but a non-granted owner redirecting where
|
||||
// a shared multi-tenant deployment resolves auth is not something to allow silently.
|
||||
privilegedEnvKeys: ['OMP_AUTH_BROKER_URL', 'OMP_AUTH_BROKER_TOKEN'],
|
||||
// PI_CONFIG_DIR added for custom-model-injection.ts's omp recipe, which reuses pi's
|
||||
// dir-redirect mechanism (see the customModelInjection comment below) — already
|
||||
// reachable via the PI_ allowedPrefix (pi's own entry), so this closes the same
|
||||
// pre-existing gap for an omp session that PI's own entry closes for a pi session.
|
||||
privilegedEnvKeys: ['OMP_AUTH_BROKER_URL', 'OMP_AUTH_BROKER_TOKEN', 'PI_CONFIG_DIR'],
|
||||
// Web-researched, unverified. omp's ~/.omp tree is itself relocatable via PI_CONFIG_DIR
|
||||
// (see the DeepSeek/OMP note in CLAUDE.md), so this reuses that same redirect rather
|
||||
// than inventing an OMP-specific dir env var.
|
||||
customModelInjection: {
|
||||
kind: 'configDir',
|
||||
dirEnvVar: 'PI_CONFIG_DIR',
|
||||
fileName: 'agent/models.yml',
|
||||
template: 'omp-models-yml',
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
// `~/.omp/agent` also holds agent.db/history.db/models.db (SQLite caches) and
|
||||
|
||||
@@ -441,6 +441,37 @@ export interface CliCapabilities {
|
||||
gates: Record<string, { minVersion: string; failClosed: boolean }>;
|
||||
/** Cap on a single terminal frame, when this CLI needs a tighter one than the default. */
|
||||
maxFrameBytes?: number;
|
||||
/**
|
||||
* How this CLI is pointed at a user-supplied custom OpenAI-compatible
|
||||
* endpoint (local, e.g. llama.cpp, or cloud, e.g. Azure AI Foundry) — the
|
||||
* Custom Model Endpoint Profiles feature (`deployment_plan.md`). Declared
|
||||
* per entry, never branched on id, same as every other capability here.
|
||||
*
|
||||
* `env`: plain env vars (claude's `ANTHROPIC_BASE_URL`/`ANTHROPIC_API_KEY`/
|
||||
* `ANTHROPIC_DEFAULT_*_MODEL`). `configContentEnv`: a full config blob
|
||||
* carried in one env var (opencode's `OPENCODE_CONFIG_CONTENT`).
|
||||
* `configDir`: a generated config file under an isolated, dir-redirect-env-
|
||||
* pointed directory so the user's real CLI config is never touched
|
||||
* (codex's `CODEX_HOME`/`config.toml`, pi/omp's `PI_CONFIG_DIR`).
|
||||
* `unsupported`: no known mechanism (antigravity) — the toolbar entry
|
||||
* stays disabled for this CLI.
|
||||
*
|
||||
* Every env var name this introduces that can redirect a session's
|
||||
* traffic MUST also appear in `privilegedEnvKeys` above, exactly like
|
||||
* `DEEPSEEK_BASE_URL` — a non-granted multi-user owner redirecting a
|
||||
* session to their own endpoint is a credential-exfiltration path, not
|
||||
* just a mischief redirect.
|
||||
*/
|
||||
customModelInjection:
|
||||
| { kind: 'env'; baseUrlVar: string; apiKeyVar: string; modelVars: string[] }
|
||||
| { kind: 'configContentEnv'; envVar: string; template: 'opencode-json' }
|
||||
| {
|
||||
kind: 'configDir';
|
||||
dirEnvVar: string;
|
||||
fileName: string;
|
||||
template: 'codex-toml' | 'pi-models-json' | 'omp-models-yml';
|
||||
}
|
||||
| { kind: 'unsupported' };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
/**
|
||||
* @fileoverview Read/write-array store for user-configured custom OpenAI-compatible
|
||||
* model endpoints (local or cloud — deployment_plan.md). Same shape as
|
||||
* `remote-hosts.ts` / `webview-store.ts`: `~/.codeman/custom-model-hosts.json`
|
||||
* holding a plain array, read/written whole.
|
||||
*/
|
||||
|
||||
import { existsSync, mkdirSync } from 'node:fs';
|
||||
import fs from 'node:fs/promises';
|
||||
import { join } from 'node:path';
|
||||
|
||||
const CUSTOM_MODEL_HOSTS_FILE = 'custom-model-hosts.json';
|
||||
|
||||
export type CustomModelAuthStyle = 'bearer' | 'api-key';
|
||||
|
||||
export interface CustomModelHost {
|
||||
id: string;
|
||||
label: string;
|
||||
/** Root URL, local or cloud — e.g. "http://192.168.1.50:8080" or an Azure AI Foundry URL. */
|
||||
baseUrl: string;
|
||||
apiKey?: string;
|
||||
/**
|
||||
* Defaults to 'bearer' (the common `Authorization: Bearer` convention — matches
|
||||
* llama.cpp, OpenAI-compatible servers, and most gateways). Pick 'api-key' for
|
||||
* endpoints that specifically want the `api-key` header, e.g. Azure AI Foundry.
|
||||
*
|
||||
* ⚠️ There is deliberately NO 'both' option. An earlier design sent BOTH headers
|
||||
* on every discovery request on the theory that an unused header is harmless —
|
||||
* live-tested against a real llama-swap server, sending both reliably HUNG the
|
||||
* request indefinitely (reproduced 3× — Bearer alone: ~500ms, api-key alone:
|
||||
* ~600ms, both together: no response inside a 15s timeout). Whatever auth
|
||||
* middleware some servers run apparently does not handle two simultaneous
|
||||
* credential conventions gracefully, so "send everything and let the server
|
||||
* ignore what it doesn't need" is not a safe default — it can silently turn a
|
||||
* working endpoint into one that always times out.
|
||||
*/
|
||||
authStyle?: CustomModelAuthStyle;
|
||||
models?: string[];
|
||||
lastDiscoveredAt?: string;
|
||||
}
|
||||
|
||||
export function customModelHostsPath(configDir: string): string {
|
||||
return join(configDir, CUSTOM_MODEL_HOSTS_FILE);
|
||||
}
|
||||
|
||||
export async function readCustomModelHosts(configDir: string): Promise<CustomModelHost[]> {
|
||||
try {
|
||||
const raw = await fs.readFile(customModelHostsPath(configDir), 'utf-8');
|
||||
const parsed = JSON.parse(raw);
|
||||
return Array.isArray(parsed) ? (parsed as CustomModelHost[]) : [];
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function writeCustomModelHosts(configDir: string, hosts: CustomModelHost[]): Promise<void> {
|
||||
if (!existsSync(configDir)) mkdirSync(configDir, { recursive: true });
|
||||
await fs.writeFile(customModelHostsPath(configDir), JSON.stringify(hosts, null, 2));
|
||||
}
|
||||
@@ -0,0 +1,188 @@
|
||||
/**
|
||||
* @fileoverview Pure builder for the Custom Model Endpoint Profiles feature
|
||||
* (deployment_plan.md): turns a CLI registry entry's
|
||||
* `capabilities.customModelInjection` declaration, a configured endpoint,
|
||||
* and a chosen model id into the concrete env vars / config-file content
|
||||
* that would redirect that CLI's session at the endpoint.
|
||||
*
|
||||
* No IO here on purpose (mirrors `session-cli-builder.ts`) — a caller
|
||||
* writes `ConfigDirInjection.files` to disk under an isolated per-session
|
||||
* directory and points `dirEnvVar` at it; this module only computes what
|
||||
* those files/env vars should contain.
|
||||
*
|
||||
* Confidence: `claude` and `opencode` are verified end-to-end against a real
|
||||
* llama-swap server (a real "hello world" reply came back). `codex`'s
|
||||
* config.toml STRUCTURE is now verified (an earlier `[model].default` table
|
||||
* shape was rejected by a real codex binary with "invalid type: map,
|
||||
* expected a string" — caught by `scripts/test-local-llm-harnesses.mjs`),
|
||||
* but `wire_api = "responses"` is the only value codex still accepts
|
||||
* (support for `"chat"` was dropped in Feb 2026), and a plain OpenAI
|
||||
* Chat-Completions server (llama.cpp, llama-swap, most local setups) does
|
||||
* NOT implement the Responses API — so codex may still fail at the
|
||||
* PROTOCOL level even with a correctly-shaped config file. That gap is
|
||||
* real and current, not a stale warning; see deployment_plan.md. The rest
|
||||
* (gemini/pi/grok/deepseek/omp) have their ONE-SHOT INVOCATION flags
|
||||
* confirmed against real installed binaries' own `--help` output, but
|
||||
* their custom-endpoint env/config conventions remain web-researched,
|
||||
* unverified.
|
||||
*/
|
||||
|
||||
import type { CliEntry } from './config/cli-registry/types.js';
|
||||
|
||||
export interface CustomModelEndpoint {
|
||||
id: string;
|
||||
label: string;
|
||||
/** Root URL, no trailing slash required — e.g. "http://192.168.1.50:8080" or an Azure AI Foundry URL. */
|
||||
baseUrl: string;
|
||||
/** Falls back to a harmless placeholder for endpoints (llama.cpp) that don't check it. */
|
||||
apiKey?: string;
|
||||
}
|
||||
|
||||
export interface EnvInjection {
|
||||
kind: 'env';
|
||||
/** Ready to merge into a session's envOverrides. */
|
||||
envOverrides: Record<string, string>;
|
||||
}
|
||||
|
||||
export interface ConfigDirInjection {
|
||||
kind: 'configDir';
|
||||
/** Env var that must be set to the directory the caller writes `files` under. */
|
||||
dirEnvVar: string;
|
||||
files: Array<{ relPath: string; content: string }>;
|
||||
/**
|
||||
* Env vars the written config file REFERENCES by name rather than embedding a
|
||||
* literal value (codex's `env_key = "..."` convention: config.toml never carries
|
||||
* the API key itself, only the name of an env var codex reads it from). Merge
|
||||
* these into the session's envOverrides alongside `dirEnvVar` — never skip them,
|
||||
* or the config points at a credential that was never actually set.
|
||||
*/
|
||||
extraEnv?: Record<string, string>;
|
||||
}
|
||||
|
||||
export interface UnsupportedInjection {
|
||||
kind: 'unsupported';
|
||||
}
|
||||
|
||||
export type CustomModelInjectionResult = EnvInjection | ConfigDirInjection | UnsupportedInjection;
|
||||
|
||||
const DEFAULT_API_KEY = 'local-dummy-key';
|
||||
|
||||
/** Normalizes a base URL to end in exactly one trailing `/v1`, for CLIs whose config expects the OpenAI-style suffix. */
|
||||
export function withV1Suffix(baseUrl: string): string {
|
||||
const trimmed = baseUrl.replace(/\/+$/, '');
|
||||
return /\/v1$/.test(trimmed) ? trimmed : `${trimmed}/v1`;
|
||||
}
|
||||
|
||||
/** JSON-escapes a string for embedding in a TOML/YAML double-quoted scalar — a safe superset of both grammars' basic escapes. */
|
||||
function quoted(value: string): string {
|
||||
return JSON.stringify(value);
|
||||
}
|
||||
|
||||
export function buildCustomModelInjection(
|
||||
entry: Pick<CliEntry, 'capabilities'>,
|
||||
endpoint: CustomModelEndpoint,
|
||||
modelId: string
|
||||
): CustomModelInjectionResult {
|
||||
const cap = entry.capabilities.customModelInjection;
|
||||
const apiKey = endpoint.apiKey?.trim() || DEFAULT_API_KEY;
|
||||
|
||||
switch (cap.kind) {
|
||||
case 'env': {
|
||||
const envOverrides: Record<string, string> = {
|
||||
[cap.baseUrlVar]: endpoint.baseUrl,
|
||||
[cap.apiKeyVar]: apiKey,
|
||||
};
|
||||
for (const modelVar of cap.modelVars) envOverrides[modelVar] = modelId;
|
||||
return { kind: 'env', envOverrides };
|
||||
}
|
||||
|
||||
case 'configContentEnv': {
|
||||
const content = renderConfigContent(cap.template, endpoint, modelId, apiKey);
|
||||
return { kind: 'env', envOverrides: { [cap.envVar]: content } };
|
||||
}
|
||||
|
||||
case 'configDir': {
|
||||
const { content, extraEnv } = renderConfigFile(cap.template, endpoint, modelId, apiKey);
|
||||
return { kind: 'configDir', dirEnvVar: cap.dirEnvVar, files: [{ relPath: cap.fileName, content }], extraEnv };
|
||||
}
|
||||
|
||||
case 'unsupported':
|
||||
return { kind: 'unsupported' };
|
||||
}
|
||||
}
|
||||
|
||||
function renderConfigContent(
|
||||
template: 'opencode-json',
|
||||
endpoint: CustomModelEndpoint,
|
||||
modelId: string,
|
||||
apiKey: string
|
||||
): string {
|
||||
switch (template) {
|
||||
case 'opencode-json':
|
||||
return JSON.stringify({
|
||||
$schema: 'https://opencode.ai/config.json',
|
||||
provider: {
|
||||
custom: {
|
||||
options: { baseURL: withV1Suffix(endpoint.baseUrl), apiKey },
|
||||
models: { [modelId]: {} },
|
||||
},
|
||||
},
|
||||
model: `custom/${modelId}`,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const CODEX_API_KEY_ENV_VAR = 'CODEMAN_CUSTOM_MODEL_API_KEY';
|
||||
|
||||
function renderConfigFile(
|
||||
template: 'codex-toml' | 'pi-models-json' | 'omp-models-yml',
|
||||
endpoint: CustomModelEndpoint,
|
||||
modelId: string,
|
||||
apiKey: string
|
||||
): { content: string; extraEnv?: Record<string, string> } {
|
||||
const baseUrl = withV1Suffix(endpoint.baseUrl);
|
||||
switch (template) {
|
||||
case 'codex-toml': {
|
||||
// Verified against real codex (>= Feb 2026): `model` is a top-level STRING, never
|
||||
// a `[model].default` table — codex rejects that with "invalid type: map, expected
|
||||
// a string" (caught by scripts/test-local-llm-harnesses.mjs against a real llama-swap
|
||||
// server). The API key is NEVER a literal TOML field: codex's schema only supports
|
||||
// `env_key`, the NAME of an env var it reads the credential from at runtime, so the
|
||||
// actual value must ride along as an extra env var, never embedded in the file.
|
||||
// ⚠️ `wire_api = "responses"` is the only value codex still accepts (it dropped
|
||||
// `"chat"` support in Feb 2026) — a plain OpenAI Chat-Completions server (llama.cpp,
|
||||
// llama-swap, most local setups) does NOT implement the Responses API, so this
|
||||
// recipe may still fail at the PROTOCOL level even though the file now parses
|
||||
// correctly. That is a real, currently-unresolved compatibility gap, not a syntax
|
||||
// bug — track it before calling codex support done.
|
||||
const content = [
|
||||
`model = ${quoted(modelId)}`,
|
||||
`model_provider = "custom"`,
|
||||
'',
|
||||
'[model_providers.custom]',
|
||||
`name = "Custom Endpoint"`,
|
||||
`base_url = ${quoted(baseUrl)}`,
|
||||
`env_key = ${quoted(CODEX_API_KEY_ENV_VAR)}`,
|
||||
`wire_api = "responses"`,
|
||||
'',
|
||||
].join('\n');
|
||||
return { content, extraEnv: { [CODEX_API_KEY_ENV_VAR]: apiKey } };
|
||||
}
|
||||
case 'pi-models-json':
|
||||
return {
|
||||
content: JSON.stringify(
|
||||
{
|
||||
providers: {
|
||||
custom: { baseUrl, apiKey, api: 'openai-completions', models: { [modelId]: {} } },
|
||||
},
|
||||
},
|
||||
null,
|
||||
2
|
||||
),
|
||||
};
|
||||
case 'omp-models-yml':
|
||||
return {
|
||||
content: `providers:\n custom:\n baseUrl: ${quoted(baseUrl)}\n apiKey: ${quoted(apiKey)}\n models:\n - ${quoted(modelId)}\n`,
|
||||
};
|
||||
}
|
||||
}
|
||||
+83
-1
@@ -577,6 +577,14 @@ export class Session extends EventEmitter {
|
||||
// the CLAUDE_CODE_EFFORT_LEVEL env var, which would hard-lock the session.
|
||||
private _effort: EffortLevel | undefined;
|
||||
|
||||
// Custom Model Endpoint Profiles (deployment_plan.md). `envKeys` and `configDir` are
|
||||
// internal bookkeeping ONLY (never surfaced via toState()/customModel getter): they are
|
||||
// what setCustomModel() needs to undo a previous injection (remove exactly the env keys
|
||||
// it added, delete a previous isolated config dir) without guessing what it once wrote.
|
||||
private _customModel:
|
||||
| { endpointId: string; modelId: string; label?: string; envKeys: string[]; configDir?: string }
|
||||
| undefined;
|
||||
|
||||
// tmux history-limit (scrollback lines) allocated when this session's pane is created.
|
||||
private readonly _tmuxHistoryLimit: number;
|
||||
|
||||
@@ -1238,6 +1246,40 @@ export class Session extends EventEmitter {
|
||||
}
|
||||
}
|
||||
|
||||
// Custom Model Endpoint Profiles (deployment_plan.md) — public-safe subset only
|
||||
// (never envKeys/configDir, which are internal bookkeeping for setCustomModel below).
|
||||
get customModel(): { endpointId: string; modelId: string; label?: string } | undefined {
|
||||
if (!this._customModel) return undefined;
|
||||
const { endpointId, modelId, label } = this._customModel;
|
||||
return { endpointId, modelId, label };
|
||||
}
|
||||
|
||||
/**
|
||||
* Update this session's custom-model selection and merge the endpoint's injected env
|
||||
* vars into `_envOverrides` — first UNDOING whatever the previous selection injected
|
||||
* (removing exactly those env keys), so switching endpoints, or clearing back to the
|
||||
* harness's native cloud default, never leaves a stale key behind. Synchronous and
|
||||
* side-effect-free beyond mutating state, matching `setNice`/`setColor` above — this
|
||||
* class does no file IO, so it returns the PREVIOUS `configDir` (if any) for the
|
||||
* caller to clean up on disk (custom-model-injection.ts's configDir kind).
|
||||
*/
|
||||
setCustomModel(
|
||||
next: { endpointId: string; modelId: string; label?: string; envKeys: string[]; configDir?: string } | undefined,
|
||||
envOverrides?: Record<string, string>
|
||||
): string | undefined {
|
||||
const previousConfigDir = this._customModel?.configDir;
|
||||
if (this._customModel) {
|
||||
for (const key of this._customModel.envKeys) {
|
||||
if (this._envOverrides) delete this._envOverrides[key];
|
||||
}
|
||||
}
|
||||
this._customModel = next;
|
||||
if (envOverrides && Object.keys(envOverrides).length > 0) {
|
||||
this._envOverrides = { ...(this._envOverrides ?? {}), ...envOverrides };
|
||||
}
|
||||
return previousConfigDir;
|
||||
}
|
||||
|
||||
// Token tracking getters and setters
|
||||
get totalTokens(): number {
|
||||
return this._totalInputTokens + this._totalOutputTokens;
|
||||
@@ -1478,6 +1520,7 @@ export class Session extends EventEmitter {
|
||||
ompConfig: this._ompConfig,
|
||||
resumeSessionId: this._resumeSessionId,
|
||||
effort: this._effort,
|
||||
customModel: this.customModel,
|
||||
// COD-118: runtime-only — surfaced so the frontend can require explicit user
|
||||
// intent before restarting a crash-looped session. Deliberately NOT restored
|
||||
// by the constructor: a Codeman restart starts with a fresh breaker so boot
|
||||
@@ -1710,10 +1753,49 @@ export class Session extends EventEmitter {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Kill and relaunch this session's CLI process IN PLACE — same pane, same tmux
|
||||
* session, fresh env/args from current state. Custom Model Endpoint Profiles
|
||||
* (deployment_plan.md) is the first caller: after `setCustomModel()` merges new
|
||||
* env vars into `_envOverrides`, the running CLI process still has the OLD env
|
||||
* (inherited at its own process start, not live-reloaded), so switching a
|
||||
* session's model/endpoint requires this restart to actually take effect.
|
||||
*
|
||||
* A GENERALIZED {@link reattachRemote} with the `!this._remote` guard dropped —
|
||||
* `_buildRespawnPaneOptions()` already passes `remote: this._remote` through
|
||||
* unconditionally, so `mux.respawnPane()` builds the right command either way
|
||||
* (a local session gets `respawn-pane -k` + the real launch line, which is the
|
||||
* kill-and-relaunch this method exists for; a remote session gets the existing
|
||||
* reattach-to-durable-tmux behavior). Deliberately does NOT check `isBusy()` —
|
||||
* that's the caller's job (mirrors `/interactive`'s guard), since a raw restart
|
||||
* primitive shouldn't itself decide when it's safe to use.
|
||||
*
|
||||
* @returns true if the pane was respawned, false otherwise (no mux session, or
|
||||
* the mux session is gone — see {@link reattachRemote} for that reasoning).
|
||||
*/
|
||||
async restartCli(): Promise<boolean> {
|
||||
if (!this._useMux || !this._mux || !this._muxSession) return false;
|
||||
const mux = this._mux;
|
||||
|
||||
if (!mux.muxSessionExists(this._muxSession.muxName)) {
|
||||
console.log('[Session] restartCli: mux session gone, skipping:', this._muxSession.muxName);
|
||||
return false;
|
||||
}
|
||||
|
||||
this._pinOmpRespawnId();
|
||||
const newPid = await mux.respawnPane(this._buildRespawnPaneOptions());
|
||||
if (!newPid) {
|
||||
console.error('[Session] restartCli: respawnPane failed for', this._muxSession.muxName);
|
||||
return false;
|
||||
}
|
||||
console.log('[Session] restartCli: restarted CLI for', this._muxSession.muxName, 'pid', newPid);
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Assemble the {@link RespawnPaneOptions} for this session. Single source of
|
||||
* truth shared by interactive start, shell start (via their inline copies),
|
||||
* and {@link reattachRemote} so the remote reattach path can never drift from
|
||||
* {@link reattachRemote}, and {@link restartCli} so no respawn path can drift from
|
||||
* the spawn path.
|
||||
*/
|
||||
private _buildRespawnPaneOptions(): import('./mux-interface.js').RespawnPaneOptions {
|
||||
|
||||
@@ -677,6 +677,13 @@ export interface SessionState {
|
||||
resumeSessionId?: string;
|
||||
/** Claude CLI effort level (soft default via --settings, switchable in-session via /effort) */
|
||||
effort?: EffortLevel;
|
||||
/**
|
||||
* Custom Model Endpoint Profiles (deployment_plan.md): the custom OpenAI-compatible
|
||||
* endpoint (local or cloud) this session's CLI is currently pointed at, if any.
|
||||
* Undefined = the harness's native cloud default. No secrets here — the endpoint's
|
||||
* base URL/api key live only in Session._envOverrides, never in this public state.
|
||||
*/
|
||||
customModel?: { endpointId: string; modelId: string; label?: string };
|
||||
/** Sanitized per-session attachment history. */
|
||||
attachmentHistory?: SessionAttachmentHistoryItem[];
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,128 @@
|
||||
/**
|
||||
* @fileoverview Custom Model Endpoint Profiles CRUD + discovery
|
||||
* (deployment_plan.md). Endpoints are machine-level infra, like remote/docker
|
||||
* hosts, so writes are admin-only in multi-user mode
|
||||
* (`case-routes.ts`'s `/api/remote-hosts` is the pattern this mirrors).
|
||||
*
|
||||
* Discovery (`POST /:id/discover-models`) fetches `${baseUrl}/v1/models`.
|
||||
* `isBlockedWebviewUrl()` is the same synchronous hostname/link-local/cloud-
|
||||
* metadata check `webview-egress-policy.ts` uses for saved dashboard URLs —
|
||||
* reused here as a save-time and discover-time guard. It does NOT re-check
|
||||
* the DNS-RESOLVED address the way `webviewFetch()`'s undici lookup hook
|
||||
* does; wiring that dispatcher-level guard here is a followup, not done in
|
||||
* this pass, since this route is already admin-only in multi-user mode.
|
||||
*/
|
||||
|
||||
import type { FastifyInstance, FastifyRequest } from 'fastify';
|
||||
import { ApiErrorCode, createErrorResponse, type ApiResponse } from '../../types.js';
|
||||
import { isAdmin, parseBody } from '../route-helpers.js';
|
||||
import { isMultiUserMode } from '../../config/multiuser.js';
|
||||
import { getDataDir } from '../../config/instance.js';
|
||||
import { isBlockedWebviewUrl } from '../webview-egress-policy.js';
|
||||
import { CustomModelHostSchema } from '../schemas.js';
|
||||
import { readCustomModelHosts, writeCustomModelHosts, type CustomModelHost } from '../../custom-model-hosts.js';
|
||||
|
||||
const CODEMAN_CONFIG_DIR = getDataDir();
|
||||
const DISCOVER_TIMEOUT_MS = 8000;
|
||||
|
||||
function adminOnly(req: FastifyRequest, reply: { code: (n: number) => unknown }): ApiResponse<never> | null {
|
||||
if (!isMultiUserMode() || isAdmin(req)) return null;
|
||||
reply.code(403);
|
||||
return createErrorResponse(ApiErrorCode.FORBIDDEN, 'Admin only in multi-user mode');
|
||||
}
|
||||
|
||||
async function discoverModels(host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>): Promise<string[]> {
|
||||
const headers: Record<string, string> = {};
|
||||
const apiKey = host.apiKey?.trim();
|
||||
// Exactly ONE header, never both — see custom-model-hosts.ts's CustomModelAuthStyle
|
||||
// doc comment for why: sending both reliably HANGS some real servers.
|
||||
const style = host.authStyle ?? 'bearer';
|
||||
if (apiKey && style === 'bearer') headers.Authorization = `Bearer ${apiKey}`;
|
||||
if (apiKey && style === 'api-key') headers['api-key'] = apiKey;
|
||||
|
||||
const res = await fetch(`${host.baseUrl.replace(/\/+$/, '')}/v1/models`, {
|
||||
headers,
|
||||
signal: AbortSignal.timeout(DISCOVER_TIMEOUT_MS),
|
||||
});
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
||||
const body = (await res.json()) as { data?: Array<{ id?: unknown }> };
|
||||
return (body.data ?? []).map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
|
||||
}
|
||||
|
||||
export function registerCustomModelRoutes(app: FastifyInstance): void {
|
||||
app.get('/api/model-endpoints', async (req) =>
|
||||
isMultiUserMode() && !isAdmin(req) ? [] : readCustomModelHosts(CODEMAN_CONFIG_DIR)
|
||||
);
|
||||
|
||||
app.post('/api/model-endpoints', async (req, reply): Promise<ApiResponse<{ host: CustomModelHost }>> => {
|
||||
const denied = adminOnly(req, reply);
|
||||
if (denied) return denied;
|
||||
const host = parseBody(CustomModelHostSchema, req.body);
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
if (hosts.some((item) => item.id === host.id)) {
|
||||
return createErrorResponse(ApiErrorCode.ALREADY_EXISTS, 'Model endpoint already exists');
|
||||
}
|
||||
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, [...hosts, host]);
|
||||
return { success: true, data: { host } };
|
||||
});
|
||||
|
||||
app.put('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ host: CustomModelHost }>> => {
|
||||
const denied = adminOnly(req, reply);
|
||||
if (denied) return denied;
|
||||
const { id } = req.params as { id: string };
|
||||
const host = parseBody(CustomModelHostSchema, { ...(req.body as object), id });
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
const index = hosts.findIndex((item) => item.id === id);
|
||||
if (index === -1) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
const next = [...hosts];
|
||||
next[index] = host;
|
||||
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
|
||||
return { success: true, data: { host } };
|
||||
});
|
||||
|
||||
app.delete('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ id: string }>> => {
|
||||
const denied = adminOnly(req, reply);
|
||||
if (denied) return denied;
|
||||
const { id } = req.params as { id: string };
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
await writeCustomModelHosts(
|
||||
CODEMAN_CONFIG_DIR,
|
||||
hosts.filter((item) => item.id !== id)
|
||||
);
|
||||
return { success: true, data: { id } };
|
||||
});
|
||||
|
||||
app.post(
|
||||
'/api/model-endpoints/:id/discover-models',
|
||||
async (req, reply): Promise<ApiResponse<{ models: string[] }>> => {
|
||||
const denied = adminOnly(req, reply);
|
||||
if (denied) return denied;
|
||||
const { id } = req.params as { id: string };
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
const index = hosts.findIndex((item) => item.id === id);
|
||||
if (index === -1) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
const host = hosts[index];
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
try {
|
||||
const models = await discoverModels(host);
|
||||
const next = [...hosts];
|
||||
next[index] = { ...host, models, lastDiscoveredAt: new Date().toISOString() };
|
||||
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
|
||||
return { success: true, data: { models } };
|
||||
} catch (err) {
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.OPERATION_FAILED,
|
||||
`Could not reach endpoint: ${err instanceof Error ? err.message : String(err)}`
|
||||
);
|
||||
}
|
||||
}
|
||||
);
|
||||
}
|
||||
@@ -27,3 +27,4 @@ export { registerWsRoutes } from './ws-routes.js';
|
||||
export { registerVoiceRoutes } from './voice-routes.js';
|
||||
export { registerWebviewRoutes, tryWebviewRefererFallback } from './webview-routes.js';
|
||||
export { registerTabLayoutRoutes } from './tab-layout-routes.js';
|
||||
export { registerCustomModelRoutes } from './custom-model-routes.js';
|
||||
|
||||
@@ -8,7 +8,7 @@ import { FastifyInstance, type FastifyReply } from 'fastify';
|
||||
import { z } from 'zod';
|
||||
import { join, dirname, extname, basename } from 'node:path';
|
||||
import { homedir } from 'node:os';
|
||||
import { existsSync, statSync, mkdirSync, writeFileSync } from 'node:fs';
|
||||
import { existsSync, statSync, mkdirSync, writeFileSync, rmSync } from 'node:fs';
|
||||
import { execFile } from 'node:child_process';
|
||||
import fs from 'node:fs/promises';
|
||||
import { randomBytes } from 'node:crypto';
|
||||
@@ -51,7 +51,10 @@ import {
|
||||
SessionOrderUpdateSchema,
|
||||
SessionWaitQuerySchema,
|
||||
SessionWaitOutputQuerySchema,
|
||||
CustomModelSelectionSchema,
|
||||
} from '../schemas.js';
|
||||
import { readCustomModelHosts } from '../../custom-model-hosts.js';
|
||||
import { buildCustomModelInjection } from '../../custom-model-injection.js';
|
||||
import { ownerLayoutKey } from '../../tab-layout-persistence.js';
|
||||
import { TabLayoutValidationError } from '../../tab-layout.js';
|
||||
import {
|
||||
@@ -1161,6 +1164,88 @@ export function registerSessionRoutes(
|
||||
return { color: session.color };
|
||||
});
|
||||
|
||||
// ========== Custom Model Endpoint Profiles (deployment_plan.md) ==========
|
||||
//
|
||||
// Applies (or clears) a session's custom OpenAI-compatible endpoint selection and
|
||||
// RESTARTS the pane's CLI process — these harnesses read endpoint config at process
|
||||
// start, not per-turn, so a live hot-swap isn't possible (confirmed with the
|
||||
// maintainer). Endpoints come from the admin-configured custom-model-hosts store
|
||||
// (chunk 3's CRUD routes), never raw client-supplied env — that's what keeps this
|
||||
// route safe to let any session owner call for their own session, unlike the
|
||||
// generic envOverrides field the privilegedEnvKeys clamp exists to guard.
|
||||
app.post('/api/sessions/:id/custom-model', async (req) => {
|
||||
const { id } = req.params as { id: string };
|
||||
const body = parseBody(CustomModelSelectionSchema, req.body, 'Invalid request body');
|
||||
const session = findSessionOrFail(ctx, id, req);
|
||||
|
||||
if (session.isBusy()) {
|
||||
return createErrorResponse(ApiErrorCode.SESSION_BUSY, 'Session is busy');
|
||||
}
|
||||
|
||||
if ('clear' in body) {
|
||||
const previousConfigDir = session.setCustomModel(undefined);
|
||||
if (previousConfigDir) rmSync(previousConfigDir, { recursive: true, force: true });
|
||||
const restarted = await session.restartCli();
|
||||
persistAndBroadcastSession(ctx, session);
|
||||
return { customModel: session.customModel, restarted };
|
||||
}
|
||||
|
||||
const entry = getCli(session.mode);
|
||||
if (!entry) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, `No CLI registry entry for mode ${session.mode}`);
|
||||
}
|
||||
if (entry.capabilities.customModelInjection.kind === 'unsupported') {
|
||||
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${session.mode} has no known custom-model mechanism`);
|
||||
}
|
||||
|
||||
const hosts = await readCustomModelHosts(getDataDir());
|
||||
const endpoint = hosts.find((h) => h.id === body.endpointId);
|
||||
if (!endpoint) {
|
||||
return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
}
|
||||
|
||||
const injection = buildCustomModelInjection(entry, endpoint, body.modelId);
|
||||
|
||||
let envOverrides: Record<string, string>;
|
||||
let envKeys: string[];
|
||||
let configDir: string | undefined;
|
||||
|
||||
if (injection.kind === 'env') {
|
||||
envOverrides = injection.envOverrides;
|
||||
envKeys = Object.keys(injection.envOverrides);
|
||||
} else if (injection.kind === 'configDir') {
|
||||
// Isolated per-session dir — never the user's real CLI config path.
|
||||
configDir = join(dataPath('custom-model-configs'), session.id);
|
||||
for (const file of injection.files) {
|
||||
const filePath = join(configDir, file.relPath);
|
||||
mkdirSync(dirname(filePath), { recursive: true });
|
||||
writeFileSync(filePath, file.content, 'utf8');
|
||||
}
|
||||
// extraEnv: vars the written config file REFERENCES by name (codex's `env_key`
|
||||
// convention) rather than embedding a literal value — must ride alongside
|
||||
// dirEnvVar or the config points at a credential that was never actually set.
|
||||
envOverrides = { [injection.dirEnvVar]: configDir, ...injection.extraEnv };
|
||||
envKeys = [injection.dirEnvVar, ...Object.keys(injection.extraEnv ?? {})];
|
||||
} else {
|
||||
// 'unsupported' is already handled above; this keeps the switch exhaustive.
|
||||
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${session.mode} has no known custom-model mechanism`);
|
||||
}
|
||||
|
||||
const previousConfigDir = session.setCustomModel(
|
||||
{ endpointId: endpoint.id, modelId: body.modelId, label: endpoint.label, envKeys, configDir },
|
||||
envOverrides
|
||||
);
|
||||
// Clean up the OLD config dir on disk, unless the new one happens to reuse the same
|
||||
// path (same session, configDir kind again) — never delete the dir we just wrote.
|
||||
if (previousConfigDir && previousConfigDir !== configDir) {
|
||||
rmSync(previousConfigDir, { recursive: true, force: true });
|
||||
}
|
||||
|
||||
const restarted = await session.restartCli();
|
||||
persistAndBroadcastSession(ctx, session);
|
||||
return { customModel: session.customModel, restarted };
|
||||
});
|
||||
|
||||
// ========== Delete Session ==========
|
||||
|
||||
app.delete('/api/sessions/:id', async (req) => {
|
||||
|
||||
@@ -741,6 +741,29 @@ export const RemoteHostSchema = z.object({
|
||||
commands: RemoteCommandOverridesSchema,
|
||||
});
|
||||
|
||||
// Custom Model Endpoint Profiles (deployment_plan.md) — a user-configured custom
|
||||
// OpenAI-compatible endpoint, local (llama.cpp) or cloud (Azure AI Foundry, etc.).
|
||||
export const CustomModelHostSchema = z.object({
|
||||
id: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid endpoint id'),
|
||||
label: z.string().min(1).max(100),
|
||||
baseUrl: z.string().url().max(2048),
|
||||
apiKey: z.string().max(4096).optional(),
|
||||
// No 'both': live-tested against a real server, sending both auth header
|
||||
// conventions on one request reliably HANGS it — see custom-model-hosts.ts.
|
||||
authStyle: z.enum(['bearer', 'api-key']).optional(),
|
||||
models: z.array(z.string().max(200)).max(200).optional(),
|
||||
lastDiscoveredAt: z.string().max(64).optional(),
|
||||
});
|
||||
|
||||
/** POST /api/sessions/:id/custom-model — apply or clear a session's custom-model selection. */
|
||||
export const CustomModelSelectionSchema = z.union([
|
||||
z.object({
|
||||
endpointId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid endpoint id'),
|
||||
modelId: z.string().min(1).max(200),
|
||||
}),
|
||||
z.object({ clear: z.literal(true) }),
|
||||
]);
|
||||
|
||||
export const RemoteCaseLinkSchema = z.object({
|
||||
name: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid case name format'),
|
||||
hostId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid remote host id'),
|
||||
@@ -1239,6 +1262,13 @@ export const SettingsUpdateSchema = z
|
||||
* stored profiles stay until DELETE /api/sessions/:id/intent.
|
||||
*/
|
||||
readMyMindEnabled: z.boolean().optional(),
|
||||
/**
|
||||
* Custom Model Endpoint Profiles (deployment_plan.md): the toolbar picker that lets a
|
||||
* session point at a user-configured custom OpenAI-compatible endpoint (local or
|
||||
* cloud) instead of its native cloud backend. SYNCED, default OFF — endpoint entry,
|
||||
* discovery, and the extra toolbar surface are all opt-in.
|
||||
*/
|
||||
customModelEndpointsEnabled: z.boolean().optional(),
|
||||
/**
|
||||
* Read My Mind predictor model override. Empty/absent = the AI-checker
|
||||
* default (opus: prediction quality is the product and it runs only on an
|
||||
|
||||
@@ -179,6 +179,7 @@ import {
|
||||
registerVoiceRoutes,
|
||||
registerWebviewRoutes,
|
||||
registerTabLayoutRoutes,
|
||||
registerCustomModelRoutes,
|
||||
tryWebviewRefererFallback,
|
||||
} from './routes/index.js';
|
||||
import { CronService } from '../cron/cron-service.js';
|
||||
@@ -1051,6 +1052,7 @@ export class WebServer extends EventEmitter {
|
||||
registerOrchestratorRoutes(this.app, ctx);
|
||||
registerWebviewRoutes(this.app, ctx, this.basePath);
|
||||
registerTabLayoutRoutes(this.app, ctx);
|
||||
registerCustomModelRoutes(this.app);
|
||||
|
||||
// Cron: build the service from the same context, recompute
|
||||
// due times for any persisted jobs, then expose it to its routes.
|
||||
|
||||
Reference in New Issue
Block a user