mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-03 14:09:42 +02:00
Merge pull request #393 from opticon454/feature-custom-llm-server-support
feat: Custom Model Endpoint Profiles (local or cloud, all harnesses)
This commit is contained in:
@@ -317,6 +317,34 @@ const capabilitiesSchema = z
|
||||
privilegedEnvKeys: z.array(envName).max(8),
|
||||
gates: z.record(z.string(), z.object({ minVersion: z.string().max(20), failClosed: z.boolean() }).strict()),
|
||||
maxFrameBytes: z.number().int().positive().optional(),
|
||||
customModelInjection: z.discriminatedUnion('kind', [
|
||||
z
|
||||
.object({
|
||||
kind: z.literal('env'),
|
||||
baseUrlVar: envName,
|
||||
apiKeyVar: envName,
|
||||
// Empty is valid: deepseek's model routing is a profile-composition concern, not
|
||||
// an env var, so it declares baseUrl/apiKey injection with no model var at all.
|
||||
modelVars: z.array(envName).max(8),
|
||||
})
|
||||
.strict(),
|
||||
z
|
||||
.object({
|
||||
kind: z.literal('configContentEnv'),
|
||||
envVar: envName,
|
||||
template: z.literal('opencode-json'),
|
||||
})
|
||||
.strict(),
|
||||
z
|
||||
.object({
|
||||
kind: z.literal('configDir'),
|
||||
dirEnvVar: envName,
|
||||
fileName: z.string().min(1).max(80),
|
||||
template: z.enum(['codex-toml', 'pi-models-json', 'omp-models-yml', 'grok-toml']),
|
||||
})
|
||||
.strict(),
|
||||
z.object({ kind: z.literal('unsupported') }).strict(),
|
||||
]),
|
||||
})
|
||||
.strict();
|
||||
|
||||
|
||||
@@ -189,6 +189,12 @@ const CLAUDE: CliEntry = {
|
||||
unset: ['CLAUDECODE'],
|
||||
tmuxSetenvKeys: [],
|
||||
dockerExecEnvNames: [],
|
||||
// Deliberately excludes ANTHROPIC_* (base URL / API key / default-model overrides):
|
||||
// custom-model-injection.ts's claude recipe uses those names, but they must reach a
|
||||
// session ONLY through the admin-configured, SSRF-guarded custom-model route, never
|
||||
// through a plain client-supplied envOverrides field. Widening this prefix would let
|
||||
// any session-create caller redirect a session's Anthropic traffic and credentials to
|
||||
// an arbitrary, unvalidated URL.
|
||||
allowedPrefixes: ['CLAUDE_CODE_'],
|
||||
allowedKeys: ['CLAUDE_CONFIG_DIR'],
|
||||
},
|
||||
@@ -220,8 +226,28 @@ const CLAUDE: CliEntry = {
|
||||
statusLineTelemetry: true,
|
||||
model: { source: 'claude-settings-file' },
|
||||
privilegedParams: [],
|
||||
privilegedEnvKeys: [],
|
||||
// ANTHROPIC_* is NOT in allowedPrefixes/allowedKeys above (deliberately — see the
|
||||
// allowedPrefixes comment nearby), so these are unreachable via plain envOverrides
|
||||
// today; listed here only so the dedicated custom-model route (deployment_plan.md
|
||||
// chunk 5) clamps them for a non-granted multi-user owner the same way every other
|
||||
// CLI's injection vars are clamped, the day that route widens who can set them.
|
||||
privilegedEnvKeys: [
|
||||
'ANTHROPIC_BASE_URL',
|
||||
'ANTHROPIC_API_KEY',
|
||||
'ANTHROPIC_DEFAULT_SONNET_MODEL',
|
||||
'ANTHROPIC_DEFAULT_HAIKU_MODEL',
|
||||
'ANTHROPIC_DEFAULT_OPUS_MODEL',
|
||||
],
|
||||
gates: { nameFlag: { minVersion: '2.1.224', failClosed: true } },
|
||||
// Custom Model Endpoint Profiles (deployment_plan.md) — verified by hand against a real
|
||||
// llama.cpp server. Claude reads these at process start only, so switching requires a
|
||||
// respawn, never a live hot-swap.
|
||||
customModelInjection: {
|
||||
kind: 'env',
|
||||
baseUrlVar: 'ANTHROPIC_BASE_URL',
|
||||
apiKeyVar: 'ANTHROPIC_API_KEY',
|
||||
modelVars: ['ANTHROPIC_DEFAULT_SONNET_MODEL', 'ANTHROPIC_DEFAULT_HAIKU_MODEL', 'ANTHROPIC_DEFAULT_OPUS_MODEL'],
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
// Mirrors the local default so the remote/in-container agent runs non-interactively
|
||||
@@ -287,6 +313,7 @@ const SHELL: CliEntry = {
|
||||
privilegedParams: [],
|
||||
privilegedEnvKeys: [],
|
||||
gates: {},
|
||||
customModelInjection: { kind: 'unsupported' }, // a raw shell has no "model" concept
|
||||
},
|
||||
overlays: {
|
||||
// No `remote` entry: defaultRemoteCommandForMode special-cases kind==='shell' directly
|
||||
@@ -366,6 +393,15 @@ const OPENCODE: CliEntry = {
|
||||
...agentDefaults(),
|
||||
altScreen: 'strip-mux-only',
|
||||
echo: { policy: 'buffer', anchor: { kind: 'cursor' }, predictProfile: undefined },
|
||||
// Verified by hand against a real llama.cpp server. Reuses the SAME env var opencode's
|
||||
// own `env.configContentVar` already declares — the builder in custom-model-injection.ts
|
||||
// must merge into whatever opencode config Codeman would otherwise send, not clobber it.
|
||||
customModelInjection: { kind: 'configContentEnv', envVar: 'OPENCODE_CONFIG_CONTENT', template: 'opencode-json' },
|
||||
// OPENCODE_CONFIG_CONTENT already matches the OPENCODE_ allowedPrefix above, so it was
|
||||
// ALREADY reachable via plain envOverrides before this feature existed — it replaces
|
||||
// opencode's whole config, provider api keys included, so a non-granted multi-user owner
|
||||
// sending it is a pre-existing credential-redirection gap, not one this feature opens.
|
||||
privilegedEnvKeys: ['OPENCODE_CONFIG_CONTENT'],
|
||||
},
|
||||
overlays: {
|
||||
credStore: { rel: '.config/opencode', seedWhole: true },
|
||||
@@ -455,6 +491,23 @@ const CODEX: CliEntry = {
|
||||
// `dangerouslyBypassApprovals` on the wire), so it is the one that would have caught a
|
||||
// regression; `schema.ts` now rejects a name that is not a declared param.
|
||||
privilegedParams: [{ param: 'bypassApprovals', clampTo: false }],
|
||||
// Verified by hand against a real llama.cpp server. Written to an isolated CODEX_HOME
|
||||
// so the user's real ~/.codex/config.toml is never touched.
|
||||
customModelInjection: {
|
||||
kind: 'configDir',
|
||||
dirEnvVar: 'CODEX_HOME',
|
||||
fileName: 'config.toml',
|
||||
template: 'codex-toml',
|
||||
},
|
||||
// CODEX_HOME already matches the CODEX_ allowedPrefix above, so it was ALREADY
|
||||
// reachable via plain envOverrides before this feature existed. It is arguably
|
||||
// MORE sensitive than a bare base-url var: a redirected CODEX_HOME points codex at a
|
||||
// config.toml a non-granted owner fully controls, which can restate sandbox/approval
|
||||
// policy INSIDE that file — a path the argv-level `bypassApprovals` clamp above
|
||||
// cannot see or stop.
|
||||
// CODEMAN_CUSTOM_MODEL_API_KEY: the credential config.toml's env_key references
|
||||
// (see custom-model-injection.ts) — same reasoning as CODEX_HOME above.
|
||||
privilegedEnvKeys: ['CODEX_HOME', 'CODEMAN_CUSTOM_MODEL_API_KEY'],
|
||||
},
|
||||
overlays: {
|
||||
credStore: {
|
||||
@@ -538,6 +591,20 @@ const GEMINI: CliEntry = {
|
||||
// MATERIALIZE a config (not just touch an already-sent one) or a non-granted owner who
|
||||
// sends no geminiConfig at all would still get yolo for free.
|
||||
privilegedParams: [{ param: 'approvalMode', clampTo: 'auto_edit', materializeWhenAbsent: true }],
|
||||
// Web-researched, unverified — needs a restart to pick up (CLI reads these at process
|
||||
// start). Confirm the exact model-override env var name against the installed
|
||||
// gemini-cli version before shipping.
|
||||
customModelInjection: {
|
||||
kind: 'env',
|
||||
baseUrlVar: 'GOOGLE_GEMINI_BASE_URL',
|
||||
apiKeyVar: 'GEMINI_API_KEY',
|
||||
modelVars: ['GEMINI_MODEL'],
|
||||
},
|
||||
// All three already match the GEMINI_/GOOGLE_ allowedPrefixes above, so they were
|
||||
// ALREADY reachable via plain envOverrides before this feature existed — a non-granted
|
||||
// multi-user owner redirecting a gemini session's endpoint/credentials is a
|
||||
// pre-existing gap this feature's analysis surfaced, not one it opens.
|
||||
privilegedEnvKeys: ['GOOGLE_GEMINI_BASE_URL', 'GEMINI_API_KEY', 'GEMINI_MODEL'],
|
||||
},
|
||||
overlays: {
|
||||
credStore: { rel: '.gemini', seedWhole: true }, // also covers antigravity — see its own entry
|
||||
@@ -603,6 +670,10 @@ const ANTIGRAVITY: CliEntry = {
|
||||
// Like codex: an ABSENT config already defaults safe (no bypass flag), so only a
|
||||
// SENT config needs the flag forced off — nothing is materialized.
|
||||
privilegedParams: [{ param: 'dangerouslySkipPermissions', clampTo: false }],
|
||||
// No known CLI/env/config mechanism — Antigravity's own docs describe a GUI-only
|
||||
// custom-endpoint setting and explicitly say it "cannot currently" become the core
|
||||
// reasoning model. Toolbar entry stays disabled for this mode.
|
||||
customModelInjection: { kind: 'unsupported' },
|
||||
},
|
||||
overlays: {
|
||||
// No credStore of its own: agy nests its whole state under ~/.gemini/antigravity-cli/,
|
||||
@@ -693,6 +764,30 @@ const PI: CliEntry = {
|
||||
// just answer "yes" to, so omitting --approve is not itself a clamp — MATERIALIZE
|
||||
// approveProjectTrust:false so buildPiCommand emits --no-approve outright.
|
||||
privilegedParams: [{ param: 'approveProjectTrust', clampTo: false, materializeWhenAbsent: true }],
|
||||
// CORRECTED after live-testing: `PI_CONFIG_DIR` does NOT exist anywhere in pi's own
|
||||
// bundled source (grepped the installed package directly) — it does nothing for pi
|
||||
// itself, despite being a real Codeman env var that OTHER things (omp) read. The
|
||||
// confirmed working redirect is `HOME` itself: pi hardcodes `~/.pi/agent/models.json`
|
||||
// with no dedicated override, so redirecting the CHILD PROCESS's HOME is what
|
||||
// actually relocates it (verified: a model written under an isolated HOME's
|
||||
// `.pi/agent/models.json` shows up in `pi --list-models` and answers a real prompt
|
||||
// against a real llama-swap server; PI_CONFIG_DIR alone left it silently unable to
|
||||
// see any provider). ⚠️ This is a bigger blast radius than a dedicated config-dir
|
||||
// var: it also redirects pi's real sessions/auth/extensions for the DURATION of a
|
||||
// custom-model session, not just its provider config — document this trade-off
|
||||
// wherever this capability is surfaced.
|
||||
customModelInjection: {
|
||||
kind: 'configDir',
|
||||
dirEnvVar: 'HOME',
|
||||
fileName: '.pi/agent/models.json',
|
||||
template: 'pi-models-json',
|
||||
},
|
||||
// HOME is not `PI_`-prefixed, so unlike the old (wrong) PI_CONFIG_DIR guess this was
|
||||
// never reachable via the generic envOverrides allowlist at all — listed here anyway,
|
||||
// matching the documented pattern for every other CLI's dir-redirect var, since a
|
||||
// redirected HOME is at least as sensitive as CODEX_HOME/GROK_HOME (pi executes
|
||||
// repo-local .pi/extensions TypeScript — see the External CLI modes note in CLAUDE.md).
|
||||
privilegedEnvKeys: ['HOME'],
|
||||
},
|
||||
overlays: {
|
||||
credStore: {
|
||||
@@ -788,6 +883,24 @@ const GROK: CliEntry = {
|
||||
// already its safe interactive ask-mode, so the multi-user clamp only needs to force an
|
||||
// EXPLICITLY-SENT bypass flag back off — nothing is materialized when config is absent.
|
||||
privilegedParams: [{ param: 'alwaysApprove', clampTo: false }],
|
||||
// CORRECTED after live-testing against a real grok binary: the original `env` kind
|
||||
// (GROK_BASE_URL/GROK_MODEL/XAI_API_KEY) produced "Not signed in" — those env vars
|
||||
// are NOT grok's real custom-endpoint mechanism. The real one (verified against
|
||||
// xAI's own docs) is a `[model.<name>]` block in a config.toml under GROK_HOME,
|
||||
// the same configDir shape as codex/pi/omp. `api_backend = "chat_completions"` is
|
||||
// explicitly supported (unlike codex, which dropped it) — grok CAN talk to a plain
|
||||
// OpenAI Chat-Completions server directly.
|
||||
customModelInjection: {
|
||||
kind: 'configDir',
|
||||
dirEnvVar: 'GROK_HOME',
|
||||
fileName: 'config.toml',
|
||||
template: 'grok-toml',
|
||||
},
|
||||
// GROK_HOME already matches the GROK_ allowedPrefix above, so it was ALREADY
|
||||
// reachable via plain envOverrides before this feature existed — same reasoning
|
||||
// as CODEX_HOME: a redirected config dir can restate policy the argv-level
|
||||
// `alwaysApprove` clamp above cannot see.
|
||||
privilegedEnvKeys: ['GROK_HOME'],
|
||||
},
|
||||
overlays: {
|
||||
// ~/.grok also holds sessions/, memory/, downloads/ (the ~160MB binary), completions/,
|
||||
@@ -943,7 +1056,23 @@ const DEEPSEEK: CliEntry = {
|
||||
// The half no other CLI needs. `DSH_*` is an allowlisted envOverrides prefix and
|
||||
// applyEnvOverrides() runs LAST, so without this a non-granted owner could send
|
||||
// DSH_PERMISSION_MODE on the same request and land after the config clamp.
|
||||
// ⚠️ DEEPSEEK_API_KEY deliberately stays OUT of this list (see the docstring on
|
||||
// clampEnvOverridesForOwner() in session-routes.ts): _configureCliEnv() forwards the
|
||||
// SERVER's own key into every dsh pane, so DEEPSEEK_BASE_URL is the exfiltration
|
||||
// vector, not the key itself — a non-granted owner supplying THEIR OWN key removes
|
||||
// privilege rather than granting it, and clamping it here was a real regression
|
||||
// (test/deepseek-mode.test.ts) fixed before this shipped.
|
||||
privilegedEnvKeys: ['DSH_PERMISSION_MODE', 'DSH_HOME', 'DEEPSEEK_BASE_URL'],
|
||||
// Web-researched, unverified, partial: reuses the already-existing DEEPSEEK_BASE_URL/
|
||||
// DEEPSEEK_API_KEY keys above. No modelVars — dsh's model is a profile-composition
|
||||
// entry (see `model: { source: 'none' }` above), not an env var, so forcing a specific
|
||||
// model name may not fully work; verify against a real profile before shipping.
|
||||
customModelInjection: {
|
||||
kind: 'env',
|
||||
baseUrlVar: 'DEEPSEEK_BASE_URL',
|
||||
apiKeyVar: 'DEEPSEEK_API_KEY',
|
||||
modelVars: [],
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
// No credStore: dsh keeps everything under $DSH_HOME (default ~/.dsh), which is
|
||||
@@ -1045,7 +1174,22 @@ const OMP: CliEntry = {
|
||||
// Where omp resolves its auth from. No known concrete exfiltration path today (omp
|
||||
// forwards no operator-held key into a pane), but a non-granted owner redirecting where
|
||||
// a shared multi-tenant deployment resolves auth is not something to allow silently.
|
||||
privilegedEnvKeys: ['OMP_AUTH_BROKER_URL', 'OMP_AUTH_BROKER_TOKEN'],
|
||||
// HOME added for custom-model-injection.ts's omp recipe (see below). Unlike pi,
|
||||
// PI_CONFIG_DIR genuinely IS one of the env vars omp reads (per the DeepSeek/OMP
|
||||
// note in CLAUDE.md) — but live-testing this feature found it did NOT relocate
|
||||
// omp's model config the way expected, while redirecting HOME itself (like pi)
|
||||
// worked immediately (verified end-to-end: a real "hello world" reply came back).
|
||||
privilegedEnvKeys: ['OMP_AUTH_BROKER_URL', 'OMP_AUTH_BROKER_TOKEN', 'HOME'],
|
||||
// Verified end-to-end against a real llama-swap server (live-tested, not just
|
||||
// researched — a real "hello world" reply came back). Same HOME-redirect mechanism
|
||||
// as pi (see its customModelInjection comment for the full reasoning) — omp hardcodes
|
||||
// `~/.omp/agent/models.yml` with no dedicated config-dir override either.
|
||||
customModelInjection: {
|
||||
kind: 'configDir',
|
||||
dirEnvVar: 'HOME',
|
||||
fileName: '.omp/agent/models.yml',
|
||||
template: 'omp-models-yml',
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
// `~/.omp/agent` also holds agent.db/history.db/models.db (SQLite caches) and
|
||||
|
||||
@@ -457,6 +457,46 @@ export interface CliCapabilities {
|
||||
gates: Record<string, { minVersion: string; failClosed: boolean }>;
|
||||
/** Cap on a single terminal frame, when this CLI needs a tighter one than the default. */
|
||||
maxFrameBytes?: number;
|
||||
/**
|
||||
* How this CLI is pointed at a user-supplied custom OpenAI-compatible
|
||||
* endpoint (local, e.g. llama.cpp, or cloud, e.g. Azure AI Foundry) — the
|
||||
* Custom Model Endpoint Profiles feature (`deployment_plan.md`). Declared
|
||||
* per entry, never branched on id, same as every other capability here.
|
||||
*
|
||||
* `env`: plain env vars (claude's `ANTHROPIC_BASE_URL`/`ANTHROPIC_API_KEY`/
|
||||
* `ANTHROPIC_DEFAULT_*_MODEL`). `configContentEnv`: a full config blob
|
||||
* carried in one env var (opencode's `OPENCODE_CONFIG_CONTENT`).
|
||||
* `configDir`: a generated config file under an isolated, dir-redirect-env-
|
||||
* pointed directory so the user's real CLI config is never touched
|
||||
* (codex's `CODEX_HOME`/`config.toml`, pi/omp's `PI_CONFIG_DIR`, grok's
|
||||
* `GROK_HOME`/`config.toml`). `unsupported`: no known mechanism
|
||||
* (antigravity) — the toolbar entry stays disabled for this CLI.
|
||||
*
|
||||
* ⚠️ grok was ORIGINALLY declared as `env` kind (`GROK_BASE_URL`/
|
||||
* `GROK_MODEL`/`XAI_API_KEY`) — that recipe was WRONG, not just unverified:
|
||||
* live-tested against a real grok binary, it produced "Not signed in",
|
||||
* because those env vars are not grok's real custom-endpoint mechanism at
|
||||
* all. The real one is a `[model.<name>]` block in a `config.toml` under
|
||||
* `GROK_HOME` (verified against xAI's own docs), same shape as codex/pi/
|
||||
* omp — this is why the confidence table in deployment_plan.md exists:
|
||||
* "researched" web docs can still be plausible-sounding and wrong.
|
||||
*
|
||||
* Every env var name this introduces that can redirect a session's
|
||||
* traffic MUST also appear in `privilegedEnvKeys` above, exactly like
|
||||
* `DEEPSEEK_BASE_URL` — a non-granted multi-user owner redirecting a
|
||||
* session to their own endpoint is a credential-exfiltration path, not
|
||||
* just a mischief redirect.
|
||||
*/
|
||||
customModelInjection:
|
||||
| { kind: 'env'; baseUrlVar: string; apiKeyVar: string; modelVars: string[] }
|
||||
| { kind: 'configContentEnv'; envVar: string; template: 'opencode-json' }
|
||||
| {
|
||||
kind: 'configDir';
|
||||
dirEnvVar: string;
|
||||
fileName: string;
|
||||
template: 'codex-toml' | 'pi-models-json' | 'omp-models-yml' | 'grok-toml';
|
||||
}
|
||||
| { kind: 'unsupported' };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
/**
|
||||
* @fileoverview Read/write-array store for user-configured custom OpenAI-compatible
|
||||
* model endpoints (local or cloud — deployment_plan.md). Same shape as
|
||||
* `remote-hosts.ts` / `webview-store.ts`: `~/.codeman/custom-model-hosts.json`
|
||||
* holding a plain array, read/written whole.
|
||||
*/
|
||||
|
||||
import { existsSync, mkdirSync } from 'node:fs';
|
||||
import fs from 'node:fs/promises';
|
||||
import { join } from 'node:path';
|
||||
|
||||
const CUSTOM_MODEL_HOSTS_FILE = 'custom-model-hosts.json';
|
||||
|
||||
export type CustomModelAuthStyle = 'bearer' | 'api-key';
|
||||
|
||||
export interface CustomModelHost {
|
||||
id: string;
|
||||
label: string;
|
||||
/** Root URL, local or cloud — e.g. "http://192.168.1.50:8080" or an Azure AI Foundry URL. */
|
||||
baseUrl: string;
|
||||
apiKey?: string;
|
||||
/**
|
||||
* Defaults to 'bearer' (the common `Authorization: Bearer` convention — matches
|
||||
* llama.cpp, OpenAI-compatible servers, and most gateways). Pick 'api-key' for
|
||||
* endpoints that specifically want the `api-key` header, e.g. Azure AI Foundry.
|
||||
*
|
||||
* ⚠️ There is deliberately NO 'both' option. An earlier design sent BOTH headers
|
||||
* on every discovery request on the theory that an unused header is harmless —
|
||||
* live-tested against a real llama-swap server, sending both reliably HUNG the
|
||||
* request indefinitely (reproduced 3× — Bearer alone: ~500ms, api-key alone:
|
||||
* ~600ms, both together: no response inside a 15s timeout). Whatever auth
|
||||
* middleware some servers run apparently does not handle two simultaneous
|
||||
* credential conventions gracefully, so "send everything and let the server
|
||||
* ignore what it doesn't need" is not a safe default — it can silently turn a
|
||||
* working endpoint into one that always times out.
|
||||
*/
|
||||
authStyle?: CustomModelAuthStyle;
|
||||
models?: string[];
|
||||
lastDiscoveredAt?: string;
|
||||
}
|
||||
|
||||
export function customModelHostsPath(configDir: string): string {
|
||||
return join(configDir, CUSTOM_MODEL_HOSTS_FILE);
|
||||
}
|
||||
|
||||
export async function readCustomModelHosts(configDir: string): Promise<CustomModelHost[]> {
|
||||
try {
|
||||
const raw = await fs.readFile(customModelHostsPath(configDir), 'utf-8');
|
||||
const parsed = JSON.parse(raw);
|
||||
return Array.isArray(parsed) ? (parsed as CustomModelHost[]) : [];
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function writeCustomModelHosts(configDir: string, hosts: CustomModelHost[]): Promise<void> {
|
||||
if (!existsSync(configDir)) mkdirSync(configDir, { recursive: true });
|
||||
await fs.writeFile(customModelHostsPath(configDir), JSON.stringify(hosts, null, 2));
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
/**
|
||||
* @fileoverview The one IO wrapper around `custom-model-injection.ts`'s pure
|
||||
* `ConfigDirInjection` output — deliberately split out so that file, the
|
||||
* discovery routes, and `scripts/test-local-llm-harnesses.ts` (via tsx) can
|
||||
* all share EXACTLY one "write these files, merge this env" implementation.
|
||||
* Before this existed, the route and the standalone script each carried
|
||||
* their own copy of this logic, which is exactly the kind of drift the CLI
|
||||
* registry's "declare once, consume everywhere" design exists to prevent —
|
||||
* see deployment_plan.md and the "dynamic to support cli-registry changes"
|
||||
* requirement it was written against.
|
||||
*/
|
||||
|
||||
import { mkdirSync, writeFileSync, rmSync } from 'node:fs';
|
||||
import { join, dirname } from 'node:path';
|
||||
import type { ConfigDirInjection } from './custom-model-injection.js';
|
||||
|
||||
/**
|
||||
* Writes a `ConfigDirInjection`'s files under `baseDir` and returns the full
|
||||
* envOverrides object a caller should merge into the session/process env
|
||||
* (the dir-redirect var plus any `extraEnv` the config file references by
|
||||
* name). Never touches anything outside `baseDir` — the caller is
|
||||
* responsible for choosing an isolated directory (never the user's real
|
||||
* `~/.codex`, `~/.pi`, etc.).
|
||||
*/
|
||||
export function applyConfigDirInjection(baseDir: string, injection: ConfigDirInjection): Record<string, string> {
|
||||
for (const file of injection.files) {
|
||||
const filePath = join(baseDir, file.relPath);
|
||||
mkdirSync(dirname(filePath), { recursive: true });
|
||||
writeFileSync(filePath, file.content, 'utf8');
|
||||
}
|
||||
return { [injection.dirEnvVar]: baseDir, ...injection.extraEnv };
|
||||
}
|
||||
|
||||
/** Best-effort recursive removal of a previously-written configDir. Never throws. */
|
||||
export function removeConfigDir(dir: string | undefined): void {
|
||||
if (!dir) return;
|
||||
try {
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
} catch {
|
||||
// best-effort cleanup only
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,229 @@
|
||||
/**
|
||||
* @fileoverview Pure builder for the Custom Model Endpoint Profiles feature
|
||||
* (deployment_plan.md): turns a CLI registry entry's
|
||||
* `capabilities.customModelInjection` declaration, a configured endpoint,
|
||||
* and a chosen model id into the concrete env vars / config-file content
|
||||
* that would redirect that CLI's session at the endpoint.
|
||||
*
|
||||
* No IO here on purpose (mirrors `session-cli-builder.ts`) — a caller
|
||||
* writes `ConfigDirInjection.files` to disk under an isolated per-session
|
||||
* directory and points `dirEnvVar` at it; this module only computes what
|
||||
* those files/env vars should contain.
|
||||
*
|
||||
* Confidence: `claude` and `opencode` are verified end-to-end against a real
|
||||
* llama-swap server (a real "hello world" reply came back). `codex`'s
|
||||
* config.toml STRUCTURE is now verified (an earlier `[model].default` table
|
||||
* shape was rejected by a real codex binary with "invalid type: map,
|
||||
* expected a string" — caught by `scripts/test-local-llm-harnesses.ts`),
|
||||
* but `wire_api = "responses"` is the only value codex still accepts
|
||||
* (support for `"chat"` was dropped in Feb 2026), and a plain OpenAI
|
||||
* Chat-Completions server (llama.cpp, llama-swap, most local setups) does
|
||||
* NOT implement the Responses API — so codex may still fail at the
|
||||
* PROTOCOL level even with a correctly-shaped config file. That gap is
|
||||
* real and current, not a stale warning; see deployment_plan.md. The rest
|
||||
* (gemini/pi/grok/deepseek/omp) have their ONE-SHOT INVOCATION flags
|
||||
* confirmed against real installed binaries' own `--help` output, but
|
||||
* their custom-endpoint env/config conventions remain web-researched,
|
||||
* unverified.
|
||||
*/
|
||||
|
||||
import type { CliEntry } from './config/cli-registry/types.js';
|
||||
|
||||
export interface CustomModelEndpoint {
|
||||
id: string;
|
||||
label: string;
|
||||
/** Root URL, no trailing slash required — e.g. "http://192.168.1.50:8080" or an Azure AI Foundry URL. */
|
||||
baseUrl: string;
|
||||
/** Falls back to a harmless placeholder for endpoints (llama.cpp) that don't check it. */
|
||||
apiKey?: string;
|
||||
}
|
||||
|
||||
export interface EnvInjection {
|
||||
kind: 'env';
|
||||
/** Ready to merge into a session's envOverrides. */
|
||||
envOverrides: Record<string, string>;
|
||||
}
|
||||
|
||||
export interface ConfigDirInjection {
|
||||
kind: 'configDir';
|
||||
/** Env var that must be set to the directory the caller writes `files` under. */
|
||||
dirEnvVar: string;
|
||||
files: Array<{ relPath: string; content: string }>;
|
||||
/**
|
||||
* Env vars the written config file REFERENCES by name rather than embedding a
|
||||
* literal value (codex's `env_key = "..."` convention: config.toml never carries
|
||||
* the API key itself, only the name of an env var codex reads it from). Merge
|
||||
* these into the session's envOverrides alongside `dirEnvVar` — never skip them,
|
||||
* or the config points at a credential that was never actually set.
|
||||
*/
|
||||
extraEnv?: Record<string, string>;
|
||||
}
|
||||
|
||||
export interface UnsupportedInjection {
|
||||
kind: 'unsupported';
|
||||
}
|
||||
|
||||
export type CustomModelInjectionResult = EnvInjection | ConfigDirInjection | UnsupportedInjection;
|
||||
|
||||
const DEFAULT_API_KEY = 'local-dummy-key';
|
||||
|
||||
/** Normalizes a base URL to end in exactly one trailing `/v1`, for CLIs whose config expects the OpenAI-style suffix. */
|
||||
export function withV1Suffix(baseUrl: string): string {
|
||||
const trimmed = baseUrl.replace(/\/+$/, '');
|
||||
return /\/v1$/.test(trimmed) ? trimmed : `${trimmed}/v1`;
|
||||
}
|
||||
|
||||
/** JSON-escapes a string for embedding in a TOML/YAML double-quoted scalar — a safe superset of both grammars' basic escapes. */
|
||||
function quoted(value: string): string {
|
||||
return JSON.stringify(value);
|
||||
}
|
||||
|
||||
export function buildCustomModelInjection(
|
||||
entry: Pick<CliEntry, 'capabilities'>,
|
||||
endpoint: CustomModelEndpoint,
|
||||
modelId: string
|
||||
): CustomModelInjectionResult {
|
||||
const cap = entry.capabilities.customModelInjection;
|
||||
const apiKey = endpoint.apiKey?.trim() || DEFAULT_API_KEY;
|
||||
|
||||
switch (cap.kind) {
|
||||
case 'env': {
|
||||
const envOverrides: Record<string, string> = {
|
||||
[cap.baseUrlVar]: endpoint.baseUrl,
|
||||
[cap.apiKeyVar]: apiKey,
|
||||
};
|
||||
for (const modelVar of cap.modelVars) envOverrides[modelVar] = modelId;
|
||||
return { kind: 'env', envOverrides };
|
||||
}
|
||||
|
||||
case 'configContentEnv': {
|
||||
const content = renderConfigContent(cap.template, endpoint, modelId, apiKey);
|
||||
return { kind: 'env', envOverrides: { [cap.envVar]: content } };
|
||||
}
|
||||
|
||||
case 'configDir': {
|
||||
const { content, extraEnv } = renderConfigFile(cap.template, endpoint, modelId, apiKey);
|
||||
return { kind: 'configDir', dirEnvVar: cap.dirEnvVar, files: [{ relPath: cap.fileName, content }], extraEnv };
|
||||
}
|
||||
|
||||
case 'unsupported':
|
||||
return { kind: 'unsupported' };
|
||||
}
|
||||
}
|
||||
|
||||
function renderConfigContent(
|
||||
template: 'opencode-json',
|
||||
endpoint: CustomModelEndpoint,
|
||||
modelId: string,
|
||||
apiKey: string
|
||||
): string {
|
||||
switch (template) {
|
||||
case 'opencode-json':
|
||||
return JSON.stringify({
|
||||
$schema: 'https://opencode.ai/config.json',
|
||||
provider: {
|
||||
custom: {
|
||||
options: { baseURL: withV1Suffix(endpoint.baseUrl), apiKey },
|
||||
models: { [modelId]: {} },
|
||||
},
|
||||
},
|
||||
model: `custom/${modelId}`,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const CODEX_API_KEY_ENV_VAR = 'CODEMAN_CUSTOM_MODEL_API_KEY';
|
||||
|
||||
/** The `[model.<name>]` block name grok's config.toml uses for the injected model — also
|
||||
* what `-m <name>` in the standalone script's ONE_SHOT argv must reference to select it. */
|
||||
export const GROK_CUSTOM_MODEL_NAME = 'codeman-custom';
|
||||
|
||||
function renderConfigFile(
|
||||
template: 'codex-toml' | 'pi-models-json' | 'omp-models-yml' | 'grok-toml',
|
||||
endpoint: CustomModelEndpoint,
|
||||
modelId: string,
|
||||
apiKey: string
|
||||
): { content: string; extraEnv?: Record<string, string> } {
|
||||
const baseUrl = withV1Suffix(endpoint.baseUrl);
|
||||
switch (template) {
|
||||
case 'codex-toml': {
|
||||
// Verified against real codex (>= Feb 2026): `model` is a top-level STRING, never
|
||||
// a `[model].default` table — codex rejects that with "invalid type: map, expected
|
||||
// a string" (caught by scripts/test-local-llm-harnesses.ts against a real llama-swap
|
||||
// server). The API key is NEVER a literal TOML field: codex's schema only supports
|
||||
// `env_key`, the NAME of an env var it reads the credential from at runtime, so the
|
||||
// actual value must ride along as an extra env var, never embedded in the file.
|
||||
// ⚠️ `wire_api = "responses"` is the only value codex still accepts (it dropped
|
||||
// `"chat"` support in Feb 2026) — a plain OpenAI Chat-Completions server (llama.cpp,
|
||||
// llama-swap, most local setups) does NOT implement the Responses API, so this
|
||||
// recipe may still fail at the PROTOCOL level even though the file now parses
|
||||
// correctly. That is a real, currently-unresolved compatibility gap, not a syntax
|
||||
// bug — track it before calling codex support done.
|
||||
const content = [
|
||||
`model = ${quoted(modelId)}`,
|
||||
`model_provider = "custom"`,
|
||||
'',
|
||||
'[model_providers.custom]',
|
||||
`name = "Custom Endpoint"`,
|
||||
`base_url = ${quoted(baseUrl)}`,
|
||||
`env_key = ${quoted(CODEX_API_KEY_ENV_VAR)}`,
|
||||
`wire_api = "responses"`,
|
||||
'',
|
||||
].join('\n');
|
||||
return { content, extraEnv: { [CODEX_API_KEY_ENV_VAR]: apiKey } };
|
||||
}
|
||||
case 'pi-models-json':
|
||||
// Verified against pi's OWN bundled docs (models.md): `models` is an ARRAY of
|
||||
// `{id: "..."}` objects, NOT an object keyed by model id — the earlier shape here
|
||||
// silently loaded zero models ("No models available"), confirmed live. `authHeader:
|
||||
// true` is required too: pi does not automatically send `Authorization: Bearer
|
||||
// <apiKey>` just because `apiKey` is set (per the same doc) — without it, a real
|
||||
// (non-llama.cpp) endpoint that actually checks the key would reject every request.
|
||||
return {
|
||||
content: JSON.stringify(
|
||||
{
|
||||
providers: {
|
||||
custom: {
|
||||
baseUrl,
|
||||
apiKey,
|
||||
api: 'openai-completions',
|
||||
authHeader: true,
|
||||
models: [{ id: modelId }],
|
||||
},
|
||||
},
|
||||
},
|
||||
null,
|
||||
2
|
||||
),
|
||||
};
|
||||
case 'omp-models-yml':
|
||||
// Mirrors the pi-models-json fix above (omp shares pi's config lineage per
|
||||
// CLAUDE.md — it reads several of pi's own env vars): a flat list of bare model
|
||||
// name strings under `models` is UNCONFIRMED against real omp docs (none are
|
||||
// bundled with the binary) — this now matches pi's `{id: "..."}` object-list
|
||||
// shape and adds `authHeader: true` on the same reasoning, but has not itself
|
||||
// been live-tested the way pi's fix was. Verify before raising its confidence.
|
||||
return {
|
||||
content: `providers:\n custom:\n baseUrl: ${quoted(baseUrl)}\n apiKey: ${quoted(apiKey)}\n api: openai-completions\n authHeader: true\n models:\n - id: ${quoted(modelId)}\n`,
|
||||
};
|
||||
case 'grok-toml': {
|
||||
// Verified against xAI's own docs (docs.x.ai/build/settings/reference): a
|
||||
// `[model.<name>]` block, NOT plain env vars — an earlier `env`-kind recipe for
|
||||
// grok was wrong, not just unverified (see the customModelInjection doc comment
|
||||
// in cli-registry/types.ts). `api_backend = "chat_completions"` is explicitly
|
||||
// supported (unlike codex, which dropped it after Feb 2026), so this one CAN
|
||||
// talk to a plain OpenAI-compatible server directly. `env_key` reuses grok's own
|
||||
// documented fallback var name (XAI_API_KEY) rather than inventing a new one.
|
||||
const content = [
|
||||
`[model.${GROK_CUSTOM_MODEL_NAME}]`,
|
||||
`model = ${quoted(modelId)}`,
|
||||
`base_url = ${quoted(baseUrl)}`,
|
||||
`name = "Custom Endpoint"`,
|
||||
`env_key = "XAI_API_KEY"`,
|
||||
`api_backend = "chat_completions"`,
|
||||
'',
|
||||
].join('\n');
|
||||
return { content, extraEnv: { XAI_API_KEY: apiKey } };
|
||||
}
|
||||
}
|
||||
}
|
||||
+83
-1
@@ -577,6 +577,14 @@ export class Session extends EventEmitter {
|
||||
// the CLAUDE_CODE_EFFORT_LEVEL env var, which would hard-lock the session.
|
||||
private _effort: EffortLevel | undefined;
|
||||
|
||||
// Custom Model Endpoint Profiles (deployment_plan.md). `envKeys` and `configDir` are
|
||||
// internal bookkeeping ONLY (never surfaced via toState()/customModel getter): they are
|
||||
// what setCustomModel() needs to undo a previous injection (remove exactly the env keys
|
||||
// it added, delete a previous isolated config dir) without guessing what it once wrote.
|
||||
private _customModel:
|
||||
| { endpointId: string; modelId: string; label?: string; envKeys: string[]; configDir?: string }
|
||||
| undefined;
|
||||
|
||||
// tmux history-limit (scrollback lines) allocated when this session's pane is created.
|
||||
private readonly _tmuxHistoryLimit: number;
|
||||
|
||||
@@ -1238,6 +1246,40 @@ export class Session extends EventEmitter {
|
||||
}
|
||||
}
|
||||
|
||||
// Custom Model Endpoint Profiles (deployment_plan.md) — public-safe subset only
|
||||
// (never envKeys/configDir, which are internal bookkeeping for setCustomModel below).
|
||||
get customModel(): { endpointId: string; modelId: string; label?: string } | undefined {
|
||||
if (!this._customModel) return undefined;
|
||||
const { endpointId, modelId, label } = this._customModel;
|
||||
return { endpointId, modelId, label };
|
||||
}
|
||||
|
||||
/**
|
||||
* Update this session's custom-model selection and merge the endpoint's injected env
|
||||
* vars into `_envOverrides` — first UNDOING whatever the previous selection injected
|
||||
* (removing exactly those env keys), so switching endpoints, or clearing back to the
|
||||
* harness's native cloud default, never leaves a stale key behind. Synchronous and
|
||||
* side-effect-free beyond mutating state, matching `setNice`/`setColor` above — this
|
||||
* class does no file IO, so it returns the PREVIOUS `configDir` (if any) for the
|
||||
* caller to clean up on disk (custom-model-injection.ts's configDir kind).
|
||||
*/
|
||||
setCustomModel(
|
||||
next: { endpointId: string; modelId: string; label?: string; envKeys: string[]; configDir?: string } | undefined,
|
||||
envOverrides?: Record<string, string>
|
||||
): string | undefined {
|
||||
const previousConfigDir = this._customModel?.configDir;
|
||||
if (this._customModel) {
|
||||
for (const key of this._customModel.envKeys) {
|
||||
if (this._envOverrides) delete this._envOverrides[key];
|
||||
}
|
||||
}
|
||||
this._customModel = next;
|
||||
if (envOverrides && Object.keys(envOverrides).length > 0) {
|
||||
this._envOverrides = { ...(this._envOverrides ?? {}), ...envOverrides };
|
||||
}
|
||||
return previousConfigDir;
|
||||
}
|
||||
|
||||
// Token tracking getters and setters
|
||||
get totalTokens(): number {
|
||||
return this._totalInputTokens + this._totalOutputTokens;
|
||||
@@ -1478,6 +1520,7 @@ export class Session extends EventEmitter {
|
||||
ompConfig: this._ompConfig,
|
||||
resumeSessionId: this._resumeSessionId,
|
||||
effort: this._effort,
|
||||
customModel: this.customModel,
|
||||
// COD-118: runtime-only — surfaced so the frontend can require explicit user
|
||||
// intent before restarting a crash-looped session. Deliberately NOT restored
|
||||
// by the constructor: a Codeman restart starts with a fresh breaker so boot
|
||||
@@ -1710,10 +1753,49 @@ export class Session extends EventEmitter {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Kill and relaunch this session's CLI process IN PLACE — same pane, same tmux
|
||||
* session, fresh env/args from current state. Custom Model Endpoint Profiles
|
||||
* (deployment_plan.md) is the first caller: after `setCustomModel()` merges new
|
||||
* env vars into `_envOverrides`, the running CLI process still has the OLD env
|
||||
* (inherited at its own process start, not live-reloaded), so switching a
|
||||
* session's model/endpoint requires this restart to actually take effect.
|
||||
*
|
||||
* A GENERALIZED {@link reattachRemote} with the `!this._remote` guard dropped —
|
||||
* `_buildRespawnPaneOptions()` already passes `remote: this._remote` through
|
||||
* unconditionally, so `mux.respawnPane()` builds the right command either way
|
||||
* (a local session gets `respawn-pane -k` + the real launch line, which is the
|
||||
* kill-and-relaunch this method exists for; a remote session gets the existing
|
||||
* reattach-to-durable-tmux behavior). Deliberately does NOT check `isBusy()` —
|
||||
* that's the caller's job (mirrors `/interactive`'s guard), since a raw restart
|
||||
* primitive shouldn't itself decide when it's safe to use.
|
||||
*
|
||||
* @returns true if the pane was respawned, false otherwise (no mux session, or
|
||||
* the mux session is gone — see {@link reattachRemote} for that reasoning).
|
||||
*/
|
||||
async restartCli(): Promise<boolean> {
|
||||
if (!this._useMux || !this._mux || !this._muxSession) return false;
|
||||
const mux = this._mux;
|
||||
|
||||
if (!mux.muxSessionExists(this._muxSession.muxName)) {
|
||||
console.log('[Session] restartCli: mux session gone, skipping:', this._muxSession.muxName);
|
||||
return false;
|
||||
}
|
||||
|
||||
this._pinOmpRespawnId();
|
||||
const newPid = await mux.respawnPane(this._buildRespawnPaneOptions());
|
||||
if (!newPid) {
|
||||
console.error('[Session] restartCli: respawnPane failed for', this._muxSession.muxName);
|
||||
return false;
|
||||
}
|
||||
console.log('[Session] restartCli: restarted CLI for', this._muxSession.muxName, 'pid', newPid);
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Assemble the {@link RespawnPaneOptions} for this session. Single source of
|
||||
* truth shared by interactive start, shell start (via their inline copies),
|
||||
* and {@link reattachRemote} so the remote reattach path can never drift from
|
||||
* {@link reattachRemote}, and {@link restartCli} so no respawn path can drift from
|
||||
* the spawn path.
|
||||
*/
|
||||
private _buildRespawnPaneOptions(): import('./mux-interface.js').RespawnPaneOptions {
|
||||
|
||||
@@ -677,6 +677,13 @@ export interface SessionState {
|
||||
resumeSessionId?: string;
|
||||
/** Claude CLI effort level (soft default via --settings, switchable in-session via /effort) */
|
||||
effort?: EffortLevel;
|
||||
/**
|
||||
* Custom Model Endpoint Profiles (deployment_plan.md): the custom OpenAI-compatible
|
||||
* endpoint (local or cloud) this session's CLI is currently pointed at, if any.
|
||||
* Undefined = the harness's native cloud default. No secrets here — the endpoint's
|
||||
* base URL/api key live only in Session._envOverrides, never in this public state.
|
||||
*/
|
||||
customModel?: { endpointId: string; modelId: string; label?: string };
|
||||
/** Sanitized per-session attachment history. */
|
||||
attachmentHistory?: SessionAttachmentHistoryItem[];
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,128 @@
|
||||
/**
|
||||
* @fileoverview Custom Model Endpoint Profiles CRUD + discovery
|
||||
* (deployment_plan.md). Endpoints are machine-level infra, like remote/docker
|
||||
* hosts, so writes are admin-only in multi-user mode
|
||||
* (`case-routes.ts`'s `/api/remote-hosts` is the pattern this mirrors).
|
||||
*
|
||||
* Discovery (`POST /:id/discover-models`) fetches `${baseUrl}/v1/models`.
|
||||
* `isBlockedWebviewUrl()` is the same synchronous hostname/link-local/cloud-
|
||||
* metadata check `webview-egress-policy.ts` uses for saved dashboard URLs —
|
||||
* reused here as a save-time and discover-time guard. It does NOT re-check
|
||||
* the DNS-RESOLVED address the way `webviewFetch()`'s undici lookup hook
|
||||
* does; wiring that dispatcher-level guard here is a followup, not done in
|
||||
* this pass, since this route is already admin-only in multi-user mode.
|
||||
*/
|
||||
|
||||
import type { FastifyInstance, FastifyRequest } from 'fastify';
|
||||
import { ApiErrorCode, createErrorResponse, type ApiResponse } from '../../types.js';
|
||||
import { isAdmin, parseBody } from '../route-helpers.js';
|
||||
import { isMultiUserMode } from '../../config/multiuser.js';
|
||||
import { getDataDir } from '../../config/instance.js';
|
||||
import { isBlockedWebviewUrl } from '../webview-egress-policy.js';
|
||||
import { CustomModelHostSchema } from '../schemas.js';
|
||||
import { readCustomModelHosts, writeCustomModelHosts, type CustomModelHost } from '../../custom-model-hosts.js';
|
||||
|
||||
const CODEMAN_CONFIG_DIR = getDataDir();
|
||||
const DISCOVER_TIMEOUT_MS = 8000;
|
||||
|
||||
function adminOnly(req: FastifyRequest, reply: { code: (n: number) => unknown }): ApiResponse<never> | null {
|
||||
if (!isMultiUserMode() || isAdmin(req)) return null;
|
||||
reply.code(403);
|
||||
return createErrorResponse(ApiErrorCode.FORBIDDEN, 'Admin only in multi-user mode');
|
||||
}
|
||||
|
||||
async function discoverModels(host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>): Promise<string[]> {
|
||||
const headers: Record<string, string> = {};
|
||||
const apiKey = host.apiKey?.trim();
|
||||
// Exactly ONE header, never both — see custom-model-hosts.ts's CustomModelAuthStyle
|
||||
// doc comment for why: sending both reliably HANGS some real servers.
|
||||
const style = host.authStyle ?? 'bearer';
|
||||
if (apiKey && style === 'bearer') headers.Authorization = `Bearer ${apiKey}`;
|
||||
if (apiKey && style === 'api-key') headers['api-key'] = apiKey;
|
||||
|
||||
const res = await fetch(`${host.baseUrl.replace(/\/+$/, '')}/v1/models`, {
|
||||
headers,
|
||||
signal: AbortSignal.timeout(DISCOVER_TIMEOUT_MS),
|
||||
});
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
||||
const body = (await res.json()) as { data?: Array<{ id?: unknown }> };
|
||||
return (body.data ?? []).map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
|
||||
}
|
||||
|
||||
export function registerCustomModelRoutes(app: FastifyInstance): void {
|
||||
app.get('/api/model-endpoints', async (req) =>
|
||||
isMultiUserMode() && !isAdmin(req) ? [] : readCustomModelHosts(CODEMAN_CONFIG_DIR)
|
||||
);
|
||||
|
||||
app.post('/api/model-endpoints', async (req, reply): Promise<ApiResponse<{ host: CustomModelHost }>> => {
|
||||
const denied = adminOnly(req, reply);
|
||||
if (denied) return denied;
|
||||
const host = parseBody(CustomModelHostSchema, req.body);
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
if (hosts.some((item) => item.id === host.id)) {
|
||||
return createErrorResponse(ApiErrorCode.ALREADY_EXISTS, 'Model endpoint already exists');
|
||||
}
|
||||
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, [...hosts, host]);
|
||||
return { success: true, data: { host } };
|
||||
});
|
||||
|
||||
app.put('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ host: CustomModelHost }>> => {
|
||||
const denied = adminOnly(req, reply);
|
||||
if (denied) return denied;
|
||||
const { id } = req.params as { id: string };
|
||||
const host = parseBody(CustomModelHostSchema, { ...(req.body as object), id });
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
const index = hosts.findIndex((item) => item.id === id);
|
||||
if (index === -1) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
const next = [...hosts];
|
||||
next[index] = host;
|
||||
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
|
||||
return { success: true, data: { host } };
|
||||
});
|
||||
|
||||
app.delete('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ id: string }>> => {
|
||||
const denied = adminOnly(req, reply);
|
||||
if (denied) return denied;
|
||||
const { id } = req.params as { id: string };
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
await writeCustomModelHosts(
|
||||
CODEMAN_CONFIG_DIR,
|
||||
hosts.filter((item) => item.id !== id)
|
||||
);
|
||||
return { success: true, data: { id } };
|
||||
});
|
||||
|
||||
app.post(
|
||||
'/api/model-endpoints/:id/discover-models',
|
||||
async (req, reply): Promise<ApiResponse<{ models: string[] }>> => {
|
||||
const denied = adminOnly(req, reply);
|
||||
if (denied) return denied;
|
||||
const { id } = req.params as { id: string };
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
const index = hosts.findIndex((item) => item.id === id);
|
||||
if (index === -1) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
const host = hosts[index];
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
try {
|
||||
const models = await discoverModels(host);
|
||||
const next = [...hosts];
|
||||
next[index] = { ...host, models, lastDiscoveredAt: new Date().toISOString() };
|
||||
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
|
||||
return { success: true, data: { models } };
|
||||
} catch (err) {
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.OPERATION_FAILED,
|
||||
`Could not reach endpoint: ${err instanceof Error ? err.message : String(err)}`
|
||||
);
|
||||
}
|
||||
}
|
||||
);
|
||||
}
|
||||
@@ -27,3 +27,4 @@ export { registerWsRoutes } from './ws-routes.js';
|
||||
export { registerVoiceRoutes } from './voice-routes.js';
|
||||
export { registerWebviewRoutes, tryWebviewRefererFallback } from './webview-routes.js';
|
||||
export { registerTabLayoutRoutes } from './tab-layout-routes.js';
|
||||
export { registerCustomModelRoutes } from './custom-model-routes.js';
|
||||
|
||||
@@ -51,7 +51,11 @@ import {
|
||||
SessionOrderUpdateSchema,
|
||||
SessionWaitQuerySchema,
|
||||
SessionWaitOutputQuerySchema,
|
||||
CustomModelSelectionSchema,
|
||||
} from '../schemas.js';
|
||||
import { readCustomModelHosts } from '../../custom-model-hosts.js';
|
||||
import { buildCustomModelInjection } from '../../custom-model-injection.js';
|
||||
import { applyConfigDirInjection, removeConfigDir } from '../../custom-model-injection-apply.js';
|
||||
import { ownerLayoutKey } from '../../tab-layout-persistence.js';
|
||||
import { TabLayoutValidationError } from '../../tab-layout.js';
|
||||
import {
|
||||
@@ -1152,6 +1156,80 @@ export function registerSessionRoutes(
|
||||
return { color: session.color };
|
||||
});
|
||||
|
||||
// ========== Custom Model Endpoint Profiles (deployment_plan.md) ==========
|
||||
//
|
||||
// Applies (or clears) a session's custom OpenAI-compatible endpoint selection and
|
||||
// RESTARTS the pane's CLI process — these harnesses read endpoint config at process
|
||||
// start, not per-turn, so a live hot-swap isn't possible (confirmed with the
|
||||
// maintainer). Endpoints come from the admin-configured custom-model-hosts store
|
||||
// (chunk 3's CRUD routes), never raw client-supplied env — that's what keeps this
|
||||
// route safe to let any session owner call for their own session, unlike the
|
||||
// generic envOverrides field the privilegedEnvKeys clamp exists to guard.
|
||||
app.post('/api/sessions/:id/custom-model', async (req) => {
|
||||
const { id } = req.params as { id: string };
|
||||
const body = parseBody(CustomModelSelectionSchema, req.body, 'Invalid request body');
|
||||
const session = findSessionOrFail(ctx, id, req);
|
||||
|
||||
if (session.isBusy()) {
|
||||
return createErrorResponse(ApiErrorCode.SESSION_BUSY, 'Session is busy');
|
||||
}
|
||||
|
||||
if ('clear' in body) {
|
||||
const previousConfigDir = session.setCustomModel(undefined);
|
||||
removeConfigDir(previousConfigDir);
|
||||
const restarted = await session.restartCli();
|
||||
persistAndBroadcastSession(ctx, session);
|
||||
return { customModel: session.customModel, restarted };
|
||||
}
|
||||
|
||||
const entry = getCli(session.mode);
|
||||
if (!entry) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, `No CLI registry entry for mode ${session.mode}`);
|
||||
}
|
||||
if (entry.capabilities.customModelInjection.kind === 'unsupported') {
|
||||
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${session.mode} has no known custom-model mechanism`);
|
||||
}
|
||||
|
||||
const hosts = await readCustomModelHosts(getDataDir());
|
||||
const endpoint = hosts.find((h) => h.id === body.endpointId);
|
||||
if (!endpoint) {
|
||||
return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
}
|
||||
|
||||
const injection = buildCustomModelInjection(entry, endpoint, body.modelId);
|
||||
|
||||
let envOverrides: Record<string, string>;
|
||||
let envKeys: string[];
|
||||
let configDir: string | undefined;
|
||||
|
||||
if (injection.kind === 'env') {
|
||||
envOverrides = injection.envOverrides;
|
||||
envKeys = Object.keys(injection.envOverrides);
|
||||
} else if (injection.kind === 'configDir') {
|
||||
// Isolated per-session dir — never the user's real CLI config path.
|
||||
configDir = join(dataPath('custom-model-configs'), session.id);
|
||||
envOverrides = applyConfigDirInjection(configDir, injection);
|
||||
envKeys = Object.keys(envOverrides);
|
||||
} else {
|
||||
// 'unsupported' is already handled above; this keeps the switch exhaustive.
|
||||
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${session.mode} has no known custom-model mechanism`);
|
||||
}
|
||||
|
||||
const previousConfigDir = session.setCustomModel(
|
||||
{ endpointId: endpoint.id, modelId: body.modelId, label: endpoint.label, envKeys, configDir },
|
||||
envOverrides
|
||||
);
|
||||
// Clean up the OLD config dir on disk, unless the new one happens to reuse the same
|
||||
// path (same session, configDir kind again) — never delete the dir we just wrote.
|
||||
if (previousConfigDir && previousConfigDir !== configDir) {
|
||||
removeConfigDir(previousConfigDir);
|
||||
}
|
||||
|
||||
const restarted = await session.restartCli();
|
||||
persistAndBroadcastSession(ctx, session);
|
||||
return { customModel: session.customModel, restarted };
|
||||
});
|
||||
|
||||
// ========== Delete Session ==========
|
||||
|
||||
app.delete('/api/sessions/:id', async (req) => {
|
||||
|
||||
@@ -739,6 +739,29 @@ export const RemoteHostSchema = z.object({
|
||||
commands: RemoteCommandOverridesSchema,
|
||||
});
|
||||
|
||||
// Custom Model Endpoint Profiles (deployment_plan.md) — a user-configured custom
|
||||
// OpenAI-compatible endpoint, local (llama.cpp) or cloud (Azure AI Foundry, etc.).
|
||||
export const CustomModelHostSchema = z.object({
|
||||
id: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid endpoint id'),
|
||||
label: z.string().min(1).max(100),
|
||||
baseUrl: z.string().url().max(2048),
|
||||
apiKey: z.string().max(4096).optional(),
|
||||
// No 'both': live-tested against a real server, sending both auth header
|
||||
// conventions on one request reliably HANGS it — see custom-model-hosts.ts.
|
||||
authStyle: z.enum(['bearer', 'api-key']).optional(),
|
||||
models: z.array(z.string().max(200)).max(200).optional(),
|
||||
lastDiscoveredAt: z.string().max(64).optional(),
|
||||
});
|
||||
|
||||
/** POST /api/sessions/:id/custom-model — apply or clear a session's custom-model selection. */
|
||||
export const CustomModelSelectionSchema = z.union([
|
||||
z.object({
|
||||
endpointId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid endpoint id'),
|
||||
modelId: z.string().min(1).max(200),
|
||||
}),
|
||||
z.object({ clear: z.literal(true) }),
|
||||
]);
|
||||
|
||||
export const RemoteCaseLinkSchema = z.object({
|
||||
name: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid case name format'),
|
||||
hostId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid remote host id'),
|
||||
@@ -1237,6 +1260,13 @@ export const SettingsUpdateSchema = z
|
||||
* stored profiles stay until DELETE /api/sessions/:id/intent.
|
||||
*/
|
||||
readMyMindEnabled: z.boolean().optional(),
|
||||
/**
|
||||
* Custom Model Endpoint Profiles (deployment_plan.md): the toolbar picker that lets a
|
||||
* session point at a user-configured custom OpenAI-compatible endpoint (local or
|
||||
* cloud) instead of its native cloud backend. SYNCED, default OFF — endpoint entry,
|
||||
* discovery, and the extra toolbar surface are all opt-in.
|
||||
*/
|
||||
customModelEndpointsEnabled: z.boolean().optional(),
|
||||
/**
|
||||
* Read My Mind predictor model override. Empty/absent = the AI-checker
|
||||
* default (opus: prediction quality is the product and it runs only on an
|
||||
|
||||
@@ -185,6 +185,7 @@ import {
|
||||
registerVoiceRoutes,
|
||||
registerWebviewRoutes,
|
||||
registerTabLayoutRoutes,
|
||||
registerCustomModelRoutes,
|
||||
tryWebviewRefererFallback,
|
||||
} from './routes/index.js';
|
||||
import { isLostWebviewFrameNavigation } from './webview-proxy.js';
|
||||
@@ -1070,6 +1071,7 @@ export class WebServer extends EventEmitter {
|
||||
registerOrchestratorRoutes(this.app, ctx);
|
||||
registerWebviewRoutes(this.app, ctx, this.basePath);
|
||||
registerTabLayoutRoutes(this.app, ctx);
|
||||
registerCustomModelRoutes(this.app);
|
||||
|
||||
// Cron: build the service from the same context, recompute
|
||||
// due times for any persisted jobs, then expose it to its routes.
|
||||
|
||||
Reference in New Issue
Block a user