Merge pull request #430 from opticon454/custom-model-run-menu

This commit is contained in:
Codeman maintainer
2026-09-19 12:25:11 +02:00
43 changed files with 7304 additions and 144 deletions
+29
View File
@@ -340,6 +340,35 @@ const capabilitiesSchema = z
// an env var, so it declares baseUrl/apiKey injection with no model var at all.
modelVars: z.array(envName).max(8),
launchModel: launchModelTemplate,
// Optional: the env var to carry a discovered per-model context-window size
// (claude's CLAUDE_CODE_MAX_CONTEXT_TOKENS), and/or the env var that isolates
// this session's config/credential directory from the user's real one (claude's
// CLAUDE_CONFIG_DIR) so an injected API key never collides with a stored OAuth
// session. See the customModelInjection doc comment in cli-registry/types.ts.
contextLengthVar: envName.optional(),
configDirVar: envName.optional(),
// Relative path, WITHIN the isolated configDirVar directory, of a trust-dialog
// seed file the CLI itself owns the shape of — claude's `.claude.json`
// `customApiKeyResponses.approved` list, the same field an interactive "Detected
// a custom API key — use it?" prompt writes to on a real terminal. Only makes
// sense alongside configDirVar (an isolated, otherwise-empty directory has none
// of a real profile's prior approvals), and only implemented for the
// 'claude-api-key-responses' shape today — see custom-model-injection-apply.ts.
apiKeyTrustFile: z
.object({ relPath: z.string().min(1).max(80), shape: z.literal('claude-api-key-responses') })
.strict()
.optional(),
// An isolated config directory replays the CLI's whole first-run sequence (theme
// picker, security notes, per-project trust dialog, bypass-permissions warning)
// on every launch, same root cause as apiKeyTrustFile above — this reuses that
// same file to pre-seed the state a real, already-onboarded profile carries. See
// the customModelInjection doc comment in cli-registry/types.ts.
skipFirstRunPrompts: z.boolean().optional(),
// DeepSeek-only, confirmed by reading its own bundled SDK source: it concatenates
// "/chat/completions" onto baseUrlVar's value with no "/v1" of its own, while
// llama-swap/llama.cpp only serves the "/v1/..." path — claude/gemini must NOT
// get this. See the customModelInjection doc comment in cli-registry/types.ts.
appendV1Suffix: z.boolean().optional(),
})
.strict(),
z
+62 -7
View File
@@ -228,15 +228,31 @@ const CLAUDE: CliEntry = {
privilegedParams: [],
// ANTHROPIC_* is NOT in allowedPrefixes/allowedKeys above (deliberately — see the
// allowedPrefixes comment nearby), so these are unreachable via plain envOverrides
// today; listed here only so the dedicated custom-model route (docs/custom-model-endpoints-plan.md
// chunk 5) clamps them for a non-granted multi-user owner the same way every other
// CLI's injection vars are clamped, the day that route widens who can set them.
// today. privilegedEnvKeys has exactly one consumer, ownerClampedEnvKeys() in
// session-env-clamp.ts, which feeds the generic envOverrides clamp on
// POST /api/sessions, POST /api/quick-start and reboot-restore — no custom-model
// route reads this field at all, and the values it injects are merged in AFTER
// that clamp runs regardless of what's listed here.
privilegedEnvKeys: [
'ANTHROPIC_BASE_URL',
'ANTHROPIC_API_KEY',
'ANTHROPIC_DEFAULT_SONNET_MODEL',
'ANTHROPIC_DEFAULT_HAIKU_MODEL',
'ANTHROPIC_DEFAULT_OPUS_MODEL',
// CLAUDE_CODE_MAX_CONTEXT_TOKENS already matches the CLAUDE_CODE_* allowedPrefix, and
// CLAUDE_CONFIG_DIR is already an allowed exact key (docs/wiki/Agent-CLIs.md), so both
// were already reachable via plain envOverrides before this pair existed and this
// feature does not strictly need either listed. They stay listed anyway, because
// types.ts's rule ("every traffic-redirecting var this feature introduces MUST also
// appear in privilegedEnvKeys") is meant to hold literally, not with an exception
// carved out for the two vars that happen not to need it today. The real
// consequence lands on the GENERIC envOverrides clamp above, not on this feature:
// a non-granted multi-user owner can no longer set CLAUDE_CONFIG_DIR through
// envOverrides at all (the per-client-account override, #255), and a PERSISTED one
// is now stripped on reboot-restore for such an owner too — see
// session-env-clamp.ts's own fileoverview.
'CLAUDE_CODE_MAX_CONTEXT_TOKENS',
'CLAUDE_CONFIG_DIR',
],
gates: { nameFlag: { minVersion: '2.1.224', failClosed: true } },
// Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md) — verified by hand against a real
@@ -247,6 +263,32 @@ const CLAUDE: CliEntry = {
baseUrlVar: 'ANTHROPIC_BASE_URL',
apiKeyVar: 'ANTHROPIC_API_KEY',
modelVars: ['ANTHROPIC_DEFAULT_SONNET_MODEL', 'ANTHROPIC_DEFAULT_HAIKU_MODEL', 'ANTHROPIC_DEFAULT_OPUS_MODEL'],
// Verified via Claude Code's own docs: CLAUDE_CODE_MAX_CONTEXT_TOKENS overrides the
// assumed context window and applies directly for a model name Claude Code doesn't
// recognize as one of its own — exactly the custom-model case. Without it, Claude Code
// assumes a large (200k) window for any unrecognized model id and never compacts,
// eventually overflowing a much smaller real local context (see plan doc reasoning
// above the interface for the confirmed failure).
contextLengthVar: 'CLAUDE_CODE_MAX_CONTEXT_TOKENS',
// Isolates this session's config/credential directory so an injected ANTHROPIC_API_KEY
// never shares a directory with a stored claude.ai OAuth login — see the doc comment on
// customModelInjection in cli-registry/types.ts for the traded-off side effect.
configDirVar: 'CLAUDE_CONFIG_DIR',
// ⚠️ Required alongside configDirVar, not optional in practice: verified live that an
// isolated, otherwise-empty config directory makes claude stop at an interactive
// "Detected a custom API key — use it?" prompt on EVERY launch, defaulting to "No" with
// no one at the TTY to answer — silently refusing the very key this feature injected.
// Pre-seeding this file's customApiKeyResponses.approved list (verified against a real
// ~/.claude.json after answering the prompt once by hand) answers it in advance instead.
apiKeyTrustFile: { relPath: '.claude.json', shape: 'claude-api-key-responses' },
// ⚠️ Same isolated-directory root cause, one step further: verified live that on top
// of the API-key prompt above, a fresh CLAUDE_CONFIG_DIR also replays claude's ENTIRE
// first-run sequence on every launch — the theme picker, the security-notes screen,
// the per-project "trust this folder?" dialog, and (running with
// --dangerously-skip-permissions) a one-time bypass-permissions warning — none of
// which a real, already-onboarded profile shows again. Pre-seeds that same
// already-onboarded state instead of leaving a human to click through it.
skipFirstRunPrompts: true,
},
},
overlays: {
@@ -1071,15 +1113,28 @@ const DEEPSEEK: CliEntry = {
// privilege rather than granting it, and clamping it here was a real regression
// (test/deepseek-mode.test.ts) fixed before this shipped.
privilegedEnvKeys: ['DSH_PERMISSION_MODE', 'DSH_HOME', 'DEEPSEEK_BASE_URL'],
// Web-researched, unverified, partial: reuses the already-existing DEEPSEEK_BASE_URL/
// DEEPSEEK_API_KEY keys above. No modelVars — dsh's model is a profile-composition
// entry (see `model: { source: 'none' }` above), not an env var, so forcing a specific
// model name may not fully work; verify against a real profile before shipping.
// Reuses the already-existing DEEPSEEK_BASE_URL/DEEPSEEK_API_KEY keys above. No
// modelVars — dsh's model is a profile-composition entry (see `model: { source: 'none'
// }` above), not an env var, so forcing a specific model name may not fully work;
// verify against a real profile before shipping.
//
// ⚠️ appendV1Suffix is REQUIRED, not optional-nice-to-have: without it every request
// 404s. Confirmed live and by reading dsh's own bundled source
// (@deepseek-ai/dsh-llm-deepseek): it builds the request URL as
// `${DEEPSEEK_BASE_URL}/chat/completions` with no "/v1" of its own (its real public
// API, https://api.deepseek.com, expects the caller's base URL to already carry any
// needed prefix), while llama-swap/llama.cpp only serves the OpenAI-conventional
// "/v1/chat/completions" — a bare POST to ".../chat/completions" 404s live, and the
// 404 reported here originally ("dsh: HTTP_404: DeepSeek API error (HTTP 404)")
// matches dsh's own error-message template for exactly this failure. See the
// customModelInjection doc comment in cli-registry/types.ts for the full reasoning,
// including why claude/gemini must NOT get this.
customModelInjection: {
kind: 'env',
baseUrlVar: 'DEEPSEEK_BASE_URL',
apiKeyVar: 'DEEPSEEK_API_KEY',
modelVars: [],
appendV1Suffix: true,
},
},
overlays: {
+66 -1
View File
@@ -496,9 +496,74 @@ export interface CliCapabilities {
* declares). Absent = the config alone selects the model (claude's env vars,
* opencode's blob, codex's top-level `model` key). Applied by the session's
* respawn options through the entry's `legacyConfigField`, never by id.
*
* `contextLengthVar` (env kind only): the env var a discovered per-model context-window
* size is written to when known (claude's `CLAUDE_CODE_MAX_CONTEXT_TOKENS`) — without it,
* a CLI that assumes a large default window for an unrecognized model name keeps sending
* full-size prompts against a much smaller local server and eventually overflows its real
* context (verified: a 33.7K-token system prompt against a 16384-token llama-swap model).
* Absent when the CLI has no such override, or the value is unknown for this model.
*
* `configDirVar` (env kind only): the env var that redirects this session's config/
* credential directory to an isolated, per-session one (claude's `CLAUDE_CONFIG_DIR`), so
* an injected API key never coexists with a stored claude.ai OAuth session in the same
* directory — the CLI still warns "both claude.ai and ANTHROPIC_API_KEY set" when they
* share a directory even though the API key wins for actual requests. Isolating it trades
* that cosmetic warning for a documented side effect: a relocated config directory writes
* transcripts outside `~/.claude/projects`, blinding the response viewer, subagent
* windows, and Read My Mind for that session (see docs/wiki/Agent-CLIs.md).
*
* `apiKeyTrustFile` (env kind only, alongside configDirVar): an isolated config directory
* has none of a real profile's prior "detected a custom API key, use it?" approvals, so
* without this the CLI stops and asks interactively on every single launch — with no one
* at a TTY to answer, that's a hang, not a warning (confirmed live: claude's own default
* answer, "No", would silently refuse to use the very key this feature just injected).
* `relPath`/`shape` name the file (claude's `.claude.json`) and its
* `customApiKeyResponses.approved` field this pre-seeds — the exact field a real answered
* prompt itself writes to, so this isn't bypassing the check, just answering it the same
* way a one-off prior approval on a shared profile already would.
*
* `skipFirstRunPrompts` (env kind only, alongside apiKeyTrustFile): an isolated config
* directory is not just missing API-key approvals — it is a brand-new profile as far as
* the CLI is concerned, so it also replays its ENTIRE first-run sequence on every launch:
* the theme picker, the security-notes screen, the per-project "trust this folder?"
* dialog, and (running with a bypass-permissions flag) a one-time warning about it —
* confirmed live, none of which a real, long-used profile ever shows again. `true`
* pre-seeds the same state a real profile accumulates from having answered all of that
* once: `hasCompletedOnboarding` and the launching session's own project entry in the
* `apiKeyTrustFile` (claude's `.claude.json`), plus `skipDangerousModePermissionPrompt`
* in claude's `settings.json` — see `seedFirstRunState`/`seedSkipBypassPermissionsPrompt`
* in custom-model-injection-apply.ts. Requires `apiKeyTrustFile` to be set too, since it
* reuses that file.
*
* `appendV1Suffix` (env kind only): the raw `endpoint.baseUrl` gets `withV1Suffix()`
* applied before being written to `baseUrlVar`, instead of being used verbatim.
* DeepSeek needs this and claude/gemini must NOT get it — a per-CLI asymmetry confirmed
* by reading each SDK's own request-building source, not assumed: DeepSeek Harness's
* bundled `@deepseek-ai/dsh-llm-deepseek` concatenates `${connection.baseURL}/chat/
* completions` with no `/v1` insertion of its own (its real public API base,
* `https://api.deepseek.com`, expects the caller's base URL to already carry any
* needed prefix), while llama-swap/llama.cpp only ever serves the OpenAI-conventional
* `/v1/chat/completions` — confirmed live: a bare `POST <baseUrl>/chat/completions`
* 404s, `POST <baseUrl>/v1/chat/completions` succeeds, and the harness's own error
* message template (`DeepSeek API error (HTTP ${status})`) reproduces the exact
* `HTTP_404` this feature originally shipped with unexplained. Claude Code's own SDK,
* by contrast, was already confirmed working end-to-end against the RAW `baseUrl` with
* no suffix — appending one there would be wrong, not just redundant.
*/
customModelInjection:
| { kind: 'env'; baseUrlVar: string; apiKeyVar: string; modelVars: string[]; launchModel?: string }
| {
kind: 'env';
baseUrlVar: string;
apiKeyVar: string;
modelVars: string[];
launchModel?: string;
contextLengthVar?: string;
apiKeyTrustFile?: { relPath: string; shape: 'claude-api-key-responses' };
configDirVar?: string;
skipFirstRunPrompts?: boolean;
appendV1Suffix?: boolean;
}
| { kind: 'configContentEnv'; envVar: string; template: 'opencode-json'; launchModel?: string }
| {
kind: 'configDir';