mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-05 23:19:43 +02:00
fix(custom-model): root-cause and fix DeepSeek's HTTP_404 (missing /v1)
DeepSeek Harness's own bundled provider module
(@deepseek-ai/dsh-llm-deepseek) builds its request URL as
`${DEEPSEEK_BASE_URL}/chat/completions` with no `/v1` insertion of its
own (its real public API, https://api.deepseek.com, expects the
caller's base URL to already carry any needed prefix), while
llama-swap/llama.cpp only ever serves the OpenAI-conventional
`/v1/chat/completions`.
Confirmed two ways:
- Installed the real @deepseek-ai/dsh package (all its actual
published dependencies) into a scratch dir purely to read
dsh-llm-deepseek's source: `fetch(`${connection.baseURL}/chat/
completions`, ...)`, baseURL read straight from DEEPSEEK_BASE_URL —
the same grep-the-real-source bar pi/grok's fixes were held to.
- Live against the test-picker's llama-swap: `POST <baseUrl>/chat/
completions` -> 404, `POST <baseUrl>/v1/chat/completions` -> 200,
same endpoint. dsh's own error template ("DeepSeek API error (HTTP
${status})") reproduces the originally-reported
"dsh: HTTP_404: DeepSeek API error (HTTP 404)" exactly.
- New registry field `appendV1Suffix` (env kind only, deepseek's entry
alone — claude/gemini must NOT get it, since claude was already
confirmed working against the unmodified baseUrl). When set,
buildCustomModelInjection runs endpoint.baseUrl through the same
withV1Suffix() helper configDir-kind CLIs (pi/grok/codex) already
use, instead of writing it verbatim.
Not yet re-run end-to-end through a real dsh binary — no install
available in this environment (not in PATH, and the test-picker
container doesn't bundle it) — so this is source-confirmed and
live-verified at the HTTP level, not yet promoted to "verified"
alongside claude/opencode/pi/grok/omp. Docs (custom-model-endpoints.md,
the plan doc's confidence table, the wiki page, CLAUDE.md) all updated
to reflect this precisely rather than leaving the old "root cause not
identified" claim in place.
2 new/updated tests for the /v1 suffix (including idempotency against
a baseUrl that already ends in /v1) plus a corrected mock-server
contract test. Typecheck/lint clean; full suite shows no new
regressions.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RqZeHrRS6DYcGcGX2p9EwG
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
8520925e76
commit
e2034177c5
@@ -364,6 +364,11 @@ const capabilitiesSchema = z
|
||||
// same file to pre-seed the state a real, already-onboarded profile carries. See
|
||||
// the customModelInjection doc comment in cli-registry/types.ts.
|
||||
skipFirstRunPrompts: z.boolean().optional(),
|
||||
// DeepSeek-only, confirmed by reading its own bundled SDK source: it concatenates
|
||||
// "/chat/completions" onto baseUrlVar's value with no "/v1" of its own, while
|
||||
// llama-swap/llama.cpp only serves the "/v1/..." path — claude/gemini must NOT
|
||||
// get this. See the customModelInjection doc comment in cli-registry/types.ts.
|
||||
appendV1Suffix: z.boolean().optional(),
|
||||
})
|
||||
.strict(),
|
||||
z
|
||||
|
||||
@@ -1103,15 +1103,28 @@ const DEEPSEEK: CliEntry = {
|
||||
// privilege rather than granting it, and clamping it here was a real regression
|
||||
// (test/deepseek-mode.test.ts) fixed before this shipped.
|
||||
privilegedEnvKeys: ['DSH_PERMISSION_MODE', 'DSH_HOME', 'DEEPSEEK_BASE_URL'],
|
||||
// Web-researched, unverified, partial: reuses the already-existing DEEPSEEK_BASE_URL/
|
||||
// DEEPSEEK_API_KEY keys above. No modelVars — dsh's model is a profile-composition
|
||||
// entry (see `model: { source: 'none' }` above), not an env var, so forcing a specific
|
||||
// model name may not fully work; verify against a real profile before shipping.
|
||||
// Reuses the already-existing DEEPSEEK_BASE_URL/DEEPSEEK_API_KEY keys above. No
|
||||
// modelVars — dsh's model is a profile-composition entry (see `model: { source: 'none'
|
||||
// }` above), not an env var, so forcing a specific model name may not fully work;
|
||||
// verify against a real profile before shipping.
|
||||
//
|
||||
// ⚠️ appendV1Suffix is REQUIRED, not optional-nice-to-have: without it every request
|
||||
// 404s. Confirmed live and by reading dsh's own bundled source
|
||||
// (@deepseek-ai/dsh-llm-deepseek): it builds the request URL as
|
||||
// `${DEEPSEEK_BASE_URL}/chat/completions` with no "/v1" of its own (its real public
|
||||
// API, https://api.deepseek.com, expects the caller's base URL to already carry any
|
||||
// needed prefix), while llama-swap/llama.cpp only serves the OpenAI-conventional
|
||||
// "/v1/chat/completions" — a bare POST to ".../chat/completions" 404s live, and the
|
||||
// 404 reported here originally ("dsh: HTTP_404: DeepSeek API error (HTTP 404)")
|
||||
// matches dsh's own error-message template for exactly this failure. See the
|
||||
// customModelInjection doc comment in cli-registry/types.ts for the full reasoning,
|
||||
// including why claude/gemini must NOT get this.
|
||||
customModelInjection: {
|
||||
kind: 'env',
|
||||
baseUrlVar: 'DEEPSEEK_BASE_URL',
|
||||
apiKeyVar: 'DEEPSEEK_API_KEY',
|
||||
modelVars: [],
|
||||
appendV1Suffix: true,
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
|
||||
@@ -535,6 +535,21 @@ export interface CliCapabilities {
|
||||
* in claude's `settings.json` — see `seedFirstRunState`/`seedSkipBypassPermissionsPrompt`
|
||||
* in custom-model-injection-apply.ts. Requires `apiKeyTrustFile` to be set too, since it
|
||||
* reuses that file.
|
||||
*
|
||||
* `appendV1Suffix` (env kind only): the raw `endpoint.baseUrl` gets `withV1Suffix()`
|
||||
* applied before being written to `baseUrlVar`, instead of being used verbatim.
|
||||
* DeepSeek needs this and claude/gemini must NOT get it — a per-CLI asymmetry confirmed
|
||||
* by reading each SDK's own request-building source, not assumed: DeepSeek Harness's
|
||||
* bundled `@deepseek-ai/dsh-llm-deepseek` concatenates `${connection.baseURL}/chat/
|
||||
* completions` with no `/v1` insertion of its own (its real public API base,
|
||||
* `https://api.deepseek.com`, expects the caller's base URL to already carry any
|
||||
* needed prefix), while llama-swap/llama.cpp only ever serves the OpenAI-conventional
|
||||
* `/v1/chat/completions` — confirmed live: a bare `POST <baseUrl>/chat/completions`
|
||||
* 404s, `POST <baseUrl>/v1/chat/completions` succeeds, and the harness's own error
|
||||
* message template (`DeepSeek API error (HTTP ${status})`) reproduces the exact
|
||||
* `HTTP_404` this feature originally shipped with unexplained. Claude Code's own SDK,
|
||||
* by contrast, was already confirmed working end-to-end against the RAW `baseUrl` with
|
||||
* no suffix — appending one there would be wrong, not just redundant.
|
||||
*/
|
||||
customModelInjection:
|
||||
| {
|
||||
@@ -547,6 +562,7 @@ export interface CliCapabilities {
|
||||
apiKeyTrustFile?: { relPath: string; shape: 'claude-api-key-responses' };
|
||||
configDirVar?: string;
|
||||
skipFirstRunPrompts?: boolean;
|
||||
appendV1Suffix?: boolean;
|
||||
}
|
||||
| { kind: 'configContentEnv'; envVar: string; template: 'opencode-json'; launchModel?: string }
|
||||
| {
|
||||
|
||||
@@ -121,7 +121,7 @@ export function buildCustomModelInjection(
|
||||
switch (cap.kind) {
|
||||
case 'env': {
|
||||
const envOverrides: Record<string, string> = {
|
||||
[cap.baseUrlVar]: endpoint.baseUrl,
|
||||
[cap.baseUrlVar]: cap.appendV1Suffix ? withV1Suffix(endpoint.baseUrl) : endpoint.baseUrl,
|
||||
[cap.apiKeyVar]: apiKey,
|
||||
};
|
||||
for (const modelVar of cap.modelVars) envOverrides[modelVar] = modelId;
|
||||
|
||||
Reference in New Issue
Block a user