mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-09-30 12:39:42 +02:00
Point any Codeman-supported harness (Claude, opencode, Codex, Gemini, Pi, Grok, DeepSeek, OMP) at a custom OpenAI-compatible endpoint instead of its native cloud backend, for a given session. Covers local hardware (llama.cpp, Ollama, vLLM, DGX Spark, Strix Halo) and cloud (Azure AI Foundry, OpenRouter). Off by default (customModelEndpointsEnabled, synced, default OFF). - Registry: capabilities.customModelInjection per CLI entry (env / configContentEnv / configDir / unsupported kinds) - Pure injection builder (custom-model-injection.ts) turning an endpoint + model id into the real env vars / config content per CLI - Endpoint store + CRUD routes (custom-model-hosts.ts, custom-model-routes.ts), discovery via GET /v1/models, SSRF-guarded - Session integration: Session.setCustomModel()/restartCli() (POST /api/sessions/:id/custom-model), reusing the existing respawn-pane -k primitive to restart the CLI process with new env - Multi-user hardening: every new redirect-capable env var added to its CLI's privilegedEnvKeys, closing a pre-existing gap where several were already reachable via the generic envOverrides field's prefix allowlist - Standalone scripts/test-local-llm-harnesses.mjs: spawns real CLI binaries against a real endpoint outside the web UI, independent of tmux/sessions - Mock-server contract tests (test/fixtures/mock-openai-server.ts) replaying every CLI's injected values through a real HTTP shape Real end-to-end validation against a live llama-swap server (inside a codeman/agent:llm-test Docker image with all 9 CLI binaries) found and fixed three real bugs before they shipped: - Codex's config.toml schema was wrong ([model].default table instead of a top-level model string + [model_providers.custom]); fixing it then surfaced a genuine, documented protocol incompatibility (Codex only speaks the Responses API since Feb 2026, which llama.cpp/llama-swap don't implement) - Claude Code's async session-title-generation call validates ANTHROPIC_DEFAULT_HAIKU_MODEL against its own internal model list and hangs the whole -p invocation on an unrecognized name; documented for chunk 6, worked around in the standalone script only (--bare is NOT safe for a real interactive session, which needs hooks) - The discovery route's authStyle: 'both' option (send both Authorization and api-key headers) reliably hung a real server; removed the option entirely rather than just changing the default Status: draft. Chunk 6 (frontend toolbar/settings UI) not yet built — see PR.md and deployment_plan.md for the full chunk breakdown and confidence table. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017HqNWfmtBU2KN29SvSVWB3
119 lines
4.0 KiB
TypeScript
119 lines
4.0 KiB
TypeScript
/**
|
|
* @fileoverview In-process fake OpenAI/Anthropic-compatible HTTP server for the
|
|
* Custom Model Endpoint Profiles contract tests (deployment_plan.md chunk 7).
|
|
*
|
|
* No external deps — plain `node:http`. Captures every request it receives
|
|
* (method, path, headers, parsed JSON body) so a test can assert the injected
|
|
* base URL / API key / model actually reached the right place, with the right
|
|
* auth header, in the shape a real llama.cpp/Azure/etc. endpoint would see it.
|
|
*
|
|
* Serves the request shapes this feature's recipes produce: OpenAI-style
|
|
* `POST /v1/chat/completions` (opencode/pi/grok/omp/gemini's compat
|
|
* endpoint), Anthropic-style `POST /v1/messages` (claude's ANTHROPIC_BASE_URL
|
|
* traffic), OpenAI's newer `POST /v1/responses` (codex's actual wire protocol
|
|
* as of Feb 2026 — it dropped chat-completions support), plus `GET /v1/models`
|
|
* for the discovery route's own tests.
|
|
*/
|
|
|
|
import { createServer, type IncomingMessage, type Server } from 'node:http';
|
|
import { AddressInfo } from 'node:net';
|
|
|
|
export interface CapturedRequest {
|
|
method: string;
|
|
path: string;
|
|
headers: Record<string, string | string[] | undefined>;
|
|
body: unknown;
|
|
}
|
|
|
|
export interface MockOpenAiServer {
|
|
baseUrl: string;
|
|
requests: CapturedRequest[];
|
|
close(): Promise<void>;
|
|
}
|
|
|
|
function readJsonBody(req: IncomingMessage): Promise<unknown> {
|
|
return new Promise((resolve) => {
|
|
const chunks: Buffer[] = [];
|
|
req.on('data', (c) => chunks.push(c));
|
|
req.on('end', () => {
|
|
const raw = Buffer.concat(chunks).toString('utf8');
|
|
if (!raw) return resolve(undefined);
|
|
try {
|
|
resolve(JSON.parse(raw));
|
|
} catch {
|
|
resolve(raw);
|
|
}
|
|
});
|
|
});
|
|
}
|
|
|
|
/** Starts the mock server on a random free port and resolves once it's listening. */
|
|
export async function startMockOpenAiServer(): Promise<MockOpenAiServer> {
|
|
const requests: CapturedRequest[] = [];
|
|
|
|
const server: Server = createServer((req, res) => {
|
|
void (async () => {
|
|
const body = await readJsonBody(req);
|
|
const path = (req.url ?? '').split('?')[0];
|
|
requests.push({ method: req.method ?? 'GET', path, headers: req.headers, body });
|
|
|
|
res.setHeader('content-type', 'application/json');
|
|
|
|
if (path === '/v1/models' && req.method === 'GET') {
|
|
res.writeHead(200);
|
|
res.end(JSON.stringify({ data: [{ id: 'qwen3' }, { id: 'llama3' }] }));
|
|
return;
|
|
}
|
|
|
|
if (path === '/v1/chat/completions' && req.method === 'POST') {
|
|
res.writeHead(200);
|
|
res.end(
|
|
JSON.stringify({
|
|
id: 'mock-completion',
|
|
choices: [{ index: 0, message: { role: 'assistant', content: 'hello world' } }],
|
|
})
|
|
);
|
|
return;
|
|
}
|
|
|
|
if (path === '/v1/messages' && req.method === 'POST') {
|
|
res.writeHead(200);
|
|
res.end(
|
|
JSON.stringify({
|
|
id: 'mock-message',
|
|
role: 'assistant',
|
|
content: [{ type: 'text', text: 'hello world' }],
|
|
})
|
|
);
|
|
return;
|
|
}
|
|
|
|
// Codex's real wire protocol (verified against a live binary: it dropped
|
|
// wire_api="chat" support in Feb 2026, so its config.toml always says
|
|
// wire_api="responses") — a different shape from OpenAI's chat-completions.
|
|
if (path === '/v1/responses' && req.method === 'POST') {
|
|
res.writeHead(200);
|
|
res.end(
|
|
JSON.stringify({
|
|
id: 'mock-response',
|
|
output: [{ type: 'message', role: 'assistant', content: [{ type: 'output_text', text: 'hello world' }] }],
|
|
})
|
|
);
|
|
return;
|
|
}
|
|
|
|
res.writeHead(404);
|
|
res.end(JSON.stringify({ error: 'not found in mock server', path }));
|
|
})();
|
|
});
|
|
|
|
await new Promise<void>((resolve) => server.listen(0, '127.0.0.1', resolve));
|
|
const { port } = server.address() as AddressInfo;
|
|
|
|
return {
|
|
baseUrl: `http://127.0.0.1:${port}`,
|
|
requests,
|
|
close: () => new Promise<void>((resolve, reject) => server.close((err) => (err ? reject(err) : resolve()))),
|
|
};
|
|
}
|