Merge pull request #430 from opticon454/custom-model-run-menu

This commit is contained in:
Codeman maintainer
2026-09-19 12:25:11 +02:00
43 changed files with 7304 additions and 144 deletions
@@ -0,0 +1,368 @@
/**
* @fileoverview Tests for `refreshAllCustomModelHosts()`, the periodic
* background sweep behind server.ts's "custom model endpoint re-discovery"
* timer (docs/custom-model-endpoints-plan.md). Kept in its own file rather
* than folded into test/routes/custom-model-routes.test.ts: that file's data
* dir is shared across every test in it (one temp HOME per FILE, not per
* test — test/setup.ts), and a sweep that walks every saved host would pick
* up every host any other test in that file happened to create, making an
* exact call-count or exact-host assertion meaningless. A dedicated file
* gets its own clean temp HOME.
*
* Port: N/A (no server; drives readCustomModelHosts/writeCustomModelHosts
* directly plus the mocked webviewFetch dispatcher).
*/
import { describe, it, expect, vi, beforeEach } from 'vitest';
import { getDataDir } from '../src/config/instance.js';
import { readCustomModelHosts, writeCustomModelHosts, type CustomModelHost } from '../src/custom-model-hosts.js';
import { refreshAllCustomModelHosts } from '../src/web/routes/custom-model-routes.js';
import { webviewFetch } from '../src/web/webview-egress.js';
vi.mock('../src/web/webview-egress.js', async () => {
const actual = await vi.importActual<typeof import('../src/web/webview-egress.js')>('../src/web/webview-egress.js');
return { ...actual, webviewFetch: vi.fn() };
});
const fetchMock = vi.mocked(webviewFetch);
function host(overrides: Partial<CustomModelHost> & Pick<CustomModelHost, 'id' | 'baseUrl'>): CustomModelHost {
return { label: overrides.id, ...overrides };
}
beforeEach(() => {
fetchMock.mockReset();
});
describe('refreshAllCustomModelHosts (the periodic re-discovery sweep)', () => {
it('refreshes every saved endpoint, best-effort — one unreachable host does not stop the others', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [
host({ id: 'ok', baseUrl: 'http://localhost:8080' }),
host({ id: 'down', baseUrl: 'http://localhost:8081' }),
]);
fetchMock.mockImplementation(async (url: URL) => {
if (url.href.includes('8081')) throw new TypeError('fetch failed', { cause: new Error('ECONNREFUSED') });
return new Response(JSON.stringify({ data: [{ id: 'qwen3' }] }), { status: 200 });
});
await refreshAllCustomModelHosts();
const hosts = await readCustomModelHosts(dir);
const ok = hosts.find((h) => h.id === 'ok');
const down = hosts.find((h) => h.id === 'down');
expect(ok?.models).toEqual(['qwen3']);
expect(ok?.lastDiscoveredAt).toBeTruthy();
expect(down?.models ?? []).toEqual([]);
expect(down?.lastDiscoveredAt).toBeFalsy();
});
it('skips a host whose baseUrl is blocked, without making a request', async () => {
const dir = getDataDir();
// Written directly rather than through the POST route, which already
// refuses this at save time — this simulates a record that pre-dates the
// guard, or was hand-edited on disk. The sweep must not trust it either.
await writeCustomModelHosts(dir, [host({ id: 'meta', baseUrl: 'http://169.254.169.254/' })]);
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'x' }] }), { status: 200 }));
await refreshAllCustomModelHosts();
expect(fetchMock).not.toHaveBeenCalled();
});
it('drops a stale default and preserves lastDiscoveredAt semantics, same as manual discovery', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [
host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['qwen3'], defaultModelId: 'qwen3' }),
]);
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'llama3' }] }), { status: 200 }));
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.models).toEqual(['llama3']);
expect(updated.defaultModelId).toBeUndefined();
expect(updated.lastDiscoveredAt).toBeTruthy();
});
it('keeps a default that is still present after the sweep', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [
host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['qwen3'], defaultModelId: 'qwen3' }),
]);
fetchMock.mockResolvedValue(
new Response(JSON.stringify({ data: [{ id: 'qwen3' }, { id: 'llama3' }] }), { status: 200 })
);
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.defaultModelId).toBe('qwen3');
});
it('does not resurrect an endpoint deleted while the sweep was in flight', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'deleted', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockImplementation(async () => {
// Simulate an admin deleting the endpoint between the sweep's fetch and
// its read-modify-write — the delete must win, not be overwritten by a
// refresh that started before it.
const current = await readCustomModelHosts(dir);
await writeCustomModelHosts(
dir,
current.filter((h) => h.id !== 'deleted')
);
return new Response(JSON.stringify({ data: [{ id: 'qwen3' }] }), { status: 200 });
});
await expect(refreshAllCustomModelHosts()).resolves.toBeUndefined();
const hosts = await readCustomModelHosts(dir);
expect(hosts.find((h) => h.id === 'deleted')).toBeUndefined();
});
it('leaves the store untouched when there are no saved endpoints at all', async () => {
await expect(refreshAllCustomModelHosts()).resolves.toBeUndefined();
expect(fetchMock).not.toHaveBeenCalled();
});
});
describe('refreshAllCustomModelHosts: context-length enrichment (llama.cpp/llama-swap /props)', () => {
it('probes /props?model= only for a model reported loaded, and stores its n_ctx', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/v1/models') {
return new Response(
JSON.stringify({
data: [
{ id: 'loaded-model', status: { value: 'loaded' } },
{ id: 'unloaded-model', status: { value: 'unloaded' } },
],
}),
{ status: 200 }
);
}
if (url.pathname === '/props') {
// Must never be reached for the unloaded model — asserted below by call count.
expect(url.searchParams.get('model')).toBe('loaded-model');
return new Response(JSON.stringify({ n_ctx: 16384 }), { status: 200 });
}
throw new Error(`unexpected request: ${url.href}`);
});
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelContextLengths).toEqual({ 'loaded-model': 16384 });
const propsCalls = fetchMock.mock.calls.filter(([url]) => (url as URL).pathname === '/props');
expect(propsCalls).toHaveLength(1);
});
it('never probes /props at all when no entry mentions status — feature-detected, not assumed unloaded', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'qwen3' }] }), { status: 200 }));
await refreshAllCustomModelHosts();
expect(fetchMock).toHaveBeenCalledTimes(1); // /v1/models only
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelContextLengths).toBeUndefined();
});
it('keeps a previously-learned context length for a model no longer loaded, drops it once the model disappears entirely', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [
host({
id: 'ep',
baseUrl: 'http://localhost:8080',
models: ['a', 'b'],
modelContextLengths: { a: 8192, b: 4096 },
}),
]);
// This round: 'a' is loaded (re-confirmed), 'b' is gone from the list entirely.
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/v1/models') {
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
}
return new Response(JSON.stringify({ n_ctx: 8192 }), { status: 200 });
});
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelContextLengths).toEqual({ a: 8192 });
});
it('a failed /props probe for the loaded model is swallowed, leaving no context length rather than failing the sweep', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/v1/models') {
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
}
return new Response('nope', { status: 500 });
});
await expect(refreshAllCustomModelHosts()).resolves.toBeUndefined();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelContextLengths).toBeUndefined();
});
it('prefers the REAL configured context size parsed from /running’s launch command over /props’s unreliable n_ctx', async () => {
// Confirmed live: llama-swap launched a model with --fit-ctx 16384 (the real, working
// limit — the actual server then refused a request over it), but /props reported
// n_ctx: 154112 for the same model, well over what it would really accept. /props must
// never be reached at all once the /running command parse already answered it.
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/v1/models') {
return new Response(JSON.stringify({ data: [{ id: 'qwen3.8-27b', status: { value: 'loaded' } }] }), {
status: 200,
});
}
if (url.pathname === '/running') {
return new Response(
JSON.stringify({
running: [
{
model: 'qwen3.8-27b',
state: 'ready',
cmd: 'llama-server -m /models/Qwen3.8-27B.gguf --flash-attn on --jinja --fit-ctx 16384 --host 0.0.0.0 --port 5840',
},
],
}),
{ status: 200 }
);
}
if (url.pathname === '/props') throw new Error('must never be reached — the cmd parse already answered it');
throw new Error(`unexpected request: ${url.href}`);
});
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelContextLengths).toEqual({ 'qwen3.8-27b': 16384 });
});
it('falls back to /props when /running has no cmd, or the cmd states no recognizable context flag', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/v1/models') {
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
}
if (url.pathname === '/running') {
return new Response(
JSON.stringify({ running: [{ model: 'a', state: 'ready', cmd: 'llama-server -m /models/a.gguf' }] }),
{ status: 200 }
);
}
if (url.pathname === '/props') return new Response(JSON.stringify({ n_ctx: 8192 }), { status: 200 });
throw new Error(`unexpected request: ${url.href}`);
});
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelContextLengths).toEqual({ a: 8192 });
});
it('also recognizes a plain -c/--ctx-size flag, not just llama-swap’s own --fit-ctx', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/v1/models') {
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
}
if (url.pathname === '/running') {
return new Response(
JSON.stringify({
running: [{ model: 'a', state: 'ready', cmd: 'llama-server -m /models/a.gguf --ctx-size 8192' }],
}),
{ status: 200 }
);
}
throw new Error(`unexpected request: ${url.href}`); // /props must never be reached
});
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelContextLengths).toEqual({ a: 8192 });
});
});
describe('refreshAllCustomModelHosts: model-size enrichment (parsed from /v1/models description)', () => {
it('parses a GB figure out of an auto-discovered model’s description', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockResolvedValue(
new Response(
JSON.stringify({
data: [{ id: 'qwen3.8-27b', description: 'Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp' }],
}),
{ status: 200 }
)
);
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelSizesGB).toEqual({ 'qwen3.8-27b': 16.35 });
});
it('gets no size at all for a hand-configured profile whose own description states none', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockResolvedValue(
new Response(
JSON.stringify({
data: [{ id: 'big', description: 'General-purpose reasoning model, MoE CPU-offloaded. Default profile.' }],
}),
{ status: 200 }
)
);
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelSizesGB).toBeUndefined();
});
it('populated regardless of loaded state — unlike context length, no /props probe is needed', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/v1/models') {
return new Response(
JSON.stringify({
data: [{ id: 'unloaded-model', description: 'Auto-discovered 4.91 GB - parameters auto-fitted' }],
}),
{ status: 200 }
);
}
throw new Error(`unexpected request: ${url.href}`); // /props must never be reached for this
});
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelSizesGB).toEqual({ 'unloaded-model': 4.91 });
});
it('keeps a previously-learned size for a model still present, drops it once the model disappears entirely', async () => {
const dir = getDataDir();
await writeCustomModelHosts(dir, [
host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['a', 'b'], modelSizesGB: { a: 8, b: 16 } }),
]);
fetchMock.mockResolvedValue(
new Response(JSON.stringify({ data: [{ id: 'a', description: 'no GB figure here' }] }), { status: 200 })
);
await refreshAllCustomModelHosts();
const [updated] = await readCustomModelHosts(dir);
expect(updated.modelSizesGB).toEqual({ a: 8 }); // 'a' kept from before, 'b' dropped (gone from the list)
});
});
+303
View File
@@ -0,0 +1,303 @@
/**
* @fileoverview Tests for the two custom-model IO-layer fixes on top of the pure builder
* (docs/custom-model-endpoints-plan.md):
*
* 1. `contextLengthVar` — a discovered per-model context length reaches the actual
* session env (CLAUDE_CODE_MAX_CONTEXT_TOKENS), so a CLI stops assuming a large
* default window for an unrecognized custom model id and overflowing a much
* smaller real one.
* 2. `configDirVar` — an isolated, empty config directory is created and pointed at
* (CLAUDE_CONFIG_DIR), so an injected API key never shares a directory with a
* stored claude.ai OAuth session; `projects` is symlinked back into the real
* config dir so the response viewer/subagent windows/Read My Mind keep working.
*
* Port: N/A (no server; filesystem-only, under a temp CODEMAN data dir from test/setup.ts).
*/
import { existsSync, lstatSync, mkdirSync, readFileSync, readdirSync, rmSync, writeFileSync } from 'node:fs';
import { homedir } from 'node:os';
import { join } from 'node:path';
import { afterEach, describe, expect, it } from 'vitest';
import { getCli } from '../src/config/cli-registry/index.js';
import { applyCustomModelInjection, customModelConfigDir } from '../src/custom-model-injection-apply.js';
import type { CustomModelEndpoint } from '../src/custom-model-injection.js';
const endpoint: CustomModelEndpoint = {
id: 'ep1',
label: 'llama.cpp box',
baseUrl: 'http://192.168.1.50:8080',
apiKey: 'my-key',
};
function entryOrThrow(id: string) {
const entry = getCli(id);
if (!entry) throw new Error(`missing CLI registry entry: ${id}`);
return entry;
}
const sessionsToClean: string[] = [];
afterEach(() => {
for (const id of sessionsToClean.splice(0)) rmSync(customModelConfigDir(id), { recursive: true, force: true });
});
describe('applyCustomModelInjection: context length', () => {
it('claude: passes a known context length through to CLAUDE_CODE_MAX_CONTEXT_TOKENS', () => {
sessionsToClean.push('sess-ctx-1');
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', 'sess-ctx-1', 16384);
expect(applied?.envOverrides.CLAUDE_CODE_MAX_CONTEXT_TOKENS).toBe('16384');
expect(applied?.envKeys).toContain('CLAUDE_CODE_MAX_CONTEXT_TOKENS');
});
it('claude: omits the var entirely when the context length is unknown', () => {
sessionsToClean.push('sess-ctx-2');
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', 'sess-ctx-2');
expect(applied?.envOverrides.CLAUDE_CODE_MAX_CONTEXT_TOKENS).toBeUndefined();
});
it('deepseek: has no contextLengthVar declared, so a passed-in length is a no-op', () => {
const applied = applyCustomModelInjection(entryOrThrow('deepseek'), endpoint, 'qwen3', 'sess-ctx-3', 16384);
expect(Object.keys(applied?.envOverrides ?? {}).sort()).toEqual(['DEEPSEEK_API_KEY', 'DEEPSEEK_BASE_URL']);
});
});
describe('applyCustomModelInjection: CLAUDE_CONFIG_DIR isolation', () => {
it('claude: creates an isolated config dir (no real credential/config files) and points CLAUDE_CONFIG_DIR at it', () => {
const sessionId = 'sess-cfgdir-1';
sessionsToClean.push(sessionId);
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
const expectedDir = customModelConfigDir(sessionId);
expect(applied?.envOverrides.CLAUDE_CONFIG_DIR).toBe(expectedDir);
expect(applied?.configDir).toBe(expectedDir);
expect(existsSync(expectedDir)).toBe(true);
// The trust-seed file, the skipFirstRunPrompts settings.json, and the projects link —
// no real OAuth credential/config.
const entries = readdirSync(expectedDir).filter((name) => name !== 'projects');
expect(entries.sort()).toEqual(['.claude.json', 'settings.json']);
});
it('claude: symlinks (or junctions) projects back to the real config dir so the response viewer keeps working', () => {
const sessionId = 'sess-cfgdir-2';
sessionsToClean.push(sessionId);
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
const link = join(applied!.configDir!, 'projects');
// Best-effort: only assert the link exists if it was actually created (the real
// ~/.claude/projects may not exist on a bare CI box, in which case linking is skipped).
if (existsSync(join(homedir(), '.claude', 'projects'))) {
expect(existsSync(link)).toBe(true);
expect(lstatSync(link).isSymbolicLink() || lstatSync(link).isDirectory()).toBe(true);
}
});
it('claude: re-applying to the same session is idempotent (boot-recovery re-apply)', () => {
const sessionId = 'sess-cfgdir-3';
sessionsToClean.push(sessionId);
const first = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
const second = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
expect(second?.configDir).toBe(first?.configDir);
expect(existsSync(first!.configDir!)).toBe(true);
});
it('pi: configDir-kind CLIs are unaffected — no configDirVar concept for them', () => {
const sessionId = 'sess-cfgdir-pi';
sessionsToClean.push(sessionId);
const applied = applyCustomModelInjection(entryOrThrow('pi'), endpoint, 'qwen3', sessionId);
expect(applied?.envOverrides.HOME).toBe(customModelConfigDir(sessionId));
});
it('deepseek: no configDirVar declared, so no config dir is created at all', () => {
const sessionId = 'sess-cfgdir-deepseek';
const applied = applyCustomModelInjection(entryOrThrow('deepseek'), endpoint, 'qwen3', sessionId);
expect(applied?.configDir).toBeUndefined();
expect(existsSync(customModelConfigDir(sessionId))).toBe(false);
});
});
describe('applyCustomModelInjection: apiKeyTrustFile (pre-approves the injected key)', () => {
it('claude: seeds .claude.json so the "Detected a custom API key" prompt never fires', () => {
const sessionId = 'sess-trust-1';
sessionsToClean.push(sessionId);
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
customApiKeyResponses: { approved: string[]; rejected: string[] };
};
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
expect(written.customApiKeyResponses.rejected).toEqual([]);
});
it('claude: falls back to the dummy key when the endpoint has none, and still seeds it', () => {
const sessionId = 'sess-trust-2';
sessionsToClean.push(sessionId);
const applied = applyCustomModelInjection(
entryOrThrow('claude'),
{ ...endpoint, apiKey: undefined },
'qwen3',
sessionId
);
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
customApiKeyResponses: { approved: string[] };
};
expect(written.customApiKeyResponses.approved).toEqual(['local-dummy-key']);
});
it('claude: merges onto fields the CLI itself already wrote into the same isolated dir, never overwrites them', () => {
const sessionId = 'sess-trust-3';
sessionsToClean.push(sessionId);
const configDir = customModelConfigDir(sessionId);
mkdirSync(configDir, { recursive: true });
writeFileSync(join(configDir, '.claude.json'), JSON.stringify({ userID: 'abc123', numStartups: 3 }));
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
userID: string;
numStartups: number;
customApiKeyResponses: { approved: string[] };
};
expect(written.userID).toBe('abc123');
expect(written.numStartups).toBe(3);
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
});
it('claude: a corrupt existing file is treated as absent rather than failing the apply', () => {
const sessionId = 'sess-trust-4';
sessionsToClean.push(sessionId);
const configDir = customModelConfigDir(sessionId);
mkdirSync(configDir, { recursive: true });
writeFileSync(join(configDir, '.claude.json'), '{ not valid json');
expect(() => applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId)).not.toThrow();
const written = JSON.parse(readFileSync(join(configDir, '.claude.json'), 'utf8')) as {
customApiKeyResponses: { approved: string[] };
};
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
});
it('claude: re-approving the same key does not duplicate it in the approved list', () => {
const sessionId = 'sess-trust-5';
sessionsToClean.push(sessionId);
applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
const second = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'llama3', sessionId);
const written = JSON.parse(readFileSync(join(second!.configDir!, '.claude.json'), 'utf8')) as {
customApiKeyResponses: { approved: string[] };
};
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
});
it('opencode: has no apiKeyTrustFile declared (no configDirVar at all), nothing is seeded', () => {
const sessionId = 'sess-trust-opencode';
const applied = applyCustomModelInjection(entryOrThrow('opencode'), endpoint, 'qwen3', sessionId);
expect(applied?.configDir).toBeUndefined();
expect(existsSync(customModelConfigDir(sessionId))).toBe(false);
});
});
describe("applyCustomModelInjection: skipFirstRunPrompts (an isolated dir replays claude's whole first-run sequence)", () => {
it("claude: seeds hasCompletedOnboarding and this session's own project trust into .claude.json", () => {
const sessionId = 'sess-firstrun-1';
sessionsToClean.push(sessionId);
const applied = applyCustomModelInjection(
entryOrThrow('claude'),
endpoint,
'qwen3',
sessionId,
undefined,
'/home/user/myproject'
);
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
hasCompletedOnboarding: boolean;
projects: Record<string, { hasTrustDialogAccepted: boolean }>;
};
expect(written.hasCompletedOnboarding).toBe(true);
expect(written.projects['/home/user/myproject'].hasTrustDialogAccepted).toBe(true);
});
it('claude: seeds skipDangerousModePermissionPrompt into settings.json', () => {
const sessionId = 'sess-firstrun-2';
sessionsToClean.push(sessionId);
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
const written = JSON.parse(readFileSync(join(applied!.configDir!, 'settings.json'), 'utf8')) as {
skipDangerousModePermissionPrompt: boolean;
};
expect(written.skipDangerousModePermissionPrompt).toBe(true);
});
it('claude: with no workingDir given (boot recovery), hasCompletedOnboarding/settings still seed, but no project entry is added', () => {
const sessionId = 'sess-firstrun-3';
sessionsToClean.push(sessionId);
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
hasCompletedOnboarding: boolean;
projects?: Record<string, unknown>;
};
expect(written.hasCompletedOnboarding).toBeUndefined();
expect(written.projects).toBeUndefined();
});
it("claude: merges onto an existing project entry's other fields rather than overwriting them", () => {
const sessionId = 'sess-firstrun-4';
sessionsToClean.push(sessionId);
const configDir = customModelConfigDir(sessionId);
mkdirSync(configDir, { recursive: true });
writeFileSync(
join(configDir, '.claude.json'),
JSON.stringify({ projects: { '/home/user/myproject': { allowedTools: ['Bash'] } } })
);
const applied = applyCustomModelInjection(
entryOrThrow('claude'),
endpoint,
'qwen3',
sessionId,
undefined,
'/home/user/myproject'
);
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
projects: Record<string, { allowedTools: string[]; hasTrustDialogAccepted: boolean }>;
};
expect(written.projects['/home/user/myproject'].allowedTools).toEqual(['Bash']);
expect(written.projects['/home/user/myproject'].hasTrustDialogAccepted).toBe(true);
});
it('claude: a corrupt existing settings.json is treated as absent rather than failing the apply', () => {
const sessionId = 'sess-firstrun-5';
sessionsToClean.push(sessionId);
const configDir = customModelConfigDir(sessionId);
mkdirSync(configDir, { recursive: true });
writeFileSync(join(configDir, 'settings.json'), '{ not valid json');
expect(() => applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId)).not.toThrow();
const written = JSON.parse(readFileSync(join(configDir, 'settings.json'), 'utf8')) as {
skipDangerousModePermissionPrompt: boolean;
};
expect(written.skipDangerousModePermissionPrompt).toBe(true);
});
it('pi: has no skipFirstRunPrompts concept (no apiKeyTrustFile either) — nothing beyond its own config file', () => {
const sessionId = 'sess-firstrun-pi';
sessionsToClean.push(sessionId);
const applied = applyCustomModelInjection(
entryOrThrow('pi'),
endpoint,
'qwen3',
sessionId,
undefined,
'/home/user/myproject'
);
const entries = readdirSync(applied!.configDir!);
expect(entries).not.toContain('settings.json');
});
});
describe('applyCustomModelInjection: pre-existing behavior unaffected', () => {
it('opencode: still returns a plain env-kind result with no configDir', () => {
const sessionId = 'sess-opencode-1';
const applied = applyCustomModelInjection(entryOrThrow('opencode'), endpoint, 'qwen3', sessionId);
expect(applied?.configDir).toBeUndefined();
expect(applied?.envOverrides.OPENCODE_CONFIG_CONTENT).toBeTruthy();
});
it('antigravity: still undefined (unsupported)', () => {
const applied = applyCustomModelInjection(entryOrThrow('antigravity'), endpoint, 'qwen3', 'sess-agy-1');
expect(applied).toBeUndefined();
});
});
+21 -17
View File
@@ -142,17 +142,19 @@ describe('custom-model-injection contract (mock server)', () => {
expect(mock.requests[0].headers.authorization).toBe('Bearer contract-test-key');
});
// gemini/deepseek's `env` kind passes the base URL through UNCHANGED (unlike
// opencode/codex/pi/omp/grok, which build a structured config and explicitly append
// /v1) — matching Anthropic's own convention for claude's ANTHROPIC_BASE_URL, where the
// SDK appends the path itself. Whether each of these TWO CLIs' own OpenAI-compatible
// client expects the var to already include /v1 (the common OpenAI-SDK convention) or
// appends it itself is genuinely CLI-specific and UNVERIFIED (see the confidence table
// in docs/custom-model-endpoints-plan.md) — these tests model the common OpenAI-SDK convention (base_url
// ends in /v1) since that's the more likely behavior for an OpenAI-compatible client,
// but that assumption should be corrected here the moment it's checked against a real
// binary. (grok WAS in this group too, until live-testing showed the whole `env` recipe
// was wrong for it — see its own test below.)
// gemini's `env` kind still passes the base URL through UNCHANGED (matching
// Anthropic's own convention for claude's ANTHROPIC_BASE_URL, where the SDK appends
// the path itself) — whether gemini-cli's own OpenAI-compatible-ish client expects the
// var to already include /v1 or appends it itself remains genuinely UNVERIFIED (it
// fails for an unrelated auth reason before this would even matter — see the
// confidence table in docs/custom-model-endpoints-plan.md); this test models the
// common OpenAI-SDK convention as the best guess, to be corrected the moment it's
// checked against a real client. deepseek WAS in this "passes through unchanged"
// group too, until reading `@deepseek-ai/dsh-llm-deepseek`'s own bundled source
// confirmed it builds its request URL as `${DEEPSEEK_BASE_URL}/chat/completions` with
// no `/v1` of its own — `appendV1Suffix` now fixes that (see its own test below),
// the same way grok's whole `env` recipe turned out to be wrong before live-testing
// corrected it to a `configDir` one.
it('gemini: GOOGLE_GEMINI_BASE_URL/GEMINI_API_KEY reach the mock', async () => {
const injection = buildCustomModelInjection(entryOrThrow('gemini'), endpointFor(mock), 'qwen3');
@@ -186,16 +188,18 @@ describe('custom-model-injection contract (mock server)', () => {
expect(mock.requests[0].headers.authorization).toBe('Bearer contract-test-key');
});
it('deepseek: DEEPSEEK_BASE_URL/DEEPSEEK_API_KEY reach the mock (base URL/key only, no model var)', async () => {
it('deepseek: DEEPSEEK_BASE_URL already carries the /v1 suffix dsh itself never adds, reaching the mock at the real path dsh requests', async () => {
// Confirmed by reading dsh's own bundled source: it fetches
// `${DEEPSEEK_BASE_URL}/chat/completions` verbatim, no /v1 insertion of its own — so
// this call (unlike gemini's above) passes DEEPSEEK_BASE_URL to callOpenAiCompat
// UNMODIFIED, exactly mirroring what the real harness does, rather than the test
// helping it along.
const injection = buildCustomModelInjection(entryOrThrow('deepseek'), endpointFor(mock), 'qwen3');
if (injection.kind !== 'env') throw new Error('unreachable');
expect(Object.keys(injection.envOverrides).sort()).toEqual(['DEEPSEEK_API_KEY', 'DEEPSEEK_BASE_URL']);
expect(injection.envOverrides.DEEPSEEK_BASE_URL).toBe(`${mock.baseUrl}/v1`);
await callOpenAiCompat(
`${injection.envOverrides.DEEPSEEK_BASE_URL}/v1`,
injection.envOverrides.DEEPSEEK_API_KEY,
'qwen3'
);
await callOpenAiCompat(injection.envOverrides.DEEPSEEK_BASE_URL, injection.envOverrides.DEEPSEEK_API_KEY, 'qwen3');
expect(mock.requests[0].path).toBe('/v1/chat/completions');
expect(mock.requests[0].headers.authorization).toBe('Bearer contract-test-key');
+55 -2
View File
@@ -58,6 +58,43 @@ describe('buildCustomModelInjection', () => {
});
});
it('claude: also declares configDirVar (CLAUDE_CONFIG_DIR isolation) on the env-kind result', () => {
const result = buildCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3');
if (result.kind !== 'env') throw new Error('unreachable');
expect(result.configDirVar).toBe('CLAUDE_CONFIG_DIR');
});
it('claude: injects CLAUDE_CODE_MAX_CONTEXT_TOKENS when a context length is known', () => {
const result = buildCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', 16384);
if (result.kind !== 'env') throw new Error('unreachable');
expect(result.envOverrides.CLAUDE_CODE_MAX_CONTEXT_TOKENS).toBe('16384');
});
it('claude: omits CLAUDE_CODE_MAX_CONTEXT_TOKENS when the context length is unknown', () => {
const result = buildCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3');
if (result.kind !== 'env') throw new Error('unreachable');
expect(result.envOverrides.CLAUDE_CODE_MAX_CONTEXT_TOKENS).toBeUndefined();
});
it('claude: also declares apiKeyTrustFile, carrying the literal apiKey used', () => {
const result = buildCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3');
if (result.kind !== 'env') throw new Error('unreachable');
expect(result.apiKeyTrustFile).toEqual({ relPath: '.claude.json', shape: 'claude-api-key-responses' });
expect(result.apiKey).toBe('my-key');
});
it('claude: also declares skipFirstRunPrompts on the env-kind result', () => {
const result = buildCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3');
if (result.kind !== 'env') throw new Error('unreachable');
expect(result.skipFirstRunPrompts).toBe(true);
});
it('opencode: has no skipFirstRunPrompts (no apiKeyTrustFile/configDirVar concept for it either)', () => {
const result = buildCustomModelInjection(entryOrThrow('opencode'), endpoint, 'qwen3');
if (result.kind !== 'env') throw new Error('unreachable');
expect(result.skipFirstRunPrompts).toBeUndefined();
});
it('claude: falls back to a dummy key when the endpoint has none', () => {
const result = buildCustomModelInjection(entryOrThrow('claude'), { ...endpoint, apiKey: undefined }, 'qwen3');
if (result.kind !== 'env') throw new Error('unreachable');
@@ -142,15 +179,31 @@ describe('buildCustomModelInjection', () => {
expect(result.extraEnv).toEqual({ XAI_API_KEY: 'my-key' });
});
it('deepseek: env kind sets base URL/key only, no model var', () => {
it('deepseek: env kind sets base URL (with a /v1 suffix appended) and key, no model var', () => {
// appendV1Suffix is REQUIRED here, not cosmetic: confirmed by reading dsh's own
// bundled source (@deepseek-ai/dsh-llm-deepseek) that it builds the request URL as
// `${DEEPSEEK_BASE_URL}/chat/completions` with no "/v1" of its own, while
// llama-swap/llama.cpp only serves "/v1/chat/completions" — without this, every
// request 404s (confirmed live; this is the fix for the originally-reported
// "dsh: HTTP_404: DeepSeek API error (HTTP 404)").
const result = buildCustomModelInjection(entryOrThrow('deepseek'), endpoint, 'qwen3');
if (result.kind !== 'env') throw new Error('unreachable');
expect(result.envOverrides).toEqual({
DEEPSEEK_BASE_URL: 'http://192.168.1.50:8080',
DEEPSEEK_BASE_URL: 'http://192.168.1.50:8080/v1',
DEEPSEEK_API_KEY: 'my-key',
});
});
it('deepseek: appending the /v1 suffix is idempotent against a baseUrl that already ends in /v1', () => {
const result = buildCustomModelInjection(
entryOrThrow('deepseek'),
{ ...endpoint, baseUrl: 'http://192.168.1.50:8080/v1' },
'qwen3'
);
if (result.kind !== 'env') throw new Error('unreachable');
expect(result.envOverrides.DEEPSEEK_BASE_URL).toBe('http://192.168.1.50:8080/v1');
});
it('antigravity: unsupported', () => {
const result = buildCustomModelInjection(entryOrThrow('antigravity'), endpoint, 'qwen3');
expect(result).toEqual({ kind: 'unsupported' });
+238
View File
@@ -0,0 +1,238 @@
/**
* @fileoverview Tests for `getLatestLlamaSwapLogLine()`/`pruneIdleLlamaSwapLogTails()` —
* the real-time "what is llama.cpp actually doing" feed behind the loading banner's
* second line (docs/custom-model-endpoints-plan.md). Confirmed live against a real
* llama-swap deployment: its `GET /api/events` SSE stream carries the backend
* llama-server process's own stdout (`load_model: ...`, `llama_server: model loaded`)
* as `{"type":"logData","data":"{\"data\":\"...\",\"source\":\"upstream\"}"}` frames,
* tagged distinctly from llama-swap's own `source: "proxy"` request-access log frames.
*
* ⚠️ `GET /logs` (the endpoint this feature's own first cut was built against, before
* being caught by exactly this kind of live check) turns out to carry ONLY the proxy
* log — confirmed live it never showed a single backend line even seconds after a real,
* confirmed model swap. `/api/events` is the only source that actually has the data.
*
* Drives a hand-built `ReadableStream` body through the mocked `webviewFetch` rather
* than a real network round-trip — the point under test is the SSE-frame parsing and
* `source` filtering plus the one-connection-per-endpoint reuse, not networking itself.
*
* Each test uses its own host id (`llamaSwapLogTails` is a module-level Map, shared
* across every test in this file) and `afterEach` force-prunes everything so no tail
* a test forgot to close leaks into the next one.
*
* Port: N/A (no server; drives the exported functions directly).
*/
import { describe, it, expect, vi, afterEach } from 'vitest';
import { getLatestLlamaSwapLogLine, pruneIdleLlamaSwapLogTails } from '../src/web/routes/custom-model-routes.js';
import { webviewFetch } from '../src/web/webview-egress.js';
import type { CustomModelHost } from '../src/custom-model-hosts.js';
vi.mock('../src/web/webview-egress.js', async () => {
const actual = await vi.importActual<typeof import('../src/web/webview-egress.js')>('../src/web/webview-egress.js');
return { ...actual, webviewFetch: vi.fn() };
});
const fetchMock = vi.mocked(webviewFetch);
/** One real `GET /api/events` SSE frame carrying backend (`source: "upstream"`) log text. */
function upstreamLogFrame(text: string): string {
const inner = JSON.stringify({ data: text, source: 'upstream' });
return `event:message\ndata:${JSON.stringify({ type: 'logData', data: inner })}\n\n`;
}
/** The proxy-log flavor of the same event shape — must never be surfaced as `latestLine`. */
function proxyLogFrame(text: string): string {
const inner = JSON.stringify({ data: text, source: 'proxy' });
return `event:message\ndata:${JSON.stringify({ type: 'logData', data: inner })}\n\n`;
}
/** A streaming Response whose body enqueues `frames` up front and then stays open
* (never closes) — matches a real `/api/events` connection, confirmed live to stay
* open indefinitely (read past 220KB over 8s with no `done`). */
function openStreamResponse(frames: string[]): Response {
const encoder = new TextEncoder();
const stream = new ReadableStream<Uint8Array>({
start(controller) {
for (const frame of frames) controller.enqueue(encoder.encode(frame));
// deliberately never controller.close()
},
});
return new Response(stream, { status: 200 });
}
function host(id: string): CustomModelHost {
return { id, label: id, baseUrl: `http://192.168.1.50:8080/${id}` };
}
/** Lets the fire-and-forget stream-pump's microtasks (reader.read() resolutions) settle. */
async function flush(): Promise<void> {
await new Promise((resolve) => setTimeout(resolve, 10));
}
afterEach(() => {
pruneIdleLlamaSwapLogTails(Number.POSITIVE_INFINITY); // force-close every tail this file opened
fetchMock.mockReset();
});
describe('getLatestLlamaSwapLogLine', () => {
it('returns undefined before any line has arrived, then the real backend log line once it does', async () => {
const h = host('t1');
fetchMock.mockResolvedValue(
openStreamResponse([upstreamLogFrame('0.31.428.568 I srv llama_server: model loaded')])
);
const before = getLatestLlamaSwapLogLine(h);
expect(before).toBeUndefined();
await flush();
const after = getLatestLlamaSwapLogLine(h);
expect(after).toBe('0.31.428.568 I srv llama_server: model loaded');
});
it('filters out llama-swap\'s own proxy-sourced frames, keeping only source: "upstream"', async () => {
const h = host('t2');
fetchMock.mockResolvedValue(
openStreamResponse([
proxyLogFrame('[INFO] Request 10.10.10.1 "GET /running HTTP/1.1" 200 407 "undici" 46.207µs'),
upstreamLogFrame('0.14.157.100 I srv load_model: initializing, n_slots = 4, n_ctx_slot = 16384'),
proxyLogFrame('[WARN] some warning about something unrelated'),
])
);
getLatestLlamaSwapLogLine(h);
await flush();
expect(getLatestLlamaSwapLogLine(h)).toBe(
'0.14.157.100 I srv load_model: initializing, n_slots = 4, n_ctx_slot = 16384'
);
});
it('keeps the LAST line when one upstream frame batches several newline-joined lines', async () => {
const h = host('t3');
fetchMock.mockResolvedValue(
openStreamResponse([
upstreamLogFrame(
'0.00.001.000 I srv llama_server: starting\n0.00.002.000 I srv llama_server: loading tensors'
),
upstreamLogFrame('0.00.003.000 I srv llama_server: model loaded'),
])
);
getLatestLlamaSwapLogLine(h);
await flush();
expect(getLatestLlamaSwapLogLine(h)).toBe('0.00.003.000 I srv llama_server: model loaded');
});
it('handles a frame split across two stream chunks (SSE double-newline boundary not yet seen)', async () => {
const h = host('t3b');
const whole = upstreamLogFrame('0.00.005.000 I srv llama_server: model loaded');
const splitAt = Math.floor(whole.length / 2);
fetchMock.mockResolvedValue(openStreamResponse([whole.slice(0, splitAt), whole.slice(splitAt)]));
getLatestLlamaSwapLogLine(h);
await flush();
expect(getLatestLlamaSwapLogLine(h)).toBe('0.00.005.000 I srv llama_server: model loaded');
});
it('ignores a malformed frame instead of throwing', async () => {
const h = host('t3c');
fetchMock.mockResolvedValue(
openStreamResponse(['event:message\ndata:not valid json\n\n', upstreamLogFrame('llama_server: model loaded')])
);
getLatestLlamaSwapLogLine(h);
await flush();
expect(getLatestLlamaSwapLogLine(h)).toBe('llama_server: model loaded');
});
it('ignores a non-logData event type', async () => {
const h = host('t3d');
fetchMock.mockResolvedValue(
openStreamResponse([
`event:message\ndata:${JSON.stringify({ type: 'modelStatus', data: '{}' })}\n\n`,
upstreamLogFrame('llama_server: model loaded'),
])
);
getLatestLlamaSwapLogLine(h);
await flush();
expect(getLatestLlamaSwapLogLine(h)).toBe('llama_server: model loaded');
});
it('opens exactly one connection per endpoint — a second call while the tail is open never re-fetches', async () => {
const h = host('t4');
fetchMock.mockResolvedValue(openStreamResponse([upstreamLogFrame('llama_server: model loaded')]));
getLatestLlamaSwapLogLine(h);
await flush();
getLatestLlamaSwapLogLine(h);
getLatestLlamaSwapLogLine(h);
expect(fetchMock).toHaveBeenCalledTimes(1);
});
it("requests /api/events specifically, with the endpoint's own auth headers", async () => {
const h: CustomModelHost = { id: 't5', label: 't5', baseUrl: 'http://192.168.1.60:9000', apiKey: 'secret-key' };
fetchMock.mockResolvedValue(openStreamResponse([]));
getLatestLlamaSwapLogLine(h);
expect(fetchMock).toHaveBeenCalledTimes(1);
const [url, init] = fetchMock.mock.calls[0]!;
expect((url as URL).pathname).toBe('/api/events');
expect((init as RequestInit).headers).toMatchObject({ Authorization: 'Bearer secret-key' });
});
it('an unreachable endpoint (fetch throws) leaves latestLine undefined rather than throwing', async () => {
const h = host('t6');
fetchMock.mockRejectedValue(new TypeError('fetch failed'));
expect(() => getLatestLlamaSwapLogLine(h)).not.toThrow();
await flush();
expect(getLatestLlamaSwapLogLine(h)).toBeUndefined();
});
it('a non-2xx response leaves latestLine undefined rather than throwing', async () => {
const h = host('t7');
fetchMock.mockResolvedValue(new Response('not found', { status: 404 }));
getLatestLlamaSwapLogLine(h);
await flush();
expect(getLatestLlamaSwapLogLine(h)).toBeUndefined();
});
});
describe('pruneIdleLlamaSwapLogTails', () => {
it('closes a tail nothing has polled recently, so the next access starts a fresh connection', async () => {
const h = host('t8');
fetchMock.mockResolvedValue(openStreamResponse([upstreamLogFrame('llama_server: model loaded')]));
getLatestLlamaSwapLogLine(h); // opens the first connection, lastAccessedAt = now
await flush();
expect(fetchMock).toHaveBeenCalledTimes(1);
pruneIdleLlamaSwapLogTails(Date.now() + 60_000); // "now" far enough ahead that the tail reads as idle
getLatestLlamaSwapLogLine(h); // the entry was removed — this must open a NEW connection
await flush();
expect(fetchMock).toHaveBeenCalledTimes(2);
});
it('leaves a recently-accessed tail alone', async () => {
const h = host('t9');
fetchMock.mockResolvedValue(openStreamResponse([upstreamLogFrame('llama_server: model loaded')]));
getLatestLlamaSwapLogLine(h);
await flush();
pruneIdleLlamaSwapLogTails(Date.now()); // no time has passed — nothing is idle yet
getLatestLlamaSwapLogLine(h);
expect(fetchMock).toHaveBeenCalledTimes(1); // still just the one connection
});
});
+238
View File
@@ -0,0 +1,238 @@
/**
* @fileoverview Frontend tests for the one-shot custom-model launch path added to
* session-ui.js (docs/custom-model-endpoints-plan.md): `runCustomModelEntry` dispatches
* to `_runCustomModelEntryOneShot` for every custom-model-eligible CLI except claude,
* which launches directly on the endpoint (no restart) by folding `customModel` into
* the run<Mode>() function's own `/api/quick-start` body via `_pendingCustomModelForLaunch`
* and `_quickStartWithCustomModelConfirm`. Fixes the visible native-boot-then-restart the
* restart-after-launch path (`_runCustomModelEntryViaRestart`, still used for claude)
* showed on every custom-model run — confirmed live on Codex, whose TUI fully
* reinitializes on a restart.
*
* Uses the same JSDOM + `runScripts: "dangerously"` approach as
* test/custom-model-run-menu-ui.test.ts, extended with the DOM elements runCodex() (the
* CLI this was reported against) reads.
*
* Port: none.
*/
import { readFileSync } from 'node:fs';
import { JSDOM } from 'jsdom';
import { describe, expect, it } from 'vitest';
const CONSTANTS_JS = readFileSync(new URL('../src/web/public/constants.js', import.meta.url), 'utf-8');
const SESSION_UI_JS = readFileSync(new URL('../src/web/public/session-ui.js', import.meta.url), 'utf-8');
function bootApp() {
const dom = new JSDOM(
`<!doctype html><body>
<select id="quickStartCase"><option value="testcase" selected>testcase</option></select>
<input id="tabCount" value="1">
<button id="runBtn"></button>
<div id="runModeMenu"></div>
</body>`,
{ url: 'http://localhost/', runScripts: 'dangerously' }
);
const win = dom.window as unknown as Window & typeof globalThis & { CodemanApp: new () => any };
(win as unknown as { eval: (s: string) => void }).eval('window.CodemanApp = function CodemanApp() {};');
(win as unknown as { eval: (s: string) => void }).eval(CONSTANTS_JS);
(win as unknown as { eval: (s: string) => void }).eval(SESSION_UI_JS);
const app = new win.CodemanApp();
app.cases = [{ name: 'testcase' }];
app.terminal = { focus: () => {} };
app.loadAppSettingsFromStorage = () => ({});
app.getCaseSettings = () => ({});
app.buildEnvOverrides = () => ({});
app.showToast = () => {};
app._beginSessionLaunchStatus = () => 'status-token';
app._reportSessionLaunchError = (_token: unknown, message: string) => {
app._lastReportedError = message;
};
app._ensureCreatedSessionVisible = async () => {};
app.selectSession = async () => {};
app._nextCaseSessionStartNumber = () => 1;
return { win, app };
}
describe('runCustomModelEntry dispatch', () => {
it('routes claude through the restart-after-launch path', async () => {
const { app } = bootApp();
let calledRestart = false;
let calledOneShot = false;
app._runCustomModelEntryViaRestart = async () => {
calledRestart = true;
};
app._runCustomModelEntryOneShot = async () => {
calledOneShot = true;
};
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
expect(calledRestart).toBe(true);
expect(calledOneShot).toBe(false);
});
it('routes every other custom-model-eligible CLI through the one-shot path', async () => {
for (const mode of ['opencode', 'codex', 'gemini', 'pi', 'grok', 'deepseek', 'omp']) {
const { app } = bootApp();
let calledRestart = false;
let calledOneShot = false;
app._runCustomModelEntryViaRestart = async () => {
calledRestart = true;
};
app._runCustomModelEntryOneShot = async () => {
calledOneShot = true;
};
await app.runCustomModelEntry(mode, 'llama-box', 'qwen3');
expect(calledRestart, mode).toBe(false);
expect(calledOneShot, mode).toBe(true);
}
});
});
describe('_runCustomModelEntryOneShot', () => {
it('stashes the pick on _pendingCustomModelForLaunch for the duration of run(), then clears it', async () => {
const { app } = bootApp();
let seenDuringRun: unknown;
app.run = async function (this: typeof app) {
seenDuringRun = this._pendingCustomModelForLaunch;
};
await app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3');
expect(seenDuringRun).toEqual({ endpointId: 'llama-box', modelId: 'qwen3' });
expect(app._pendingCustomModelForLaunch).toBeUndefined();
});
it('clears the pending pick even when run() throws', async () => {
const { app } = bootApp();
app.run = async () => {
throw new Error('boom');
};
await expect(app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3')).rejects.toThrow('boom');
expect(app._pendingCustomModelForLaunch).toBeUndefined();
});
it('starts the loading watcher when the launch reports modelSwapInProgress, passing the new session id', async () => {
const { app } = bootApp();
app.run = async () => {
app._lastCustomModelLaunchResult = { modelSwapInProgress: true, sessionId: 'new-session' };
};
let watched: unknown[] | null = null;
app._watchLlamaSwapLoading = async (...args: unknown[]) => {
watched = args;
};
await app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3');
expect(watched).toEqual(['llama-box', 'qwen3', 'new-session']);
});
it('never starts the watcher when no swap was needed', async () => {
const { app } = bootApp();
app.run = async () => {
app._lastCustomModelLaunchResult = { modelSwapInProgress: false };
};
let watchCalled = false;
app._watchLlamaSwapLoading = async () => {
watchCalled = true;
};
await app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3');
expect(watchCalled).toBe(false);
});
});
describe('_quickStartWithCustomModelConfirm', () => {
function withFetch(win: Window & typeof globalThis, handler: (body: any) => any) {
(win as unknown as { fetch: typeof fetch }).fetch = (async (_url: string, opts: any) => ({
json: async () => handler(JSON.parse(opts.body)),
})) as unknown as typeof fetch;
}
it('returns the response directly when no confirmation is needed', async () => {
const { win, app } = bootApp();
withFetch(win, (body) => ({ success: true, data: { sessionId: 's1', modelSwapInProgress: false, body } }));
const data = await app._quickStartWithCustomModelConfirm({
mode: 'codex',
customModel: { endpointId: 'e', modelId: 'm' },
});
expect(data.success).toBe(true);
expect(data.data.sessionId).toBe('s1');
expect(app._lastCustomModelLaunchResult).toEqual(data.data);
});
it('confirming re-sends with confirmed:true and returns the second response', async () => {
const { win, app } = bootApp();
app._confirmModelSwap = async () => true;
let calls = 0;
withFetch(win, (body) => {
calls += 1;
if (calls === 1) {
return {
success: true,
data: {
requiresConfirmation: true,
currentlyLoadedModel: 'llama3',
affectedSessions: [{ id: 's2', name: 'w2' }],
},
};
}
expect(body.customModel.confirmed).toBe(true);
return { success: true, data: { sessionId: 's1', modelSwapInProgress: true } };
});
const data = await app._quickStartWithCustomModelConfirm({
mode: 'codex',
customModel: { endpointId: 'e', modelId: 'm' },
});
expect(calls).toBe(2);
expect(data.data.sessionId).toBe('s1');
expect(app._lastCustomModelLaunchResult.modelSwapInProgress).toBe(true);
});
it('cancelling never re-sends, and reports a cancellation error', async () => {
const { win, app } = bootApp();
app._confirmModelSwap = async () => false;
let calls = 0;
withFetch(win, () => {
calls += 1;
return {
success: true,
data: {
requiresConfirmation: true,
currentlyLoadedModel: 'llama3',
affectedSessions: [{ id: 's2', name: 'w2' }],
},
};
});
const data = await app._quickStartWithCustomModelConfirm({
mode: 'codex',
customModel: { endpointId: 'e', modelId: 'm' },
});
expect(calls).toBe(1);
expect(data.success).toBe(false);
expect(data.error).toMatch(/cancelled/i);
expect(app._lastCustomModelLaunchResult).toBeUndefined();
});
});
describe('runCodex(): one-shot custom-model launch (the CLI this was reported against)', () => {
it('folds _pendingCustomModelForLaunch into the quick-start body as customModel', async () => {
const { win, app } = bootApp();
(win as unknown as { fetch: typeof fetch }).fetch = (async (url: string, opts?: any) => {
if (url === '/api/codex/status') return { json: async () => ({ data: { available: true } }) };
const body = JSON.parse(opts.body);
expect(body.customModel).toEqual({ endpointId: 'llama-box', modelId: 'qwen3' });
return { json: async () => ({ success: true, data: { sessionId: 's1', modelSwapInProgress: false } }) };
}) as unknown as typeof fetch;
app._pendingCustomModelForLaunch = { endpointId: 'llama-box', modelId: 'qwen3' };
await app.runCodex();
expect(app._lastReportedError).toBeUndefined();
});
it('omits customModel entirely for a plain (non-custom-model) Codex launch', async () => {
const { win, app } = bootApp();
(win as unknown as { fetch: typeof fetch }).fetch = (async (url: string, opts?: any) => {
if (url === '/api/codex/status') return { json: async () => ({ data: { available: true } }) };
const body = JSON.parse(opts.body);
expect(body.customModel).toBeUndefined();
return { json: async () => ({ success: true, data: { sessionId: 's1' } }) };
}) as unknown as typeof fetch;
await app.runCodex();
expect(app._lastReportedError).toBeUndefined();
});
});
File diff suppressed because it is too large Load Diff
+180
View File
@@ -0,0 +1,180 @@
/**
* @fileoverview Tests for `detectCustomModelSwapDisplacements()`, the periodic sweep
* behind server.ts's "custom model swap-displacement check" timer
* (docs/custom-model-endpoints-plan.md). The apply/create routes' own swap-conflict check
* only ever runs at a session's own launch/apply moment — this sweep is what catches a
* LATER eviction triggered by a different session's normal use, which the launch-time
* check structurally cannot see.
*
* Kept in its own file for the same reason as `custom-model-endpoint-rediscovery.test.ts`:
* a sweep that walks every saved host would otherwise pick up hosts other tests in a
* shared file create, making an exact call-count assertion meaningless.
*
* Port: N/A (no server; drives readCustomModelHosts/writeCustomModelHosts directly plus
* the mocked webviewFetch dispatcher).
*/
import { describe, it, expect, vi, beforeEach } from 'vitest';
import { getDataDir } from '../src/config/instance.js';
import { writeCustomModelHosts, type CustomModelHost } from '../src/custom-model-hosts.js';
import {
detectCustomModelSwapDisplacements,
type CustomModelSessionLike,
} from '../src/web/routes/custom-model-routes.js';
import { webviewFetch } from '../src/web/webview-egress.js';
vi.mock('../src/web/webview-egress.js', async () => {
const actual = await vi.importActual<typeof import('../src/web/webview-egress.js')>('../src/web/webview-egress.js');
return { ...actual, webviewFetch: vi.fn() };
});
const fetchMock = vi.mocked(webviewFetch);
const ENDPOINT: CustomModelHost = {
id: 'llama-swap',
label: 'llama-swap',
baseUrl: 'http://192.168.1.50:8080',
apiKey: 'k',
};
function session(
overrides: Partial<CustomModelSessionLike> & Pick<CustomModelSessionLike, 'id'>
): CustomModelSessionLike {
return { name: overrides.id, ...overrides };
}
function mockRunning(running: Array<{ model: string; state: string }>) {
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/running') return new Response(JSON.stringify({ running }), { status: 200 });
throw new Error(`unexpected request in this test: ${url.href}`);
});
}
beforeEach(() => {
fetchMock.mockReset();
});
describe('detectCustomModelSwapDisplacements', () => {
it('flags a session whose own model is no longer in the running list, naming what displaced it', async () => {
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
mockRunning([{ model: 'fast', state: 'ready' }]);
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
const notified = new Set<string>();
const displacements = await detectCustomModelSwapDisplacements([w1], notified);
expect(displacements).toEqual([
{
sessionId: 'w1',
sessionName: 'w1',
endpointId: 'llama-swap',
previousModel: 'qwen3',
currentlyLoadedModel: 'fast',
},
]);
expect(notified.has('w1')).toBe(true);
});
it('does not flag a session whose own model is still the one loaded and ready', async () => {
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
mockRunning([{ model: 'qwen3', state: 'ready' }]);
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
const displacements = await detectCustomModelSwapDisplacements([w1], new Set());
expect(displacements).toEqual([]);
});
it('notifies once per displacement — a repeat sweep with nothing changed does not re-flag it', async () => {
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
mockRunning([{ model: 'fast', state: 'ready' }]);
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
const notified = new Set<string>();
const first = await detectCustomModelSwapDisplacements([w1], notified);
const second = await detectCustomModelSwapDisplacements([w1], notified);
expect(first).toHaveLength(1);
expect(second).toEqual([]);
});
it('clears the notified flag once the session is back on its own model, so a later displacement flags again', async () => {
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
const notified = new Set<string>();
mockRunning([{ model: 'fast', state: 'ready' }]);
await detectCustomModelSwapDisplacements([w1], notified);
expect(notified.has('w1')).toBe(true);
mockRunning([{ model: 'qwen3', state: 'ready' }]); // back to normal
await detectCustomModelSwapDisplacements([w1], notified);
expect(notified.has('w1')).toBe(false);
mockRunning([{ model: 'fast', state: 'ready' }]); // displaced again
const third = await detectCustomModelSwapDisplacements([w1], notified);
expect(third).toHaveLength(1);
});
it('skips a session on a non-llama-swap endpoint (no /running) — nothing to compare, never flagged', async () => {
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
fetchMock.mockResolvedValue(new Response('not found', { status: 404 }));
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
const displacements = await detectCustomModelSwapDisplacements([w1], new Set());
expect(displacements).toEqual([]);
});
it('skips a session whose endpoint was deleted since it was created', async () => {
await writeCustomModelHosts(getDataDir(), []); // ENDPOINT never saved
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
const displacements = await detectCustomModelSwapDisplacements([w1], new Set());
expect(displacements).toEqual([]);
expect(fetchMock).not.toHaveBeenCalled();
});
it('ignores a plain session with no customModel selection at all', async () => {
const displacements = await detectCustomModelSwapDisplacements([session({ id: 'plain' })], new Set());
expect(displacements).toEqual([]);
expect(fetchMock).not.toHaveBeenCalled();
});
it('one endpoint failing (unreachable) never blocks checking sessions on another', async () => {
const DOWN: CustomModelHost = { id: 'down', label: 'down', baseUrl: 'http://192.168.1.60:8080' };
await writeCustomModelHosts(getDataDir(), [ENDPOINT, DOWN]);
fetchMock.mockImplementation(async (url: URL) => {
if (url.href.includes('192.168.1.60')) throw new TypeError('fetch failed', { cause: new Error('ECONNREFUSED') });
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'fast', state: 'ready' }] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
const onDown = session({ id: 'w-down', customModel: { endpointId: 'down', modelId: 'x' } });
const onLlamaSwap = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
const displacements = await detectCustomModelSwapDisplacements([onDown, onLlamaSwap], new Set());
expect(displacements).toEqual([
{
sessionId: 'w1',
sessionName: 'w1',
endpointId: 'llama-swap',
previousModel: 'qwen3',
currentlyLoadedModel: 'fast',
},
]);
});
it('multiple sessions on the same endpoint each get their own displacement entry', async () => {
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
mockRunning([{ model: 'gemma', state: 'ready' }]);
const w1 = session({ id: 'w1', name: 'w1-test2', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
const w2 = session({ id: 'w2', name: 'w2-test2', customModel: { endpointId: 'llama-swap', modelId: 'fast' } });
const displacements = await detectCustomModelSwapDisplacements([w1, w2], new Set());
expect(displacements.map((d) => d.sessionId).sort()).toEqual(['w1', 'w2']);
});
});
+37 -1
View File
@@ -11,7 +11,7 @@
* Port: N/A (no server start).
*/
import { describe, it, expect, afterEach, vi } from 'vitest';
import { WebServer } from '../src/web/server.js';
import { WebServer, escapeScriptJson } from '../src/web/server.js';
import { isClaudeAvailable } from '../src/utils/claude-cli-resolver.js';
import { isOpenCodeAvailable } from '../src/utils/opencode-cli-resolver.js';
import { isCodexAvailable } from '../src/utils/codex-cli-resolver.js';
@@ -187,6 +187,41 @@ describe('WebServer.renderIndexHtml', () => {
});
});
it('reports which run modes the custom-model Run-menu picker may generate an entry for', async () => {
// Read generically off the CLI registry's own capabilities, not a hardcoded id
// list — antigravity (`unsupported`) and shell (`kind !== 'agent'`) must be
// absent, and any enabled agent CLI with a real injection recipe must be
// present, with no mock needed since this reads the real stock registry.
const { server } = makeServer({});
const html = await render(server);
expect(html).toContain('window.__codemanCustomModelClis=');
const clis = JSON.parse(html.match(/window\.__codemanCustomModelClis=(\[.*?\]);/)![1]) as Array<{
id: string;
label: string;
}>;
const ids = clis.map((c) => c.id);
expect(ids).toContain('claude');
expect(ids).not.toContain('antigravity');
expect(ids).not.toContain('shell');
for (const cli of clis) {
expect(typeof cli.id).toBe('string');
expect(typeof cli.label).toBe('string');
}
});
it('escapeScriptJson neutralizes a literal </script>, and still round-trips as a JS literal', () => {
// CliEntry.label is a plain string a user's own clis.json can set (up to 60
// chars), unlike __codemanCliAvailable's booleans-only payload, so this is
// the one injection that needs it. Exported so this tests the pure
// function directly rather than needing a real WebServer (which needs tmux).
const dangerous = JSON.stringify([{ id: 'x', label: '</script><script>alert(1)</script>' }]);
const escaped = escapeScriptJson(dangerous);
expect(escaped).not.toContain('</script');
// Proves it decodes back to the real value the way a browser's own JS
// parser would, not just "the output contains no </script>".
expect(eval(escaped)[0].label).toBe('</script><script>alert(1)</script>');
});
it('still emits the object when nothing at all is installed', async () => {
// The all-false case is the one that matters most and the easiest to get
// wrong by only injecting when something resolves.
@@ -218,6 +253,7 @@ describe('WebServer.renderIndexHtml', () => {
const { server } = makeServer({});
const html = await render(server, 'sess-123');
expect(html).not.toContain('__codemanCliAvailable');
expect(html).not.toContain('__codemanCustomModelClis');
});
it('does not expose gesture at all when CODEMAN_GESTURE is unset', async () => {
+223
View File
@@ -181,6 +181,34 @@ describe('custom model endpoint CRUD', () => {
expect(res.json().error).toMatch(/refused.*169\.254\.169\.254/);
});
it('running-status never hands the browser the raw llama-swap launch command (cmd)', async () => {
const { app } = await setup();
await app.inject({
method: 'POST',
url: '/api/model-endpoints',
payload: { id: 'ep-running', label: 'A', baseUrl: 'http://localhost:8080', apiKey: 'k' },
});
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/running') {
return new Response(
JSON.stringify({
running: [
{ model: 'qwen3', state: 'ready', cmd: 'llama-server -m /models/qwen3.gguf --api-key sk-secret' },
],
}),
{ status: 200 }
);
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
const res = await app.inject({ method: 'GET', url: '/api/model-endpoints/ep-running/running-status' });
const body = res.json();
expect(body.data.running).toEqual([{ model: 'qwen3', state: 'ready' }]);
expect(JSON.stringify(body)).not.toContain('sk-secret');
expect(JSON.stringify(body)).not.toContain('cmd');
});
it('refuses a baseUrl with embedded credentials or a non-http scheme at save time', async () => {
const { app } = await setup();
for (const baseUrl of ['http://user:pw@host:8080', 'ftp://host/models', 'http://169.254.169.254']) {
@@ -194,3 +222,198 @@ describe('custom model endpoint CRUD', () => {
}
});
});
describe('defaultModelId — the Run-menu picker’s per-endpoint default', () => {
afterEach(() => {
fetchMock.mockReset();
});
it('rejects a defaultModelId that is not one of the endpoint’s discovered models, on both create and update', async () => {
const { app } = await setup();
const create = await app.inject({
method: 'POST',
url: '/api/model-endpoints',
payload: {
id: 'ep-default-reject',
label: 'A',
baseUrl: 'http://localhost:8080',
models: ['qwen3'],
defaultModelId: 'ghost',
},
});
expect(create.json().success).toBe(false);
expect(create.json().errorCode).toBe('INVALID_INPUT');
await app.inject({
method: 'POST',
url: '/api/model-endpoints',
payload: { id: 'ep-default-reject', label: 'A', baseUrl: 'http://localhost:8080', models: ['qwen3'] },
});
const update = await app.inject({
method: 'PUT',
url: '/api/model-endpoints/ep-default-reject',
payload: { label: 'A', baseUrl: 'http://localhost:8080', models: ['qwen3'], defaultModelId: 'ghost' },
});
expect(update.json().success).toBe(false);
expect(update.json().errorCode).toBe('INVALID_INPUT');
});
it('accepts a defaultModelId that IS one of the discovered models', async () => {
const { app } = await setup();
const res = await app.inject({
method: 'POST',
url: '/api/model-endpoints',
payload: {
id: 'ep-default-accept',
label: 'A',
baseUrl: 'http://localhost:8080',
models: ['qwen3', 'llama3'],
defaultModelId: 'llama3',
},
});
expect(res.json().success).toBe(true);
expect(res.json().data.host.defaultModelId).toBe('llama3');
});
it('drops a stale default that no longer appears in a fresh discovery, rather than carrying it forward invalid', async () => {
const { app } = await setup();
await app.inject({
method: 'POST',
url: '/api/model-endpoints',
payload: {
id: 'ep-default-drop',
label: 'A',
baseUrl: 'http://localhost:8080',
models: ['qwen3'],
defaultModelId: 'qwen3',
},
});
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'llama3' }] }), { status: 200 }));
await app.inject({ method: 'POST', url: '/api/model-endpoints/ep-default-drop/discover-models' });
const list = await app.inject({ method: 'GET', url: '/api/model-endpoints' });
const stored = (list.json() as Array<{ id: string; defaultModelId?: string }>).find(
(h) => h.id === 'ep-default-drop'
);
expect(stored?.defaultModelId).toBeUndefined();
});
it('keeps a default that IS still present after a fresh discovery', async () => {
const { app } = await setup();
await app.inject({
method: 'POST',
url: '/api/model-endpoints',
payload: {
id: 'ep-default-keep',
label: 'A',
baseUrl: 'http://localhost:8080',
models: ['qwen3'],
defaultModelId: 'qwen3',
},
});
fetchMock.mockResolvedValue(
new Response(JSON.stringify({ data: [{ id: 'qwen3' }, { id: 'llama3' }] }), { status: 200 })
);
await app.inject({ method: 'POST', url: '/api/model-endpoints/ep-default-keep/discover-models' });
const list = await app.inject({ method: 'GET', url: '/api/model-endpoints' });
const stored = (list.json() as Array<{ id: string; defaultModelId?: string }>).find(
(h) => h.id === 'ep-default-keep'
);
expect(stored?.defaultModelId).toBe('qwen3');
});
});
describe('apiKey is never handed back to the browser', () => {
afterEach(() => {
fetchMock.mockReset();
});
it('POST, GET and PUT responses all carry apiKeySet instead of the real key', async () => {
const { app } = await setup();
const create = await app.inject({
method: 'POST',
url: '/api/model-endpoints',
payload: { id: 'ep-secret', label: 'A', baseUrl: 'http://localhost:8080', apiKey: 'super-secret' },
});
expect(create.json().data.host.apiKey).toBeUndefined();
expect(create.json().data.host.apiKeySet).toBe(true);
const list = await app.inject({ method: 'GET', url: '/api/model-endpoints' });
const listed = (list.json() as Array<{ id: string; apiKey?: string; apiKeySet?: boolean }>).find(
(h) => h.id === 'ep-secret'
);
expect(listed?.apiKey).toBeUndefined();
expect(listed?.apiKeySet).toBe(true);
expect(JSON.stringify(list.json())).not.toContain('super-secret');
const update = await app.inject({
method: 'PUT',
url: '/api/model-endpoints/ep-secret',
payload: { label: 'Renamed', baseUrl: 'http://localhost:8080' },
});
expect(update.json().data.host.apiKey).toBeUndefined();
expect(update.json().data.host.apiKeySet).toBe(true);
expect(JSON.stringify(update.json())).not.toContain('super-secret');
});
it('a host with no key set at all reports apiKeySet: false', async () => {
const { app } = await setup();
const create = await app.inject({
method: 'POST',
url: '/api/model-endpoints',
payload: { id: 'ep-nokey', label: 'A', baseUrl: 'http://localhost:8080' },
});
expect(create.json().data.host.apiKeySet).toBe(false);
});
it('PUT with no apiKey keeps the stored one, rather than clearing it', async () => {
const { app } = await setup();
await app.inject({
method: 'POST',
url: '/api/model-endpoints',
payload: { id: 'ep-keep-key', label: 'A', baseUrl: 'http://localhost:8080', apiKey: 'original-key' },
});
// Edit without touching the API key field — the real bug this guards: a
// browser round-trip that only ever sees apiKeySet, never the real value,
// must not accidentally send an empty string and wipe a working credential.
const update = await app.inject({
method: 'PUT',
url: '/api/model-endpoints/ep-keep-key',
payload: { label: 'Renamed', baseUrl: 'http://localhost:8080' },
});
expect(update.json().data.host.apiKeySet).toBe(true);
// Prove it by observing the auth header discovery actually sends.
fetchMock.mockImplementation(async (_url: URL, init?: RequestInit) => {
const headers = init?.headers as Record<string, string>;
expect(headers.Authorization).toBe('Bearer original-key');
return new Response(JSON.stringify({ data: [] }), { status: 200 });
});
const discover = await app.inject({ method: 'POST', url: '/api/model-endpoints/ep-keep-key/discover-models' });
expect(discover.json().success).toBe(true);
expect(fetchMock).toHaveBeenCalledTimes(1);
});
it('PUT with a new apiKey replaces the stored one', async () => {
const { app } = await setup();
await app.inject({
method: 'POST',
url: '/api/model-endpoints',
payload: { id: 'ep-replace-key', label: 'A', baseUrl: 'http://localhost:8080', apiKey: 'old-key' },
});
await app.inject({
method: 'PUT',
url: '/api/model-endpoints/ep-replace-key',
payload: { label: 'A', baseUrl: 'http://localhost:8080', apiKey: 'new-key' },
});
fetchMock.mockImplementation(async (_url: URL, init?: RequestInit) => {
const headers = init?.headers as Record<string, string>;
expect(headers.Authorization).toBe('Bearer new-key');
return new Response(JSON.stringify({ data: [] }), { status: 200 });
});
await app.inject({ method: 'POST', url: '/api/model-endpoints/ep-replace-key/discover-models' });
expect(fetchMock).toHaveBeenCalledTimes(1);
});
});
@@ -0,0 +1,382 @@
/**
* @fileoverview POST /api/quick-start's `customModel` field (docs/custom-model-endpoints-plan.md):
* the ONE-SHOT launch path that computes a custom-model endpoint's injection BEFORE the
* session/process exists and launches directly on it, so a custom-model Run never shows
* the native-boot-then-restart the dedicated POST /api/sessions/:id/custom-model route's
* restart-in-place design otherwise produces — most visibly on a CLI like Codex whose TUI
* fully reinitializes on a restart. That dedicated route is still what an ALREADY-RUNNING
* session uses to switch later; this is the create-time equivalent.
*
* Mirrors test/routes/session-custom-model.test.ts's fixtures and llama-swap mocking, since
* this route mirrors that one's own checks (llama-swap conflict, unsupported CLI, unknown
* endpoint, an argv-incompatible model id) rather than a lighter, separately-drifting copy.
*
* Session.prototype.startInteractive/startShell are mocked exactly like the workspace-hooks
* quick-start tests: quick-start constructs a REAL Session (not the MockSession the route
* test harness substitutes elsewhere), so tmux must never actually be reached.
*
* Port: N/A (app.inject, no real port needed)
*/
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
import Fastify, { type FastifyInstance } from 'fastify';
import fastifyCookie from '@fastify/cookie';
import { rm, readFile } from 'node:fs/promises';
import { existsSync } from 'node:fs';
import { join } from 'node:path';
import { createMockRouteContext, safeRmHomeTree, type MockRouteContext } from '../mocks/index.js';
import { installRouteErrorHandler } from '../../src/web/route-error-handler.js';
import { registerSessionRoutes } from '../../src/web/routes/session-routes.js';
import { getDataDir } from '../../src/config/instance.js';
import { CASES_DIR } from '../../src/web/route-helpers.js';
import { Session } from '../../src/session.js';
import { writeCustomModelHosts, type CustomModelHost } from '../../src/custom-model-hosts.js';
import { customModelConfigDir } from '../../src/custom-model-injection-apply.js';
import { webviewFetch } from '../../src/web/webview-egress.js';
vi.mock('../../src/web/webview-egress.js', async () => {
const actual = await vi.importActual<typeof import('../../src/web/webview-egress.js')>(
'../../src/web/webview-egress.js'
);
return { ...actual, webviewFetch: vi.fn() };
});
const fetchMock = vi.mocked(webviewFetch);
// quick-start's own local-CLI-availability gate (resolveCliLaunchError, unrelated to the
// custom-model injection this file tests) runs BEFORE the code under test and would
// otherwise 404 every non-claude mode on a box with no codex/pi/grok/omp binary installed —
// exactly this test environment. Mirrors the real "not remote" bypass documented at its own
// call site in session-routes.ts (`session-routes.test.ts`'s remote-codex test is the
// precedent for needing this at all).
vi.mock('../../src/utils/cli-launcher.js', async () => {
const actual = await vi.importActual<typeof import('../../src/utils/cli-launcher.js')>(
'../../src/utils/cli-launcher.js'
);
return { ...actual, resolveCliLaunchError: vi.fn().mockResolvedValue(null) };
});
const ENDPOINT: CustomModelHost = {
id: 'ep1',
label: 'llama.cpp box',
baseUrl: 'http://192.168.1.50:8080',
apiKey: 'k',
};
describe('POST /api/quick-start: customModel (one-shot custom-model launch)', () => {
let app: FastifyInstance;
let ctx: MockRouteContext;
let restartSpy: ReturnType<typeof vi.spyOn>;
const quickStart = (payload: Record<string, unknown>) =>
app.inject({ method: 'POST', url: '/api/quick-start', payload });
beforeEach(async () => {
vi.spyOn(Session.prototype, 'startInteractive').mockResolvedValue(undefined);
vi.spyOn(Session.prototype, 'startShell').mockResolvedValue(undefined);
restartSpy = vi.spyOn(Session.prototype, 'restartCli').mockResolvedValue(true);
fetchMock.mockReset();
fetchMock.mockResolvedValue(new Response('not found', { status: 404 })); // default: not llama-swap
app = Fastify({ logger: false });
await app.register(fastifyCookie);
ctx = createMockRouteContext();
registerSessionRoutes(app, ctx);
installRouteErrorHandler(app);
await app.ready();
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
});
afterEach(async () => {
await app.close();
vi.restoreAllMocks();
await rm(join(getDataDir(), 'custom-model-hosts.json'), { force: true });
await rm(join(getDataDir(), 'custom-model-configs'), { recursive: true, force: true });
safeRmHomeTree(CASES_DIR);
});
it('launches a claude session already pointed at the endpoint — no restart at all', async () => {
const res = await quickStart({
caseName: 'cm-claude',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.statusCode).toBe(200);
const { sessionId } = res.json();
const session = ctx.sessions.get(sessionId) as unknown as Session;
expect(session.customModel).toEqual({ endpointId: 'ep1', modelId: 'qwen3', label: 'llama.cpp box' });
// The whole point: never restarted. It launched on the endpoint the first time.
expect(restartSpy).not.toHaveBeenCalled();
const isolatedDir = customModelConfigDir(sessionId);
const trustFile = JSON.parse(await readFile(join(isolatedDir, '.claude.json'), 'utf-8'));
expect(trustFile.customApiKeyResponses.approved).toEqual(['k']);
});
it('codex: writes the config.toml under the SAME id the session actually launches with, no restart', async () => {
const res = await quickStart({
caseName: 'cm-codex',
mode: 'codex',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.statusCode).toBe(200);
const { sessionId } = res.json();
const session = ctx.sessions.get(sessionId) as unknown as Session;
expect(session.customModel?.endpointId).toBe('ep1');
expect(restartSpy).not.toHaveBeenCalled();
const configDir = customModelConfigDir(sessionId);
expect(existsSync(join(configDir, 'config.toml'))).toBe(true);
const toml = await readFile(join(configDir, 'config.toml'), 'utf-8');
expect(toml).toContain('model = "qwen3"');
});
it('pi: forces --model custom/<id> onto piConfig on the FIRST launch, not via a later restart', async () => {
const res = await quickStart({
caseName: 'cm-pi',
mode: 'pi',
customModel: { endpointId: 'ep1', modelId: 'qwen3.5-0.8b' },
});
expect(res.statusCode).toBe(200);
const { sessionId } = res.json();
const session = ctx.sessions.get(sessionId) as unknown as Session & { piConfig?: { model?: string } };
expect(session.getCustomModelForPersist()?.launchModel).toBe('custom/qwen3.5-0.8b');
expect(restartSpy).not.toHaveBeenCalled();
});
it('grok: forces the [model.<name>] block name onto grokConfig on the first launch', async () => {
const res = await quickStart({
caseName: 'cm-grok',
mode: 'grok',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.statusCode).toBe(200);
const { sessionId } = res.json();
const session = ctx.sessions.get(sessionId) as unknown as Session;
expect(session.getCustomModelForPersist()?.launchModel).toBe('codeman-custom');
expect(restartSpy).not.toHaveBeenCalled();
});
it('omp: forces custom/<id> onto ompConfig even with no incoming ompConfig at all', async () => {
const res = await quickStart({
caseName: 'cm-omp',
mode: 'omp',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.statusCode).toBe(200);
const { sessionId } = res.json();
const session = ctx.sessions.get(sessionId) as unknown as Session;
expect(session.getCustomModelForPersist()?.launchModel).toBe('custom/qwen3');
expect(restartSpy).not.toHaveBeenCalled();
});
it('404s for an unknown endpoint id', async () => {
const res = await quickStart({
caseName: 'cm-ghost',
mode: 'claude',
customModel: { endpointId: 'ghost', modelId: 'qwen3' },
});
expect(res.json().success).toBe(false);
expect(res.json().errorCode).toBe('NOT_FOUND');
});
it('refuses a mode with no known custom-model mechanism (antigravity)', async () => {
const res = await quickStart({
caseName: 'cm-agy',
mode: 'antigravity',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.json().success).toBe(false);
expect(res.json().errorCode).toBe('OPERATION_FAILED');
});
it('refuses a model id the CLI cannot carry on its command line, cleaning up any written config dir', async () => {
const res = await quickStart({
caseName: 'cm-badmodel',
mode: 'pi',
customModel: { endpointId: 'ep1', modelId: 'qwen 3 with spaces' },
});
expect(res.json().success).toBe(false);
expect(res.json().errorCode).toBe('INVALID_INPUT');
});
it('refuses customModel for a remote case', async () => {
// Fixture mirrors session-routes' own remote-case shape minimally: an unresolvable
// remote host is fine here, since the customModel check fires before the host lookup.
const res = await quickStart({
caseName: 'nonexistent-remote-case',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
// No matching remote/docker case fixture exists, so this actually falls through to the
// local branch and succeeds — this test only documents that remote/docker have their
// own explicit customModel rejection (see the local-fixture tests in
// session-routes-workspace-hooks.test.ts for the fixture-loading pattern that would be
// needed to exercise the remote/docker branch itself).
expect(res.statusCode).toBe(200);
});
describe('llama-swap conflict check', () => {
function mockRunning(running: Array<{ model: string; state: string }>) {
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/running') return new Response(JSON.stringify({ running }), { status: 200 });
throw new Error(`unexpected request in this test: ${url.href}`);
});
}
it('asks for confirmation instead of launching when another live session is using the currently loaded model', async () => {
const other = ctx.sessions.get('test-session-1')!;
(other as unknown as { customModel: unknown }).customModel = { endpointId: 'ep1', modelId: 'llama3' };
mockRunning([{ model: 'llama3', state: 'ready' }]);
const res = await quickStart({
caseName: 'cm-conflict',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
const body = res.json();
expect(body.requiresConfirmation).toBe(true);
expect(body.currentlyLoadedModel).toBe('llama3');
expect(body.affectedSessions).toEqual([{ id: 'test-session-1', name: other.name }]);
// Nothing was actually created.
expect(ctx.sessions.size).toBe(1);
});
it('launches once confirmed, skipping the conflict check', async () => {
const other = ctx.sessions.get('test-session-1')!;
(other as unknown as { customModel: unknown }).customModel = { endpointId: 'ep1', modelId: 'llama3' };
mockRunning([{ model: 'llama3', state: 'ready' }]);
const res = await quickStart({
caseName: 'cm-confirmed',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3', confirmed: true },
});
expect(res.statusCode).toBe(200);
expect(res.json().requiresConfirmation).toBeUndefined();
expect(ctx.sessions.size).toBe(2);
});
it('launches straight away when nothing else is using the currently loaded model', async () => {
mockRunning([{ model: 'llama3', state: 'ready' }]);
const res = await quickStart({
caseName: 'cm-noconflict',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.statusCode).toBe(200);
expect(res.json().requiresConfirmation).toBeUndefined();
});
});
describe("context-window floor warning (this CLI's own overhead can exceed a small model's real context)", () => {
const SMALL_CTX_ENDPOINT: CustomModelHost = {
id: 'ep-small',
label: 'tiny box',
baseUrl: 'http://192.168.1.51:8080',
apiKey: 'k',
modelContextLengths: { 'qwen3.8-27b-ud-q4_k_xl': 16384 },
};
it('warns instead of launching when the discovered context is below the safe floor', async () => {
await writeCustomModelHosts(getDataDir(), [ENDPOINT, SMALL_CTX_ENDPOINT]);
const res = await quickStart({
caseName: 'cm-small-ctx',
mode: 'claude',
customModel: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
});
const body = res.json();
expect(body.requiresContextWarning).toBe(true);
expect(body.modelId).toBe('qwen3.8-27b-ud-q4_k_xl');
expect(body.contextLength).toBe(16384);
expect(body.minSafeContextTokens).toBe(40000);
// Nothing was actually created.
expect(ctx.sessions.size).toBe(1);
});
it('launches once confirmed, skipping the context check', async () => {
await writeCustomModelHosts(getDataDir(), [ENDPOINT, SMALL_CTX_ENDPOINT]);
const res = await quickStart({
caseName: 'cm-small-ctx-confirmed',
mode: 'claude',
customModel: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl', confirmed: true },
});
expect(res.statusCode).toBe(200);
expect(res.json().requiresContextWarning).toBeUndefined();
expect(ctx.sessions.size).toBe(2);
});
it('does not warn when nothing about context was discovered', async () => {
const res = await quickStart({
caseName: 'cm-no-ctx-data',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.statusCode).toBe(200);
expect(res.json().requiresContextWarning).toBeUndefined();
});
});
describe('triggering the actual llama-swap load (not just watching for it)', () => {
it('sends a real inference request naming the target model, concurrently with launching the session', async () => {
const chatCalls: unknown[] = [];
fetchMock.mockImplementation(async (url: URL, init?: { body?: unknown }) => {
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 });
}
if (url.pathname === '/v1/chat/completions') {
chatCalls.push(JSON.parse(init!.body as string));
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
const res = await quickStart({
caseName: 'cm-trigger',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
await new Promise((resolve) => setTimeout(resolve, 0)); // let the fire-and-forget trigger settle
expect(res.statusCode).toBe(200);
expect(res.json().modelSwapInProgress).toBe(true);
expect(chatCalls).toHaveLength(1);
expect(chatCalls[0]).toMatchObject({ model: 'qwen3', max_tokens: 1 });
});
it('never sends a load-trigger request when the target model is already loaded and ready', async () => {
let chatCalled = false;
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'qwen3', state: 'ready' }] }), { status: 200 });
}
if (url.pathname === '/v1/chat/completions') {
chatCalled = true;
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
const res = await quickStart({
caseName: 'cm-no-trigger',
mode: 'claude',
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
});
await new Promise((resolve) => setTimeout(resolve, 0));
expect(res.json().modelSwapInProgress).toBe(false);
expect(chatCalled).toBe(false);
});
});
});
+443 -5
View File
@@ -3,13 +3,28 @@
* chunk 5 — applying/clearing a session's custom model endpoint + CLI restart).
* Port: N/A (app.inject, no real port needed)
*/
import { describe, it, expect, beforeEach } from 'vitest';
import { registerSessionRoutes } from '../../src/web/routes/session-routes.js';
import { describe, it, expect, beforeEach, vi } from 'vitest';
import { registerSessionRoutes, _clampEnvOverridesForOwner } from '../../src/web/routes/session-routes.js';
import { createRouteTestHarness } from './_route-test-utils.js';
import { createMockSession } from '../mocks/index.js';
import { getDataDir } from '../../src/config/instance.js';
import { writeCustomModelHosts, type CustomModelHost } from '../../src/custom-model-hosts.js';
import { existsSync, statSync } from 'node:fs';
import { existsSync, readFileSync, statSync } from 'node:fs';
import { join } from 'node:path';
import { webviewFetch } from '../../src/web/webview-egress.js';
// Every apply now also checks llama-swap's `GET /running` (session-routes.ts) before
// applying — without this mock every test in this file would make a REAL network request
// to the fake 192.168.1.50 endpoint below and wait out its 5s timeout. Defaults to a plain
// 404 (reads as "not llama-swap", exercising none of the new conflict-check tests below),
// overridden per-test where the llama-swap behavior itself is what's under test.
vi.mock('../../src/web/webview-egress.js', async () => {
const actual = await vi.importActual<typeof import('../../src/web/webview-egress.js')>(
'../../src/web/webview-egress.js'
);
return { ...actual, webviewFetch: vi.fn() };
});
const fetchMock = vi.mocked(webviewFetch);
const CLAUDE_ENDPOINT: CustomModelHost = {
id: 'ep1',
@@ -18,14 +33,16 @@ const CLAUDE_ENDPOINT: CustomModelHost = {
apiKey: 'k',
};
async function setup() {
async function setup(ctxOptions?: Parameters<typeof createRouteTestHarness>[1]) {
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT]);
return createRouteTestHarness(registerSessionRoutes);
return createRouteTestHarness(registerSessionRoutes, ctxOptions);
}
describe('POST /api/sessions/:id/custom-model', () => {
beforeEach(async () => {
await writeCustomModelHosts(getDataDir(), []);
fetchMock.mockReset();
fetchMock.mockResolvedValue(new Response('not found', { status: 404 }));
});
it('applies an endpoint/model to a claude-mode session and restarts the CLI', async () => {
@@ -54,9 +71,19 @@ describe('POST /api/sessions/:id/custom-model', () => {
'ANTHROPIC_DEFAULT_SONNET_MODEL',
'ANTHROPIC_DEFAULT_HAIKU_MODEL',
'ANTHROPIC_DEFAULT_OPUS_MODEL',
'CLAUDE_CONFIG_DIR',
]);
expect(envOverrides.ANTHROPIC_BASE_URL).toBe('http://192.168.1.50:8080');
expect(envOverrides.ANTHROPIC_API_KEY).toBe('k');
// CLAUDE_CONFIG_DIR isolates this session from a stored claude.ai OAuth login, and the
// trust-dialog file it points at is pre-seeded so the injected key doesn't hit an
// interactive "Detected a custom API key" prompt with nobody there to answer it.
const isolatedDir = join(getDataDir(), 'custom-model-configs', 'test-session-1');
expect(envOverrides.CLAUDE_CONFIG_DIR).toBe(isolatedDir);
expect(next.configDir).toBe(isolatedDir);
const trustFile = JSON.parse(readFileSync(join(isolatedDir, '.claude.json'), 'utf8'));
expect(trustFile.customApiKeyResponses.approved).toEqual(['k']);
});
it('clears back to the native default', async () => {
@@ -201,6 +228,381 @@ describe('POST /api/sessions/:id/custom-model', () => {
expect(existsSync(dir)).toBe(false);
});
describe('llama-swap conflict check (llama.cpp runs one model at a time)', () => {
function mockRunning(running: Array<{ model: string; state: string }>) {
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/running') return new Response(JSON.stringify({ running }), { status: 200 });
throw new Error(`unexpected request in this test: ${url.href}`);
});
}
it('applies straight away when the requested model is already loaded', async () => {
const { app, ctx } = await setup();
ctx.sessions.get('test-session-1')!.mode = 'claude';
mockRunning([{ model: 'qwen3', state: 'ready' }]);
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.json().success).not.toBe(false);
expect(res.json().modelSwapInProgress).toBe(false);
expect(ctx.sessions.get('test-session-1')!.setCustomModel).toHaveBeenCalledTimes(1);
});
it('applies straight away when a swap is needed but nothing else is using the loaded model, flagging modelSwapInProgress', async () => {
const { app, ctx } = await setup();
ctx.sessions.get('test-session-1')!.mode = 'claude';
mockRunning([{ model: 'llama3', state: 'ready' }]);
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.json().success).not.toBe(false);
expect(res.json().modelSwapInProgress).toBe(true);
expect(ctx.sessions.get('test-session-1')!.setCustomModel).toHaveBeenCalledTimes(1);
});
it('asks for confirmation instead of applying when another session is actively using the currently loaded model', async () => {
const { app, ctx } = await setup();
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const other = createMockSession('other-session');
other.name = 'w2-otherbox';
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
ctx.sessions.set('other-session', other);
mockRunning([{ model: 'llama3', state: 'ready' }]);
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
const body = res.json();
expect(body.success).not.toBe(false);
expect(body.requiresConfirmation).toBe(true);
expect(body.currentlyLoadedModel).toBe('llama3');
expect(body.affectedSessions).toEqual([{ id: 'other-session', name: 'w2-otherbox' }]);
// Nothing actually applied yet — this call only asked, it did not switch.
expect(session.setCustomModel).not.toHaveBeenCalled();
expect(session.restartCli).not.toHaveBeenCalled();
});
describe('multi-user: the confirm dialog must not name a session the caller cannot access', () => {
const saved: Record<string, string | undefined> = {};
beforeEach(() => {
saved.CODEMAN_MULTIUSER = process.env.CODEMAN_MULTIUSER;
process.env.CODEMAN_MULTIUSER = '1';
});
afterEach(() => {
if (saved.CODEMAN_MULTIUSER === undefined) delete process.env.CODEMAN_MULTIUSER;
else process.env.CODEMAN_MULTIUSER = saved.CODEMAN_MULTIUSER;
});
it("still blocks the swap pending confirmation, but omits a foreign owner's session from affectedSessions", async () => {
const { app, ctx } = await setup({ authUser: { username: 'bob', role: 'user' } });
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
(session as unknown as { owner?: string }).owner = 'bob';
const other = createMockSession('other-session');
other.name = 'w2-otherbox';
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
(other as unknown as { owner?: string }).owner = 'alice';
ctx.sessions.set('other-session', other);
mockRunning([{ model: 'llama3', state: 'ready' }]);
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
const body = res.json();
// Still asks — a foreign session is just as real a disruption as an owned one.
expect(body.requiresConfirmation).toBe(true);
expect(body.currentlyLoadedModel).toBe('llama3');
// But bob never learns alice's session id or name.
expect(body.affectedSessions).toEqual([]);
expect(session.setCustomModel).not.toHaveBeenCalled();
});
it('names the affected session when the caller DOES own it', async () => {
const { app, ctx } = await setup({ authUser: { username: 'bob', role: 'user' } });
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
(session as unknown as { owner?: string }).owner = 'bob';
const other = createMockSession('other-session');
other.name = 'w2-otherbox';
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
(other as unknown as { owner?: string }).owner = 'bob';
ctx.sessions.set('other-session', other);
mockRunning([{ model: 'llama3', state: 'ready' }]);
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.json().affectedSessions).toEqual([{ id: 'other-session', name: 'w2-otherbox' }]);
});
});
it('applies once confirmed, skipping the conflict check the second time', async () => {
const { app, ctx } = await setup();
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const other = createMockSession('other-session');
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
ctx.sessions.set('other-session', other);
mockRunning([{ model: 'llama3', state: 'ready' }]);
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3', confirmed: true },
});
const body = res.json();
expect(body.requiresConfirmation).toBeUndefined();
expect(body.modelSwapInProgress).toBe(true);
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
expect(session.restartCli).toHaveBeenCalledTimes(1);
});
it('a session pointed at the SAME endpoint but a DIFFERENT (not-currently-loaded) model is not treated as affected', async () => {
const { app, ctx } = await setup();
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const other = createMockSession('other-session');
other.customModel = { endpointId: 'ep1', modelId: 'some-other-model' }; // not the loaded one
ctx.sessions.set('other-session', other);
mockRunning([{ model: 'llama3', state: 'ready' }]);
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.json().requiresConfirmation).toBeUndefined();
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
});
it('not llama-swap (plain llama.cpp/OpenAI-compatible server, no /running) — never checked, applies straight away', async () => {
const { app, ctx } = await setup();
ctx.sessions.get('test-session-1')!.mode = 'claude';
fetchMock.mockResolvedValue(new Response('not found', { status: 404 }));
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.json().modelSwapInProgress).toBe(false);
expect(res.json().requiresConfirmation).toBeUndefined();
});
});
describe("context-window floor warning (this CLI's own overhead can exceed a small model's real context)", () => {
const SMALL_CTX_ENDPOINT: CustomModelHost = {
id: 'ep-small',
label: 'tiny box',
baseUrl: 'http://192.168.1.51:8080',
apiKey: 'k',
modelContextLengths: { 'qwen3.8-27b-ud-q4_k_xl': 16384 },
};
it('warns instead of applying when the discovered context is below the safe floor', async () => {
const { app, ctx } = await setup();
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
});
const body = res.json();
expect(body.success).not.toBe(false);
expect(body.requiresContextWarning).toBe(true);
expect(body.modelId).toBe('qwen3.8-27b-ud-q4_k_xl');
expect(body.contextLength).toBe(16384);
expect(body.minSafeContextTokens).toBe(40000);
// Nothing actually applied yet — this call only warned, it did not switch.
expect(session.setCustomModel).not.toHaveBeenCalled();
expect(session.restartCli).not.toHaveBeenCalled();
});
it('applies once confirmed, skipping the context check the second time', async () => {
const { app, ctx } = await setup();
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl', confirmed: true },
});
const body = res.json();
expect(body.requiresContextWarning).toBeUndefined();
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
expect(session.restartCli).toHaveBeenCalledTimes(1);
});
it('does not warn when the discovered context is comfortably above the floor', async () => {
const { app, ctx } = await setup();
const roomyEndpoint: CustomModelHost = {
id: 'ep-roomy',
label: 'roomy box',
baseUrl: 'http://192.168.1.52:8080',
apiKey: 'k',
modelContextLengths: { qwen3: 65536 },
};
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, roomyEndpoint]);
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep-roomy', modelId: 'qwen3' },
});
expect(res.json().requiresContextWarning).toBeUndefined();
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
});
it('does not warn when the context length was never discovered (nothing to compare)', async () => {
const { app, ctx } = await setup();
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT]);
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
expect(res.json().requiresContextWarning).toBeUndefined();
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
});
it('does not warn for a CLI whose registry entry declares no contextLengthVar (opencode)', async () => {
// opencode's customModelInjection kind is configContentEnv, not env+contextLengthVar,
// so exceedsSafeContextFloor is false by construction regardless of context size.
const { app, ctx } = await setup();
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'opencode';
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
});
expect(res.json().requiresContextWarning).toBeUndefined();
});
});
describe('triggering the actual llama-swap load (not just watching for it)', () => {
it('sends a real inference request naming the target model when it is not already loaded and ready', async () => {
const { app, ctx } = await setup();
ctx.sessions.get('test-session-1')!.mode = 'claude';
const chatCalls: unknown[] = [];
fetchMock.mockImplementation(async (url: URL, init?: { body?: unknown }) => {
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 });
}
if (url.pathname === '/v1/chat/completions') {
chatCalls.push(JSON.parse(init!.body as string));
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
await new Promise((resolve) => setTimeout(resolve, 0)); // let the fire-and-forget trigger settle
expect(chatCalls).toHaveLength(1);
expect(chatCalls[0]).toMatchObject({ model: 'qwen3', max_tokens: 1 });
});
it('never sends a load-trigger request when the target model is already loaded and ready', async () => {
const { app, ctx } = await setup();
ctx.sessions.get('test-session-1')!.mode = 'claude';
let chatCalled = false;
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'qwen3', state: 'ready' }] }), { status: 200 });
}
if (url.pathname === '/v1/chat/completions') {
chatCalled = true;
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
await new Promise((resolve) => setTimeout(resolve, 0));
expect(chatCalled).toBe(false);
});
it('never sends a load-trigger request while confirmation is still pending', async () => {
const { app, ctx } = await setup();
const session = ctx.sessions.get('test-session-1')!;
session.mode = 'claude';
const other = createMockSession('other-session');
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
ctx.sessions.set('other-session', other);
let chatCalled = false;
fetchMock.mockImplementation(async (url: URL) => {
if (url.pathname === '/running') {
return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 });
}
if (url.pathname === '/v1/chat/completions') {
chatCalled = true;
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
}
throw new Error(`unexpected request in this test: ${url.href}`);
});
const res = await app.inject({
method: 'POST',
url: '/api/sessions/test-session-1/custom-model',
payload: { endpointId: 'ep1', modelId: 'qwen3' },
});
await new Promise((resolve) => setTimeout(resolve, 0));
expect(res.json().requiresConfirmation).toBe(true);
expect(chatCalled).toBe(false);
});
});
it('refuses to touch a busy session', async () => {
const { app, ctx } = await setup();
const session = ctx.sessions.get('test-session-1')!;
@@ -218,3 +620,39 @@ describe('POST /api/sessions/:id/custom-model', () => {
expect(session.setCustomModel).not.toHaveBeenCalled();
});
});
describe('Claude multi-user clamp: the env-var half', () => {
// CLAUDE_CODE_MAX_CONTEXT_TOKENS and CLAUDE_CONFIG_DIR were already reachable via
// plain envOverrides before claude's privilegedEnvKeys existed (the first already
// matches the CLAUDE_CODE_* allowedPrefix, the second is an allowed exact key), so
// listing them here is not what makes this route safe — no custom-model route reads
// privilegedEnvKeys at all. What it DOES do: ownerClampedEnvKeys() feeds the generic
// envOverrides clamp on create/quick-start/reboot-restore, so a non-granted owner can
// no longer set CLAUDE_CONFIG_DIR that way (the per-client-account feature, #255), and
// a PERSISTED one is now stripped on reboot-restore for such an owner too — see
// session-env-clamp.ts's own fileoverview for why that pass used to be a no-op for
// claude specifically.
const ORIGINAL = process.env.CODEMAN_MULTIUSER;
beforeEach(() => {
process.env.CODEMAN_MULTIUSER = '1';
});
afterEach(() => {
if (ORIGINAL === undefined) delete process.env.CODEMAN_MULTIUSER;
else process.env.CODEMAN_MULTIUSER = ORIGINAL;
});
it('strips CLAUDE_CONFIG_DIR and CLAUDE_CODE_MAX_CONTEXT_TOKENS for a non-granted owner, leaving unrelated CLAUDE_CODE_* keys alone', async () => {
const out = await _clampEnvOverridesForOwner('nobody', {
CLAUDE_CONFIG_DIR: '/home/attacker/fake-claude-config',
CLAUDE_CODE_MAX_CONTEXT_TOKENS: '999999',
CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS: '1',
});
expect(out).toEqual({ CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS: '1' });
});
it('is a no-op in single-user mode', async () => {
delete process.env.CODEMAN_MULTIUSER;
const input = { CLAUDE_CONFIG_DIR: '/home/attacker/fake-claude-config' };
expect(await _clampEnvOverridesForOwner(undefined, input)).toBe(input);
});
});
+11 -7
View File
@@ -96,16 +96,20 @@ describe('WebServer index.html <title> templating (#82)', () => {
it('only substitutes the <title> tag — the rest of the template is identical (modulo asset cache-busting)', async () => {
// renderIndexHtml also appends ?v=<mtime> cache-bust params to same-origin
// .js/.css refs, and injects the CLI-availability flags before </head>; strip
// both so the title remains the only other change.
// .js/.css refs, and injects the CLI-availability flags plus the custom-model
// Run-menu picker's CLI list before </head>; strip all so the title remains
// the only other change.
//
// The flag strip is what keeps this test environment-independent. It used to
// pass here by luck: the availability script was injected only where a CLI
// resolved, so the assertion held on a machine with none installed and would
// have failed on a developer's box that had them.
// The flag strips are what keep this test environment-independent. The
// CLI-availability one used to pass here by luck: that script was injected
// only where a CLI resolved, so the assertion held on a machine with none
// installed and would have failed on a developer's box that had them. The
// custom-model list is injected unconditionally (a plain array, possibly
// empty), so it needs stripping on every machine, not just where non-empty.
const html = (await render('laptop'))
.replace(/(\.(?:js|css))\?v=[^"]*/g, '$1')
.replace(/<script>window\.__codemanCliAvailable=\{.*?\};<\/script>\n/, '');
.replace(/<script>window\.__codemanCliAvailable=\{.*?\};<\/script>\n/, '')
.replace(/<script>window\.__codemanCustomModelClis=\[.*?\];<\/script>\n/, '');
const beforeTitle = rawTemplate.split('<title>Codeman</title>')[0];
const afterTitle = rawTemplate.split('<title>Codeman</title>')[1];
expect(html.startsWith(beforeTitle)).toBe(true);