mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-06 23:49:41 +02:00
Merge pull request #430 from opticon454/custom-model-run-menu
This commit is contained in:
@@ -0,0 +1,368 @@
|
||||
/**
|
||||
* @fileoverview Tests for `refreshAllCustomModelHosts()`, the periodic
|
||||
* background sweep behind server.ts's "custom model endpoint re-discovery"
|
||||
* timer (docs/custom-model-endpoints-plan.md). Kept in its own file rather
|
||||
* than folded into test/routes/custom-model-routes.test.ts: that file's data
|
||||
* dir is shared across every test in it (one temp HOME per FILE, not per
|
||||
* test — test/setup.ts), and a sweep that walks every saved host would pick
|
||||
* up every host any other test in that file happened to create, making an
|
||||
* exact call-count or exact-host assertion meaningless. A dedicated file
|
||||
* gets its own clean temp HOME.
|
||||
*
|
||||
* Port: N/A (no server; drives readCustomModelHosts/writeCustomModelHosts
|
||||
* directly plus the mocked webviewFetch dispatcher).
|
||||
*/
|
||||
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
||||
import { getDataDir } from '../src/config/instance.js';
|
||||
import { readCustomModelHosts, writeCustomModelHosts, type CustomModelHost } from '../src/custom-model-hosts.js';
|
||||
import { refreshAllCustomModelHosts } from '../src/web/routes/custom-model-routes.js';
|
||||
import { webviewFetch } from '../src/web/webview-egress.js';
|
||||
|
||||
vi.mock('../src/web/webview-egress.js', async () => {
|
||||
const actual = await vi.importActual<typeof import('../src/web/webview-egress.js')>('../src/web/webview-egress.js');
|
||||
return { ...actual, webviewFetch: vi.fn() };
|
||||
});
|
||||
|
||||
const fetchMock = vi.mocked(webviewFetch);
|
||||
|
||||
function host(overrides: Partial<CustomModelHost> & Pick<CustomModelHost, 'id' | 'baseUrl'>): CustomModelHost {
|
||||
return { label: overrides.id, ...overrides };
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
fetchMock.mockReset();
|
||||
});
|
||||
|
||||
describe('refreshAllCustomModelHosts (the periodic re-discovery sweep)', () => {
|
||||
it('refreshes every saved endpoint, best-effort — one unreachable host does not stop the others', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [
|
||||
host({ id: 'ok', baseUrl: 'http://localhost:8080' }),
|
||||
host({ id: 'down', baseUrl: 'http://localhost:8081' }),
|
||||
]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.href.includes('8081')) throw new TypeError('fetch failed', { cause: new Error('ECONNREFUSED') });
|
||||
return new Response(JSON.stringify({ data: [{ id: 'qwen3' }] }), { status: 200 });
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const hosts = await readCustomModelHosts(dir);
|
||||
const ok = hosts.find((h) => h.id === 'ok');
|
||||
const down = hosts.find((h) => h.id === 'down');
|
||||
expect(ok?.models).toEqual(['qwen3']);
|
||||
expect(ok?.lastDiscoveredAt).toBeTruthy();
|
||||
expect(down?.models ?? []).toEqual([]);
|
||||
expect(down?.lastDiscoveredAt).toBeFalsy();
|
||||
});
|
||||
|
||||
it('skips a host whose baseUrl is blocked, without making a request', async () => {
|
||||
const dir = getDataDir();
|
||||
// Written directly rather than through the POST route, which already
|
||||
// refuses this at save time — this simulates a record that pre-dates the
|
||||
// guard, or was hand-edited on disk. The sweep must not trust it either.
|
||||
await writeCustomModelHosts(dir, [host({ id: 'meta', baseUrl: 'http://169.254.169.254/' })]);
|
||||
|
||||
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'x' }] }), { status: 200 }));
|
||||
await refreshAllCustomModelHosts();
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('drops a stale default and preserves lastDiscoveredAt semantics, same as manual discovery', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [
|
||||
host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['qwen3'], defaultModelId: 'qwen3' }),
|
||||
]);
|
||||
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'llama3' }] }), { status: 200 }));
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.models).toEqual(['llama3']);
|
||||
expect(updated.defaultModelId).toBeUndefined();
|
||||
expect(updated.lastDiscoveredAt).toBeTruthy();
|
||||
});
|
||||
|
||||
it('keeps a default that is still present after the sweep', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [
|
||||
host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['qwen3'], defaultModelId: 'qwen3' }),
|
||||
]);
|
||||
fetchMock.mockResolvedValue(
|
||||
new Response(JSON.stringify({ data: [{ id: 'qwen3' }, { id: 'llama3' }] }), { status: 200 })
|
||||
);
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.defaultModelId).toBe('qwen3');
|
||||
});
|
||||
|
||||
it('does not resurrect an endpoint deleted while the sweep was in flight', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'deleted', baseUrl: 'http://localhost:8080' })]);
|
||||
|
||||
fetchMock.mockImplementation(async () => {
|
||||
// Simulate an admin deleting the endpoint between the sweep's fetch and
|
||||
// its read-modify-write — the delete must win, not be overwritten by a
|
||||
// refresh that started before it.
|
||||
const current = await readCustomModelHosts(dir);
|
||||
await writeCustomModelHosts(
|
||||
dir,
|
||||
current.filter((h) => h.id !== 'deleted')
|
||||
);
|
||||
return new Response(JSON.stringify({ data: [{ id: 'qwen3' }] }), { status: 200 });
|
||||
});
|
||||
|
||||
await expect(refreshAllCustomModelHosts()).resolves.toBeUndefined();
|
||||
const hosts = await readCustomModelHosts(dir);
|
||||
expect(hosts.find((h) => h.id === 'deleted')).toBeUndefined();
|
||||
});
|
||||
|
||||
it('leaves the store untouched when there are no saved endpoints at all', async () => {
|
||||
await expect(refreshAllCustomModelHosts()).resolves.toBeUndefined();
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
|
||||
describe('refreshAllCustomModelHosts: context-length enrichment (llama.cpp/llama-swap /props)', () => {
|
||||
it('probes /props?model= only for a model reported loaded, and stores its n_ctx', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
data: [
|
||||
{ id: 'loaded-model', status: { value: 'loaded' } },
|
||||
{ id: 'unloaded-model', status: { value: 'unloaded' } },
|
||||
],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
if (url.pathname === '/props') {
|
||||
// Must never be reached for the unloaded model — asserted below by call count.
|
||||
expect(url.searchParams.get('model')).toBe('loaded-model');
|
||||
return new Response(JSON.stringify({ n_ctx: 16384 }), { status: 200 });
|
||||
}
|
||||
throw new Error(`unexpected request: ${url.href}`);
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ 'loaded-model': 16384 });
|
||||
const propsCalls = fetchMock.mock.calls.filter(([url]) => (url as URL).pathname === '/props');
|
||||
expect(propsCalls).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('never probes /props at all when no entry mentions status — feature-detected, not assumed unloaded', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'qwen3' }] }), { status: 200 }));
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1); // /v1/models only
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toBeUndefined();
|
||||
});
|
||||
|
||||
it('keeps a previously-learned context length for a model no longer loaded, drops it once the model disappears entirely', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [
|
||||
host({
|
||||
id: 'ep',
|
||||
baseUrl: 'http://localhost:8080',
|
||||
models: ['a', 'b'],
|
||||
modelContextLengths: { a: 8192, b: 4096 },
|
||||
}),
|
||||
]);
|
||||
// This round: 'a' is loaded (re-confirmed), 'b' is gone from the list entirely.
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
|
||||
}
|
||||
return new Response(JSON.stringify({ n_ctx: 8192 }), { status: 200 });
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ a: 8192 });
|
||||
});
|
||||
|
||||
it('a failed /props probe for the loaded model is swallowed, leaving no context length rather than failing the sweep', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
|
||||
}
|
||||
return new Response('nope', { status: 500 });
|
||||
});
|
||||
|
||||
await expect(refreshAllCustomModelHosts()).resolves.toBeUndefined();
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toBeUndefined();
|
||||
});
|
||||
|
||||
it('prefers the REAL configured context size parsed from /running’s launch command over /props’s unreliable n_ctx', async () => {
|
||||
// Confirmed live: llama-swap launched a model with --fit-ctx 16384 (the real, working
|
||||
// limit — the actual server then refused a request over it), but /props reported
|
||||
// n_ctx: 154112 for the same model, well over what it would really accept. /props must
|
||||
// never be reached at all once the /running command parse already answered it.
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'qwen3.8-27b', status: { value: 'loaded' } }] }), {
|
||||
status: 200,
|
||||
});
|
||||
}
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
running: [
|
||||
{
|
||||
model: 'qwen3.8-27b',
|
||||
state: 'ready',
|
||||
cmd: 'llama-server -m /models/Qwen3.8-27B.gguf --flash-attn on --jinja --fit-ctx 16384 --host 0.0.0.0 --port 5840',
|
||||
},
|
||||
],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
if (url.pathname === '/props') throw new Error('must never be reached — the cmd parse already answered it');
|
||||
throw new Error(`unexpected request: ${url.href}`);
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ 'qwen3.8-27b': 16384 });
|
||||
});
|
||||
|
||||
it('falls back to /props when /running has no cmd, or the cmd states no recognizable context flag', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(
|
||||
JSON.stringify({ running: [{ model: 'a', state: 'ready', cmd: 'llama-server -m /models/a.gguf' }] }),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
if (url.pathname === '/props') return new Response(JSON.stringify({ n_ctx: 8192 }), { status: 200 });
|
||||
throw new Error(`unexpected request: ${url.href}`);
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ a: 8192 });
|
||||
});
|
||||
|
||||
it('also recognizes a plain -c/--ctx-size flag, not just llama-swap’s own --fit-ctx', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
running: [{ model: 'a', state: 'ready', cmd: 'llama-server -m /models/a.gguf --ctx-size 8192' }],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
throw new Error(`unexpected request: ${url.href}`); // /props must never be reached
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ a: 8192 });
|
||||
});
|
||||
});
|
||||
|
||||
describe('refreshAllCustomModelHosts: model-size enrichment (parsed from /v1/models description)', () => {
|
||||
it('parses a GB figure out of an auto-discovered model’s description', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockResolvedValue(
|
||||
new Response(
|
||||
JSON.stringify({
|
||||
data: [{ id: 'qwen3.8-27b', description: 'Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp' }],
|
||||
}),
|
||||
{ status: 200 }
|
||||
)
|
||||
);
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelSizesGB).toEqual({ 'qwen3.8-27b': 16.35 });
|
||||
});
|
||||
|
||||
it('gets no size at all for a hand-configured profile whose own description states none', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockResolvedValue(
|
||||
new Response(
|
||||
JSON.stringify({
|
||||
data: [{ id: 'big', description: 'General-purpose reasoning model, MoE CPU-offloaded. Default profile.' }],
|
||||
}),
|
||||
{ status: 200 }
|
||||
)
|
||||
);
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelSizesGB).toBeUndefined();
|
||||
});
|
||||
|
||||
it('populated regardless of loaded state — unlike context length, no /props probe is needed', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
data: [{ id: 'unloaded-model', description: 'Auto-discovered 4.91 GB - parameters auto-fitted' }],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
throw new Error(`unexpected request: ${url.href}`); // /props must never be reached for this
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelSizesGB).toEqual({ 'unloaded-model': 4.91 });
|
||||
});
|
||||
|
||||
it('keeps a previously-learned size for a model still present, drops it once the model disappears entirely', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [
|
||||
host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['a', 'b'], modelSizesGB: { a: 8, b: 16 } }),
|
||||
]);
|
||||
fetchMock.mockResolvedValue(
|
||||
new Response(JSON.stringify({ data: [{ id: 'a', description: 'no GB figure here' }] }), { status: 200 })
|
||||
);
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelSizesGB).toEqual({ a: 8 }); // 'a' kept from before, 'b' dropped (gone from the list)
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,303 @@
|
||||
/**
|
||||
* @fileoverview Tests for the two custom-model IO-layer fixes on top of the pure builder
|
||||
* (docs/custom-model-endpoints-plan.md):
|
||||
*
|
||||
* 1. `contextLengthVar` — a discovered per-model context length reaches the actual
|
||||
* session env (CLAUDE_CODE_MAX_CONTEXT_TOKENS), so a CLI stops assuming a large
|
||||
* default window for an unrecognized custom model id and overflowing a much
|
||||
* smaller real one.
|
||||
* 2. `configDirVar` — an isolated, empty config directory is created and pointed at
|
||||
* (CLAUDE_CONFIG_DIR), so an injected API key never shares a directory with a
|
||||
* stored claude.ai OAuth session; `projects` is symlinked back into the real
|
||||
* config dir so the response viewer/subagent windows/Read My Mind keep working.
|
||||
*
|
||||
* Port: N/A (no server; filesystem-only, under a temp CODEMAN data dir from test/setup.ts).
|
||||
*/
|
||||
import { existsSync, lstatSync, mkdirSync, readFileSync, readdirSync, rmSync, writeFileSync } from 'node:fs';
|
||||
import { homedir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { afterEach, describe, expect, it } from 'vitest';
|
||||
import { getCli } from '../src/config/cli-registry/index.js';
|
||||
import { applyCustomModelInjection, customModelConfigDir } from '../src/custom-model-injection-apply.js';
|
||||
import type { CustomModelEndpoint } from '../src/custom-model-injection.js';
|
||||
|
||||
const endpoint: CustomModelEndpoint = {
|
||||
id: 'ep1',
|
||||
label: 'llama.cpp box',
|
||||
baseUrl: 'http://192.168.1.50:8080',
|
||||
apiKey: 'my-key',
|
||||
};
|
||||
|
||||
function entryOrThrow(id: string) {
|
||||
const entry = getCli(id);
|
||||
if (!entry) throw new Error(`missing CLI registry entry: ${id}`);
|
||||
return entry;
|
||||
}
|
||||
|
||||
const sessionsToClean: string[] = [];
|
||||
afterEach(() => {
|
||||
for (const id of sessionsToClean.splice(0)) rmSync(customModelConfigDir(id), { recursive: true, force: true });
|
||||
});
|
||||
|
||||
describe('applyCustomModelInjection: context length', () => {
|
||||
it('claude: passes a known context length through to CLAUDE_CODE_MAX_CONTEXT_TOKENS', () => {
|
||||
sessionsToClean.push('sess-ctx-1');
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', 'sess-ctx-1', 16384);
|
||||
expect(applied?.envOverrides.CLAUDE_CODE_MAX_CONTEXT_TOKENS).toBe('16384');
|
||||
expect(applied?.envKeys).toContain('CLAUDE_CODE_MAX_CONTEXT_TOKENS');
|
||||
});
|
||||
|
||||
it('claude: omits the var entirely when the context length is unknown', () => {
|
||||
sessionsToClean.push('sess-ctx-2');
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', 'sess-ctx-2');
|
||||
expect(applied?.envOverrides.CLAUDE_CODE_MAX_CONTEXT_TOKENS).toBeUndefined();
|
||||
});
|
||||
|
||||
it('deepseek: has no contextLengthVar declared, so a passed-in length is a no-op', () => {
|
||||
const applied = applyCustomModelInjection(entryOrThrow('deepseek'), endpoint, 'qwen3', 'sess-ctx-3', 16384);
|
||||
expect(Object.keys(applied?.envOverrides ?? {}).sort()).toEqual(['DEEPSEEK_API_KEY', 'DEEPSEEK_BASE_URL']);
|
||||
});
|
||||
});
|
||||
|
||||
describe('applyCustomModelInjection: CLAUDE_CONFIG_DIR isolation', () => {
|
||||
it('claude: creates an isolated config dir (no real credential/config files) and points CLAUDE_CONFIG_DIR at it', () => {
|
||||
const sessionId = 'sess-cfgdir-1';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const expectedDir = customModelConfigDir(sessionId);
|
||||
expect(applied?.envOverrides.CLAUDE_CONFIG_DIR).toBe(expectedDir);
|
||||
expect(applied?.configDir).toBe(expectedDir);
|
||||
expect(existsSync(expectedDir)).toBe(true);
|
||||
// The trust-seed file, the skipFirstRunPrompts settings.json, and the projects link —
|
||||
// no real OAuth credential/config.
|
||||
const entries = readdirSync(expectedDir).filter((name) => name !== 'projects');
|
||||
expect(entries.sort()).toEqual(['.claude.json', 'settings.json']);
|
||||
});
|
||||
|
||||
it('claude: symlinks (or junctions) projects back to the real config dir so the response viewer keeps working', () => {
|
||||
const sessionId = 'sess-cfgdir-2';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const link = join(applied!.configDir!, 'projects');
|
||||
// Best-effort: only assert the link exists if it was actually created (the real
|
||||
// ~/.claude/projects may not exist on a bare CI box, in which case linking is skipped).
|
||||
if (existsSync(join(homedir(), '.claude', 'projects'))) {
|
||||
expect(existsSync(link)).toBe(true);
|
||||
expect(lstatSync(link).isSymbolicLink() || lstatSync(link).isDirectory()).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it('claude: re-applying to the same session is idempotent (boot-recovery re-apply)', () => {
|
||||
const sessionId = 'sess-cfgdir-3';
|
||||
sessionsToClean.push(sessionId);
|
||||
const first = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const second = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
expect(second?.configDir).toBe(first?.configDir);
|
||||
expect(existsSync(first!.configDir!)).toBe(true);
|
||||
});
|
||||
|
||||
it('pi: configDir-kind CLIs are unaffected — no configDirVar concept for them', () => {
|
||||
const sessionId = 'sess-cfgdir-pi';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('pi'), endpoint, 'qwen3', sessionId);
|
||||
expect(applied?.envOverrides.HOME).toBe(customModelConfigDir(sessionId));
|
||||
});
|
||||
|
||||
it('deepseek: no configDirVar declared, so no config dir is created at all', () => {
|
||||
const sessionId = 'sess-cfgdir-deepseek';
|
||||
const applied = applyCustomModelInjection(entryOrThrow('deepseek'), endpoint, 'qwen3', sessionId);
|
||||
expect(applied?.configDir).toBeUndefined();
|
||||
expect(existsSync(customModelConfigDir(sessionId))).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('applyCustomModelInjection: apiKeyTrustFile (pre-approves the injected key)', () => {
|
||||
it('claude: seeds .claude.json so the "Detected a custom API key" prompt never fires', () => {
|
||||
const sessionId = 'sess-trust-1';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
customApiKeyResponses: { approved: string[]; rejected: string[] };
|
||||
};
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
|
||||
expect(written.customApiKeyResponses.rejected).toEqual([]);
|
||||
});
|
||||
|
||||
it('claude: falls back to the dummy key when the endpoint has none, and still seeds it', () => {
|
||||
const sessionId = 'sess-trust-2';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(
|
||||
entryOrThrow('claude'),
|
||||
{ ...endpoint, apiKey: undefined },
|
||||
'qwen3',
|
||||
sessionId
|
||||
);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
customApiKeyResponses: { approved: string[] };
|
||||
};
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['local-dummy-key']);
|
||||
});
|
||||
|
||||
it('claude: merges onto fields the CLI itself already wrote into the same isolated dir, never overwrites them', () => {
|
||||
const sessionId = 'sess-trust-3';
|
||||
sessionsToClean.push(sessionId);
|
||||
const configDir = customModelConfigDir(sessionId);
|
||||
mkdirSync(configDir, { recursive: true });
|
||||
writeFileSync(join(configDir, '.claude.json'), JSON.stringify({ userID: 'abc123', numStartups: 3 }));
|
||||
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
userID: string;
|
||||
numStartups: number;
|
||||
customApiKeyResponses: { approved: string[] };
|
||||
};
|
||||
expect(written.userID).toBe('abc123');
|
||||
expect(written.numStartups).toBe(3);
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
|
||||
});
|
||||
|
||||
it('claude: a corrupt existing file is treated as absent rather than failing the apply', () => {
|
||||
const sessionId = 'sess-trust-4';
|
||||
sessionsToClean.push(sessionId);
|
||||
const configDir = customModelConfigDir(sessionId);
|
||||
mkdirSync(configDir, { recursive: true });
|
||||
writeFileSync(join(configDir, '.claude.json'), '{ not valid json');
|
||||
|
||||
expect(() => applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId)).not.toThrow();
|
||||
const written = JSON.parse(readFileSync(join(configDir, '.claude.json'), 'utf8')) as {
|
||||
customApiKeyResponses: { approved: string[] };
|
||||
};
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
|
||||
});
|
||||
|
||||
it('claude: re-approving the same key does not duplicate it in the approved list', () => {
|
||||
const sessionId = 'sess-trust-5';
|
||||
sessionsToClean.push(sessionId);
|
||||
applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const second = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'llama3', sessionId);
|
||||
const written = JSON.parse(readFileSync(join(second!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
customApiKeyResponses: { approved: string[] };
|
||||
};
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
|
||||
});
|
||||
|
||||
it('opencode: has no apiKeyTrustFile declared (no configDirVar at all), nothing is seeded', () => {
|
||||
const sessionId = 'sess-trust-opencode';
|
||||
const applied = applyCustomModelInjection(entryOrThrow('opencode'), endpoint, 'qwen3', sessionId);
|
||||
expect(applied?.configDir).toBeUndefined();
|
||||
expect(existsSync(customModelConfigDir(sessionId))).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("applyCustomModelInjection: skipFirstRunPrompts (an isolated dir replays claude's whole first-run sequence)", () => {
|
||||
it("claude: seeds hasCompletedOnboarding and this session's own project trust into .claude.json", () => {
|
||||
const sessionId = 'sess-firstrun-1';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(
|
||||
entryOrThrow('claude'),
|
||||
endpoint,
|
||||
'qwen3',
|
||||
sessionId,
|
||||
undefined,
|
||||
'/home/user/myproject'
|
||||
);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
hasCompletedOnboarding: boolean;
|
||||
projects: Record<string, { hasTrustDialogAccepted: boolean }>;
|
||||
};
|
||||
expect(written.hasCompletedOnboarding).toBe(true);
|
||||
expect(written.projects['/home/user/myproject'].hasTrustDialogAccepted).toBe(true);
|
||||
});
|
||||
|
||||
it('claude: seeds skipDangerousModePermissionPrompt into settings.json', () => {
|
||||
const sessionId = 'sess-firstrun-2';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, 'settings.json'), 'utf8')) as {
|
||||
skipDangerousModePermissionPrompt: boolean;
|
||||
};
|
||||
expect(written.skipDangerousModePermissionPrompt).toBe(true);
|
||||
});
|
||||
|
||||
it('claude: with no workingDir given (boot recovery), hasCompletedOnboarding/settings still seed, but no project entry is added', () => {
|
||||
const sessionId = 'sess-firstrun-3';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
hasCompletedOnboarding: boolean;
|
||||
projects?: Record<string, unknown>;
|
||||
};
|
||||
expect(written.hasCompletedOnboarding).toBeUndefined();
|
||||
expect(written.projects).toBeUndefined();
|
||||
});
|
||||
|
||||
it("claude: merges onto an existing project entry's other fields rather than overwriting them", () => {
|
||||
const sessionId = 'sess-firstrun-4';
|
||||
sessionsToClean.push(sessionId);
|
||||
const configDir = customModelConfigDir(sessionId);
|
||||
mkdirSync(configDir, { recursive: true });
|
||||
writeFileSync(
|
||||
join(configDir, '.claude.json'),
|
||||
JSON.stringify({ projects: { '/home/user/myproject': { allowedTools: ['Bash'] } } })
|
||||
);
|
||||
|
||||
const applied = applyCustomModelInjection(
|
||||
entryOrThrow('claude'),
|
||||
endpoint,
|
||||
'qwen3',
|
||||
sessionId,
|
||||
undefined,
|
||||
'/home/user/myproject'
|
||||
);
|
||||
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
projects: Record<string, { allowedTools: string[]; hasTrustDialogAccepted: boolean }>;
|
||||
};
|
||||
expect(written.projects['/home/user/myproject'].allowedTools).toEqual(['Bash']);
|
||||
expect(written.projects['/home/user/myproject'].hasTrustDialogAccepted).toBe(true);
|
||||
});
|
||||
|
||||
it('claude: a corrupt existing settings.json is treated as absent rather than failing the apply', () => {
|
||||
const sessionId = 'sess-firstrun-5';
|
||||
sessionsToClean.push(sessionId);
|
||||
const configDir = customModelConfigDir(sessionId);
|
||||
mkdirSync(configDir, { recursive: true });
|
||||
writeFileSync(join(configDir, 'settings.json'), '{ not valid json');
|
||||
|
||||
expect(() => applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId)).not.toThrow();
|
||||
const written = JSON.parse(readFileSync(join(configDir, 'settings.json'), 'utf8')) as {
|
||||
skipDangerousModePermissionPrompt: boolean;
|
||||
};
|
||||
expect(written.skipDangerousModePermissionPrompt).toBe(true);
|
||||
});
|
||||
|
||||
it('pi: has no skipFirstRunPrompts concept (no apiKeyTrustFile either) — nothing beyond its own config file', () => {
|
||||
const sessionId = 'sess-firstrun-pi';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(
|
||||
entryOrThrow('pi'),
|
||||
endpoint,
|
||||
'qwen3',
|
||||
sessionId,
|
||||
undefined,
|
||||
'/home/user/myproject'
|
||||
);
|
||||
const entries = readdirSync(applied!.configDir!);
|
||||
expect(entries).not.toContain('settings.json');
|
||||
});
|
||||
});
|
||||
|
||||
describe('applyCustomModelInjection: pre-existing behavior unaffected', () => {
|
||||
it('opencode: still returns a plain env-kind result with no configDir', () => {
|
||||
const sessionId = 'sess-opencode-1';
|
||||
const applied = applyCustomModelInjection(entryOrThrow('opencode'), endpoint, 'qwen3', sessionId);
|
||||
expect(applied?.configDir).toBeUndefined();
|
||||
expect(applied?.envOverrides.OPENCODE_CONFIG_CONTENT).toBeTruthy();
|
||||
});
|
||||
|
||||
it('antigravity: still undefined (unsupported)', () => {
|
||||
const applied = applyCustomModelInjection(entryOrThrow('antigravity'), endpoint, 'qwen3', 'sess-agy-1');
|
||||
expect(applied).toBeUndefined();
|
||||
});
|
||||
});
|
||||
@@ -142,17 +142,19 @@ describe('custom-model-injection contract (mock server)', () => {
|
||||
expect(mock.requests[0].headers.authorization).toBe('Bearer contract-test-key');
|
||||
});
|
||||
|
||||
// gemini/deepseek's `env` kind passes the base URL through UNCHANGED (unlike
|
||||
// opencode/codex/pi/omp/grok, which build a structured config and explicitly append
|
||||
// /v1) — matching Anthropic's own convention for claude's ANTHROPIC_BASE_URL, where the
|
||||
// SDK appends the path itself. Whether each of these TWO CLIs' own OpenAI-compatible
|
||||
// client expects the var to already include /v1 (the common OpenAI-SDK convention) or
|
||||
// appends it itself is genuinely CLI-specific and UNVERIFIED (see the confidence table
|
||||
// in docs/custom-model-endpoints-plan.md) — these tests model the common OpenAI-SDK convention (base_url
|
||||
// ends in /v1) since that's the more likely behavior for an OpenAI-compatible client,
|
||||
// but that assumption should be corrected here the moment it's checked against a real
|
||||
// binary. (grok WAS in this group too, until live-testing showed the whole `env` recipe
|
||||
// was wrong for it — see its own test below.)
|
||||
// gemini's `env` kind still passes the base URL through UNCHANGED (matching
|
||||
// Anthropic's own convention for claude's ANTHROPIC_BASE_URL, where the SDK appends
|
||||
// the path itself) — whether gemini-cli's own OpenAI-compatible-ish client expects the
|
||||
// var to already include /v1 or appends it itself remains genuinely UNVERIFIED (it
|
||||
// fails for an unrelated auth reason before this would even matter — see the
|
||||
// confidence table in docs/custom-model-endpoints-plan.md); this test models the
|
||||
// common OpenAI-SDK convention as the best guess, to be corrected the moment it's
|
||||
// checked against a real client. deepseek WAS in this "passes through unchanged"
|
||||
// group too, until reading `@deepseek-ai/dsh-llm-deepseek`'s own bundled source
|
||||
// confirmed it builds its request URL as `${DEEPSEEK_BASE_URL}/chat/completions` with
|
||||
// no `/v1` of its own — `appendV1Suffix` now fixes that (see its own test below),
|
||||
// the same way grok's whole `env` recipe turned out to be wrong before live-testing
|
||||
// corrected it to a `configDir` one.
|
||||
|
||||
it('gemini: GOOGLE_GEMINI_BASE_URL/GEMINI_API_KEY reach the mock', async () => {
|
||||
const injection = buildCustomModelInjection(entryOrThrow('gemini'), endpointFor(mock), 'qwen3');
|
||||
@@ -186,16 +188,18 @@ describe('custom-model-injection contract (mock server)', () => {
|
||||
expect(mock.requests[0].headers.authorization).toBe('Bearer contract-test-key');
|
||||
});
|
||||
|
||||
it('deepseek: DEEPSEEK_BASE_URL/DEEPSEEK_API_KEY reach the mock (base URL/key only, no model var)', async () => {
|
||||
it('deepseek: DEEPSEEK_BASE_URL already carries the /v1 suffix dsh itself never adds, reaching the mock at the real path dsh requests', async () => {
|
||||
// Confirmed by reading dsh's own bundled source: it fetches
|
||||
// `${DEEPSEEK_BASE_URL}/chat/completions` verbatim, no /v1 insertion of its own — so
|
||||
// this call (unlike gemini's above) passes DEEPSEEK_BASE_URL to callOpenAiCompat
|
||||
// UNMODIFIED, exactly mirroring what the real harness does, rather than the test
|
||||
// helping it along.
|
||||
const injection = buildCustomModelInjection(entryOrThrow('deepseek'), endpointFor(mock), 'qwen3');
|
||||
if (injection.kind !== 'env') throw new Error('unreachable');
|
||||
expect(Object.keys(injection.envOverrides).sort()).toEqual(['DEEPSEEK_API_KEY', 'DEEPSEEK_BASE_URL']);
|
||||
expect(injection.envOverrides.DEEPSEEK_BASE_URL).toBe(`${mock.baseUrl}/v1`);
|
||||
|
||||
await callOpenAiCompat(
|
||||
`${injection.envOverrides.DEEPSEEK_BASE_URL}/v1`,
|
||||
injection.envOverrides.DEEPSEEK_API_KEY,
|
||||
'qwen3'
|
||||
);
|
||||
await callOpenAiCompat(injection.envOverrides.DEEPSEEK_BASE_URL, injection.envOverrides.DEEPSEEK_API_KEY, 'qwen3');
|
||||
|
||||
expect(mock.requests[0].path).toBe('/v1/chat/completions');
|
||||
expect(mock.requests[0].headers.authorization).toBe('Bearer contract-test-key');
|
||||
|
||||
@@ -58,6 +58,43 @@ describe('buildCustomModelInjection', () => {
|
||||
});
|
||||
});
|
||||
|
||||
it('claude: also declares configDirVar (CLAUDE_CONFIG_DIR isolation) on the env-kind result', () => {
|
||||
const result = buildCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3');
|
||||
if (result.kind !== 'env') throw new Error('unreachable');
|
||||
expect(result.configDirVar).toBe('CLAUDE_CONFIG_DIR');
|
||||
});
|
||||
|
||||
it('claude: injects CLAUDE_CODE_MAX_CONTEXT_TOKENS when a context length is known', () => {
|
||||
const result = buildCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', 16384);
|
||||
if (result.kind !== 'env') throw new Error('unreachable');
|
||||
expect(result.envOverrides.CLAUDE_CODE_MAX_CONTEXT_TOKENS).toBe('16384');
|
||||
});
|
||||
|
||||
it('claude: omits CLAUDE_CODE_MAX_CONTEXT_TOKENS when the context length is unknown', () => {
|
||||
const result = buildCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3');
|
||||
if (result.kind !== 'env') throw new Error('unreachable');
|
||||
expect(result.envOverrides.CLAUDE_CODE_MAX_CONTEXT_TOKENS).toBeUndefined();
|
||||
});
|
||||
|
||||
it('claude: also declares apiKeyTrustFile, carrying the literal apiKey used', () => {
|
||||
const result = buildCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3');
|
||||
if (result.kind !== 'env') throw new Error('unreachable');
|
||||
expect(result.apiKeyTrustFile).toEqual({ relPath: '.claude.json', shape: 'claude-api-key-responses' });
|
||||
expect(result.apiKey).toBe('my-key');
|
||||
});
|
||||
|
||||
it('claude: also declares skipFirstRunPrompts on the env-kind result', () => {
|
||||
const result = buildCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3');
|
||||
if (result.kind !== 'env') throw new Error('unreachable');
|
||||
expect(result.skipFirstRunPrompts).toBe(true);
|
||||
});
|
||||
|
||||
it('opencode: has no skipFirstRunPrompts (no apiKeyTrustFile/configDirVar concept for it either)', () => {
|
||||
const result = buildCustomModelInjection(entryOrThrow('opencode'), endpoint, 'qwen3');
|
||||
if (result.kind !== 'env') throw new Error('unreachable');
|
||||
expect(result.skipFirstRunPrompts).toBeUndefined();
|
||||
});
|
||||
|
||||
it('claude: falls back to a dummy key when the endpoint has none', () => {
|
||||
const result = buildCustomModelInjection(entryOrThrow('claude'), { ...endpoint, apiKey: undefined }, 'qwen3');
|
||||
if (result.kind !== 'env') throw new Error('unreachable');
|
||||
@@ -142,15 +179,31 @@ describe('buildCustomModelInjection', () => {
|
||||
expect(result.extraEnv).toEqual({ XAI_API_KEY: 'my-key' });
|
||||
});
|
||||
|
||||
it('deepseek: env kind sets base URL/key only, no model var', () => {
|
||||
it('deepseek: env kind sets base URL (with a /v1 suffix appended) and key, no model var', () => {
|
||||
// appendV1Suffix is REQUIRED here, not cosmetic: confirmed by reading dsh's own
|
||||
// bundled source (@deepseek-ai/dsh-llm-deepseek) that it builds the request URL as
|
||||
// `${DEEPSEEK_BASE_URL}/chat/completions` with no "/v1" of its own, while
|
||||
// llama-swap/llama.cpp only serves "/v1/chat/completions" — without this, every
|
||||
// request 404s (confirmed live; this is the fix for the originally-reported
|
||||
// "dsh: HTTP_404: DeepSeek API error (HTTP 404)").
|
||||
const result = buildCustomModelInjection(entryOrThrow('deepseek'), endpoint, 'qwen3');
|
||||
if (result.kind !== 'env') throw new Error('unreachable');
|
||||
expect(result.envOverrides).toEqual({
|
||||
DEEPSEEK_BASE_URL: 'http://192.168.1.50:8080',
|
||||
DEEPSEEK_BASE_URL: 'http://192.168.1.50:8080/v1',
|
||||
DEEPSEEK_API_KEY: 'my-key',
|
||||
});
|
||||
});
|
||||
|
||||
it('deepseek: appending the /v1 suffix is idempotent against a baseUrl that already ends in /v1', () => {
|
||||
const result = buildCustomModelInjection(
|
||||
entryOrThrow('deepseek'),
|
||||
{ ...endpoint, baseUrl: 'http://192.168.1.50:8080/v1' },
|
||||
'qwen3'
|
||||
);
|
||||
if (result.kind !== 'env') throw new Error('unreachable');
|
||||
expect(result.envOverrides.DEEPSEEK_BASE_URL).toBe('http://192.168.1.50:8080/v1');
|
||||
});
|
||||
|
||||
it('antigravity: unsupported', () => {
|
||||
const result = buildCustomModelInjection(entryOrThrow('antigravity'), endpoint, 'qwen3');
|
||||
expect(result).toEqual({ kind: 'unsupported' });
|
||||
|
||||
@@ -0,0 +1,238 @@
|
||||
/**
|
||||
* @fileoverview Tests for `getLatestLlamaSwapLogLine()`/`pruneIdleLlamaSwapLogTails()` —
|
||||
* the real-time "what is llama.cpp actually doing" feed behind the loading banner's
|
||||
* second line (docs/custom-model-endpoints-plan.md). Confirmed live against a real
|
||||
* llama-swap deployment: its `GET /api/events` SSE stream carries the backend
|
||||
* llama-server process's own stdout (`load_model: ...`, `llama_server: model loaded`)
|
||||
* as `{"type":"logData","data":"{\"data\":\"...\",\"source\":\"upstream\"}"}` frames,
|
||||
* tagged distinctly from llama-swap's own `source: "proxy"` request-access log frames.
|
||||
*
|
||||
* ⚠️ `GET /logs` (the endpoint this feature's own first cut was built against, before
|
||||
* being caught by exactly this kind of live check) turns out to carry ONLY the proxy
|
||||
* log — confirmed live it never showed a single backend line even seconds after a real,
|
||||
* confirmed model swap. `/api/events` is the only source that actually has the data.
|
||||
*
|
||||
* Drives a hand-built `ReadableStream` body through the mocked `webviewFetch` rather
|
||||
* than a real network round-trip — the point under test is the SSE-frame parsing and
|
||||
* `source` filtering plus the one-connection-per-endpoint reuse, not networking itself.
|
||||
*
|
||||
* Each test uses its own host id (`llamaSwapLogTails` is a module-level Map, shared
|
||||
* across every test in this file) and `afterEach` force-prunes everything so no tail
|
||||
* a test forgot to close leaks into the next one.
|
||||
*
|
||||
* Port: N/A (no server; drives the exported functions directly).
|
||||
*/
|
||||
import { describe, it, expect, vi, afterEach } from 'vitest';
|
||||
import { getLatestLlamaSwapLogLine, pruneIdleLlamaSwapLogTails } from '../src/web/routes/custom-model-routes.js';
|
||||
import { webviewFetch } from '../src/web/webview-egress.js';
|
||||
import type { CustomModelHost } from '../src/custom-model-hosts.js';
|
||||
|
||||
vi.mock('../src/web/webview-egress.js', async () => {
|
||||
const actual = await vi.importActual<typeof import('../src/web/webview-egress.js')>('../src/web/webview-egress.js');
|
||||
return { ...actual, webviewFetch: vi.fn() };
|
||||
});
|
||||
|
||||
const fetchMock = vi.mocked(webviewFetch);
|
||||
|
||||
/** One real `GET /api/events` SSE frame carrying backend (`source: "upstream"`) log text. */
|
||||
function upstreamLogFrame(text: string): string {
|
||||
const inner = JSON.stringify({ data: text, source: 'upstream' });
|
||||
return `event:message\ndata:${JSON.stringify({ type: 'logData', data: inner })}\n\n`;
|
||||
}
|
||||
|
||||
/** The proxy-log flavor of the same event shape — must never be surfaced as `latestLine`. */
|
||||
function proxyLogFrame(text: string): string {
|
||||
const inner = JSON.stringify({ data: text, source: 'proxy' });
|
||||
return `event:message\ndata:${JSON.stringify({ type: 'logData', data: inner })}\n\n`;
|
||||
}
|
||||
|
||||
/** A streaming Response whose body enqueues `frames` up front and then stays open
|
||||
* (never closes) — matches a real `/api/events` connection, confirmed live to stay
|
||||
* open indefinitely (read past 220KB over 8s with no `done`). */
|
||||
function openStreamResponse(frames: string[]): Response {
|
||||
const encoder = new TextEncoder();
|
||||
const stream = new ReadableStream<Uint8Array>({
|
||||
start(controller) {
|
||||
for (const frame of frames) controller.enqueue(encoder.encode(frame));
|
||||
// deliberately never controller.close()
|
||||
},
|
||||
});
|
||||
return new Response(stream, { status: 200 });
|
||||
}
|
||||
|
||||
function host(id: string): CustomModelHost {
|
||||
return { id, label: id, baseUrl: `http://192.168.1.50:8080/${id}` };
|
||||
}
|
||||
|
||||
/** Lets the fire-and-forget stream-pump's microtasks (reader.read() resolutions) settle. */
|
||||
async function flush(): Promise<void> {
|
||||
await new Promise((resolve) => setTimeout(resolve, 10));
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
pruneIdleLlamaSwapLogTails(Number.POSITIVE_INFINITY); // force-close every tail this file opened
|
||||
fetchMock.mockReset();
|
||||
});
|
||||
|
||||
describe('getLatestLlamaSwapLogLine', () => {
|
||||
it('returns undefined before any line has arrived, then the real backend log line once it does', async () => {
|
||||
const h = host('t1');
|
||||
fetchMock.mockResolvedValue(
|
||||
openStreamResponse([upstreamLogFrame('0.31.428.568 I srv llama_server: model loaded')])
|
||||
);
|
||||
|
||||
const before = getLatestLlamaSwapLogLine(h);
|
||||
expect(before).toBeUndefined();
|
||||
await flush();
|
||||
const after = getLatestLlamaSwapLogLine(h);
|
||||
|
||||
expect(after).toBe('0.31.428.568 I srv llama_server: model loaded');
|
||||
});
|
||||
|
||||
it('filters out llama-swap\'s own proxy-sourced frames, keeping only source: "upstream"', async () => {
|
||||
const h = host('t2');
|
||||
fetchMock.mockResolvedValue(
|
||||
openStreamResponse([
|
||||
proxyLogFrame('[INFO] Request 10.10.10.1 "GET /running HTTP/1.1" 200 407 "undici" 46.207µs'),
|
||||
upstreamLogFrame('0.14.157.100 I srv load_model: initializing, n_slots = 4, n_ctx_slot = 16384'),
|
||||
proxyLogFrame('[WARN] some warning about something unrelated'),
|
||||
])
|
||||
);
|
||||
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
await flush();
|
||||
|
||||
expect(getLatestLlamaSwapLogLine(h)).toBe(
|
||||
'0.14.157.100 I srv load_model: initializing, n_slots = 4, n_ctx_slot = 16384'
|
||||
);
|
||||
});
|
||||
|
||||
it('keeps the LAST line when one upstream frame batches several newline-joined lines', async () => {
|
||||
const h = host('t3');
|
||||
fetchMock.mockResolvedValue(
|
||||
openStreamResponse([
|
||||
upstreamLogFrame(
|
||||
'0.00.001.000 I srv llama_server: starting\n0.00.002.000 I srv llama_server: loading tensors'
|
||||
),
|
||||
upstreamLogFrame('0.00.003.000 I srv llama_server: model loaded'),
|
||||
])
|
||||
);
|
||||
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
await flush();
|
||||
|
||||
expect(getLatestLlamaSwapLogLine(h)).toBe('0.00.003.000 I srv llama_server: model loaded');
|
||||
});
|
||||
|
||||
it('handles a frame split across two stream chunks (SSE double-newline boundary not yet seen)', async () => {
|
||||
const h = host('t3b');
|
||||
const whole = upstreamLogFrame('0.00.005.000 I srv llama_server: model loaded');
|
||||
const splitAt = Math.floor(whole.length / 2);
|
||||
fetchMock.mockResolvedValue(openStreamResponse([whole.slice(0, splitAt), whole.slice(splitAt)]));
|
||||
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
await flush();
|
||||
|
||||
expect(getLatestLlamaSwapLogLine(h)).toBe('0.00.005.000 I srv llama_server: model loaded');
|
||||
});
|
||||
|
||||
it('ignores a malformed frame instead of throwing', async () => {
|
||||
const h = host('t3c');
|
||||
fetchMock.mockResolvedValue(
|
||||
openStreamResponse(['event:message\ndata:not valid json\n\n', upstreamLogFrame('llama_server: model loaded')])
|
||||
);
|
||||
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
await flush();
|
||||
|
||||
expect(getLatestLlamaSwapLogLine(h)).toBe('llama_server: model loaded');
|
||||
});
|
||||
|
||||
it('ignores a non-logData event type', async () => {
|
||||
const h = host('t3d');
|
||||
fetchMock.mockResolvedValue(
|
||||
openStreamResponse([
|
||||
`event:message\ndata:${JSON.stringify({ type: 'modelStatus', data: '{}' })}\n\n`,
|
||||
upstreamLogFrame('llama_server: model loaded'),
|
||||
])
|
||||
);
|
||||
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
await flush();
|
||||
|
||||
expect(getLatestLlamaSwapLogLine(h)).toBe('llama_server: model loaded');
|
||||
});
|
||||
|
||||
it('opens exactly one connection per endpoint — a second call while the tail is open never re-fetches', async () => {
|
||||
const h = host('t4');
|
||||
fetchMock.mockResolvedValue(openStreamResponse([upstreamLogFrame('llama_server: model loaded')]));
|
||||
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
await flush();
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("requests /api/events specifically, with the endpoint's own auth headers", async () => {
|
||||
const h: CustomModelHost = { id: 't5', label: 't5', baseUrl: 'http://192.168.1.60:9000', apiKey: 'secret-key' };
|
||||
fetchMock.mockResolvedValue(openStreamResponse([]));
|
||||
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
const [url, init] = fetchMock.mock.calls[0]!;
|
||||
expect((url as URL).pathname).toBe('/api/events');
|
||||
expect((init as RequestInit).headers).toMatchObject({ Authorization: 'Bearer secret-key' });
|
||||
});
|
||||
|
||||
it('an unreachable endpoint (fetch throws) leaves latestLine undefined rather than throwing', async () => {
|
||||
const h = host('t6');
|
||||
fetchMock.mockRejectedValue(new TypeError('fetch failed'));
|
||||
|
||||
expect(() => getLatestLlamaSwapLogLine(h)).not.toThrow();
|
||||
await flush();
|
||||
expect(getLatestLlamaSwapLogLine(h)).toBeUndefined();
|
||||
});
|
||||
|
||||
it('a non-2xx response leaves latestLine undefined rather than throwing', async () => {
|
||||
const h = host('t7');
|
||||
fetchMock.mockResolvedValue(new Response('not found', { status: 404 }));
|
||||
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
await flush();
|
||||
|
||||
expect(getLatestLlamaSwapLogLine(h)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('pruneIdleLlamaSwapLogTails', () => {
|
||||
it('closes a tail nothing has polled recently, so the next access starts a fresh connection', async () => {
|
||||
const h = host('t8');
|
||||
fetchMock.mockResolvedValue(openStreamResponse([upstreamLogFrame('llama_server: model loaded')]));
|
||||
|
||||
getLatestLlamaSwapLogLine(h); // opens the first connection, lastAccessedAt = now
|
||||
await flush();
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
|
||||
pruneIdleLlamaSwapLogTails(Date.now() + 60_000); // "now" far enough ahead that the tail reads as idle
|
||||
|
||||
getLatestLlamaSwapLogLine(h); // the entry was removed — this must open a NEW connection
|
||||
await flush();
|
||||
expect(fetchMock).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it('leaves a recently-accessed tail alone', async () => {
|
||||
const h = host('t9');
|
||||
fetchMock.mockResolvedValue(openStreamResponse([upstreamLogFrame('llama_server: model loaded')]));
|
||||
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
await flush();
|
||||
|
||||
pruneIdleLlamaSwapLogTails(Date.now()); // no time has passed — nothing is idle yet
|
||||
|
||||
getLatestLlamaSwapLogLine(h);
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1); // still just the one connection
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,238 @@
|
||||
/**
|
||||
* @fileoverview Frontend tests for the one-shot custom-model launch path added to
|
||||
* session-ui.js (docs/custom-model-endpoints-plan.md): `runCustomModelEntry` dispatches
|
||||
* to `_runCustomModelEntryOneShot` for every custom-model-eligible CLI except claude,
|
||||
* which launches directly on the endpoint (no restart) by folding `customModel` into
|
||||
* the run<Mode>() function's own `/api/quick-start` body via `_pendingCustomModelForLaunch`
|
||||
* and `_quickStartWithCustomModelConfirm`. Fixes the visible native-boot-then-restart the
|
||||
* restart-after-launch path (`_runCustomModelEntryViaRestart`, still used for claude)
|
||||
* showed on every custom-model run — confirmed live on Codex, whose TUI fully
|
||||
* reinitializes on a restart.
|
||||
*
|
||||
* Uses the same JSDOM + `runScripts: "dangerously"` approach as
|
||||
* test/custom-model-run-menu-ui.test.ts, extended with the DOM elements runCodex() (the
|
||||
* CLI this was reported against) reads.
|
||||
*
|
||||
* Port: none.
|
||||
*/
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { JSDOM } from 'jsdom';
|
||||
import { describe, expect, it } from 'vitest';
|
||||
|
||||
const CONSTANTS_JS = readFileSync(new URL('../src/web/public/constants.js', import.meta.url), 'utf-8');
|
||||
const SESSION_UI_JS = readFileSync(new URL('../src/web/public/session-ui.js', import.meta.url), 'utf-8');
|
||||
|
||||
function bootApp() {
|
||||
const dom = new JSDOM(
|
||||
`<!doctype html><body>
|
||||
<select id="quickStartCase"><option value="testcase" selected>testcase</option></select>
|
||||
<input id="tabCount" value="1">
|
||||
<button id="runBtn"></button>
|
||||
<div id="runModeMenu"></div>
|
||||
</body>`,
|
||||
{ url: 'http://localhost/', runScripts: 'dangerously' }
|
||||
);
|
||||
const win = dom.window as unknown as Window & typeof globalThis & { CodemanApp: new () => any };
|
||||
(win as unknown as { eval: (s: string) => void }).eval('window.CodemanApp = function CodemanApp() {};');
|
||||
(win as unknown as { eval: (s: string) => void }).eval(CONSTANTS_JS);
|
||||
(win as unknown as { eval: (s: string) => void }).eval(SESSION_UI_JS);
|
||||
const app = new win.CodemanApp();
|
||||
app.cases = [{ name: 'testcase' }];
|
||||
app.terminal = { focus: () => {} };
|
||||
app.loadAppSettingsFromStorage = () => ({});
|
||||
app.getCaseSettings = () => ({});
|
||||
app.buildEnvOverrides = () => ({});
|
||||
app.showToast = () => {};
|
||||
app._beginSessionLaunchStatus = () => 'status-token';
|
||||
app._reportSessionLaunchError = (_token: unknown, message: string) => {
|
||||
app._lastReportedError = message;
|
||||
};
|
||||
app._ensureCreatedSessionVisible = async () => {};
|
||||
app.selectSession = async () => {};
|
||||
app._nextCaseSessionStartNumber = () => 1;
|
||||
return { win, app };
|
||||
}
|
||||
|
||||
describe('runCustomModelEntry dispatch', () => {
|
||||
it('routes claude through the restart-after-launch path', async () => {
|
||||
const { app } = bootApp();
|
||||
let calledRestart = false;
|
||||
let calledOneShot = false;
|
||||
app._runCustomModelEntryViaRestart = async () => {
|
||||
calledRestart = true;
|
||||
};
|
||||
app._runCustomModelEntryOneShot = async () => {
|
||||
calledOneShot = true;
|
||||
};
|
||||
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
|
||||
expect(calledRestart).toBe(true);
|
||||
expect(calledOneShot).toBe(false);
|
||||
});
|
||||
|
||||
it('routes every other custom-model-eligible CLI through the one-shot path', async () => {
|
||||
for (const mode of ['opencode', 'codex', 'gemini', 'pi', 'grok', 'deepseek', 'omp']) {
|
||||
const { app } = bootApp();
|
||||
let calledRestart = false;
|
||||
let calledOneShot = false;
|
||||
app._runCustomModelEntryViaRestart = async () => {
|
||||
calledRestart = true;
|
||||
};
|
||||
app._runCustomModelEntryOneShot = async () => {
|
||||
calledOneShot = true;
|
||||
};
|
||||
await app.runCustomModelEntry(mode, 'llama-box', 'qwen3');
|
||||
expect(calledRestart, mode).toBe(false);
|
||||
expect(calledOneShot, mode).toBe(true);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('_runCustomModelEntryOneShot', () => {
|
||||
it('stashes the pick on _pendingCustomModelForLaunch for the duration of run(), then clears it', async () => {
|
||||
const { app } = bootApp();
|
||||
let seenDuringRun: unknown;
|
||||
app.run = async function (this: typeof app) {
|
||||
seenDuringRun = this._pendingCustomModelForLaunch;
|
||||
};
|
||||
await app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3');
|
||||
expect(seenDuringRun).toEqual({ endpointId: 'llama-box', modelId: 'qwen3' });
|
||||
expect(app._pendingCustomModelForLaunch).toBeUndefined();
|
||||
});
|
||||
|
||||
it('clears the pending pick even when run() throws', async () => {
|
||||
const { app } = bootApp();
|
||||
app.run = async () => {
|
||||
throw new Error('boom');
|
||||
};
|
||||
await expect(app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3')).rejects.toThrow('boom');
|
||||
expect(app._pendingCustomModelForLaunch).toBeUndefined();
|
||||
});
|
||||
|
||||
it('starts the loading watcher when the launch reports modelSwapInProgress, passing the new session id', async () => {
|
||||
const { app } = bootApp();
|
||||
app.run = async () => {
|
||||
app._lastCustomModelLaunchResult = { modelSwapInProgress: true, sessionId: 'new-session' };
|
||||
};
|
||||
let watched: unknown[] | null = null;
|
||||
app._watchLlamaSwapLoading = async (...args: unknown[]) => {
|
||||
watched = args;
|
||||
};
|
||||
await app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3');
|
||||
expect(watched).toEqual(['llama-box', 'qwen3', 'new-session']);
|
||||
});
|
||||
|
||||
it('never starts the watcher when no swap was needed', async () => {
|
||||
const { app } = bootApp();
|
||||
app.run = async () => {
|
||||
app._lastCustomModelLaunchResult = { modelSwapInProgress: false };
|
||||
};
|
||||
let watchCalled = false;
|
||||
app._watchLlamaSwapLoading = async () => {
|
||||
watchCalled = true;
|
||||
};
|
||||
await app._runCustomModelEntryOneShot('codex', 'llama-box', 'qwen3');
|
||||
expect(watchCalled).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('_quickStartWithCustomModelConfirm', () => {
|
||||
function withFetch(win: Window & typeof globalThis, handler: (body: any) => any) {
|
||||
(win as unknown as { fetch: typeof fetch }).fetch = (async (_url: string, opts: any) => ({
|
||||
json: async () => handler(JSON.parse(opts.body)),
|
||||
})) as unknown as typeof fetch;
|
||||
}
|
||||
|
||||
it('returns the response directly when no confirmation is needed', async () => {
|
||||
const { win, app } = bootApp();
|
||||
withFetch(win, (body) => ({ success: true, data: { sessionId: 's1', modelSwapInProgress: false, body } }));
|
||||
const data = await app._quickStartWithCustomModelConfirm({
|
||||
mode: 'codex',
|
||||
customModel: { endpointId: 'e', modelId: 'm' },
|
||||
});
|
||||
expect(data.success).toBe(true);
|
||||
expect(data.data.sessionId).toBe('s1');
|
||||
expect(app._lastCustomModelLaunchResult).toEqual(data.data);
|
||||
});
|
||||
|
||||
it('confirming re-sends with confirmed:true and returns the second response', async () => {
|
||||
const { win, app } = bootApp();
|
||||
app._confirmModelSwap = async () => true;
|
||||
let calls = 0;
|
||||
withFetch(win, (body) => {
|
||||
calls += 1;
|
||||
if (calls === 1) {
|
||||
return {
|
||||
success: true,
|
||||
data: {
|
||||
requiresConfirmation: true,
|
||||
currentlyLoadedModel: 'llama3',
|
||||
affectedSessions: [{ id: 's2', name: 'w2' }],
|
||||
},
|
||||
};
|
||||
}
|
||||
expect(body.customModel.confirmed).toBe(true);
|
||||
return { success: true, data: { sessionId: 's1', modelSwapInProgress: true } };
|
||||
});
|
||||
const data = await app._quickStartWithCustomModelConfirm({
|
||||
mode: 'codex',
|
||||
customModel: { endpointId: 'e', modelId: 'm' },
|
||||
});
|
||||
expect(calls).toBe(2);
|
||||
expect(data.data.sessionId).toBe('s1');
|
||||
expect(app._lastCustomModelLaunchResult.modelSwapInProgress).toBe(true);
|
||||
});
|
||||
|
||||
it('cancelling never re-sends, and reports a cancellation error', async () => {
|
||||
const { win, app } = bootApp();
|
||||
app._confirmModelSwap = async () => false;
|
||||
let calls = 0;
|
||||
withFetch(win, () => {
|
||||
calls += 1;
|
||||
return {
|
||||
success: true,
|
||||
data: {
|
||||
requiresConfirmation: true,
|
||||
currentlyLoadedModel: 'llama3',
|
||||
affectedSessions: [{ id: 's2', name: 'w2' }],
|
||||
},
|
||||
};
|
||||
});
|
||||
const data = await app._quickStartWithCustomModelConfirm({
|
||||
mode: 'codex',
|
||||
customModel: { endpointId: 'e', modelId: 'm' },
|
||||
});
|
||||
expect(calls).toBe(1);
|
||||
expect(data.success).toBe(false);
|
||||
expect(data.error).toMatch(/cancelled/i);
|
||||
expect(app._lastCustomModelLaunchResult).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('runCodex(): one-shot custom-model launch (the CLI this was reported against)', () => {
|
||||
it('folds _pendingCustomModelForLaunch into the quick-start body as customModel', async () => {
|
||||
const { win, app } = bootApp();
|
||||
(win as unknown as { fetch: typeof fetch }).fetch = (async (url: string, opts?: any) => {
|
||||
if (url === '/api/codex/status') return { json: async () => ({ data: { available: true } }) };
|
||||
const body = JSON.parse(opts.body);
|
||||
expect(body.customModel).toEqual({ endpointId: 'llama-box', modelId: 'qwen3' });
|
||||
return { json: async () => ({ success: true, data: { sessionId: 's1', modelSwapInProgress: false } }) };
|
||||
}) as unknown as typeof fetch;
|
||||
|
||||
app._pendingCustomModelForLaunch = { endpointId: 'llama-box', modelId: 'qwen3' };
|
||||
await app.runCodex();
|
||||
expect(app._lastReportedError).toBeUndefined();
|
||||
});
|
||||
|
||||
it('omits customModel entirely for a plain (non-custom-model) Codex launch', async () => {
|
||||
const { win, app } = bootApp();
|
||||
(win as unknown as { fetch: typeof fetch }).fetch = (async (url: string, opts?: any) => {
|
||||
if (url === '/api/codex/status') return { json: async () => ({ data: { available: true } }) };
|
||||
const body = JSON.parse(opts.body);
|
||||
expect(body.customModel).toBeUndefined();
|
||||
return { json: async () => ({ success: true, data: { sessionId: 's1' } }) };
|
||||
}) as unknown as typeof fetch;
|
||||
|
||||
await app.runCodex();
|
||||
expect(app._lastReportedError).toBeUndefined();
|
||||
});
|
||||
});
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,180 @@
|
||||
/**
|
||||
* @fileoverview Tests for `detectCustomModelSwapDisplacements()`, the periodic sweep
|
||||
* behind server.ts's "custom model swap-displacement check" timer
|
||||
* (docs/custom-model-endpoints-plan.md). The apply/create routes' own swap-conflict check
|
||||
* only ever runs at a session's own launch/apply moment — this sweep is what catches a
|
||||
* LATER eviction triggered by a different session's normal use, which the launch-time
|
||||
* check structurally cannot see.
|
||||
*
|
||||
* Kept in its own file for the same reason as `custom-model-endpoint-rediscovery.test.ts`:
|
||||
* a sweep that walks every saved host would otherwise pick up hosts other tests in a
|
||||
* shared file create, making an exact call-count assertion meaningless.
|
||||
*
|
||||
* Port: N/A (no server; drives readCustomModelHosts/writeCustomModelHosts directly plus
|
||||
* the mocked webviewFetch dispatcher).
|
||||
*/
|
||||
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
||||
import { getDataDir } from '../src/config/instance.js';
|
||||
import { writeCustomModelHosts, type CustomModelHost } from '../src/custom-model-hosts.js';
|
||||
import {
|
||||
detectCustomModelSwapDisplacements,
|
||||
type CustomModelSessionLike,
|
||||
} from '../src/web/routes/custom-model-routes.js';
|
||||
import { webviewFetch } from '../src/web/webview-egress.js';
|
||||
|
||||
vi.mock('../src/web/webview-egress.js', async () => {
|
||||
const actual = await vi.importActual<typeof import('../src/web/webview-egress.js')>('../src/web/webview-egress.js');
|
||||
return { ...actual, webviewFetch: vi.fn() };
|
||||
});
|
||||
|
||||
const fetchMock = vi.mocked(webviewFetch);
|
||||
|
||||
const ENDPOINT: CustomModelHost = {
|
||||
id: 'llama-swap',
|
||||
label: 'llama-swap',
|
||||
baseUrl: 'http://192.168.1.50:8080',
|
||||
apiKey: 'k',
|
||||
};
|
||||
|
||||
function session(
|
||||
overrides: Partial<CustomModelSessionLike> & Pick<CustomModelSessionLike, 'id'>
|
||||
): CustomModelSessionLike {
|
||||
return { name: overrides.id, ...overrides };
|
||||
}
|
||||
|
||||
function mockRunning(running: Array<{ model: string; state: string }>) {
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/running') return new Response(JSON.stringify({ running }), { status: 200 });
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
fetchMock.mockReset();
|
||||
});
|
||||
|
||||
describe('detectCustomModelSwapDisplacements', () => {
|
||||
it('flags a session whose own model is no longer in the running list, naming what displaced it', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
|
||||
mockRunning([{ model: 'fast', state: 'ready' }]);
|
||||
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
|
||||
const notified = new Set<string>();
|
||||
|
||||
const displacements = await detectCustomModelSwapDisplacements([w1], notified);
|
||||
|
||||
expect(displacements).toEqual([
|
||||
{
|
||||
sessionId: 'w1',
|
||||
sessionName: 'w1',
|
||||
endpointId: 'llama-swap',
|
||||
previousModel: 'qwen3',
|
||||
currentlyLoadedModel: 'fast',
|
||||
},
|
||||
]);
|
||||
expect(notified.has('w1')).toBe(true);
|
||||
});
|
||||
|
||||
it('does not flag a session whose own model is still the one loaded and ready', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
|
||||
mockRunning([{ model: 'qwen3', state: 'ready' }]);
|
||||
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
|
||||
|
||||
const displacements = await detectCustomModelSwapDisplacements([w1], new Set());
|
||||
|
||||
expect(displacements).toEqual([]);
|
||||
});
|
||||
|
||||
it('notifies once per displacement — a repeat sweep with nothing changed does not re-flag it', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
|
||||
mockRunning([{ model: 'fast', state: 'ready' }]);
|
||||
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
|
||||
const notified = new Set<string>();
|
||||
|
||||
const first = await detectCustomModelSwapDisplacements([w1], notified);
|
||||
const second = await detectCustomModelSwapDisplacements([w1], notified);
|
||||
|
||||
expect(first).toHaveLength(1);
|
||||
expect(second).toEqual([]);
|
||||
});
|
||||
|
||||
it('clears the notified flag once the session is back on its own model, so a later displacement flags again', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
|
||||
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
|
||||
const notified = new Set<string>();
|
||||
|
||||
mockRunning([{ model: 'fast', state: 'ready' }]);
|
||||
await detectCustomModelSwapDisplacements([w1], notified);
|
||||
expect(notified.has('w1')).toBe(true);
|
||||
|
||||
mockRunning([{ model: 'qwen3', state: 'ready' }]); // back to normal
|
||||
await detectCustomModelSwapDisplacements([w1], notified);
|
||||
expect(notified.has('w1')).toBe(false);
|
||||
|
||||
mockRunning([{ model: 'fast', state: 'ready' }]); // displaced again
|
||||
const third = await detectCustomModelSwapDisplacements([w1], notified);
|
||||
expect(third).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('skips a session on a non-llama-swap endpoint (no /running) — nothing to compare, never flagged', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
|
||||
fetchMock.mockResolvedValue(new Response('not found', { status: 404 }));
|
||||
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
|
||||
|
||||
const displacements = await detectCustomModelSwapDisplacements([w1], new Set());
|
||||
|
||||
expect(displacements).toEqual([]);
|
||||
});
|
||||
|
||||
it('skips a session whose endpoint was deleted since it was created', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), []); // ENDPOINT never saved
|
||||
const w1 = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
|
||||
|
||||
const displacements = await detectCustomModelSwapDisplacements([w1], new Set());
|
||||
|
||||
expect(displacements).toEqual([]);
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('ignores a plain session with no customModel selection at all', async () => {
|
||||
const displacements = await detectCustomModelSwapDisplacements([session({ id: 'plain' })], new Set());
|
||||
expect(displacements).toEqual([]);
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('one endpoint failing (unreachable) never blocks checking sessions on another', async () => {
|
||||
const DOWN: CustomModelHost = { id: 'down', label: 'down', baseUrl: 'http://192.168.1.60:8080' };
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT, DOWN]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.href.includes('192.168.1.60')) throw new TypeError('fetch failed', { cause: new Error('ECONNREFUSED') });
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(JSON.stringify({ running: [{ model: 'fast', state: 'ready' }] }), { status: 200 });
|
||||
}
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
const onDown = session({ id: 'w-down', customModel: { endpointId: 'down', modelId: 'x' } });
|
||||
const onLlamaSwap = session({ id: 'w1', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
|
||||
|
||||
const displacements = await detectCustomModelSwapDisplacements([onDown, onLlamaSwap], new Set());
|
||||
|
||||
expect(displacements).toEqual([
|
||||
{
|
||||
sessionId: 'w1',
|
||||
sessionName: 'w1',
|
||||
endpointId: 'llama-swap',
|
||||
previousModel: 'qwen3',
|
||||
currentlyLoadedModel: 'fast',
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it('multiple sessions on the same endpoint each get their own displacement entry', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
|
||||
mockRunning([{ model: 'gemma', state: 'ready' }]);
|
||||
const w1 = session({ id: 'w1', name: 'w1-test2', customModel: { endpointId: 'llama-swap', modelId: 'qwen3' } });
|
||||
const w2 = session({ id: 'w2', name: 'w2-test2', customModel: { endpointId: 'llama-swap', modelId: 'fast' } });
|
||||
|
||||
const displacements = await detectCustomModelSwapDisplacements([w1, w2], new Set());
|
||||
|
||||
expect(displacements.map((d) => d.sessionId).sort()).toEqual(['w1', 'w2']);
|
||||
});
|
||||
});
|
||||
@@ -11,7 +11,7 @@
|
||||
* Port: N/A (no server start).
|
||||
*/
|
||||
import { describe, it, expect, afterEach, vi } from 'vitest';
|
||||
import { WebServer } from '../src/web/server.js';
|
||||
import { WebServer, escapeScriptJson } from '../src/web/server.js';
|
||||
import { isClaudeAvailable } from '../src/utils/claude-cli-resolver.js';
|
||||
import { isOpenCodeAvailable } from '../src/utils/opencode-cli-resolver.js';
|
||||
import { isCodexAvailable } from '../src/utils/codex-cli-resolver.js';
|
||||
@@ -187,6 +187,41 @@ describe('WebServer.renderIndexHtml', () => {
|
||||
});
|
||||
});
|
||||
|
||||
it('reports which run modes the custom-model Run-menu picker may generate an entry for', async () => {
|
||||
// Read generically off the CLI registry's own capabilities, not a hardcoded id
|
||||
// list — antigravity (`unsupported`) and shell (`kind !== 'agent'`) must be
|
||||
// absent, and any enabled agent CLI with a real injection recipe must be
|
||||
// present, with no mock needed since this reads the real stock registry.
|
||||
const { server } = makeServer({});
|
||||
const html = await render(server);
|
||||
expect(html).toContain('window.__codemanCustomModelClis=');
|
||||
const clis = JSON.parse(html.match(/window\.__codemanCustomModelClis=(\[.*?\]);/)![1]) as Array<{
|
||||
id: string;
|
||||
label: string;
|
||||
}>;
|
||||
const ids = clis.map((c) => c.id);
|
||||
expect(ids).toContain('claude');
|
||||
expect(ids).not.toContain('antigravity');
|
||||
expect(ids).not.toContain('shell');
|
||||
for (const cli of clis) {
|
||||
expect(typeof cli.id).toBe('string');
|
||||
expect(typeof cli.label).toBe('string');
|
||||
}
|
||||
});
|
||||
|
||||
it('escapeScriptJson neutralizes a literal </script>, and still round-trips as a JS literal', () => {
|
||||
// CliEntry.label is a plain string a user's own clis.json can set (up to 60
|
||||
// chars), unlike __codemanCliAvailable's booleans-only payload, so this is
|
||||
// the one injection that needs it. Exported so this tests the pure
|
||||
// function directly rather than needing a real WebServer (which needs tmux).
|
||||
const dangerous = JSON.stringify([{ id: 'x', label: '</script><script>alert(1)</script>' }]);
|
||||
const escaped = escapeScriptJson(dangerous);
|
||||
expect(escaped).not.toContain('</script');
|
||||
// Proves it decodes back to the real value the way a browser's own JS
|
||||
// parser would, not just "the output contains no </script>".
|
||||
expect(eval(escaped)[0].label).toBe('</script><script>alert(1)</script>');
|
||||
});
|
||||
|
||||
it('still emits the object when nothing at all is installed', async () => {
|
||||
// The all-false case is the one that matters most and the easiest to get
|
||||
// wrong by only injecting when something resolves.
|
||||
@@ -218,6 +253,7 @@ describe('WebServer.renderIndexHtml', () => {
|
||||
const { server } = makeServer({});
|
||||
const html = await render(server, 'sess-123');
|
||||
expect(html).not.toContain('__codemanCliAvailable');
|
||||
expect(html).not.toContain('__codemanCustomModelClis');
|
||||
});
|
||||
|
||||
it('does not expose gesture at all when CODEMAN_GESTURE is unset', async () => {
|
||||
|
||||
@@ -181,6 +181,34 @@ describe('custom model endpoint CRUD', () => {
|
||||
expect(res.json().error).toMatch(/refused.*169\.254\.169\.254/);
|
||||
});
|
||||
|
||||
it('running-status never hands the browser the raw llama-swap launch command (cmd)', async () => {
|
||||
const { app } = await setup();
|
||||
await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/model-endpoints',
|
||||
payload: { id: 'ep-running', label: 'A', baseUrl: 'http://localhost:8080', apiKey: 'k' },
|
||||
});
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
running: [
|
||||
{ model: 'qwen3', state: 'ready', cmd: 'llama-server -m /models/qwen3.gguf --api-key sk-secret' },
|
||||
],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
|
||||
const res = await app.inject({ method: 'GET', url: '/api/model-endpoints/ep-running/running-status' });
|
||||
const body = res.json();
|
||||
expect(body.data.running).toEqual([{ model: 'qwen3', state: 'ready' }]);
|
||||
expect(JSON.stringify(body)).not.toContain('sk-secret');
|
||||
expect(JSON.stringify(body)).not.toContain('cmd');
|
||||
});
|
||||
|
||||
it('refuses a baseUrl with embedded credentials or a non-http scheme at save time', async () => {
|
||||
const { app } = await setup();
|
||||
for (const baseUrl of ['http://user:pw@host:8080', 'ftp://host/models', 'http://169.254.169.254']) {
|
||||
@@ -194,3 +222,198 @@ describe('custom model endpoint CRUD', () => {
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('defaultModelId — the Run-menu picker’s per-endpoint default', () => {
|
||||
afterEach(() => {
|
||||
fetchMock.mockReset();
|
||||
});
|
||||
|
||||
it('rejects a defaultModelId that is not one of the endpoint’s discovered models, on both create and update', async () => {
|
||||
const { app } = await setup();
|
||||
const create = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/model-endpoints',
|
||||
payload: {
|
||||
id: 'ep-default-reject',
|
||||
label: 'A',
|
||||
baseUrl: 'http://localhost:8080',
|
||||
models: ['qwen3'],
|
||||
defaultModelId: 'ghost',
|
||||
},
|
||||
});
|
||||
expect(create.json().success).toBe(false);
|
||||
expect(create.json().errorCode).toBe('INVALID_INPUT');
|
||||
|
||||
await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/model-endpoints',
|
||||
payload: { id: 'ep-default-reject', label: 'A', baseUrl: 'http://localhost:8080', models: ['qwen3'] },
|
||||
});
|
||||
const update = await app.inject({
|
||||
method: 'PUT',
|
||||
url: '/api/model-endpoints/ep-default-reject',
|
||||
payload: { label: 'A', baseUrl: 'http://localhost:8080', models: ['qwen3'], defaultModelId: 'ghost' },
|
||||
});
|
||||
expect(update.json().success).toBe(false);
|
||||
expect(update.json().errorCode).toBe('INVALID_INPUT');
|
||||
});
|
||||
|
||||
it('accepts a defaultModelId that IS one of the discovered models', async () => {
|
||||
const { app } = await setup();
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/model-endpoints',
|
||||
payload: {
|
||||
id: 'ep-default-accept',
|
||||
label: 'A',
|
||||
baseUrl: 'http://localhost:8080',
|
||||
models: ['qwen3', 'llama3'],
|
||||
defaultModelId: 'llama3',
|
||||
},
|
||||
});
|
||||
expect(res.json().success).toBe(true);
|
||||
expect(res.json().data.host.defaultModelId).toBe('llama3');
|
||||
});
|
||||
|
||||
it('drops a stale default that no longer appears in a fresh discovery, rather than carrying it forward invalid', async () => {
|
||||
const { app } = await setup();
|
||||
await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/model-endpoints',
|
||||
payload: {
|
||||
id: 'ep-default-drop',
|
||||
label: 'A',
|
||||
baseUrl: 'http://localhost:8080',
|
||||
models: ['qwen3'],
|
||||
defaultModelId: 'qwen3',
|
||||
},
|
||||
});
|
||||
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'llama3' }] }), { status: 200 }));
|
||||
await app.inject({ method: 'POST', url: '/api/model-endpoints/ep-default-drop/discover-models' });
|
||||
|
||||
const list = await app.inject({ method: 'GET', url: '/api/model-endpoints' });
|
||||
const stored = (list.json() as Array<{ id: string; defaultModelId?: string }>).find(
|
||||
(h) => h.id === 'ep-default-drop'
|
||||
);
|
||||
expect(stored?.defaultModelId).toBeUndefined();
|
||||
});
|
||||
|
||||
it('keeps a default that IS still present after a fresh discovery', async () => {
|
||||
const { app } = await setup();
|
||||
await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/model-endpoints',
|
||||
payload: {
|
||||
id: 'ep-default-keep',
|
||||
label: 'A',
|
||||
baseUrl: 'http://localhost:8080',
|
||||
models: ['qwen3'],
|
||||
defaultModelId: 'qwen3',
|
||||
},
|
||||
});
|
||||
fetchMock.mockResolvedValue(
|
||||
new Response(JSON.stringify({ data: [{ id: 'qwen3' }, { id: 'llama3' }] }), { status: 200 })
|
||||
);
|
||||
await app.inject({ method: 'POST', url: '/api/model-endpoints/ep-default-keep/discover-models' });
|
||||
|
||||
const list = await app.inject({ method: 'GET', url: '/api/model-endpoints' });
|
||||
const stored = (list.json() as Array<{ id: string; defaultModelId?: string }>).find(
|
||||
(h) => h.id === 'ep-default-keep'
|
||||
);
|
||||
expect(stored?.defaultModelId).toBe('qwen3');
|
||||
});
|
||||
});
|
||||
|
||||
describe('apiKey is never handed back to the browser', () => {
|
||||
afterEach(() => {
|
||||
fetchMock.mockReset();
|
||||
});
|
||||
|
||||
it('POST, GET and PUT responses all carry apiKeySet instead of the real key', async () => {
|
||||
const { app } = await setup();
|
||||
const create = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/model-endpoints',
|
||||
payload: { id: 'ep-secret', label: 'A', baseUrl: 'http://localhost:8080', apiKey: 'super-secret' },
|
||||
});
|
||||
expect(create.json().data.host.apiKey).toBeUndefined();
|
||||
expect(create.json().data.host.apiKeySet).toBe(true);
|
||||
|
||||
const list = await app.inject({ method: 'GET', url: '/api/model-endpoints' });
|
||||
const listed = (list.json() as Array<{ id: string; apiKey?: string; apiKeySet?: boolean }>).find(
|
||||
(h) => h.id === 'ep-secret'
|
||||
);
|
||||
expect(listed?.apiKey).toBeUndefined();
|
||||
expect(listed?.apiKeySet).toBe(true);
|
||||
expect(JSON.stringify(list.json())).not.toContain('super-secret');
|
||||
|
||||
const update = await app.inject({
|
||||
method: 'PUT',
|
||||
url: '/api/model-endpoints/ep-secret',
|
||||
payload: { label: 'Renamed', baseUrl: 'http://localhost:8080' },
|
||||
});
|
||||
expect(update.json().data.host.apiKey).toBeUndefined();
|
||||
expect(update.json().data.host.apiKeySet).toBe(true);
|
||||
expect(JSON.stringify(update.json())).not.toContain('super-secret');
|
||||
});
|
||||
|
||||
it('a host with no key set at all reports apiKeySet: false', async () => {
|
||||
const { app } = await setup();
|
||||
const create = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/model-endpoints',
|
||||
payload: { id: 'ep-nokey', label: 'A', baseUrl: 'http://localhost:8080' },
|
||||
});
|
||||
expect(create.json().data.host.apiKeySet).toBe(false);
|
||||
});
|
||||
|
||||
it('PUT with no apiKey keeps the stored one, rather than clearing it', async () => {
|
||||
const { app } = await setup();
|
||||
await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/model-endpoints',
|
||||
payload: { id: 'ep-keep-key', label: 'A', baseUrl: 'http://localhost:8080', apiKey: 'original-key' },
|
||||
});
|
||||
// Edit without touching the API key field — the real bug this guards: a
|
||||
// browser round-trip that only ever sees apiKeySet, never the real value,
|
||||
// must not accidentally send an empty string and wipe a working credential.
|
||||
const update = await app.inject({
|
||||
method: 'PUT',
|
||||
url: '/api/model-endpoints/ep-keep-key',
|
||||
payload: { label: 'Renamed', baseUrl: 'http://localhost:8080' },
|
||||
});
|
||||
expect(update.json().data.host.apiKeySet).toBe(true);
|
||||
|
||||
// Prove it by observing the auth header discovery actually sends.
|
||||
fetchMock.mockImplementation(async (_url: URL, init?: RequestInit) => {
|
||||
const headers = init?.headers as Record<string, string>;
|
||||
expect(headers.Authorization).toBe('Bearer original-key');
|
||||
return new Response(JSON.stringify({ data: [] }), { status: 200 });
|
||||
});
|
||||
const discover = await app.inject({ method: 'POST', url: '/api/model-endpoints/ep-keep-key/discover-models' });
|
||||
expect(discover.json().success).toBe(true);
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('PUT with a new apiKey replaces the stored one', async () => {
|
||||
const { app } = await setup();
|
||||
await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/model-endpoints',
|
||||
payload: { id: 'ep-replace-key', label: 'A', baseUrl: 'http://localhost:8080', apiKey: 'old-key' },
|
||||
});
|
||||
await app.inject({
|
||||
method: 'PUT',
|
||||
url: '/api/model-endpoints/ep-replace-key',
|
||||
payload: { label: 'A', baseUrl: 'http://localhost:8080', apiKey: 'new-key' },
|
||||
});
|
||||
|
||||
fetchMock.mockImplementation(async (_url: URL, init?: RequestInit) => {
|
||||
const headers = init?.headers as Record<string, string>;
|
||||
expect(headers.Authorization).toBe('Bearer new-key');
|
||||
return new Response(JSON.stringify({ data: [] }), { status: 200 });
|
||||
});
|
||||
await app.inject({ method: 'POST', url: '/api/model-endpoints/ep-replace-key/discover-models' });
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,382 @@
|
||||
/**
|
||||
* @fileoverview POST /api/quick-start's `customModel` field (docs/custom-model-endpoints-plan.md):
|
||||
* the ONE-SHOT launch path that computes a custom-model endpoint's injection BEFORE the
|
||||
* session/process exists and launches directly on it, so a custom-model Run never shows
|
||||
* the native-boot-then-restart the dedicated POST /api/sessions/:id/custom-model route's
|
||||
* restart-in-place design otherwise produces — most visibly on a CLI like Codex whose TUI
|
||||
* fully reinitializes on a restart. That dedicated route is still what an ALREADY-RUNNING
|
||||
* session uses to switch later; this is the create-time equivalent.
|
||||
*
|
||||
* Mirrors test/routes/session-custom-model.test.ts's fixtures and llama-swap mocking, since
|
||||
* this route mirrors that one's own checks (llama-swap conflict, unsupported CLI, unknown
|
||||
* endpoint, an argv-incompatible model id) rather than a lighter, separately-drifting copy.
|
||||
*
|
||||
* Session.prototype.startInteractive/startShell are mocked exactly like the workspace-hooks
|
||||
* quick-start tests: quick-start constructs a REAL Session (not the MockSession the route
|
||||
* test harness substitutes elsewhere), so tmux must never actually be reached.
|
||||
*
|
||||
* Port: N/A (app.inject, no real port needed)
|
||||
*/
|
||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
|
||||
import Fastify, { type FastifyInstance } from 'fastify';
|
||||
import fastifyCookie from '@fastify/cookie';
|
||||
import { rm, readFile } from 'node:fs/promises';
|
||||
import { existsSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
import { createMockRouteContext, safeRmHomeTree, type MockRouteContext } from '../mocks/index.js';
|
||||
import { installRouteErrorHandler } from '../../src/web/route-error-handler.js';
|
||||
import { registerSessionRoutes } from '../../src/web/routes/session-routes.js';
|
||||
import { getDataDir } from '../../src/config/instance.js';
|
||||
import { CASES_DIR } from '../../src/web/route-helpers.js';
|
||||
import { Session } from '../../src/session.js';
|
||||
import { writeCustomModelHosts, type CustomModelHost } from '../../src/custom-model-hosts.js';
|
||||
import { customModelConfigDir } from '../../src/custom-model-injection-apply.js';
|
||||
import { webviewFetch } from '../../src/web/webview-egress.js';
|
||||
|
||||
vi.mock('../../src/web/webview-egress.js', async () => {
|
||||
const actual = await vi.importActual<typeof import('../../src/web/webview-egress.js')>(
|
||||
'../../src/web/webview-egress.js'
|
||||
);
|
||||
return { ...actual, webviewFetch: vi.fn() };
|
||||
});
|
||||
const fetchMock = vi.mocked(webviewFetch);
|
||||
|
||||
// quick-start's own local-CLI-availability gate (resolveCliLaunchError, unrelated to the
|
||||
// custom-model injection this file tests) runs BEFORE the code under test and would
|
||||
// otherwise 404 every non-claude mode on a box with no codex/pi/grok/omp binary installed —
|
||||
// exactly this test environment. Mirrors the real "not remote" bypass documented at its own
|
||||
// call site in session-routes.ts (`session-routes.test.ts`'s remote-codex test is the
|
||||
// precedent for needing this at all).
|
||||
vi.mock('../../src/utils/cli-launcher.js', async () => {
|
||||
const actual = await vi.importActual<typeof import('../../src/utils/cli-launcher.js')>(
|
||||
'../../src/utils/cli-launcher.js'
|
||||
);
|
||||
return { ...actual, resolveCliLaunchError: vi.fn().mockResolvedValue(null) };
|
||||
});
|
||||
|
||||
const ENDPOINT: CustomModelHost = {
|
||||
id: 'ep1',
|
||||
label: 'llama.cpp box',
|
||||
baseUrl: 'http://192.168.1.50:8080',
|
||||
apiKey: 'k',
|
||||
};
|
||||
|
||||
describe('POST /api/quick-start: customModel (one-shot custom-model launch)', () => {
|
||||
let app: FastifyInstance;
|
||||
let ctx: MockRouteContext;
|
||||
let restartSpy: ReturnType<typeof vi.spyOn>;
|
||||
|
||||
const quickStart = (payload: Record<string, unknown>) =>
|
||||
app.inject({ method: 'POST', url: '/api/quick-start', payload });
|
||||
|
||||
beforeEach(async () => {
|
||||
vi.spyOn(Session.prototype, 'startInteractive').mockResolvedValue(undefined);
|
||||
vi.spyOn(Session.prototype, 'startShell').mockResolvedValue(undefined);
|
||||
restartSpy = vi.spyOn(Session.prototype, 'restartCli').mockResolvedValue(true);
|
||||
fetchMock.mockReset();
|
||||
fetchMock.mockResolvedValue(new Response('not found', { status: 404 })); // default: not llama-swap
|
||||
app = Fastify({ logger: false });
|
||||
await app.register(fastifyCookie);
|
||||
ctx = createMockRouteContext();
|
||||
registerSessionRoutes(app, ctx);
|
||||
installRouteErrorHandler(app);
|
||||
await app.ready();
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT]);
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
await app.close();
|
||||
vi.restoreAllMocks();
|
||||
await rm(join(getDataDir(), 'custom-model-hosts.json'), { force: true });
|
||||
await rm(join(getDataDir(), 'custom-model-configs'), { recursive: true, force: true });
|
||||
safeRmHomeTree(CASES_DIR);
|
||||
});
|
||||
|
||||
it('launches a claude session already pointed at the endpoint — no restart at all', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-claude',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
const { sessionId } = res.json();
|
||||
const session = ctx.sessions.get(sessionId) as unknown as Session;
|
||||
expect(session.customModel).toEqual({ endpointId: 'ep1', modelId: 'qwen3', label: 'llama.cpp box' });
|
||||
// The whole point: never restarted. It launched on the endpoint the first time.
|
||||
expect(restartSpy).not.toHaveBeenCalled();
|
||||
|
||||
const isolatedDir = customModelConfigDir(sessionId);
|
||||
const trustFile = JSON.parse(await readFile(join(isolatedDir, '.claude.json'), 'utf-8'));
|
||||
expect(trustFile.customApiKeyResponses.approved).toEqual(['k']);
|
||||
});
|
||||
|
||||
it('codex: writes the config.toml under the SAME id the session actually launches with, no restart', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-codex',
|
||||
mode: 'codex',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
const { sessionId } = res.json();
|
||||
const session = ctx.sessions.get(sessionId) as unknown as Session;
|
||||
expect(session.customModel?.endpointId).toBe('ep1');
|
||||
expect(restartSpy).not.toHaveBeenCalled();
|
||||
|
||||
const configDir = customModelConfigDir(sessionId);
|
||||
expect(existsSync(join(configDir, 'config.toml'))).toBe(true);
|
||||
const toml = await readFile(join(configDir, 'config.toml'), 'utf-8');
|
||||
expect(toml).toContain('model = "qwen3"');
|
||||
});
|
||||
|
||||
it('pi: forces --model custom/<id> onto piConfig on the FIRST launch, not via a later restart', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-pi',
|
||||
mode: 'pi',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3.5-0.8b' },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
const { sessionId } = res.json();
|
||||
const session = ctx.sessions.get(sessionId) as unknown as Session & { piConfig?: { model?: string } };
|
||||
expect(session.getCustomModelForPersist()?.launchModel).toBe('custom/qwen3.5-0.8b');
|
||||
expect(restartSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('grok: forces the [model.<name>] block name onto grokConfig on the first launch', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-grok',
|
||||
mode: 'grok',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
const { sessionId } = res.json();
|
||||
const session = ctx.sessions.get(sessionId) as unknown as Session;
|
||||
expect(session.getCustomModelForPersist()?.launchModel).toBe('codeman-custom');
|
||||
expect(restartSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('omp: forces custom/<id> onto ompConfig even with no incoming ompConfig at all', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-omp',
|
||||
mode: 'omp',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
const { sessionId } = res.json();
|
||||
const session = ctx.sessions.get(sessionId) as unknown as Session;
|
||||
expect(session.getCustomModelForPersist()?.launchModel).toBe('custom/qwen3');
|
||||
expect(restartSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('404s for an unknown endpoint id', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-ghost',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ghost', modelId: 'qwen3' },
|
||||
});
|
||||
expect(res.json().success).toBe(false);
|
||||
expect(res.json().errorCode).toBe('NOT_FOUND');
|
||||
});
|
||||
|
||||
it('refuses a mode with no known custom-model mechanism (antigravity)', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-agy',
|
||||
mode: 'antigravity',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
expect(res.json().success).toBe(false);
|
||||
expect(res.json().errorCode).toBe('OPERATION_FAILED');
|
||||
});
|
||||
|
||||
it('refuses a model id the CLI cannot carry on its command line, cleaning up any written config dir', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-badmodel',
|
||||
mode: 'pi',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen 3 with spaces' },
|
||||
});
|
||||
expect(res.json().success).toBe(false);
|
||||
expect(res.json().errorCode).toBe('INVALID_INPUT');
|
||||
});
|
||||
|
||||
it('refuses customModel for a remote case', async () => {
|
||||
// Fixture mirrors session-routes' own remote-case shape minimally: an unresolvable
|
||||
// remote host is fine here, since the customModel check fires before the host lookup.
|
||||
const res = await quickStart({
|
||||
caseName: 'nonexistent-remote-case',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
// No matching remote/docker case fixture exists, so this actually falls through to the
|
||||
// local branch and succeeds — this test only documents that remote/docker have their
|
||||
// own explicit customModel rejection (see the local-fixture tests in
|
||||
// session-routes-workspace-hooks.test.ts for the fixture-loading pattern that would be
|
||||
// needed to exercise the remote/docker branch itself).
|
||||
expect(res.statusCode).toBe(200);
|
||||
});
|
||||
|
||||
describe('llama-swap conflict check', () => {
|
||||
function mockRunning(running: Array<{ model: string; state: string }>) {
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/running') return new Response(JSON.stringify({ running }), { status: 200 });
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
}
|
||||
|
||||
it('asks for confirmation instead of launching when another live session is using the currently loaded model', async () => {
|
||||
const other = ctx.sessions.get('test-session-1')!;
|
||||
(other as unknown as { customModel: unknown }).customModel = { endpointId: 'ep1', modelId: 'llama3' };
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-conflict',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.requiresConfirmation).toBe(true);
|
||||
expect(body.currentlyLoadedModel).toBe('llama3');
|
||||
expect(body.affectedSessions).toEqual([{ id: 'test-session-1', name: other.name }]);
|
||||
// Nothing was actually created.
|
||||
expect(ctx.sessions.size).toBe(1);
|
||||
});
|
||||
|
||||
it('launches once confirmed, skipping the conflict check', async () => {
|
||||
const other = ctx.sessions.get('test-session-1')!;
|
||||
(other as unknown as { customModel: unknown }).customModel = { endpointId: 'ep1', modelId: 'llama3' };
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-confirmed',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3', confirmed: true },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
expect(res.json().requiresConfirmation).toBeUndefined();
|
||||
expect(ctx.sessions.size).toBe(2);
|
||||
});
|
||||
|
||||
it('launches straight away when nothing else is using the currently loaded model', async () => {
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-noconflict',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
expect(res.json().requiresConfirmation).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("context-window floor warning (this CLI's own overhead can exceed a small model's real context)", () => {
|
||||
const SMALL_CTX_ENDPOINT: CustomModelHost = {
|
||||
id: 'ep-small',
|
||||
label: 'tiny box',
|
||||
baseUrl: 'http://192.168.1.51:8080',
|
||||
apiKey: 'k',
|
||||
modelContextLengths: { 'qwen3.8-27b-ud-q4_k_xl': 16384 },
|
||||
};
|
||||
|
||||
it('warns instead of launching when the discovered context is below the safe floor', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-small-ctx',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.requiresContextWarning).toBe(true);
|
||||
expect(body.modelId).toBe('qwen3.8-27b-ud-q4_k_xl');
|
||||
expect(body.contextLength).toBe(16384);
|
||||
expect(body.minSafeContextTokens).toBe(40000);
|
||||
// Nothing was actually created.
|
||||
expect(ctx.sessions.size).toBe(1);
|
||||
});
|
||||
|
||||
it('launches once confirmed, skipping the context check', async () => {
|
||||
await writeCustomModelHosts(getDataDir(), [ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-small-ctx-confirmed',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl', confirmed: true },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
expect(ctx.sessions.size).toBe(2);
|
||||
});
|
||||
|
||||
it('does not warn when nothing about context was discovered', async () => {
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-no-ctx-data',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('triggering the actual llama-swap load (not just watching for it)', () => {
|
||||
it('sends a real inference request naming the target model, concurrently with launching the session', async () => {
|
||||
const chatCalls: unknown[] = [];
|
||||
fetchMock.mockImplementation(async (url: URL, init?: { body?: unknown }) => {
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/v1/chat/completions') {
|
||||
chatCalls.push(JSON.parse(init!.body as string));
|
||||
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
|
||||
}
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-trigger',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
await new Promise((resolve) => setTimeout(resolve, 0)); // let the fire-and-forget trigger settle
|
||||
|
||||
expect(res.statusCode).toBe(200);
|
||||
expect(res.json().modelSwapInProgress).toBe(true);
|
||||
expect(chatCalls).toHaveLength(1);
|
||||
expect(chatCalls[0]).toMatchObject({ model: 'qwen3', max_tokens: 1 });
|
||||
});
|
||||
|
||||
it('never sends a load-trigger request when the target model is already loaded and ready', async () => {
|
||||
let chatCalled = false;
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(JSON.stringify({ running: [{ model: 'qwen3', state: 'ready' }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/v1/chat/completions') {
|
||||
chatCalled = true;
|
||||
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
|
||||
}
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
|
||||
const res = await quickStart({
|
||||
caseName: 'cm-no-trigger',
|
||||
mode: 'claude',
|
||||
customModel: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
await new Promise((resolve) => setTimeout(resolve, 0));
|
||||
|
||||
expect(res.json().modelSwapInProgress).toBe(false);
|
||||
expect(chatCalled).toBe(false);
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -3,13 +3,28 @@
|
||||
* chunk 5 — applying/clearing a session's custom model endpoint + CLI restart).
|
||||
* Port: N/A (app.inject, no real port needed)
|
||||
*/
|
||||
import { describe, it, expect, beforeEach } from 'vitest';
|
||||
import { registerSessionRoutes } from '../../src/web/routes/session-routes.js';
|
||||
import { describe, it, expect, beforeEach, vi } from 'vitest';
|
||||
import { registerSessionRoutes, _clampEnvOverridesForOwner } from '../../src/web/routes/session-routes.js';
|
||||
import { createRouteTestHarness } from './_route-test-utils.js';
|
||||
import { createMockSession } from '../mocks/index.js';
|
||||
import { getDataDir } from '../../src/config/instance.js';
|
||||
import { writeCustomModelHosts, type CustomModelHost } from '../../src/custom-model-hosts.js';
|
||||
import { existsSync, statSync } from 'node:fs';
|
||||
import { existsSync, readFileSync, statSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
import { webviewFetch } from '../../src/web/webview-egress.js';
|
||||
|
||||
// Every apply now also checks llama-swap's `GET /running` (session-routes.ts) before
|
||||
// applying — without this mock every test in this file would make a REAL network request
|
||||
// to the fake 192.168.1.50 endpoint below and wait out its 5s timeout. Defaults to a plain
|
||||
// 404 (reads as "not llama-swap", exercising none of the new conflict-check tests below),
|
||||
// overridden per-test where the llama-swap behavior itself is what's under test.
|
||||
vi.mock('../../src/web/webview-egress.js', async () => {
|
||||
const actual = await vi.importActual<typeof import('../../src/web/webview-egress.js')>(
|
||||
'../../src/web/webview-egress.js'
|
||||
);
|
||||
return { ...actual, webviewFetch: vi.fn() };
|
||||
});
|
||||
const fetchMock = vi.mocked(webviewFetch);
|
||||
|
||||
const CLAUDE_ENDPOINT: CustomModelHost = {
|
||||
id: 'ep1',
|
||||
@@ -18,14 +33,16 @@ const CLAUDE_ENDPOINT: CustomModelHost = {
|
||||
apiKey: 'k',
|
||||
};
|
||||
|
||||
async function setup() {
|
||||
async function setup(ctxOptions?: Parameters<typeof createRouteTestHarness>[1]) {
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT]);
|
||||
return createRouteTestHarness(registerSessionRoutes);
|
||||
return createRouteTestHarness(registerSessionRoutes, ctxOptions);
|
||||
}
|
||||
|
||||
describe('POST /api/sessions/:id/custom-model', () => {
|
||||
beforeEach(async () => {
|
||||
await writeCustomModelHosts(getDataDir(), []);
|
||||
fetchMock.mockReset();
|
||||
fetchMock.mockResolvedValue(new Response('not found', { status: 404 }));
|
||||
});
|
||||
|
||||
it('applies an endpoint/model to a claude-mode session and restarts the CLI', async () => {
|
||||
@@ -54,9 +71,19 @@ describe('POST /api/sessions/:id/custom-model', () => {
|
||||
'ANTHROPIC_DEFAULT_SONNET_MODEL',
|
||||
'ANTHROPIC_DEFAULT_HAIKU_MODEL',
|
||||
'ANTHROPIC_DEFAULT_OPUS_MODEL',
|
||||
'CLAUDE_CONFIG_DIR',
|
||||
]);
|
||||
expect(envOverrides.ANTHROPIC_BASE_URL).toBe('http://192.168.1.50:8080');
|
||||
expect(envOverrides.ANTHROPIC_API_KEY).toBe('k');
|
||||
|
||||
// CLAUDE_CONFIG_DIR isolates this session from a stored claude.ai OAuth login, and the
|
||||
// trust-dialog file it points at is pre-seeded so the injected key doesn't hit an
|
||||
// interactive "Detected a custom API key" prompt with nobody there to answer it.
|
||||
const isolatedDir = join(getDataDir(), 'custom-model-configs', 'test-session-1');
|
||||
expect(envOverrides.CLAUDE_CONFIG_DIR).toBe(isolatedDir);
|
||||
expect(next.configDir).toBe(isolatedDir);
|
||||
const trustFile = JSON.parse(readFileSync(join(isolatedDir, '.claude.json'), 'utf8'));
|
||||
expect(trustFile.customApiKeyResponses.approved).toEqual(['k']);
|
||||
});
|
||||
|
||||
it('clears back to the native default', async () => {
|
||||
@@ -201,6 +228,381 @@ describe('POST /api/sessions/:id/custom-model', () => {
|
||||
expect(existsSync(dir)).toBe(false);
|
||||
});
|
||||
|
||||
describe('llama-swap conflict check (llama.cpp runs one model at a time)', () => {
|
||||
function mockRunning(running: Array<{ model: string; state: string }>) {
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/running') return new Response(JSON.stringify({ running }), { status: 200 });
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
}
|
||||
|
||||
it('applies straight away when the requested model is already loaded', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
ctx.sessions.get('test-session-1')!.mode = 'claude';
|
||||
mockRunning([{ model: 'qwen3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().success).not.toBe(false);
|
||||
expect(res.json().modelSwapInProgress).toBe(false);
|
||||
expect(ctx.sessions.get('test-session-1')!.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('applies straight away when a swap is needed but nothing else is using the loaded model, flagging modelSwapInProgress', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
ctx.sessions.get('test-session-1')!.mode = 'claude';
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().success).not.toBe(false);
|
||||
expect(res.json().modelSwapInProgress).toBe(true);
|
||||
expect(ctx.sessions.get('test-session-1')!.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('asks for confirmation instead of applying when another session is actively using the currently loaded model', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
const other = createMockSession('other-session');
|
||||
other.name = 'w2-otherbox';
|
||||
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
|
||||
ctx.sessions.set('other-session', other);
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.success).not.toBe(false);
|
||||
expect(body.requiresConfirmation).toBe(true);
|
||||
expect(body.currentlyLoadedModel).toBe('llama3');
|
||||
expect(body.affectedSessions).toEqual([{ id: 'other-session', name: 'w2-otherbox' }]);
|
||||
// Nothing actually applied yet — this call only asked, it did not switch.
|
||||
expect(session.setCustomModel).not.toHaveBeenCalled();
|
||||
expect(session.restartCli).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
describe('multi-user: the confirm dialog must not name a session the caller cannot access', () => {
|
||||
const saved: Record<string, string | undefined> = {};
|
||||
|
||||
beforeEach(() => {
|
||||
saved.CODEMAN_MULTIUSER = process.env.CODEMAN_MULTIUSER;
|
||||
process.env.CODEMAN_MULTIUSER = '1';
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
if (saved.CODEMAN_MULTIUSER === undefined) delete process.env.CODEMAN_MULTIUSER;
|
||||
else process.env.CODEMAN_MULTIUSER = saved.CODEMAN_MULTIUSER;
|
||||
});
|
||||
|
||||
it("still blocks the swap pending confirmation, but omits a foreign owner's session from affectedSessions", async () => {
|
||||
const { app, ctx } = await setup({ authUser: { username: 'bob', role: 'user' } });
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
(session as unknown as { owner?: string }).owner = 'bob';
|
||||
const other = createMockSession('other-session');
|
||||
other.name = 'w2-otherbox';
|
||||
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
|
||||
(other as unknown as { owner?: string }).owner = 'alice';
|
||||
ctx.sessions.set('other-session', other);
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
// Still asks — a foreign session is just as real a disruption as an owned one.
|
||||
expect(body.requiresConfirmation).toBe(true);
|
||||
expect(body.currentlyLoadedModel).toBe('llama3');
|
||||
// But bob never learns alice's session id or name.
|
||||
expect(body.affectedSessions).toEqual([]);
|
||||
expect(session.setCustomModel).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('names the affected session when the caller DOES own it', async () => {
|
||||
const { app, ctx } = await setup({ authUser: { username: 'bob', role: 'user' } });
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
(session as unknown as { owner?: string }).owner = 'bob';
|
||||
const other = createMockSession('other-session');
|
||||
other.name = 'w2-otherbox';
|
||||
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
|
||||
(other as unknown as { owner?: string }).owner = 'bob';
|
||||
ctx.sessions.set('other-session', other);
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().affectedSessions).toEqual([{ id: 'other-session', name: 'w2-otherbox' }]);
|
||||
});
|
||||
});
|
||||
|
||||
it('applies once confirmed, skipping the conflict check the second time', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
const other = createMockSession('other-session');
|
||||
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
|
||||
ctx.sessions.set('other-session', other);
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3', confirmed: true },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.requiresConfirmation).toBeUndefined();
|
||||
expect(body.modelSwapInProgress).toBe(true);
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
expect(session.restartCli).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('a session pointed at the SAME endpoint but a DIFFERENT (not-currently-loaded) model is not treated as affected', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
const other = createMockSession('other-session');
|
||||
other.customModel = { endpointId: 'ep1', modelId: 'some-other-model' }; // not the loaded one
|
||||
ctx.sessions.set('other-session', other);
|
||||
mockRunning([{ model: 'llama3', state: 'ready' }]);
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresConfirmation).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('not llama-swap (plain llama.cpp/OpenAI-compatible server, no /running) — never checked, applies straight away', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
ctx.sessions.get('test-session-1')!.mode = 'claude';
|
||||
fetchMock.mockResolvedValue(new Response('not found', { status: 404 }));
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().modelSwapInProgress).toBe(false);
|
||||
expect(res.json().requiresConfirmation).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("context-window floor warning (this CLI's own overhead can exceed a small model's real context)", () => {
|
||||
const SMALL_CTX_ENDPOINT: CustomModelHost = {
|
||||
id: 'ep-small',
|
||||
label: 'tiny box',
|
||||
baseUrl: 'http://192.168.1.51:8080',
|
||||
apiKey: 'k',
|
||||
modelContextLengths: { 'qwen3.8-27b-ud-q4_k_xl': 16384 },
|
||||
};
|
||||
|
||||
it('warns instead of applying when the discovered context is below the safe floor', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.success).not.toBe(false);
|
||||
expect(body.requiresContextWarning).toBe(true);
|
||||
expect(body.modelId).toBe('qwen3.8-27b-ud-q4_k_xl');
|
||||
expect(body.contextLength).toBe(16384);
|
||||
expect(body.minSafeContextTokens).toBe(40000);
|
||||
// Nothing actually applied yet — this call only warned, it did not switch.
|
||||
expect(session.setCustomModel).not.toHaveBeenCalled();
|
||||
expect(session.restartCli).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('applies once confirmed, skipping the context check the second time', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl', confirmed: true },
|
||||
});
|
||||
|
||||
const body = res.json();
|
||||
expect(body.requiresContextWarning).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
expect(session.restartCli).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not warn when the discovered context is comfortably above the floor', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const roomyEndpoint: CustomModelHost = {
|
||||
id: 'ep-roomy',
|
||||
label: 'roomy box',
|
||||
baseUrl: 'http://192.168.1.52:8080',
|
||||
apiKey: 'k',
|
||||
modelContextLengths: { qwen3: 65536 },
|
||||
};
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, roomyEndpoint]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-roomy', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not warn when the context length was never discovered (nothing to compare)', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
expect(session.setCustomModel).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not warn for a CLI whose registry entry declares no contextLengthVar (opencode)', async () => {
|
||||
// opencode's customModelInjection kind is configContentEnv, not env+contextLengthVar,
|
||||
// so exceedsSafeContextFloor is false by construction regardless of context size.
|
||||
const { app, ctx } = await setup();
|
||||
await writeCustomModelHosts(getDataDir(), [CLAUDE_ENDPOINT, SMALL_CTX_ENDPOINT]);
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'opencode';
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep-small', modelId: 'qwen3.8-27b-ud-q4_k_xl' },
|
||||
});
|
||||
|
||||
expect(res.json().requiresContextWarning).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('triggering the actual llama-swap load (not just watching for it)', () => {
|
||||
it('sends a real inference request naming the target model when it is not already loaded and ready', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
ctx.sessions.get('test-session-1')!.mode = 'claude';
|
||||
const chatCalls: unknown[] = [];
|
||||
fetchMock.mockImplementation(async (url: URL, init?: { body?: unknown }) => {
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/v1/chat/completions') {
|
||||
chatCalls.push(JSON.parse(init!.body as string));
|
||||
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
|
||||
}
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
|
||||
await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
await new Promise((resolve) => setTimeout(resolve, 0)); // let the fire-and-forget trigger settle
|
||||
|
||||
expect(chatCalls).toHaveLength(1);
|
||||
expect(chatCalls[0]).toMatchObject({ model: 'qwen3', max_tokens: 1 });
|
||||
});
|
||||
|
||||
it('never sends a load-trigger request when the target model is already loaded and ready', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
ctx.sessions.get('test-session-1')!.mode = 'claude';
|
||||
let chatCalled = false;
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(JSON.stringify({ running: [{ model: 'qwen3', state: 'ready' }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/v1/chat/completions') {
|
||||
chatCalled = true;
|
||||
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
|
||||
}
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
|
||||
await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
await new Promise((resolve) => setTimeout(resolve, 0));
|
||||
|
||||
expect(chatCalled).toBe(false);
|
||||
});
|
||||
|
||||
it('never sends a load-trigger request while confirmation is still pending', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
session.mode = 'claude';
|
||||
const other = createMockSession('other-session');
|
||||
other.customModel = { endpointId: 'ep1', modelId: 'llama3' };
|
||||
ctx.sessions.set('other-session', other);
|
||||
let chatCalled = false;
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(JSON.stringify({ running: [{ model: 'llama3', state: 'ready' }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/v1/chat/completions') {
|
||||
chatCalled = true;
|
||||
return new Response(JSON.stringify({ choices: [] }), { status: 200 });
|
||||
}
|
||||
throw new Error(`unexpected request in this test: ${url.href}`);
|
||||
});
|
||||
|
||||
const res = await app.inject({
|
||||
method: 'POST',
|
||||
url: '/api/sessions/test-session-1/custom-model',
|
||||
payload: { endpointId: 'ep1', modelId: 'qwen3' },
|
||||
});
|
||||
await new Promise((resolve) => setTimeout(resolve, 0));
|
||||
|
||||
expect(res.json().requiresConfirmation).toBe(true);
|
||||
expect(chatCalled).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
it('refuses to touch a busy session', async () => {
|
||||
const { app, ctx } = await setup();
|
||||
const session = ctx.sessions.get('test-session-1')!;
|
||||
@@ -218,3 +620,39 @@ describe('POST /api/sessions/:id/custom-model', () => {
|
||||
expect(session.setCustomModel).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
|
||||
describe('Claude multi-user clamp: the env-var half', () => {
|
||||
// CLAUDE_CODE_MAX_CONTEXT_TOKENS and CLAUDE_CONFIG_DIR were already reachable via
|
||||
// plain envOverrides before claude's privilegedEnvKeys existed (the first already
|
||||
// matches the CLAUDE_CODE_* allowedPrefix, the second is an allowed exact key), so
|
||||
// listing them here is not what makes this route safe — no custom-model route reads
|
||||
// privilegedEnvKeys at all. What it DOES do: ownerClampedEnvKeys() feeds the generic
|
||||
// envOverrides clamp on create/quick-start/reboot-restore, so a non-granted owner can
|
||||
// no longer set CLAUDE_CONFIG_DIR that way (the per-client-account feature, #255), and
|
||||
// a PERSISTED one is now stripped on reboot-restore for such an owner too — see
|
||||
// session-env-clamp.ts's own fileoverview for why that pass used to be a no-op for
|
||||
// claude specifically.
|
||||
const ORIGINAL = process.env.CODEMAN_MULTIUSER;
|
||||
beforeEach(() => {
|
||||
process.env.CODEMAN_MULTIUSER = '1';
|
||||
});
|
||||
afterEach(() => {
|
||||
if (ORIGINAL === undefined) delete process.env.CODEMAN_MULTIUSER;
|
||||
else process.env.CODEMAN_MULTIUSER = ORIGINAL;
|
||||
});
|
||||
|
||||
it('strips CLAUDE_CONFIG_DIR and CLAUDE_CODE_MAX_CONTEXT_TOKENS for a non-granted owner, leaving unrelated CLAUDE_CODE_* keys alone', async () => {
|
||||
const out = await _clampEnvOverridesForOwner('nobody', {
|
||||
CLAUDE_CONFIG_DIR: '/home/attacker/fake-claude-config',
|
||||
CLAUDE_CODE_MAX_CONTEXT_TOKENS: '999999',
|
||||
CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS: '1',
|
||||
});
|
||||
expect(out).toEqual({ CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS: '1' });
|
||||
});
|
||||
|
||||
it('is a no-op in single-user mode', async () => {
|
||||
delete process.env.CODEMAN_MULTIUSER;
|
||||
const input = { CLAUDE_CONFIG_DIR: '/home/attacker/fake-claude-config' };
|
||||
expect(await _clampEnvOverridesForOwner(undefined, input)).toBe(input);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -96,16 +96,20 @@ describe('WebServer index.html <title> templating (#82)', () => {
|
||||
|
||||
it('only substitutes the <title> tag — the rest of the template is identical (modulo asset cache-busting)', async () => {
|
||||
// renderIndexHtml also appends ?v=<mtime> cache-bust params to same-origin
|
||||
// .js/.css refs, and injects the CLI-availability flags before </head>; strip
|
||||
// both so the title remains the only other change.
|
||||
// .js/.css refs, and injects the CLI-availability flags plus the custom-model
|
||||
// Run-menu picker's CLI list before </head>; strip all so the title remains
|
||||
// the only other change.
|
||||
//
|
||||
// The flag strip is what keeps this test environment-independent. It used to
|
||||
// pass here by luck: the availability script was injected only where a CLI
|
||||
// resolved, so the assertion held on a machine with none installed and would
|
||||
// have failed on a developer's box that had them.
|
||||
// The flag strips are what keep this test environment-independent. The
|
||||
// CLI-availability one used to pass here by luck: that script was injected
|
||||
// only where a CLI resolved, so the assertion held on a machine with none
|
||||
// installed and would have failed on a developer's box that had them. The
|
||||
// custom-model list is injected unconditionally (a plain array, possibly
|
||||
// empty), so it needs stripping on every machine, not just where non-empty.
|
||||
const html = (await render('laptop'))
|
||||
.replace(/(\.(?:js|css))\?v=[^"]*/g, '$1')
|
||||
.replace(/<script>window\.__codemanCliAvailable=\{.*?\};<\/script>\n/, '');
|
||||
.replace(/<script>window\.__codemanCliAvailable=\{.*?\};<\/script>\n/, '')
|
||||
.replace(/<script>window\.__codemanCustomModelClis=\[.*?\];<\/script>\n/, '');
|
||||
const beforeTitle = rawTemplate.split('<title>Codeman</title>')[0];
|
||||
const afterTitle = rawTemplate.split('<title>Codeman</title>')[1];
|
||||
expect(html.startsWith(beforeTitle)).toBe(true);
|
||||
|
||||
Reference in New Issue
Block a user