mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-09-30 12:39:42 +02:00
Merge pull request #459 from opticon454/feature/run-menu-picker-currently-loaded-model
feat(custom-model): promote the currently-loaded/last-used model in the Run-menu picker
This commit is contained in:
@@ -183,6 +183,28 @@ launch, with the endpoint's `defaultModelId` marked but not auto-chosen —
|
|||||||
the point of asking is letting one launch deliberately differ from the
|
the point of asking is letting one launch deliberately differ from the
|
||||||
saved default, not just confirming it.
|
saved default, not just confirming it.
|
||||||
|
|
||||||
|
The modal promotes exactly one row to the top of the list rather than
|
||||||
|
always showing raw discovery order, so the zero-wait choice is the one
|
||||||
|
under your thumb:
|
||||||
|
|
||||||
|
- **"Currently loaded"** — a model from this host's own list that
|
||||||
|
llama-swap reports `ready` right now, queried via
|
||||||
|
`GET /api/model-endpoints/:id/running-status`. Bounded client-side to
|
||||||
|
~800ms (`Promise.race`), on top of the route's own 5s server-side
|
||||||
|
timeout, so an endpoint that is asleep or firewalled cannot leave the
|
||||||
|
modal invisible for the full 5s after the Run menu has already closed.
|
||||||
|
- **"Last used"** — shown only when nothing is currently loaded: the model
|
||||||
|
actually launched last for this exact (harness, endpoint) pair, read
|
||||||
|
from the per-device `codeman:customModelLastUsed:<mode>:<endpointId>`
|
||||||
|
localStorage key. Written by `runCustomModelEntry` /
|
||||||
|
`_quickStartWithCustomModelConfirm` only once the model is actually
|
||||||
|
applied, never on the mere click — declining the context-window warning
|
||||||
|
means this exact model cannot work with this CLI at all, so promoting it
|
||||||
|
next time would be actively wrong, not just premature.
|
||||||
|
|
||||||
|
Neither tag reorders anything past that one promoted row, and both defer
|
||||||
|
to "Default" when neither applies.
|
||||||
|
|
||||||
**How the launch itself applies the endpoint depends on the harness.** For
|
**How the launch itself applies the endpoint depends on the harness.** For
|
||||||
opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP (`runCustomModelEntry` →
|
opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP (`runCustomModelEntry` →
|
||||||
`_runCustomModelEntryOneShot`), the endpoint/model is folded into the SAME
|
`_runCustomModelEntryOneShot`), the endpoint/model is folded into the SAME
|
||||||
|
|||||||
@@ -312,6 +312,8 @@
|
|||||||
'运行菜单选择器会为此端点应用该模型。请先发现可用模型。',
|
'运行菜单选择器会为此端点应用该模型。请先发现可用模型。',
|
||||||
'Custom Endpoints': '自定义端点',
|
'Custom Endpoints': '自定义端点',
|
||||||
'Choose a model': '选择模型',
|
'Choose a model': '选择模型',
|
||||||
|
'Currently loaded': '当前已加载',
|
||||||
|
'Last used': '上次使用',
|
||||||
'That endpoint no longer exists': '该端点已不存在',
|
'That endpoint no longer exists': '该端点已不存在',
|
||||||
'No models discovered for this endpoint yet': '此端点尚未发现任何模型',
|
'No models discovered for this endpoint yet': '此端点尚未发现任何模型',
|
||||||
'Subagent Options': '子智能体选项',
|
'Subagent Options': '子智能体选项',
|
||||||
|
|||||||
@@ -720,14 +720,101 @@ Object.assign(CodemanApp.prototype, {
|
|||||||
if (models.length === 1) {
|
if (models.length === 1) {
|
||||||
return this.runCustomModelEntry(mode, endpointId, models[0]);
|
return this.runCustomModelEntry(mode, endpointId, models[0]);
|
||||||
}
|
}
|
||||||
this._openCustomModelPickModal(mode, host);
|
await this._openCustomModelPickModal(mode, host);
|
||||||
},
|
},
|
||||||
|
|
||||||
/** Renders the "which model" picker for a (harness, endpoint) pair with more than one discovered model. */
|
/** localStorage key for the last model launched on a given (harness, endpoint) pair — per-device by design, like every other `codeman:*` UI preference, never synced. */
|
||||||
_openCustomModelPickModal(mode, host) {
|
_customModelLastUsedKey(mode, endpointId) {
|
||||||
|
return `codeman:customModelLastUsed:${mode}:${endpointId}`;
|
||||||
|
},
|
||||||
|
|
||||||
|
/** Reads the last model chosen for this (harness, endpoint) pair, or null. Never throws — a blocked/full localStorage just means no promotion, not a broken picker. */
|
||||||
|
_getCustomModelLastUsed(mode, endpointId) {
|
||||||
|
try {
|
||||||
|
return localStorage.getItem(this._customModelLastUsedKey(mode, endpointId));
|
||||||
|
} catch {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
},
|
||||||
|
|
||||||
|
/** Remembers `modelId` as the last one launched for this (harness, endpoint) pair. */
|
||||||
|
_setCustomModelLastUsed(mode, endpointId, modelId) {
|
||||||
|
try {
|
||||||
|
localStorage.setItem(this._customModelLastUsedKey(mode, endpointId), modelId);
|
||||||
|
} catch {
|
||||||
|
// best-effort — losing the "last used" hint is cosmetic, never worth surfacing
|
||||||
|
}
|
||||||
|
},
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Best-effort lookup of the model llama-swap currently has loaded and ready on this
|
||||||
|
* endpoint, so the picker can offer it first instead of making the user remember what
|
||||||
|
* they picked last time it mattered. Mirrors `_watchLlamaSwapLoading`'s own
|
||||||
|
* `state === 'ready'` check. Returns null for a plain (non-llama-swap) server, an
|
||||||
|
* unreachable endpoint, or a loaded model this host no longer lists as discovered —
|
||||||
|
* never throws, since a failed probe should just skip promotion, not break the picker.
|
||||||
|
*
|
||||||
|
* ⚠️ Client-side bounded to ~800ms via Promise.race, on top of (never instead of) the
|
||||||
|
* route's own 5s server-side timeout (`RUNNING_TIMEOUT_MS`, custom-model-routes.ts) —
|
||||||
|
* a saved endpoint keeps its discovered models cached, so "the box behind this endpoint
|
||||||
|
* is asleep or firewalled" is a normal way to reach this path, not an exotic one, and
|
||||||
|
* the modal must not sit invisible (Run menu already closed, nothing else on screen)
|
||||||
|
* for the full 5s a slow/dead endpoint can take. The losing side of the race is left to
|
||||||
|
* resolve on its own — `.catch(() => null)` only stops an unhandled-rejection warning
|
||||||
|
* when it eventually fails, it never cancels the in-flight fetch.
|
||||||
|
*
|
||||||
|
* `timeoutMs` exists to let a test drive this in milliseconds instead of the real
|
||||||
|
* 800 — same reasoning as `_watchLlamaSwapLoading`'s own `pollIntervalMs`: this code
|
||||||
|
* runs inside a JSDOM window's own realm, whose `setTimeout` is not the one
|
||||||
|
* `vi.useFakeTimers()` patches, so a param is the only way to test the timeout without
|
||||||
|
* actually waiting on it. Real callers never pass it.
|
||||||
|
*/
|
||||||
|
async _getCustomModelCurrentlyLoaded(host, timeoutMs = 800) {
|
||||||
|
const probe = this._apiJson(`/api/model-endpoints/${encodeURIComponent(host.id)}/running-status`).catch(
|
||||||
|
() => null
|
||||||
|
);
|
||||||
|
const timeout = new Promise((resolve) => setTimeout(() => resolve(null), timeoutMs));
|
||||||
|
const status = await Promise.race([probe, timeout]);
|
||||||
|
if (!status?.isLlamaSwap) return null;
|
||||||
|
const ready = (status.running || []).find((r) => r.state === 'ready' && (host.models || []).includes(r.model));
|
||||||
|
return ready?.model || null;
|
||||||
|
},
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Renders the "which model" picker for a (harness, endpoint) pair with more than one
|
||||||
|
* discovered model. Async since it now awaits the currently-loaded-model probe below,
|
||||||
|
* so a SECOND call (a different custom-model entry clicked while the first one's probe
|
||||||
|
* is still in flight — the probe has its own 5s timeout) must not let the first call's
|
||||||
|
* later-arriving response clobber the second's already-rendered, already-correct modal.
|
||||||
|
* `_customModelPickGeneration` is the same guard-a-mutable-counter pattern
|
||||||
|
* `_watchLlamaSwapLoading` uses for the same reason: every DOM write below, including
|
||||||
|
* `_pendingCustomModelPick` itself, stays deferred until after the await, and a call
|
||||||
|
* that finds a newer generation already claimed bails out untouched rather than only
|
||||||
|
* skipping the model-list write and leaving title/hint/`_pendingCustomModelPick`
|
||||||
|
* inconsistent with what's on screen.
|
||||||
|
*/
|
||||||
|
async _openCustomModelPickModal(mode, host) {
|
||||||
const modal = document.getElementById('customModelPickModal');
|
const modal = document.getElementById('customModelPickModal');
|
||||||
const list = document.getElementById('customModelPickList');
|
const list = document.getElementById('customModelPickList');
|
||||||
if (!modal || !list) return;
|
if (!modal || !list) return;
|
||||||
|
const generation = (this._customModelPickGeneration = (this._customModelPickGeneration || 0) + 1);
|
||||||
|
const isCurrent = () => this._customModelPickGeneration === generation;
|
||||||
|
|
||||||
|
// Whichever model llama-swap actually has loaded right now beats a merely
|
||||||
|
// remembered choice — it's what a launch would attach to with zero wait, while
|
||||||
|
// "last used" might have been swapped out by another session since. Neither
|
||||||
|
// reorders past the top: exactly one model is promoted, everything else keeps
|
||||||
|
// its discovery order.
|
||||||
|
const currentlyLoaded = await this._getCustomModelCurrentlyLoaded(host);
|
||||||
|
if (!isCurrent()) return; // a newer pick opened (and possibly already rendered) while this probe was in flight
|
||||||
|
const lastUsed = currentlyLoaded ? null : this._getCustomModelLastUsed(mode, host.id);
|
||||||
|
const promoted = currentlyLoaded || lastUsed;
|
||||||
|
const models = [...(host.models || [])];
|
||||||
|
if (promoted && models.includes(promoted)) {
|
||||||
|
models.splice(models.indexOf(promoted), 1);
|
||||||
|
models.unshift(promoted);
|
||||||
|
}
|
||||||
|
|
||||||
this._pendingCustomModelPick = { mode, endpointId: host.id };
|
this._pendingCustomModelPick = { mode, endpointId: host.id };
|
||||||
const cliLabel = (window.__codemanCustomModelClis || []).find((c) => c.id === mode)?.label || mode;
|
const cliLabel = (window.__codemanCustomModelClis || []).find((c) => c.id === mode)?.label || mode;
|
||||||
// A static title (translatable by i18n.js's exact-string walker) plus a
|
// A static title (translatable by i18n.js's exact-string walker) plus a
|
||||||
@@ -736,13 +823,14 @@ Object.assign(CodemanApp.prototype, {
|
|||||||
document.getElementById('customModelPickTitle').textContent = 'Choose a model';
|
document.getElementById('customModelPickTitle').textContent = 'Choose a model';
|
||||||
document.getElementById('customModelPickHint').textContent =
|
document.getElementById('customModelPickHint').textContent =
|
||||||
`${cliLabel} → ${host.label} — ${(host.models || []).length} models discovered.`;
|
`${cliLabel} → ${host.label} — ${(host.models || []).length} models discovered.`;
|
||||||
list.innerHTML = (host.models || [])
|
list.innerHTML = models
|
||||||
.map((m) => {
|
.map((m) => {
|
||||||
const isDefault = m === host.defaultModelId;
|
const isDefault = m === host.defaultModelId;
|
||||||
|
const tag = m === currentlyLoaded ? 'Currently loaded' : m === lastUsed ? 'Last used' : isDefault ? 'Default' : null;
|
||||||
const arg = escapeHtml(JSON.stringify(m));
|
const arg = escapeHtml(JSON.stringify(m));
|
||||||
return `
|
return `
|
||||||
<button class="run-mode-option" onclick="app.chooseCustomModelAndRun(${arg})">
|
<button class="run-mode-option" onclick="app.chooseCustomModelAndRun(${arg})">
|
||||||
<span class="run-mode-dot ${escapeHtml(mode)}"></span>${escapeHtml(m)}${isDefault ? ' <span class="set-scope">Default</span>' : ''}
|
<span class="run-mode-dot ${escapeHtml(mode)}"></span>${escapeHtml(m)}${tag ? ` <span class="set-scope">${escapeHtml(tag)}</span>` : ''}
|
||||||
</button>`;
|
</button>`;
|
||||||
})
|
})
|
||||||
.join('');
|
.join('');
|
||||||
@@ -858,6 +946,12 @@ Object.assign(CodemanApp.prototype, {
|
|||||||
* (Codex, confirmed live) than on claude's own `--resume`-based restart.
|
* (Codex, confirmed live) than on claude's own `--resume`-based restart.
|
||||||
*/
|
*/
|
||||||
async runCustomModelEntry(mode, endpointId, modelId) {
|
async runCustomModelEntry(mode, endpointId, modelId) {
|
||||||
|
// "Last used" is recorded by each path itself, ONLY once the model is actually
|
||||||
|
// applied — never here, unconditionally, on the mere attempt. A context-window
|
||||||
|
// warning or a swap-conflict question can still say no after this call, and the
|
||||||
|
// context-warning case is the one that bites: declining it means this exact
|
||||||
|
// model cannot work with this CLI at all, so promoting it as "Last used" next
|
||||||
|
// time the picker opens would be actively wrong, not just premature.
|
||||||
if (mode === 'claude') {
|
if (mode === 'claude') {
|
||||||
return this._runCustomModelEntryViaRestart(mode, endpointId, modelId);
|
return this._runCustomModelEntryViaRestart(mode, endpointId, modelId);
|
||||||
}
|
}
|
||||||
@@ -950,7 +1044,15 @@ Object.assign(CodemanApp.prototype, {
|
|||||||
answered = { ...answered, confirmedSwap: true };
|
answered = { ...answered, confirmedSwap: true };
|
||||||
data = await post({ ...bodyObj, customModel: { ...bodyObj.customModel, ...answered } });
|
data = await post({ ...bodyObj, customModel: { ...bodyObj.customModel, ...answered } });
|
||||||
}
|
}
|
||||||
this._lastCustomModelLaunchResult = data?.success !== false ? data?.data : undefined;
|
const launched = data?.success !== false;
|
||||||
|
this._lastCustomModelLaunchResult = launched ? data?.data : undefined;
|
||||||
|
// Only once actually launched, and only for a call that carried a custom-model pick
|
||||||
|
// at all — `_launchQuickStartInstances` runs every quick-start body (custom-model or
|
||||||
|
// not) through this same function, so a plain launch must not fall through here with
|
||||||
|
// an undefined endpointId/modelId that quietly no-ops the (mode, endpointId) key.
|
||||||
|
if (launched && bodyObj.customModel) {
|
||||||
|
this._setCustomModelLastUsed(bodyObj.mode, bodyObj.customModel.endpointId, bodyObj.customModel.modelId);
|
||||||
|
}
|
||||||
return data;
|
return data;
|
||||||
},
|
},
|
||||||
|
|
||||||
@@ -1078,6 +1180,11 @@ Object.assign(CodemanApp.prototype, {
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The apply has actually succeeded and both questions (if asked) are answered
|
||||||
|
// yes — only now is this a real "last used" for the picker's next open, not
|
||||||
|
// before either confirmation had a chance to decline it.
|
||||||
|
this._setCustomModelLastUsed(mode, endpointId, modelId);
|
||||||
|
|
||||||
// The apply above already succeeded — the session IS pointed at the endpoint — but
|
// The apply above already succeeded — the session IS pointed at the endpoint — but
|
||||||
// llama-swap itself may still be unloading the old model and loading this one, which
|
// llama-swap itself may still be unloading the old model and loading this one, which
|
||||||
// can take well over a minute. Without this, a prompt sent during that window either
|
// can take well over a minute. Without this, a prompt sent during that window either
|
||||||
|
|||||||
@@ -142,7 +142,7 @@ describe('_quickStartWithCustomModelConfirm', () => {
|
|||||||
})) as unknown as typeof fetch;
|
})) as unknown as typeof fetch;
|
||||||
}
|
}
|
||||||
|
|
||||||
it('returns the response directly when no confirmation is needed', async () => {
|
it('returns the response directly when no confirmation is needed, and records "last used"', async () => {
|
||||||
const { win, app } = bootApp();
|
const { win, app } = bootApp();
|
||||||
withFetch(win, (body) => ({ success: true, data: { sessionId: 's1', modelSwapInProgress: false, body } }));
|
withFetch(win, (body) => ({ success: true, data: { sessionId: 's1', modelSwapInProgress: false, body } }));
|
||||||
const data = await app._quickStartWithCustomModelConfirm({
|
const data = await app._quickStartWithCustomModelConfirm({
|
||||||
@@ -152,9 +152,17 @@ describe('_quickStartWithCustomModelConfirm', () => {
|
|||||||
expect(data.success).toBe(true);
|
expect(data.success).toBe(true);
|
||||||
expect(data.data.sessionId).toBe('s1');
|
expect(data.data.sessionId).toBe('s1');
|
||||||
expect(app._lastCustomModelLaunchResult).toEqual(data.data);
|
expect(app._lastCustomModelLaunchResult).toEqual(data.data);
|
||||||
|
expect(win.localStorage.getItem('codeman:customModelLastUsed:codex:e')).toBe('m');
|
||||||
});
|
});
|
||||||
|
|
||||||
it('confirming re-sends with confirmedSwap and returns the second response', async () => {
|
it('a plain launch with no customModel at all never touches the "last used" key (undefined endpointId/modelId would otherwise silently no-op it)', async () => {
|
||||||
|
const { win, app } = bootApp();
|
||||||
|
withFetch(win, () => ({ success: true, data: { sessionId: 's1' } }));
|
||||||
|
await app._quickStartWithCustomModelConfirm({ mode: 'codex' });
|
||||||
|
expect(win.localStorage.getItem('codeman:customModelLastUsed:codex:undefined')).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
it('confirming re-sends with confirmedSwap, returns the second response, and only THEN records "last used"', async () => {
|
||||||
const { win, app } = bootApp();
|
const { win, app } = bootApp();
|
||||||
app._confirmModelSwap = async () => true;
|
app._confirmModelSwap = async () => true;
|
||||||
let calls = 0;
|
let calls = 0;
|
||||||
@@ -183,9 +191,10 @@ describe('_quickStartWithCustomModelConfirm', () => {
|
|||||||
expect(calls).toBe(2);
|
expect(calls).toBe(2);
|
||||||
expect(data.data.sessionId).toBe('s1');
|
expect(data.data.sessionId).toBe('s1');
|
||||||
expect(app._lastCustomModelLaunchResult.modelSwapInProgress).toBe(true);
|
expect(app._lastCustomModelLaunchResult.modelSwapInProgress).toBe(true);
|
||||||
|
expect(win.localStorage.getItem('codeman:customModelLastUsed:codex:e')).toBe('m');
|
||||||
});
|
});
|
||||||
|
|
||||||
it('cancelling never re-sends, and reports a cancellation error', async () => {
|
it('cancelling never re-sends, reports a cancellation error, and must NEVER record "last used" for a launch that never happened', async () => {
|
||||||
const { win, app } = bootApp();
|
const { win, app } = bootApp();
|
||||||
app._confirmModelSwap = async () => false;
|
app._confirmModelSwap = async () => false;
|
||||||
let calls = 0;
|
let calls = 0;
|
||||||
@@ -208,6 +217,7 @@ describe('_quickStartWithCustomModelConfirm', () => {
|
|||||||
expect(data.success).toBe(false);
|
expect(data.success).toBe(false);
|
||||||
expect(data.error).toMatch(/cancelled/i);
|
expect(data.error).toMatch(/cancelled/i);
|
||||||
expect(app._lastCustomModelLaunchResult).toBeUndefined();
|
expect(app._lastCustomModelLaunchResult).toBeUndefined();
|
||||||
|
expect(win.localStorage.getItem('codeman:customModelLastUsed:codex:e')).toBeNull();
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|||||||
@@ -272,6 +272,146 @@ describe('Custom Model Endpoint Profiles: the "which model" picker', () => {
|
|||||||
expect(win.document.getElementById('customModelPickList')!.textContent).toContain('Default');
|
expect(win.document.getElementById('customModelPickList')!.textContent).toContain('Default');
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it('promotes the model llama-swap currently has loaded and ready to the top of the list, tagged', async () => {
|
||||||
|
const { win, app } = bootApp({
|
||||||
|
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }],
|
||||||
|
});
|
||||||
|
const origApiJson = app._apiJson;
|
||||||
|
app._apiJson = async (path: string) => {
|
||||||
|
if (path === '/api/model-endpoints/llama-box/running-status') {
|
||||||
|
return { isLlamaSwap: true, running: [{ model: 'phi4', state: 'ready' }] };
|
||||||
|
}
|
||||||
|
return origApiJson(path);
|
||||||
|
};
|
||||||
|
|
||||||
|
await app.selectCustomModelEntry('claude', 'llama-box');
|
||||||
|
|
||||||
|
const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')];
|
||||||
|
expect(buttons.map((b) => b.textContent)).toHaveLength(3);
|
||||||
|
expect(buttons[0].textContent).toContain('phi4');
|
||||||
|
expect(buttons[0].textContent).toContain('Currently loaded');
|
||||||
|
// Nothing else got relabelled or reordered past the promoted row.
|
||||||
|
expect(buttons[1].textContent).toContain('qwen3');
|
||||||
|
expect(buttons[2].textContent).toContain('llama3');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('falls back to the last model launched on this (harness, endpoint) pair when nothing is currently loaded', async () => {
|
||||||
|
const { win, app } = bootApp({
|
||||||
|
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }],
|
||||||
|
});
|
||||||
|
const origApiJson = app._apiJson;
|
||||||
|
app._apiJson = async (path: string) => {
|
||||||
|
if (path === '/api/model-endpoints/llama-box/running-status') return { isLlamaSwap: false, running: [] };
|
||||||
|
return origApiJson(path);
|
||||||
|
};
|
||||||
|
// Simulate a prior launch on this exact (harness, endpoint) pair having picked llama3.
|
||||||
|
win.localStorage.setItem('codeman:customModelLastUsed:claude:llama-box', 'llama3');
|
||||||
|
|
||||||
|
await app.selectCustomModelEntry('claude', 'llama-box');
|
||||||
|
|
||||||
|
const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')];
|
||||||
|
expect(buttons[0].textContent).toContain('llama3');
|
||||||
|
expect(buttons[0].textContent).toContain('Last used');
|
||||||
|
expect(buttons[0].textContent).not.toContain('Currently loaded');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('prefers the currently-loaded model over a stale "last used" entry when both are present', async () => {
|
||||||
|
const { win, app } = bootApp({
|
||||||
|
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }],
|
||||||
|
});
|
||||||
|
const origApiJson = app._apiJson;
|
||||||
|
app._apiJson = async (path: string) => {
|
||||||
|
if (path === '/api/model-endpoints/llama-box/running-status') {
|
||||||
|
return { isLlamaSwap: true, running: [{ model: 'phi4', state: 'ready' }] };
|
||||||
|
}
|
||||||
|
return origApiJson(path);
|
||||||
|
};
|
||||||
|
win.localStorage.setItem('codeman:customModelLastUsed:claude:llama-box', 'llama3');
|
||||||
|
|
||||||
|
await app.selectCustomModelEntry('claude', 'llama-box');
|
||||||
|
|
||||||
|
const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')];
|
||||||
|
expect(buttons[0].textContent).toContain('phi4');
|
||||||
|
expect(buttons[0].textContent).toContain('Currently loaded');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('is not fooled by a model llama-swap reports loaded but not yet ready, or one this host no longer lists', async () => {
|
||||||
|
const { win, app } = bootApp({
|
||||||
|
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3'] }],
|
||||||
|
});
|
||||||
|
const origApiJson = app._apiJson;
|
||||||
|
app._apiJson = async (path: string) => {
|
||||||
|
if (path === '/api/model-endpoints/llama-box/running-status') {
|
||||||
|
// "loading", not "ready" — and a model id this host's own /v1/models no longer serves.
|
||||||
|
return { isLlamaSwap: true, running: [{ model: 'ghost-model', state: 'loading' }] };
|
||||||
|
}
|
||||||
|
return origApiJson(path);
|
||||||
|
};
|
||||||
|
|
||||||
|
await app.selectCustomModelEntry('claude', 'llama-box');
|
||||||
|
|
||||||
|
const list = win.document.getElementById('customModelPickList')!;
|
||||||
|
expect(list.textContent).not.toContain('Currently loaded');
|
||||||
|
expect(list.textContent).not.toContain('ghost-model');
|
||||||
|
const buttons = [...list.querySelectorAll('button')];
|
||||||
|
expect(buttons[0].textContent).toContain('qwen3');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('remembers the launched model as "last used" only once the apply actually succeeds, not on the mere attempt', async () => {
|
||||||
|
const { win, app } = bootApp({});
|
||||||
|
app.activeSessionId = 'old-session';
|
||||||
|
// A real new session, and a real successful apply with no questions asked — the
|
||||||
|
// restart path only reaches its _setCustomModelLastUsed call past both.
|
||||||
|
app.run = async () => {
|
||||||
|
app.activeSessionId = 'new-session';
|
||||||
|
};
|
||||||
|
app._api = async () => ({
|
||||||
|
ok: true,
|
||||||
|
status: 200,
|
||||||
|
json: async () => ({ success: true, data: { customModel: { endpointId: 'llama-box' }, restarted: true } }),
|
||||||
|
});
|
||||||
|
|
||||||
|
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
|
||||||
|
|
||||||
|
expect(win.localStorage.getItem('codeman:customModelLastUsed:claude:llama-box')).toBe('qwen3');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('a slower currently-loaded probe for an earlier pick must never clobber a faster, later pick for a different endpoint', async () => {
|
||||||
|
const { win, app } = bootApp({});
|
||||||
|
const hostA = { id: 'host-a', label: 'Host A', baseUrl: 'http://a', models: ['a1', 'a2'] };
|
||||||
|
const hostB = { id: 'host-b', label: 'Host B', baseUrl: 'http://b', models: ['b1', 'b2'] };
|
||||||
|
let resolveA!: (v: unknown) => void;
|
||||||
|
const pendingA = new Promise((resolve) => {
|
||||||
|
resolveA = resolve;
|
||||||
|
});
|
||||||
|
app._apiJson = async (path: string) => {
|
||||||
|
if (path === '/api/model-endpoints/host-a/running-status') return pendingA;
|
||||||
|
if (path === '/api/model-endpoints/host-b/running-status') return { isLlamaSwap: false, running: [] };
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Host A's picker opens first but its probe never resolves until we say so below —
|
||||||
|
// Host B's opens second and resolves immediately, so it renders first.
|
||||||
|
const openA = app._openCustomModelPickModal('claude', hostA);
|
||||||
|
await app._openCustomModelPickModal('claude', hostB);
|
||||||
|
|
||||||
|
expect(win.document.getElementById('customModelPickHint')!.textContent).toContain('Host B');
|
||||||
|
expect(app._pendingCustomModelPick).toEqual({ mode: 'claude', endpointId: 'host-b' });
|
||||||
|
|
||||||
|
// Host A's probe finally answers, after Host B has already rendered.
|
||||||
|
resolveA({ isLlamaSwap: false, running: [] });
|
||||||
|
await openA;
|
||||||
|
|
||||||
|
// The late-arriving Host A response must be a no-op: still Host B on screen.
|
||||||
|
expect(win.document.getElementById('customModelPickHint')!.textContent).toContain('Host B');
|
||||||
|
expect(app._pendingCustomModelPick).toEqual({ mode: 'claude', endpointId: 'host-b' });
|
||||||
|
const listText = win.document.getElementById('customModelPickList')!.textContent;
|
||||||
|
expect(listText).toContain('b1');
|
||||||
|
expect(listText).toContain('b2');
|
||||||
|
expect(listText).not.toContain('a1');
|
||||||
|
expect(listText).not.toContain('a2');
|
||||||
|
});
|
||||||
|
|
||||||
it('picking a row in the modal closes it and launches with that exact model', async () => {
|
it('picking a row in the modal closes it and launches with that exact model', async () => {
|
||||||
const { win, app } = bootApp({
|
const { win, app } = bootApp({
|
||||||
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://localhost:8080', models: ['qwen3', 'llama3'] }],
|
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://localhost:8080', models: ['qwen3', 'llama3'] }],
|
||||||
@@ -333,6 +473,48 @@ describe('Custom Model Endpoint Profiles: the "which model" picker', () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe('Custom Model Endpoint Profiles: _getCustomModelCurrentlyLoaded is client-side bounded', () => {
|
||||||
|
// `timeoutMs` driven in milliseconds rather than the real 800 — same reasoning as
|
||||||
|
// `_watchLlamaSwapLoading`'s own `pollIntervalMs` a few describe blocks down: this
|
||||||
|
// code runs inside the JSDOM window's own realm, whose setTimeout vi.useFakeTimers()
|
||||||
|
// does not patch, so this is the only way to test the bound without actually waiting
|
||||||
|
// on it (or, worse, hanging on a promise that deliberately never resolves).
|
||||||
|
|
||||||
|
it('never lets an endpoint that never answers keep the picker waiting past the client-side bound', async () => {
|
||||||
|
const { app } = bootApp({
|
||||||
|
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3'] }],
|
||||||
|
});
|
||||||
|
// A `running-status` probe that simply never resolves — the exact shape of an
|
||||||
|
// endpoint that is asleep or firewalled, distinct from one that answers an error.
|
||||||
|
app._apiJson = () => new Promise(() => {});
|
||||||
|
|
||||||
|
const result = await app._getCustomModelCurrentlyLoaded({ id: 'llama-box', models: ['qwen3'] }, 5);
|
||||||
|
|
||||||
|
expect(result).toBeNull();
|
||||||
|
});
|
||||||
|
|
||||||
|
it('an endpoint that answers well within the bound is unaffected by it', async () => {
|
||||||
|
const { app } = bootApp({});
|
||||||
|
app._apiJson = async () => ({ isLlamaSwap: true, running: [{ model: 'qwen3', state: 'ready' }] });
|
||||||
|
|
||||||
|
const result = await app._getCustomModelCurrentlyLoaded({ id: 'llama-box', models: ['qwen3'] }, 5);
|
||||||
|
|
||||||
|
expect(result).toBe('qwen3');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('a rejected probe settles quietly to null rather than leaving an unhandled rejection once the timeout has already won the race', async () => {
|
||||||
|
const { app } = bootApp({});
|
||||||
|
app._apiJson = () => new Promise((_resolve, reject) => setTimeout(() => reject(new Error('boom')), 10));
|
||||||
|
|
||||||
|
const result = await app._getCustomModelCurrentlyLoaded({ id: 'llama-box', models: ['qwen3'] }, 2);
|
||||||
|
expect(result).toBeNull();
|
||||||
|
// Give the loser of the race a turn to actually reject and hit its own .catch —
|
||||||
|
// an unswallowed rejection here would surface as an "Unhandled Errors" failure
|
||||||
|
// for the whole test file, not a failed assertion in this test.
|
||||||
|
await new Promise((resolve) => setTimeout(resolve, 20));
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
describe('Custom Model Endpoint Profiles: applying a picked entry', () => {
|
describe('Custom Model Endpoint Profiles: applying a picked entry', () => {
|
||||||
it('does not apply the endpoint to a session that was already open when the launch fails', async () => {
|
it('does not apply the endpoint to a session that was already open when the launch fails', async () => {
|
||||||
const { app } = bootApp({});
|
const { app } = bootApp({});
|
||||||
@@ -1025,8 +1207,8 @@ describe("Custom Model Endpoint Profiles: requiresContextWarning (this CLI's own
|
|||||||
return { win, app, applyBodies };
|
return { win, app, applyBodies };
|
||||||
}
|
}
|
||||||
|
|
||||||
it('confirming the in-app context-warning modal re-sends the apply with confirmed:true', async () => {
|
it('confirming the in-app context-warning modal re-sends the apply with confirmed:true, and only THEN records "last used"', async () => {
|
||||||
const { app, applyBodies } = launchHarness([
|
const { win, app, applyBodies } = launchHarness([
|
||||||
{ requiresContextWarning: true, modelId: 'qwen3', contextLength: 16384, minSafeContextTokens: 40000 },
|
{ requiresContextWarning: true, modelId: 'qwen3', contextLength: 16384, minSafeContextTokens: 40000 },
|
||||||
{ customModel: { endpointId: 'llama-box' }, restarted: true, modelSwapInProgress: false },
|
{ customModel: { endpointId: 'llama-box' }, restarted: true, modelSwapInProgress: false },
|
||||||
]);
|
]);
|
||||||
@@ -1043,10 +1225,11 @@ describe("Custom Model Endpoint Profiles: requiresContextWarning (this CLI's own
|
|||||||
{ endpointId: 'llama-box', modelId: 'qwen3' },
|
{ endpointId: 'llama-box', modelId: 'qwen3' },
|
||||||
{ endpointId: 'llama-box', modelId: 'qwen3', confirmedContext: true },
|
{ endpointId: 'llama-box', modelId: 'qwen3', confirmedContext: true },
|
||||||
]);
|
]);
|
||||||
|
expect(win.localStorage.getItem('codeman:customModelLastUsed:claude:llama-box')).toBe('qwen3');
|
||||||
});
|
});
|
||||||
|
|
||||||
it('declining the in-app context-warning modal keeps the native backend and never re-sends the apply', async () => {
|
it('declining the in-app context-warning modal keeps the native backend, never re-sends the apply, and must NEVER record this model as "last used" — it cannot work with this CLI at all', async () => {
|
||||||
const { app, applyBodies } = launchHarness([
|
const { win, app, applyBodies } = launchHarness([
|
||||||
{ requiresContextWarning: true, modelId: 'qwen3', contextLength: 16384, minSafeContextTokens: 40000 },
|
{ requiresContextWarning: true, modelId: 'qwen3', contextLength: 16384, minSafeContextTokens: 40000 },
|
||||||
]);
|
]);
|
||||||
app._confirmContextWarning = async () => false;
|
app._confirmContextWarning = async () => false;
|
||||||
@@ -1059,6 +1242,7 @@ describe("Custom Model Endpoint Profiles: requiresContextWarning (this CLI's own
|
|||||||
|
|
||||||
expect(applyBodies).toHaveLength(1); // no second (confirmed) call
|
expect(applyBodies).toHaveLength(1); // no second (confirmed) call
|
||||||
expect(toastMessage).toMatch(/context window too small/i);
|
expect(toastMessage).toMatch(/context window too small/i);
|
||||||
|
expect(win.localStorage.getItem('codeman:customModelLastUsed:claude:llama-box')).toBeNull();
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user