mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-09-30 12:39:42 +02:00
feat(custom-model): promote the currently-loaded/last-used model in the Run-menu picker
Custom Model Endpoint Profiles' "which model" picker (session-ui.js's _openCustomModelPickModal) always listed models in their raw discovery order, so on a host with several downloaded GGUFs the user had to remember (or eyeball the "Default" tag) which one llama-swap actually had hot before picking — the whole point of the picker being fast is undone if it makes you think first. The picker now promotes exactly one model to the top of the list: - If llama-swap reports a model from this host's own list `ready` right now (via the existing GET /api/model-endpoints/:id/running-status route), that model is promoted and tagged "Currently loaded" — it's what a launch attaches to with zero wait. - Otherwise, the last model actually launched on this exact (harness, endpoint) pair is promoted and tagged "Last used", read from a new per-device localStorage key (codeman:customModelLastUsed:<mode>:<endpointId>), written by runCustomModelEntry on every launch attempt regardless of outcome. - A plain (non-llama-swap) OpenAI-compatible server, an unreachable endpoint, or a loaded-but-not-yet-ready model never promotes anything — the rest of the list keeps its discovery order. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01N6eadpRyqpA9PD3i139cSD
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
51b4a1b758
commit
458ca578e7
@@ -633,11 +633,53 @@ Object.assign(CodemanApp.prototype, {
|
||||
if (models.length === 1) {
|
||||
return this.runCustomModelEntry(mode, endpointId, models[0]);
|
||||
}
|
||||
this._openCustomModelPickModal(mode, host);
|
||||
await this._openCustomModelPickModal(mode, host);
|
||||
},
|
||||
|
||||
/** localStorage key for the last model launched on a given (harness, endpoint) pair — per-device by design, like every other `codeman:*` UI preference, never synced. */
|
||||
_customModelLastUsedKey(mode, endpointId) {
|
||||
return `codeman:customModelLastUsed:${mode}:${endpointId}`;
|
||||
},
|
||||
|
||||
/** Reads the last model chosen for this (harness, endpoint) pair, or null. Never throws — a blocked/full localStorage just means no promotion, not a broken picker. */
|
||||
_getCustomModelLastUsed(mode, endpointId) {
|
||||
try {
|
||||
return localStorage.getItem(this._customModelLastUsedKey(mode, endpointId));
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
},
|
||||
|
||||
/** Remembers `modelId` as the last one launched for this (harness, endpoint) pair. */
|
||||
_setCustomModelLastUsed(mode, endpointId, modelId) {
|
||||
try {
|
||||
localStorage.setItem(this._customModelLastUsedKey(mode, endpointId), modelId);
|
||||
} catch {
|
||||
// best-effort — losing the "last used" hint is cosmetic, never worth surfacing
|
||||
}
|
||||
},
|
||||
|
||||
/**
|
||||
* Best-effort lookup of the model llama-swap currently has loaded and ready on this
|
||||
* endpoint, so the picker can offer it first instead of making the user remember what
|
||||
* they picked last time it mattered. Mirrors `_watchLlamaSwapLoading`'s own
|
||||
* `state === 'ready'` check. Returns null for a plain (non-llama-swap) server, an
|
||||
* unreachable endpoint, or a loaded model this host no longer lists as discovered —
|
||||
* never throws, since a failed probe should just skip promotion, not break the picker.
|
||||
*/
|
||||
async _getCustomModelCurrentlyLoaded(host) {
|
||||
try {
|
||||
const status = await this._apiJson(`/api/model-endpoints/${encodeURIComponent(host.id)}/running-status`);
|
||||
if (!status?.isLlamaSwap) return null;
|
||||
const ready = (status.running || []).find((r) => r.state === 'ready' && (host.models || []).includes(r.model));
|
||||
return ready?.model || null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
},
|
||||
|
||||
/** Renders the "which model" picker for a (harness, endpoint) pair with more than one discovered model. */
|
||||
_openCustomModelPickModal(mode, host) {
|
||||
async _openCustomModelPickModal(mode, host) {
|
||||
const modal = document.getElementById('customModelPickModal');
|
||||
const list = document.getElementById('customModelPickList');
|
||||
if (!modal || !list) return;
|
||||
@@ -649,13 +691,29 @@ Object.assign(CodemanApp.prototype, {
|
||||
document.getElementById('customModelPickTitle').textContent = 'Choose a model';
|
||||
document.getElementById('customModelPickHint').textContent =
|
||||
`${cliLabel} → ${host.label} — ${(host.models || []).length} models discovered.`;
|
||||
list.innerHTML = (host.models || [])
|
||||
|
||||
// Whichever model llama-swap actually has loaded right now beats a merely
|
||||
// remembered choice — it's what a launch would attach to with zero wait, while
|
||||
// "last used" might have been swapped out by another session since. Neither
|
||||
// reorders past the top: exactly one model is promoted, everything else keeps
|
||||
// its discovery order.
|
||||
const currentlyLoaded = await this._getCustomModelCurrentlyLoaded(host);
|
||||
const lastUsed = currentlyLoaded ? null : this._getCustomModelLastUsed(mode, host.id);
|
||||
const promoted = currentlyLoaded || lastUsed;
|
||||
const models = [...(host.models || [])];
|
||||
if (promoted && models.includes(promoted)) {
|
||||
models.splice(models.indexOf(promoted), 1);
|
||||
models.unshift(promoted);
|
||||
}
|
||||
|
||||
list.innerHTML = models
|
||||
.map((m) => {
|
||||
const isDefault = m === host.defaultModelId;
|
||||
const tag = m === currentlyLoaded ? 'Currently loaded' : m === lastUsed ? 'Last used' : isDefault ? 'Default' : null;
|
||||
const arg = escapeHtml(JSON.stringify(m));
|
||||
return `
|
||||
<button class="run-mode-option" onclick="app.chooseCustomModelAndRun(${arg})">
|
||||
<span class="run-mode-dot ${escapeHtml(mode)}"></span>${escapeHtml(m)}${isDefault ? ' <span class="set-scope">Default</span>' : ''}
|
||||
<span class="run-mode-dot ${escapeHtml(mode)}"></span>${escapeHtml(m)}${tag ? ` <span class="set-scope">${escapeHtml(tag)}</span>` : ''}
|
||||
</button>`;
|
||||
})
|
||||
.join('');
|
||||
@@ -771,6 +829,10 @@ Object.assign(CodemanApp.prototype, {
|
||||
* (Codex, confirmed live) than on claude's own `--resume`-based restart.
|
||||
*/
|
||||
async runCustomModelEntry(mode, endpointId, modelId) {
|
||||
// Recorded on the attempt, not gated on success below — the picker's "Last used"
|
||||
// promotion is a convenience hint, not a launch-history log, so it should reflect
|
||||
// what the user picked even if this particular launch goes on to fail.
|
||||
this._setCustomModelLastUsed(mode, endpointId, modelId);
|
||||
if (mode === 'claude') {
|
||||
return this._runCustomModelEntryViaRestart(mode, endpointId, modelId);
|
||||
}
|
||||
|
||||
@@ -272,6 +272,101 @@ describe('Custom Model Endpoint Profiles: the "which model" picker', () => {
|
||||
expect(win.document.getElementById('customModelPickList')!.textContent).toContain('Default');
|
||||
});
|
||||
|
||||
it('promotes the model llama-swap currently has loaded and ready to the top of the list, tagged', async () => {
|
||||
const { win, app } = bootApp({
|
||||
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }],
|
||||
});
|
||||
const origApiJson = app._apiJson;
|
||||
app._apiJson = async (path: string) => {
|
||||
if (path === '/api/model-endpoints/llama-box/running-status') {
|
||||
return { isLlamaSwap: true, running: [{ model: 'phi4', state: 'ready' }] };
|
||||
}
|
||||
return origApiJson(path);
|
||||
};
|
||||
|
||||
await app.selectCustomModelEntry('claude', 'llama-box');
|
||||
|
||||
const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')];
|
||||
expect(buttons.map((b) => b.textContent)).toHaveLength(3);
|
||||
expect(buttons[0].textContent).toContain('phi4');
|
||||
expect(buttons[0].textContent).toContain('Currently loaded');
|
||||
// Nothing else got relabelled or reordered past the promoted row.
|
||||
expect(buttons[1].textContent).toContain('qwen3');
|
||||
expect(buttons[2].textContent).toContain('llama3');
|
||||
});
|
||||
|
||||
it('falls back to the last model launched on this (harness, endpoint) pair when nothing is currently loaded', async () => {
|
||||
const { win, app } = bootApp({
|
||||
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }],
|
||||
});
|
||||
const origApiJson = app._apiJson;
|
||||
app._apiJson = async (path: string) => {
|
||||
if (path === '/api/model-endpoints/llama-box/running-status') return { isLlamaSwap: false, running: [] };
|
||||
return origApiJson(path);
|
||||
};
|
||||
// Simulate a prior launch on this exact (harness, endpoint) pair having picked llama3.
|
||||
win.localStorage.setItem('codeman:customModelLastUsed:claude:llama-box', 'llama3');
|
||||
|
||||
await app.selectCustomModelEntry('claude', 'llama-box');
|
||||
|
||||
const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')];
|
||||
expect(buttons[0].textContent).toContain('llama3');
|
||||
expect(buttons[0].textContent).toContain('Last used');
|
||||
expect(buttons[0].textContent).not.toContain('Currently loaded');
|
||||
});
|
||||
|
||||
it('prefers the currently-loaded model over a stale "last used" entry when both are present', async () => {
|
||||
const { win, app } = bootApp({
|
||||
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }],
|
||||
});
|
||||
const origApiJson = app._apiJson;
|
||||
app._apiJson = async (path: string) => {
|
||||
if (path === '/api/model-endpoints/llama-box/running-status') {
|
||||
return { isLlamaSwap: true, running: [{ model: 'phi4', state: 'ready' }] };
|
||||
}
|
||||
return origApiJson(path);
|
||||
};
|
||||
win.localStorage.setItem('codeman:customModelLastUsed:claude:llama-box', 'llama3');
|
||||
|
||||
await app.selectCustomModelEntry('claude', 'llama-box');
|
||||
|
||||
const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')];
|
||||
expect(buttons[0].textContent).toContain('phi4');
|
||||
expect(buttons[0].textContent).toContain('Currently loaded');
|
||||
});
|
||||
|
||||
it('is not fooled by a model llama-swap reports loaded but not yet ready, or one this host no longer lists', async () => {
|
||||
const { win, app } = bootApp({
|
||||
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3'] }],
|
||||
});
|
||||
const origApiJson = app._apiJson;
|
||||
app._apiJson = async (path: string) => {
|
||||
if (path === '/api/model-endpoints/llama-box/running-status') {
|
||||
// "loading", not "ready" — and a model id this host's own /v1/models no longer serves.
|
||||
return { isLlamaSwap: true, running: [{ model: 'ghost-model', state: 'loading' }] };
|
||||
}
|
||||
return origApiJson(path);
|
||||
};
|
||||
|
||||
await app.selectCustomModelEntry('claude', 'llama-box');
|
||||
|
||||
const list = win.document.getElementById('customModelPickList')!;
|
||||
expect(list.textContent).not.toContain('Currently loaded');
|
||||
expect(list.textContent).not.toContain('ghost-model');
|
||||
const buttons = [...list.querySelectorAll('button')];
|
||||
expect(buttons[0].textContent).toContain('qwen3');
|
||||
});
|
||||
|
||||
it('remembers the launched model as "last used" for this (harness, endpoint) pair', async () => {
|
||||
const { win, app } = bootApp({});
|
||||
app.run = async () => {};
|
||||
app._api = async () => ({ ok: true, json: async () => ({ success: true, data: {} }) });
|
||||
|
||||
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
|
||||
|
||||
expect(win.localStorage.getItem('codeman:customModelLastUsed:claude:llama-box')).toBe('qwen3');
|
||||
});
|
||||
|
||||
it('picking a row in the modal closes it and launches with that exact model', async () => {
|
||||
const { win, app } = bootApp({
|
||||
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://localhost:8080', models: ['qwen3', 'llama3'] }],
|
||||
|
||||
Reference in New Issue
Block a user