feat(custom-model): promote the currently-loaded/last-used model in the Run-menu picker

Custom Model Endpoint Profiles' "which model" picker (session-ui.js's
_openCustomModelPickModal) always listed models in their raw discovery
order, so on a host with several downloaded GGUFs the user had to
remember (or eyeball the "Default" tag) which one llama-swap actually
had hot before picking — the whole point of the picker being fast is
undone if it makes you think first.

The picker now promotes exactly one model to the top of the list:

- If llama-swap reports a model from this host's own list `ready`
  right now (via the existing GET /api/model-endpoints/:id/running-status
  route), that model is promoted and tagged "Currently loaded" — it's
  what a launch attaches to with zero wait.
- Otherwise, the last model actually launched on this exact
  (harness, endpoint) pair is promoted and tagged "Last used", read
  from a new per-device localStorage key
  (codeman:customModelLastUsed:<mode>:<endpointId>), written by
  runCustomModelEntry on every launch attempt regardless of outcome.
- A plain (non-llama-swap) OpenAI-compatible server, an unreachable
  endpoint, or a loaded-but-not-yet-ready model never promotes
  anything — the rest of the list keeps its discovery order.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01N6eadpRyqpA9PD3i139cSD
This commit is contained in:
Devvyn
2026-09-20 19:20:45 +08:00
co-authored by Claude Sonnet 5
parent 51b4a1b758
commit 458ca578e7
2 changed files with 161 additions and 4 deletions
+66 -4
View File
@@ -633,11 +633,53 @@ Object.assign(CodemanApp.prototype, {
if (models.length === 1) {
return this.runCustomModelEntry(mode, endpointId, models[0]);
}
this._openCustomModelPickModal(mode, host);
await this._openCustomModelPickModal(mode, host);
},
/** localStorage key for the last model launched on a given (harness, endpoint) pair — per-device by design, like every other `codeman:*` UI preference, never synced. */
_customModelLastUsedKey(mode, endpointId) {
return `codeman:customModelLastUsed:${mode}:${endpointId}`;
},
/** Reads the last model chosen for this (harness, endpoint) pair, or null. Never throws — a blocked/full localStorage just means no promotion, not a broken picker. */
_getCustomModelLastUsed(mode, endpointId) {
try {
return localStorage.getItem(this._customModelLastUsedKey(mode, endpointId));
} catch {
return null;
}
},
/** Remembers `modelId` as the last one launched for this (harness, endpoint) pair. */
_setCustomModelLastUsed(mode, endpointId, modelId) {
try {
localStorage.setItem(this._customModelLastUsedKey(mode, endpointId), modelId);
} catch {
// best-effort — losing the "last used" hint is cosmetic, never worth surfacing
}
},
/**
* Best-effort lookup of the model llama-swap currently has loaded and ready on this
* endpoint, so the picker can offer it first instead of making the user remember what
* they picked last time it mattered. Mirrors `_watchLlamaSwapLoading`'s own
* `state === 'ready'` check. Returns null for a plain (non-llama-swap) server, an
* unreachable endpoint, or a loaded model this host no longer lists as discovered —
* never throws, since a failed probe should just skip promotion, not break the picker.
*/
async _getCustomModelCurrentlyLoaded(host) {
try {
const status = await this._apiJson(`/api/model-endpoints/${encodeURIComponent(host.id)}/running-status`);
if (!status?.isLlamaSwap) return null;
const ready = (status.running || []).find((r) => r.state === 'ready' && (host.models || []).includes(r.model));
return ready?.model || null;
} catch {
return null;
}
},
/** Renders the "which model" picker for a (harness, endpoint) pair with more than one discovered model. */
_openCustomModelPickModal(mode, host) {
async _openCustomModelPickModal(mode, host) {
const modal = document.getElementById('customModelPickModal');
const list = document.getElementById('customModelPickList');
if (!modal || !list) return;
@@ -649,13 +691,29 @@ Object.assign(CodemanApp.prototype, {
document.getElementById('customModelPickTitle').textContent = 'Choose a model';
document.getElementById('customModelPickHint').textContent =
`${cliLabel} → ${host.label} — ${(host.models || []).length} models discovered.`;
list.innerHTML = (host.models || [])
// Whichever model llama-swap actually has loaded right now beats a merely
// remembered choice — it's what a launch would attach to with zero wait, while
// "last used" might have been swapped out by another session since. Neither
// reorders past the top: exactly one model is promoted, everything else keeps
// its discovery order.
const currentlyLoaded = await this._getCustomModelCurrentlyLoaded(host);
const lastUsed = currentlyLoaded ? null : this._getCustomModelLastUsed(mode, host.id);
const promoted = currentlyLoaded || lastUsed;
const models = [...(host.models || [])];
if (promoted && models.includes(promoted)) {
models.splice(models.indexOf(promoted), 1);
models.unshift(promoted);
}
list.innerHTML = models
.map((m) => {
const isDefault = m === host.defaultModelId;
const tag = m === currentlyLoaded ? 'Currently loaded' : m === lastUsed ? 'Last used' : isDefault ? 'Default' : null;
const arg = escapeHtml(JSON.stringify(m));
return `
<button class="run-mode-option" onclick="app.chooseCustomModelAndRun(${arg})">
<span class="run-mode-dot ${escapeHtml(mode)}"></span>${escapeHtml(m)}${isDefault ? ' <span class="set-scope">Default</span>' : ''}
<span class="run-mode-dot ${escapeHtml(mode)}"></span>${escapeHtml(m)}${tag ? ` <span class="set-scope">${escapeHtml(tag)}</span>` : ''}
</button>`;
})
.join('');
@@ -771,6 +829,10 @@ Object.assign(CodemanApp.prototype, {
* (Codex, confirmed live) than on claude's own `--resume`-based restart.
*/
async runCustomModelEntry(mode, endpointId, modelId) {
// Recorded on the attempt, not gated on success below — the picker's "Last used"
// promotion is a convenience hint, not a launch-history log, so it should reflect
// what the user picked even if this particular launch goes on to fail.
this._setCustomModelLastUsed(mode, endpointId, modelId);
if (mode === 'claude') {
return this._runCustomModelEntryViaRestart(mode, endpointId, modelId);
}
+95
View File
@@ -272,6 +272,101 @@ describe('Custom Model Endpoint Profiles: the "which model" picker', () => {
expect(win.document.getElementById('customModelPickList')!.textContent).toContain('Default');
});
it('promotes the model llama-swap currently has loaded and ready to the top of the list, tagged', async () => {
const { win, app } = bootApp({
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }],
});
const origApiJson = app._apiJson;
app._apiJson = async (path: string) => {
if (path === '/api/model-endpoints/llama-box/running-status') {
return { isLlamaSwap: true, running: [{ model: 'phi4', state: 'ready' }] };
}
return origApiJson(path);
};
await app.selectCustomModelEntry('claude', 'llama-box');
const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')];
expect(buttons.map((b) => b.textContent)).toHaveLength(3);
expect(buttons[0].textContent).toContain('phi4');
expect(buttons[0].textContent).toContain('Currently loaded');
// Nothing else got relabelled or reordered past the promoted row.
expect(buttons[1].textContent).toContain('qwen3');
expect(buttons[2].textContent).toContain('llama3');
});
it('falls back to the last model launched on this (harness, endpoint) pair when nothing is currently loaded', async () => {
const { win, app } = bootApp({
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }],
});
const origApiJson = app._apiJson;
app._apiJson = async (path: string) => {
if (path === '/api/model-endpoints/llama-box/running-status') return { isLlamaSwap: false, running: [] };
return origApiJson(path);
};
// Simulate a prior launch on this exact (harness, endpoint) pair having picked llama3.
win.localStorage.setItem('codeman:customModelLastUsed:claude:llama-box', 'llama3');
await app.selectCustomModelEntry('claude', 'llama-box');
const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')];
expect(buttons[0].textContent).toContain('llama3');
expect(buttons[0].textContent).toContain('Last used');
expect(buttons[0].textContent).not.toContain('Currently loaded');
});
it('prefers the currently-loaded model over a stale "last used" entry when both are present', async () => {
const { win, app } = bootApp({
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }],
});
const origApiJson = app._apiJson;
app._apiJson = async (path: string) => {
if (path === '/api/model-endpoints/llama-box/running-status') {
return { isLlamaSwap: true, running: [{ model: 'phi4', state: 'ready' }] };
}
return origApiJson(path);
};
win.localStorage.setItem('codeman:customModelLastUsed:claude:llama-box', 'llama3');
await app.selectCustomModelEntry('claude', 'llama-box');
const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')];
expect(buttons[0].textContent).toContain('phi4');
expect(buttons[0].textContent).toContain('Currently loaded');
});
it('is not fooled by a model llama-swap reports loaded but not yet ready, or one this host no longer lists', async () => {
const { win, app } = bootApp({
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3'] }],
});
const origApiJson = app._apiJson;
app._apiJson = async (path: string) => {
if (path === '/api/model-endpoints/llama-box/running-status') {
// "loading", not "ready" — and a model id this host's own /v1/models no longer serves.
return { isLlamaSwap: true, running: [{ model: 'ghost-model', state: 'loading' }] };
}
return origApiJson(path);
};
await app.selectCustomModelEntry('claude', 'llama-box');
const list = win.document.getElementById('customModelPickList')!;
expect(list.textContent).not.toContain('Currently loaded');
expect(list.textContent).not.toContain('ghost-model');
const buttons = [...list.querySelectorAll('button')];
expect(buttons[0].textContent).toContain('qwen3');
});
it('remembers the launched model as "last used" for this (harness, endpoint) pair', async () => {
const { win, app } = bootApp({});
app.run = async () => {};
app._api = async () => ({ ok: true, json: async () => ({ success: true, data: {} }) });
await app.runCustomModelEntry('claude', 'llama-box', 'qwen3');
expect(win.localStorage.getItem('codeman:customModelLastUsed:claude:llama-box')).toBe('qwen3');
});
it('picking a row in the modal closes it and launches with that exact model', async () => {
const { win, app } = bootApp({
hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://localhost:8080', models: ['qwen3', 'llama3'] }],