diff --git a/src/web/public/session-ui.js b/src/web/public/session-ui.js index 3047f90c..62c6ca49 100644 --- a/src/web/public/session-ui.js +++ b/src/web/public/session-ui.js @@ -633,11 +633,53 @@ Object.assign(CodemanApp.prototype, { if (models.length === 1) { return this.runCustomModelEntry(mode, endpointId, models[0]); } - this._openCustomModelPickModal(mode, host); + await this._openCustomModelPickModal(mode, host); + }, + + /** localStorage key for the last model launched on a given (harness, endpoint) pair — per-device by design, like every other `codeman:*` UI preference, never synced. */ + _customModelLastUsedKey(mode, endpointId) { + return `codeman:customModelLastUsed:${mode}:${endpointId}`; + }, + + /** Reads the last model chosen for this (harness, endpoint) pair, or null. Never throws — a blocked/full localStorage just means no promotion, not a broken picker. */ + _getCustomModelLastUsed(mode, endpointId) { + try { + return localStorage.getItem(this._customModelLastUsedKey(mode, endpointId)); + } catch { + return null; + } + }, + + /** Remembers `modelId` as the last one launched for this (harness, endpoint) pair. */ + _setCustomModelLastUsed(mode, endpointId, modelId) { + try { + localStorage.setItem(this._customModelLastUsedKey(mode, endpointId), modelId); + } catch { + // best-effort — losing the "last used" hint is cosmetic, never worth surfacing + } + }, + + /** + * Best-effort lookup of the model llama-swap currently has loaded and ready on this + * endpoint, so the picker can offer it first instead of making the user remember what + * they picked last time it mattered. Mirrors `_watchLlamaSwapLoading`'s own + * `state === 'ready'` check. Returns null for a plain (non-llama-swap) server, an + * unreachable endpoint, or a loaded model this host no longer lists as discovered — + * never throws, since a failed probe should just skip promotion, not break the picker. + */ + async _getCustomModelCurrentlyLoaded(host) { + try { + const status = await this._apiJson(`/api/model-endpoints/${encodeURIComponent(host.id)}/running-status`); + if (!status?.isLlamaSwap) return null; + const ready = (status.running || []).find((r) => r.state === 'ready' && (host.models || []).includes(r.model)); + return ready?.model || null; + } catch { + return null; + } }, /** Renders the "which model" picker for a (harness, endpoint) pair with more than one discovered model. */ - _openCustomModelPickModal(mode, host) { + async _openCustomModelPickModal(mode, host) { const modal = document.getElementById('customModelPickModal'); const list = document.getElementById('customModelPickList'); if (!modal || !list) return; @@ -649,13 +691,29 @@ Object.assign(CodemanApp.prototype, { document.getElementById('customModelPickTitle').textContent = 'Choose a model'; document.getElementById('customModelPickHint').textContent = `${cliLabel} → ${host.label} — ${(host.models || []).length} models discovered.`; - list.innerHTML = (host.models || []) + + // Whichever model llama-swap actually has loaded right now beats a merely + // remembered choice — it's what a launch would attach to with zero wait, while + // "last used" might have been swapped out by another session since. Neither + // reorders past the top: exactly one model is promoted, everything else keeps + // its discovery order. + const currentlyLoaded = await this._getCustomModelCurrentlyLoaded(host); + const lastUsed = currentlyLoaded ? null : this._getCustomModelLastUsed(mode, host.id); + const promoted = currentlyLoaded || lastUsed; + const models = [...(host.models || [])]; + if (promoted && models.includes(promoted)) { + models.splice(models.indexOf(promoted), 1); + models.unshift(promoted); + } + + list.innerHTML = models .map((m) => { const isDefault = m === host.defaultModelId; + const tag = m === currentlyLoaded ? 'Currently loaded' : m === lastUsed ? 'Last used' : isDefault ? 'Default' : null; const arg = escapeHtml(JSON.stringify(m)); return ` `; }) .join(''); @@ -771,6 +829,10 @@ Object.assign(CodemanApp.prototype, { * (Codex, confirmed live) than on claude's own `--resume`-based restart. */ async runCustomModelEntry(mode, endpointId, modelId) { + // Recorded on the attempt, not gated on success below — the picker's "Last used" + // promotion is a convenience hint, not a launch-history log, so it should reflect + // what the user picked even if this particular launch goes on to fail. + this._setCustomModelLastUsed(mode, endpointId, modelId); if (mode === 'claude') { return this._runCustomModelEntryViaRestart(mode, endpointId, modelId); } diff --git a/test/custom-model-run-menu-ui.test.ts b/test/custom-model-run-menu-ui.test.ts index 693a2ae7..b7c292dc 100644 --- a/test/custom-model-run-menu-ui.test.ts +++ b/test/custom-model-run-menu-ui.test.ts @@ -272,6 +272,101 @@ describe('Custom Model Endpoint Profiles: the "which model" picker', () => { expect(win.document.getElementById('customModelPickList')!.textContent).toContain('Default'); }); + it('promotes the model llama-swap currently has loaded and ready to the top of the list, tagged', async () => { + const { win, app } = bootApp({ + hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }], + }); + const origApiJson = app._apiJson; + app._apiJson = async (path: string) => { + if (path === '/api/model-endpoints/llama-box/running-status') { + return { isLlamaSwap: true, running: [{ model: 'phi4', state: 'ready' }] }; + } + return origApiJson(path); + }; + + await app.selectCustomModelEntry('claude', 'llama-box'); + + const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')]; + expect(buttons.map((b) => b.textContent)).toHaveLength(3); + expect(buttons[0].textContent).toContain('phi4'); + expect(buttons[0].textContent).toContain('Currently loaded'); + // Nothing else got relabelled or reordered past the promoted row. + expect(buttons[1].textContent).toContain('qwen3'); + expect(buttons[2].textContent).toContain('llama3'); + }); + + it('falls back to the last model launched on this (harness, endpoint) pair when nothing is currently loaded', async () => { + const { win, app } = bootApp({ + hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }], + }); + const origApiJson = app._apiJson; + app._apiJson = async (path: string) => { + if (path === '/api/model-endpoints/llama-box/running-status') return { isLlamaSwap: false, running: [] }; + return origApiJson(path); + }; + // Simulate a prior launch on this exact (harness, endpoint) pair having picked llama3. + win.localStorage.setItem('codeman:customModelLastUsed:claude:llama-box', 'llama3'); + + await app.selectCustomModelEntry('claude', 'llama-box'); + + const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')]; + expect(buttons[0].textContent).toContain('llama3'); + expect(buttons[0].textContent).toContain('Last used'); + expect(buttons[0].textContent).not.toContain('Currently loaded'); + }); + + it('prefers the currently-loaded model over a stale "last used" entry when both are present', async () => { + const { win, app } = bootApp({ + hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3', 'phi4'] }], + }); + const origApiJson = app._apiJson; + app._apiJson = async (path: string) => { + if (path === '/api/model-endpoints/llama-box/running-status') { + return { isLlamaSwap: true, running: [{ model: 'phi4', state: 'ready' }] }; + } + return origApiJson(path); + }; + win.localStorage.setItem('codeman:customModelLastUsed:claude:llama-box', 'llama3'); + + await app.selectCustomModelEntry('claude', 'llama-box'); + + const buttons = [...win.document.getElementById('customModelPickList')!.querySelectorAll('button')]; + expect(buttons[0].textContent).toContain('phi4'); + expect(buttons[0].textContent).toContain('Currently loaded'); + }); + + it('is not fooled by a model llama-swap reports loaded but not yet ready, or one this host no longer lists', async () => { + const { win, app } = bootApp({ + hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://x', models: ['qwen3', 'llama3'] }], + }); + const origApiJson = app._apiJson; + app._apiJson = async (path: string) => { + if (path === '/api/model-endpoints/llama-box/running-status') { + // "loading", not "ready" — and a model id this host's own /v1/models no longer serves. + return { isLlamaSwap: true, running: [{ model: 'ghost-model', state: 'loading' }] }; + } + return origApiJson(path); + }; + + await app.selectCustomModelEntry('claude', 'llama-box'); + + const list = win.document.getElementById('customModelPickList')!; + expect(list.textContent).not.toContain('Currently loaded'); + expect(list.textContent).not.toContain('ghost-model'); + const buttons = [...list.querySelectorAll('button')]; + expect(buttons[0].textContent).toContain('qwen3'); + }); + + it('remembers the launched model as "last used" for this (harness, endpoint) pair', async () => { + const { win, app } = bootApp({}); + app.run = async () => {}; + app._api = async () => ({ ok: true, json: async () => ({ success: true, data: {} }) }); + + await app.runCustomModelEntry('claude', 'llama-box', 'qwen3'); + + expect(win.localStorage.getItem('codeman:customModelLastUsed:claude:llama-box')).toBe('qwen3'); + }); + it('picking a row in the modal closes it and launches with that exact model', async () => { const { win, app } = bootApp({ hosts: [{ id: 'llama-box', label: 'llama.cpp', baseUrl: 'http://localhost:8080', models: ['qwen3', 'llama3'] }],