diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 6f545a8..cbdfab0 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -7,10 +7,14 @@ export type { export { getAdapter } from "./agent"; export { DiffCapture } from "./diff-capture"; /** Path identity, shared so containment, diff keys and history keys agree. */ +export type { ModelProbeOptions } from "./models"; +export { listAllModels, listModels } from "./models"; export { isPathInside, pathKey, toPosixPath } from "./paths"; export type { EditPromptInput } from "./prompt"; export { buildEditPrompt, PIKA_SYSTEM_PROMPT, systemPrompt } from "./prompt"; export type { CodexConfigValue, CodexSettings } from "./providers/codex"; +/** Whether opencode can run a model id, so a caller can refuse one it cannot. */ +export { namesProvider } from "./providers/opencode-events"; export type { OpencodeSettings } from "./providers/opencode-server"; /** Stops the shared `opencode serve` child, which otherwise outlives the run. */ export { shutdownServer as shutdownOpencodeServer } from "./providers/opencode-server"; diff --git a/packages/core/src/models.test.ts b/packages/core/src/models.test.ts new file mode 100644 index 0000000..241b00d --- /dev/null +++ b/packages/core/src/models.test.ts @@ -0,0 +1,179 @@ +/** + * The translation half of the model probes. + * + * Mostly the pure functions, for the reason `opencode-events.test.ts` gives about + * its own: the probes themselves start a subprocess and talk to an account, so + * a test that covered them would be testing the machine it ran on. What is + * worth pinning is the mapping — which fields become a row, and which shapes + * are dropped on the way. + * + * The exception is the last block, which stubs `getAdapter` to cover the one + * behaviour the module's header promises and no mapping test can reach: that a + * backend which cannot be loaded at all comes back as a note rather than a + * rejection. + */ + +import { beforeEach, describe, expect, it, vi } from "vitest"; +import { getAdapter } from "./agent"; +import { + fromClaudeModels, + fromOpencodeProviders, + listAllModels, +} from "./models"; + +vi.mock("./agent", () => ({ getAdapter: vi.fn() })); + +describe("fromClaudeModels", () => { + it("takes the label live and the hint from the seed", () => { + // The split matters: the backend is the authority on what a model is + // *called*, and the seed is the only thing that knows its context window, + // because `supportedModels()` does not report one. + expect( + fromClaudeModels([{ displayName: "Opus 5", value: "claude-opus-5" }]) + ).toEqual([{ hint: "1M", id: "claude-opus-5", label: "Opus 5" }]); + }); + + it("falls back to the id when there is no display name", () => { + expect(fromClaudeModels([{ value: "some-new-model" }])[0]).toEqual({ + hint: undefined, + id: "some-new-model", + label: "some-new-model", + }); + }); + + it("drops the SDK's own default row", () => { + // Every group already leads with a synthetic Default that clears the model + // from the request. Two rows called Default, deferring to different things, + // is worse than one. + const rows = fromClaudeModels([ + { displayName: "Default (recommended)", value: "default" }, + { displayName: "Sonnet", value: "sonnet" }, + ]); + expect(rows.map((r) => r.id)).toEqual(["sonnet"]); + }); + + it("drops a row with no value at all", () => { + // Through `unknown` on purpose: the point of the test is a payload the + // declared type says cannot happen, and a direct cast is the one thing + // `tsc` will not let you write for exactly that reason. + const malformed = [{ displayName: "ghost" }] as unknown as Parameters< + typeof fromClaudeModels + >[0]; + expect(fromClaudeModels(malformed)).toEqual([]); + }); + + it("does not put the prose description in the hint", () => { + // The hint slot is a dimmed right-aligned mono cell; a sentence in it wraps + // the row. The context window comes from the seed instead, when known. + const [row] = fromClaudeModels([ + { + description: "Strongest model for coding, agents and long tasks", + displayName: "Opus", + value: "opus", + }, + ]); + expect(row.hint).not.toContain("Strongest"); + }); +}); + +describe("fromOpencodeProviders", () => { + it("joins the provider onto the model id", () => { + expect( + fromOpencodeProviders([ + { + id: "anthropic", + models: { "claude-sonnet-5": { name: "Claude Sonnet 5" } }, + name: "Anthropic", + }, + ]) + ).toEqual([ + { + hint: "Anthropic", + id: "anthropic/claude-sonnet-5", + label: "Claude Sonnet 5", + }, + ]); + }); + + it("flattens several providers and sorts them", () => { + const rows = fromOpencodeProviders([ + { id: "openai", models: { "gpt-5.6": {} } }, + { + id: "anthropic", + models: { "claude-opus-5": {}, "claude-sonnet-5": {} }, + }, + ]); + expect(rows.map((r) => r.id)).toEqual([ + "anthropic/claude-opus-5", + "anthropic/claude-sonnet-5", + "openai/gpt-5.6", + ]); + }); + + it("prefers the model's own id over its key", () => { + const [row] = fromOpencodeProviders([ + { id: "p", models: { alias: { id: "real-id" } } }, + ]); + expect(row.id).toBe("p/real-id"); + }); + + it("falls back to the provider id when it has no display name", () => { + // The hint is what tells two providers' copies of the same model apart, so + // it has to say something even when the registry gave no label. + const [row] = fromOpencodeProviders([{ id: "custom", models: { m: {} } }]); + expect(row.hint).toBe("custom"); + }); + + it("survives a provider with no models and an empty list", () => { + expect(fromOpencodeProviders([{ id: "empty" }])).toEqual([]); + expect(fromOpencodeProviders([])).toEqual([]); + }); +}); + +/* + * The one thing this module promises: it does not throw. + * + * `getAdapter` is a dynamic import, so a backend whose package is missing or + * broken rejects there rather than returning. That call used to sit *outside* + * the `try`, and `listAllModels` runs all three through `Promise.all` — so a + * single unusable backend rejected the whole catalogue and the picker showed an + * error toast with no rows, instead of two working groups and one explained gap. + */ +describe("listAllModels", () => { + const mockGetAdapter = vi.mocked(getAdapter); + + beforeEach(() => { + mockGetAdapter.mockReset(); + }); + + it("reports a backend that cannot even be loaded as a note, not a rejection", async () => { + mockGetAdapter.mockRejectedValue(new Error("Cannot find module 'codex'")); + + const groups = await listAllModels("/tmp"); + + expect(groups).toHaveLength(3); + for (const group of groups) { + expect(group.note).toBe("Cannot find module 'codex'"); + // A degraded menu still has to be a menu. + expect(group.models.length).toBeGreaterThan(0); + } + }); + + it("lets the healthy backends through when one is broken", async () => { + mockGetAdapter.mockImplementation((agent) => { + if (agent === "codex") { + return Promise.reject(new Error("broken install")); + } + return Promise.resolve({ + checkAuth: () => ({ ok: false, reason: "Not signed in" }), + } as unknown as Awaited>); + }); + + const groups = await listAllModels("/tmp"); + + const byAgent = new Map(groups.map((g) => [g.agent, g])); + expect(byAgent.get("codex")?.note).toBe("broken install"); + expect(byAgent.get("claude")?.note).toBe("Not signed in"); + expect(byAgent.get("opencode")?.note).toBe("Not signed in"); + }); +}); diff --git a/packages/core/src/models.ts b/packages/core/src/models.ts new file mode 100644 index 0000000..9d6502d --- /dev/null +++ b/packages/core/src/models.ts @@ -0,0 +1,255 @@ +/** + * Which models each backend will accept, asked of the backend itself. + * + * The three harnesses answer this question very differently, and the asymmetry + * is the whole reason this module exists: + * + * | Backend | How it enumerates | + * |----------|----------------------------------------------------------| + * | claude | `query.supportedModels()` — live, and account-aware | + * | opencode | `client.config.providers()` — live, only what is authed | + * | codex | nothing. No subcommand, no RPC, no config to read | + * + * So Claude and OpenCode are asked, and Codex is served from the generated seed + * in `@airship/protocol/models`. The seed also backs the other two whenever a + * probe fails, which is the common case on a machine that has only signed into + * one of them. + * + * Nothing here throws. A picker that cannot list models is a degraded menu; a + * picker that takes the session down with it is a bug. Every failure comes back + * as a group with a `note` explaining itself. + */ + +import type { + AgentKind, + ModelCatalogue, + ModelGroup, + ModelOption, +} from "@airship/protocol"; +import { AGENT_KINDS } from "@airship/protocol"; +import { SEED_MODELS } from "@airship/protocol/models"; +import { getAdapter } from "./agent"; +import type { OpencodeSettings } from "./providers/opencode-server"; + +/** + * How long a single backend gets to answer. + * + * Generous, because two of the three probes start a subprocess and a cold + * `opencode serve` on a slow disk is not a failure. Bounded, because the menu + * is already on screen showing the seed — this only decides how long the user + * waits before the live list replaces it. + */ +const PROBE_TIMEOUT_MS = 15_000; + +export interface ModelProbeOptions { + opencode?: OpencodeSettings; + safe?: boolean; +} + +/** The seed rows for one harness, already in wire shape. */ +function seedFor(agent: AgentKind): ModelOption[] { + return SEED_MODELS[agent].map((m) => ({ ...m })); +} + +/** + * Lose a race against the clock rather than hang the request. + * + * The losing promise is deliberately not cancelled: `supportedModels()` has no + * abort signal, and a probe that finishes late is harmless — its result is + * dropped and the SDK's own teardown still runs. Leaving it to settle is + * cheaper than inventing a cancellation path the SDKs do not offer. + */ +function withTimeout(work: Promise, label: string): Promise { + return Promise.race([ + work, + new Promise((_, reject) => + setTimeout( + () => reject(new Error(`${label} did not answer in time`)), + PROBE_TIMEOUT_MS + ).unref?.() + ), + ]); +} + +// -- Mapping ------------------------------------------------------------------ +// Pure, and separated from every probe above it so the translation can be +// tested without an SDK — the same split `opencode-events.ts` uses. + +/** The shape `query.supportedModels()` returns, as much of it as a row needs. */ +export interface ClaudeModelInfo { + description?: string; + displayName?: string; + value: string; +} + +/** + * Claude's answer → rows. + * + * Two things get dropped on the way through. + * + * `description` is prose ("Strongest model for coding, agents…") and the hint + * slot is a dimmed right-aligned mono cell, so it is dropped rather than + * truncated into it. The seed carries context windows for the ids it knows; + * anything newer simply has no hint, which reads fine. + * + * The SDK's own `default` row goes too. Every group already leads with a + * synthetic Default that means "send no model and let the daemon's resolved + * setting stand" — keeping both would put two rows labelled Default in one + * menu, disagreeing about which setting they defer to. + */ +export function fromClaudeModels(infos: ClaudeModelInfo[]): ModelOption[] { + const hints = new Map(SEED_MODELS.claude.map((m) => [m.id, m.hint])); + return infos + .filter((info) => info?.value && info.value !== "default") + .map((info) => ({ + hint: hints.get(info.value), + id: info.value, + label: info.displayName || info.value, + })); +} + +/** One provider block from `config.providers()`, as much as a row needs. */ +export interface OpencodeProvider { + id: string; + models?: Record; + name?: string; +} + +/** + * OpenCode's answer → rows. + * + * Ids are joined back into the `provider/model` form, which is the only form + * that survives: the adapter's `modelRefFor` splits on the first slash and + * drops an id it cannot attribute to a provider. The provider's display name + * becomes the hint, because with several providers configured the same model + * appears more than once and the provider is what tells them apart. + */ +export function fromOpencodeProviders( + providers: OpencodeProvider[] +): ModelOption[] { + const rows: ModelOption[] = []; + for (const provider of providers ?? []) { + for (const [modelId, model] of Object.entries(provider.models ?? {})) { + const id = model?.id ?? modelId; + rows.push({ + hint: provider.name ?? provider.id, + id: `${provider.id}/${id}`, + label: model?.name ?? id, + }); + } + } + return rows.sort((a, b) => a.id.localeCompare(b.id)); +} + +// -- Probes ------------------------------------------------------------------- + +/** + * Ask Claude, through a session that is never prompted. + * + * The same idiom `rewind()` uses: open a `query` with an empty prompt, wait for + * the one `system/init` message that means the control channel is live, ask the + * question, and break out of the iterator. `settingSources: []` keeps a + * project's own config off a call that is only reading a list. + */ +async function probeClaude(cwd: string): Promise { + const { query } = await import("@anthropic-ai/claude-agent-sdk"); + const q = query({ + options: { cwd, permissionMode: "default", settingSources: [] }, + prompt: "", + }); + for await (const msg of q) { + if (msg.type === "system" && msg.subtype === "init") { + // Returning from inside `for await` runs the iterator's `return()`, which + // is what tears the subprocess down. Deliberately not `interrupt()`: that + // needs streaming-input mode and this session has a plain string prompt. + return fromClaudeModels((await q.supportedModels()) as ClaudeModelInfo[]); + } + } + throw new Error("session did not initialize"); +} + +async function probeOpencode( + cwd: string, + opts: ModelProbeOptions +): Promise<{ default?: string; models: ModelOption[] }> { + const { acquireServer } = await import("./providers/opencode-server"); + const { client } = await acquireServer(opts.opencode, opts.safe ?? false); + const res = await client.config.providers({ directory: cwd }); + const providers = res.data?.providers ?? []; + const models = fromOpencodeProviders(providers); + // `default` is keyed by provider; the first provider's entry is the one the + // server would actually pick, so it is the only one worth surfacing. + const [first] = providers; + const fallback = first ? res.data?.default?.[first.id] : undefined; + return { + default: fallback && first ? `${first.id}/${fallback}` : undefined, + models, + }; +} + +/** + * One harness's group. + * + * Auth is checked before the probe rather than after it fails: an unsigned-in + * backend would otherwise spend the full timeout on a request that could never + * have worked, and report "did not answer in time" for what is really a missing + * credential. The distinction matters — one of those the user can fix. + */ +export async function listModels( + agent: AgentKind, + cwd: string, + opts: ModelProbeOptions = {} +): Promise { + const seed = seedFor(agent); + + // `getAdapter` and `checkAuth` are inside the `try` too, which they were not. + // `getAdapter` is a dynamic import, so a backend whose package is missing or + // broken rejects *here* — and this function is called through `Promise.all`, + // so one unusable backend took the other two down with it and the picker got + // an error toast with no rows instead of two working groups and one note. + try { + const adapter = await getAdapter(agent); + const auth = adapter.checkAuth(); + if (!auth.ok) { + return { agent, models: seed, note: auth.reason ?? "Not signed in" }; + } + + // Codex is not probed because there is nothing to probe. Said here rather + // than left as a silent fallthrough — the absence is the point. + if (agent === "codex") { + return { agent, models: seed }; + } + + if (agent === "claude") { + const models = await withTimeout(probeClaude(cwd), "claude"); + return { agent, models: models.length ? models : seed }; + } + const { default: fallback, models } = await withTimeout( + probeOpencode(cwd, opts), + "opencode" + ); + return models.length + ? { agent, default: fallback, models } + : { agent, models: seed, note: "No providers configured" }; + } catch (err) { + return { + agent, + models: seed, + note: err instanceof Error ? err.message : String(err), + }; + } +} + +/** + * Every harness, probed at once. + * + * Concurrent because the slow one is whichever backend has to start a + * subprocess, and running them in series would add those cold starts together + * behind a menu the user is already looking at. + */ +export function listAllModels( + cwd: string, + opts: ModelProbeOptions = {} +): Promise { + return Promise.all(AGENT_KINDS.map((agent) => listModels(agent, cwd, opts))); +} diff --git a/packages/core/src/providers/opencode-events.ts b/packages/core/src/providers/opencode-events.ts index 50adcf4..9bb06fb 100644 --- a/packages/core/src/providers/opencode-events.ts +++ b/packages/core/src/providers/opencode-events.ts @@ -425,6 +425,20 @@ export function modelRefFor( }; } +/** + * Whether opencode can actually run this model id. + * + * The rule the *drop* is made on, exported so the places that must refuse an id + * ahead of time — `--opencode-model` at parse time, a picked model at request + * time — decide it the same way `toModelBody` does rather than each re-spelling + * "has a slash". A ref with no `providerID` is dropped from the request body, so + * asking for one is indistinguishable from asking for nothing: the turn runs on + * the server's default, having been told otherwise. + */ +export function namesProvider(model?: string): boolean { + return modelRefFor(model)?.providerID !== undefined; +} + /** * Which session an event belongs to, or `undefined` when it belongs to none. * diff --git a/packages/core/src/providers/opencode-run.test.ts b/packages/core/src/providers/opencode-run.test.ts index 200222b..a8498bc 100644 --- a/packages/core/src/providers/opencode-run.test.ts +++ b/packages/core/src/providers/opencode-run.test.ts @@ -93,6 +93,9 @@ function scriptedClient(attempts: AttemptScript[]): { let call = 0; const script = () => attempts[Math.min(call, attempts.length - 1)]; const client: OpencodeClientLike = { + // Never called on a run — the model catalogue is a separate request. Present + // because the interface describes the real client, not one turn's slice. + config: { providers: () => Promise.resolve({ data: { providers: [] } }) }, event: { subscribe: () => { const { sse } = script(); diff --git a/packages/core/src/providers/opencode-server.ts b/packages/core/src/providers/opencode-server.ts index 7825691..be4589a 100644 --- a/packages/core/src/providers/opencode-server.ts +++ b/packages/core/src/providers/opencode-server.ts @@ -42,6 +42,29 @@ export interface OpencodeSettings { * carrying nothing but heartbeats). */ export interface OpencodeClientLike { + config: { + /** + * Which providers this server can actually reach, and each one's models. + * + * Scoped to what the user has configured or authenticated — opencode + * resolves the models.dev registry itself at startup and reports back only + * the reachable subset. That is why the picker asks the server rather than + * reading the registry a second time: this answer already knows what a + * request would be allowed to do. + * + * `default` maps a provider id to the model it falls back to. + */ + providers: (params?: { directory?: string }) => Promise<{ + data?: { + default?: Record; + providers?: { + id: string; + models?: Record; + name?: string; + }[]; + }; + }>; + }; event: { /** * The second argument is the transport options bag — `ServerSentEventsOptions` diff --git a/packages/protocol/package.json b/packages/protocol/package.json index 044d0e5..7dea36a 100644 --- a/packages/protocol/package.json +++ b/packages/protocol/package.json @@ -11,14 +11,18 @@ "./tokens": { "types": "./dist/tokens.d.ts", "import": "./dist/tokens.js" + }, + "./models": { + "types": "./dist/models.d.ts", + "import": "./dist/models.js" } }, "main": "./dist/index.js", "types": "./dist/index.d.ts", "scripts": { - "build": "tsup src/index.ts src/tokens.ts --format esm --dts --sourcemap --clean", + "build": "tsup src/index.ts src/tokens.ts src/models.ts --format esm --dts --sourcemap --clean", "clean": "node ../../scripts/clean.mjs dist .turbo", - "dev": "tsup src/index.ts src/tokens.ts --format esm --dts --sourcemap --watch", + "dev": "tsup src/index.ts src/tokens.ts src/models.ts --format esm --dts --sourcemap --watch", "test": "vitest run", "typecheck": "tsc --noEmit" }, diff --git a/packages/protocol/src/index.ts b/packages/protocol/src/index.ts index 9c4bc9e..7a8deb9 100644 --- a/packages/protocol/src/index.ts +++ b/packages/protocol/src/index.ts @@ -53,6 +53,59 @@ export const EFFORT_LEVELS = [ export const EffortSchema = z.enum(EFFORT_LEVELS); export type Effort = z.infer; +// --------------------------------------------------------------------------- +// Models +// --------------------------------------------------------------------------- + +/** + * One selectable model, shaped for the row that renders it. + * + * The seed list in `./models` uses the same shape, so a group can be filled + * from the generated catalogue or from a live probe without translating. + */ +export const ModelOptionSchema = z.object({ + /** Right-aligned dimmed hint on the row — a context window, a provider. */ + hint: z.string().optional(), + /** + * What travels as `CreateJobRequest.model`. + * + * Bare for claude and codex, whose `--model` takes an id or an alias. For + * opencode this must be the `provider/model` form: `modelRefFor` splits on + * the first slash and an id it cannot attribute to a provider is dropped + * rather than guessed at. + */ + id: z.string(), + label: z.string(), +}); +export type ModelOption = z.infer; + +/** + * One harness's models, shaped for a `MenuGroup` in the overlay's picker. + * + * A list of groups rather than a `Record`: it maps one-to-one + * onto what `createMenu` accepts, and zod v4 records over an enum require every + * key, which a partial probe result cannot promise. + */ +export const ModelGroupSchema = z.object({ + agent: AgentKindSchema, + /** The resting default — the daemon's resolved per-harness model when one is + * configured, otherwise whatever the backend reports. Leads the group. */ + default: z.string().optional(), + models: z.array(ModelOptionSchema), + /** + * Why this group is short or empty: "Not signed in", "Probe timed out". + * + * Rendered as a disabled row. A backend that cannot be reached must say so — + * an empty group with no explanation reads as a bug in the picker rather than + * as a missing credential. + */ + note: z.string().optional(), +}); +export type ModelGroup = z.infer; + +export const ModelCatalogueSchema = z.array(ModelGroupSchema); +export type ModelCatalogue = z.infer; + // --------------------------------------------------------------------------- // Element + source location // --------------------------------------------------------------------------- @@ -320,6 +373,10 @@ export const CreateJobRequestSchema = z /** Fork instead of continue — "try a different approach". */ fork: z.boolean().optional(), images: z.array(ImageInputSchema).optional(), + /** Which model that backend runs this turn on. Absent means the daemon's + * resolved default for `agent` (`--claude-model` and friends, else + * `--model`), which is what a client that predates the picker sends. */ + model: z.string().optional(), /** Direct-manipulation structural moves (drag-to-reposition in the tree). * Like `visualChanges`, these make `prompt` optional. */ moveChanges: z.array(MoveEditSchema).optional(), @@ -557,6 +614,11 @@ export interface JobHistorySummary { error?: string; filesChanged: number; jobId: string; + /** Which model ran, once resolved. Lives here rather than on `JobDiffBundle` + * so it survives `toSummary()`: reopening a thread re-seeds the picker from a + * summary, and a field stripped from the listing would never reach it. + * Absent on bundles written before the picker could choose one. */ + model?: string; parentJobId?: string; promptPreview: string; /** Agent session id — a Claude session or a Codex thread, per `agent`. Used @@ -628,6 +690,10 @@ export type ServerEvent = /** The project's design tokens, scanned from the files on disk. Answers a * `tokens` request; also pushed unprompted once the first scan completes. */ | { type: "tokens:result"; scan: TokenScanResult } + /** What each backend offers, answering a `models` request. Sent to the asking + * socket only — the probe is per-connection work, and a broadcast would repaint + * every other tab's open menu underneath its user. */ + | { type: "models:result"; catalogue: ModelCatalogue } /** The assembled instruction for a `prompt` request — the exact string the * adapter would receive. Sent to the asking socket only: a broadcast would let * one tab's composer overwrite another's preview. */ @@ -675,6 +741,19 @@ export const ClientMessageSchema = z.discriminatedUnion("type", [ refresh: z.boolean().optional(), type: z.literal("tokens"), }), + /** + * Which models each backend offers. Read-only, so the server answers it off + * the edit chain like `tokens`. + * + * Asked when the picker is first opened rather than pushed at the handshake: + * the opencode probe starts an `opencode serve` process, and booting one for + * a session that never opens the menu is a cost with no return. `refresh` + * busts the daemon's memo after signing into a backend mid-session. + */ + z.object({ + refresh: z.boolean().optional(), + type: z.literal("models"), + }), /** * Render the prompt this request would produce, without running it. Read-only, * so the server answers it off the edit chain like `tokens`. diff --git a/packages/protocol/src/models.test.ts b/packages/protocol/src/models.test.ts new file mode 100644 index 0000000..b6186bf --- /dev/null +++ b/packages/protocol/src/models.test.ts @@ -0,0 +1,70 @@ +/** + * The generated seed catalogue, checked for shape rather than for contents. + * + * Every assertion here has to survive `make models:refresh` picking up a model + * that shipped this morning. Pinning ids would invert that — the suite would go + * red on a refresh that did exactly what it was asked to, and the fix each time + * would be to edit the test to match the output, which is a gate that only ever + * agrees with itself. + * + * So this asserts the contract the rest of the code leans on: every row is + * sendable, every harness has something to show, and opencode's ids carry the + * provider its adapter needs. + */ + +import { describe, expect, it } from "vitest"; +import { AGENT_KINDS } from "./index"; +import { SEED_MODELS } from "./models"; + +const entries = Object.entries(SEED_MODELS); + +describe("SEED_MODELS", () => { + it("covers exactly the harnesses the protocol declares", () => { + // The generated module spells its keys as string literals rather than + // importing `AgentKind`, to stay free of the zod-importing barrel. This is + // what makes that safe: a fourth backend, or a rename, fails here. + expect(Object.keys(SEED_MODELS).sort()).toEqual([...AGENT_KINDS].sort()); + }); + + it.each(entries)("gives %s something to offer", (_agent, models) => { + // Codex especially: it can enumerate nothing at runtime, so an empty seed + // would leave its group permanently blank rather than merely stale. + expect(models.length).toBeGreaterThan(0); + }); + + it.each(entries)("gives every %s row an id and a label", (_agent, models) => { + for (const model of models) { + expect(model.id.trim()).not.toBe(""); + expect(model.label.trim()).not.toBe(""); + } + }); + + it.each(entries)("does not repeat an id within %s", (_agent, models) => { + const ids = models.map((m) => m.id); + expect(ids).toEqual([...new Set(ids)]); + }); + + it("keeps every opencode id in the provider/model form", () => { + // `modelRefFor` splits on the first slash and drops an id it cannot + // attribute, so a bare one here would be silently ignored at run time. + for (const model of SEED_MODELS.opencode) { + const [provider, ...rest] = model.id.split("/"); + expect(provider).not.toBe(""); + expect(rest.join("/")).not.toBe(""); + } + }); + + it("keeps claude and codex ids bare", () => { + // The mirror of the rule above: those two take an id or an alias, and a + // `provider/` prefix is not something either would recognise. + for (const model of [...SEED_MODELS.claude, ...SEED_MODELS.codex]) { + expect(model.id).not.toContain("/"); + } + }); + + it("never uses the empty id, which the menu reserves for Default", () => { + for (const [, models] of entries) { + expect(models.some((m) => m.id === "")).toBe(false); + } + }); +}); diff --git a/packages/protocol/src/models.ts b/packages/protocol/src/models.ts new file mode 100644 index 0000000..14e0bbd --- /dev/null +++ b/packages/protocol/src/models.ts @@ -0,0 +1,90 @@ +// AUTO-GENERATED from https://models.dev/models.json by scripts/gen-models.mjs. +// Do not edit by hand — edit scripts/models.curation.json and run +// `make models:refresh`. 34 models across three harnesses. + +/** + * The model list the picker paints before anything has been asked, and falls + * back to when a probe fails. + * + * This module is deliberately free of zod and of every other runtime import, + * and reaches the overlay through the `./models` export subpath alongside + * `./tokens`. That is what lets the browser bundle import the list as a value, + * rather than hand-copying it beside `AGENTS` in `app.ts`, at no cost — an + * import from the package's main entry would pull the validator in with it. + * + * (The overlay bundle does contain zod today regardless: `app.ts` takes + * `modeToSurface` and the surface constants from that main entry, which is + * enough to drag it along. This module simply does not add to that, and stays + * correct if those imports are ever moved.) + * + * It is a *seed*, never the truth. Claude and OpenCode both enumerate their own + * models at runtime and their answers win, because only they know what the user + * is signed in to. Codex can enumerate nothing at any layer, so for that + * harness this is the whole list — which is the reason the generator exists. + */ +export interface SeedModel { + /** Right-aligned dimmed hint on the row — a context window. */ + hint?: string; + /** What goes on the wire. Bare for claude and codex; `provider/model` for + * opencode, whose `modelRefFor` drops an id it cannot attribute. */ + id: string; + label: string; +} + +/* + * Keys spelled out rather than imported as `AgentKind`. That type is a + * `z.infer`, so naming it here would tie this module to the zod-importing + * barrel for no gain — `models.test.ts` asserts these keys against + * `AGENT_KINDS` instead, which catches the drift the import would have. + */ +export const SEED_MODELS: Record<"claude" | "codex" | "opencode", SeedModel[]> = + { + claude: [ + { hint: "latest", id: "opus", label: "Opus" }, + { hint: "latest", id: "sonnet", label: "Sonnet" }, + { hint: "latest", id: "haiku", label: "Haiku" }, + { hint: "latest", id: "fable", label: "Fable" }, + { hint: "1M", id: "claude-opus-5", label: "Claude Opus 5" }, + { hint: "1M", id: "claude-sonnet-5", label: "Claude Sonnet 5" }, + { hint: "1M", id: "claude-fable-5", label: "Claude Fable 5" }, + { hint: "1M", id: "claude-mythos-5", label: "Claude Mythos 5" }, + { hint: "1M", id: "claude-opus-4-8", label: "Claude Opus 4.8" }, + { hint: "1M", id: "claude-opus-4-7", label: "Claude Opus 4.7" }, + { hint: "1M", id: "claude-sonnet-4-6", label: "Claude Sonnet 4.6" }, + { hint: "1M", id: "claude-opus-4-6", label: "Claude Opus 4.6" }, + { + hint: "200K", + id: "claude-opus-4-5", + label: "Claude Opus 4.5 (latest)", + }, + { + hint: "200K", + id: "claude-haiku-4-5", + label: "Claude Haiku 4.5 (latest)", + }, + ], + codex: [ + { hint: "1.1M", id: "gpt-5.6-luna", label: "GPT-5.6 Luna" }, + { hint: "1.1M", id: "gpt-5.6-sol", label: "GPT-5.6 Sol" }, + { hint: "1.1M", id: "gpt-5.6-terra", label: "GPT-5.6 Terra" }, + { hint: "1.1M", id: "gpt-5.5", label: "GPT-5.5" }, + { hint: "1.1M", id: "gpt-5.5-pro", label: "GPT-5.5 Pro" }, + { hint: "400K", id: "gpt-5.4-mini", label: "GPT-5.4 mini" }, + { hint: "400K", id: "gpt-5.4-nano", label: "GPT-5.4 nano" }, + { hint: "1.1M", id: "gpt-5.4", label: "GPT-5.4" }, + { hint: "1.1M", id: "gpt-5.4-pro", label: "GPT-5.4 Pro" }, + { hint: "400K", id: "gpt-5.3-codex", label: "GPT-5.3 Codex" }, + ], + opencode: [ + { hint: "1M", id: "anthropic/claude-opus-5", label: "Claude Opus 5" }, + { hint: "1.1M", id: "openai/gpt-5.6-luna", label: "GPT-5.6 Luna" }, + { hint: "1.1M", id: "openai/gpt-5.6-sol", label: "GPT-5.6 Sol" }, + { hint: "1.1M", id: "openai/gpt-5.6-terra", label: "GPT-5.6 Terra" }, + { hint: "1M", id: "anthropic/claude-sonnet-5", label: "Claude Sonnet 5" }, + { hint: "1M", id: "anthropic/claude-fable-5", label: "Claude Fable 5" }, + { hint: "1M", id: "anthropic/claude-mythos-5", label: "Claude Mythos 5" }, + { hint: "1M", id: "anthropic/claude-opus-4-8", label: "Claude Opus 4.8" }, + { hint: "1.1M", id: "openai/gpt-5.5", label: "GPT-5.5" }, + { hint: "1.1M", id: "openai/gpt-5.5-pro", label: "GPT-5.5 Pro" }, + ], + }; diff --git a/packages/server/src/index.ts b/packages/server/src/index.ts index 52e1b5f..3f13841 100644 --- a/packages/server/src/index.ts +++ b/packages/server/src/index.ts @@ -10,6 +10,8 @@ import type { } from "@airship/core"; import { buildEditPrompt, + listAllModels, + namesProvider, runEdit, shutdownOpencodeServer, } from "@airship/core"; @@ -36,6 +38,7 @@ import { type ElementContext, type JobDiffBundle, type JobStatus, + type ModelCatalogue, type MoveEdit, type ReviewComment, type ServerEvent, @@ -54,10 +57,11 @@ import { createProxyServer } from "./proxy"; export type { CodexConfigValue, CodexSettings, + ModelProbeOptions, OpencodeSettings, } from "@airship/core"; /** Re-exported so the CLI depends only on @airship/server. */ -export { checkAuth } from "@airship/core"; +export { checkAuth, listModels } from "@airship/core"; export { isGitRepo } from "@airship/git"; export type { AgentKind, AirshipSurface, Effort } from "@airship/protocol"; @@ -76,7 +80,19 @@ export interface ServerOptions { maxBudgetUsd?: number; /** Claude-only turn cap. */ maxTurns?: number; + /** Cross-harness model default, from `--model`. Superseded per backend by + * `models`, and by a turn that names its own. Kept because the launch banner + * reads it, and because it is still the right answer for a single-backend run. */ model?: string; + /** + * Per-backend model defaults, already resolved by the CLI (`--claude-model` + * and friends, each falling back to `--model`). + * + * Per backend rather than one string because the overlay's picker can change + * harness mid-session: a single default would hand a `claude-opus-5` to Codex + * the first time someone switched. + */ + models?: Partial>; /** OpenCode-only passthrough knobs; opaque here by design. */ opencode?: OpencodeSettings; /** Port Airship's proxy listens on. */ @@ -114,6 +130,42 @@ export async function startServer(opts: ServerOptions): Promise { // cross-contaminate diff capture and undo baselines. let editChain: Promise = Promise.resolve(); + /** + * The model catalogue, probed once and kept. + * + * Memoized as the in-flight promise rather than its value, so two tabs + * opening their pickers together share one probe instead of racing to start + * two `opencode serve` processes. A rejection is not possible to observe + * here — `listAllModels` reports failures as groups with a `note` — but the + * memo is cleared on one anyway, so a genuinely broken probe is retried + * rather than cached forever. + */ + let catalogue: Promise | null = null; + + function modelCatalogue(refresh?: boolean): Promise { + if (refresh) { + catalogue = null; + } + catalogue ??= listAllModels(cwd, { + opencode: opts.opencode, + safe: opts.safe, + }) + .then((groups) => + // The daemon's own resolved default outranks whatever the backend + // reports: if someone launched with `--codex-model gpt-5.4`, that is + // what an unpicked turn will run, so it is what the menu must lead with. + groups.map((group) => ({ + ...group, + default: opts.models?.[group.agent] ?? group.default, + })) + ) + .catch((err) => { + catalogue = null; + throw err; + }); + return catalogue; + } + function broadcast(event: ServerEvent): void { const data = JSON.stringify(event); for (const client of clients) { @@ -170,6 +222,13 @@ export async function startServer(opts: ServerOptions): Promise { switch (parsed.type) { case "edit": { const { request } = parsed; + // Refused here rather than inside `startEdit`, so a model opencode + // cannot run neither creates a job nor takes a place in the edit chain. + const refusal = modelRefusal(request, opts); + if (refusal) { + send(ws, { message: refusal, type: "error" }); + break; + } editChain = editChain .then(() => startEdit(request)) .catch((err) => { @@ -198,6 +257,22 @@ export async function startServer(opts: ServerOptions): Promise { type: "tokens:result", }); break; + // Off `editChain` like `tokens`, and for a stronger reason: the probe + // talks to the same backends a running turn is using, and queueing it + // would leave the picker spinning for the length of an edit. Answered to + // the asking socket only — the menu that asked is the one waiting. + case "models": + modelCatalogue(parsed.refresh) + .then((result) => + send(ws, { catalogue: result, type: "models:result" }) + ) + .catch((err) => { + send(ws, { + message: `model list failed: ${err instanceof Error ? err.message : String(err)}`, + type: "error", + }); + }); + break; // Off `editChain` for the same reason as `tokens`: it only resolves // sources and renders a string. Queueing it behind a running edit would // freeze the composer's live preview for the length of a turn. It does @@ -248,7 +323,7 @@ export async function startServer(opts: ServerOptions): Promise { // the two must render the identical string. const promptInput = preparePromptInput(cwd, request); - const agent = request.agent ?? opts.agent ?? "claude"; + const { agent, model } = resolveTarget(request, opts); const resumeSessionId = resolveResume(cwd, request.parentJobId, agent); const abort = new AbortController(); @@ -269,7 +344,7 @@ export async function startServer(opts: ServerOptions): Promise { images: request.images, maxBudgetUsd: opts.maxBudgetUsd, maxTurns: opts.maxTurns, - model: opts.model, + model, opencode: opts.opencode, resumeSessionId, safe: opts.safe, @@ -304,6 +379,7 @@ export async function startServer(opts: ServerOptions): Promise { createdAt: rec.createdAt, displayPrompt, jobId: rec.jobId, + model, parentJobId: request.parentJobId, primaryElement: promptInput.element, result, @@ -632,6 +708,68 @@ function jobStatus(aborted: boolean, ok: boolean): JobStatus { */ const RESUME_WALK_LIMIT = 3; +/** + * Which backend this turn runs on, and on which model. + * + * Three steps each: what the turn asked for, else the default resolved for the + * backend that is actually running, else the cross-harness one. Unlike `agent`, + * a missing model is fine — every adapter reads an absent model as "use your + * own default". + * + * Per backend rather than one string because the overlay's picker can change + * harness mid-session, so a single default would follow it and hand Codex an id + * only Claude answers to. + * + * One function, and module-level rather than a closure, for two reasons: the + * `edit` handler validates the model that `startEdit` then sends, and resolving + * it in both places would let the checked value drift from the run one; and a + * precedence chain this load-bearing should be reachable from a test without + * standing a server up. + */ +export function resolveTarget( + request: Pick, + opts: Pick +): { agent: AgentKind; model?: string } { + const agent = request.agent ?? opts.agent ?? "claude"; + return { + agent, + model: request.model ?? opts.models?.[agent] ?? opts.model, + }; +} + +/** + * Why this turn's model cannot run, or `null` if it can. + * + * `toModelBody` drops an opencode ref that does not name a provider, which makes + * asking for one indistinguishable from asking for nothing: the turn runs on the + * server's default while the composer goes on showing what was picked. + * + * It reads `request.model` and deliberately **not** the resolved model. The + * three ways a model gets here are guarded differently on purpose, and `args.ts` + * spells out why: + * + * - `--opencode-model` names its backend, so a bare id there can only be a + * mistake — a hard error at parse time. + * - `--model` reaches all three backends, where a bare id is correct for two of + * them. It warns at launch and the turn runs on opencode's own default. + * - The picker's custom-model box had no guard at either end. It is the one door + * left, and the one where the user is choosing right now and can act on being + * told. + * + * Guarding the *resolved* model would fold the second case into the first and + * turn a documented warn-and-continue into "every edit refused". + */ +export function modelRefusal( + request: Pick, + opts: Pick +): string | null { + const { agent } = resolveTarget(request, opts); + if (agent !== "opencode" || !request.model || namesProvider(request.model)) { + return null; + } + return `opencode cannot run '${request.model}': it resolves a model through its provider, so it needs the provider/model form — try 'anthropic/${request.model}'.`; +} + /** * The session to resume, or null to start clean. * @@ -674,6 +812,7 @@ function buildBundle(args: { createdAt: number; displayPrompt: string; jobId: string; + model?: string; parentJobId?: string; primaryElement?: ElementContext; result: RunEditResult; @@ -693,6 +832,7 @@ function buildBundle(args: { filesChanged: result.diffs.length, followUps: result.followUps, jobId: args.jobId, + model: args.model, parentJobId: args.parentJobId, prompt: displayPrompt, promptPreview: preview(displayPrompt), diff --git a/packages/server/src/resolve-target.test.ts b/packages/server/src/resolve-target.test.ts new file mode 100644 index 0000000..18e6f05 --- /dev/null +++ b/packages/server/src/resolve-target.test.ts @@ -0,0 +1,151 @@ +import { describe, expect, it } from "vitest"; +import { modelRefusal, resolveTarget } from "./index"; + +/* + * The model precedence chain, which is the whole of the model workstream's + * server half. + * + * Three inputs can name a model — the turn, the per-backend default the CLI + * resolved from `--claude-model` and friends, and the cross-harness `--model` — + * and the order they win in is the only thing that decides what a turn runs on. + * None of it was covered: `serve.test.ts` pins the CLI's half of the same chain + * (how the flags collapse into `models`), and this pins what the daemon then + * does with it. + * + * The per-backend map is what makes a mid-session harness switch safe. A single + * shared default would follow the picker and hand Codex an id only Claude + * answers to, which is the bug the map exists to prevent — the last case here. + */ + +/** The CLI's resolved options, as `toServeOptions` produces them. */ +const OPTS = { + agent: "claude" as const, + model: "cross-harness", + models: { + claude: "claude-default", + codex: "codex-default", + opencode: "anthropic/opencode-default", + }, +}; + +describe("resolveTarget — the agent", () => { + it("takes the turn's agent over the daemon's", () => { + expect(resolveTarget({ agent: "codex" }, OPTS).agent).toBe("codex"); + }); + + it("falls back to the daemon's agent", () => { + expect(resolveTarget({}, OPTS).agent).toBe("claude"); + }); + + it("falls back to claude when nothing names one", () => { + expect(resolveTarget({}, {}).agent).toBe("claude"); + }); +}); + +describe("resolveTarget — the model", () => { + it("takes what the turn asked for, over every default", () => { + expect(resolveTarget({ model: "picked" }, OPTS).model).toBe("picked"); + }); + + it("falls back to the default for the backend that is running", () => { + expect(resolveTarget({ agent: "codex" }, OPTS).model).toBe("codex-default"); + }); + + it("falls back to the cross-harness model when that backend has none", () => { + const opts = { ...OPTS, models: { claude: "claude-default" } }; + + expect(resolveTarget({ agent: "codex" }, opts).model).toBe("cross-harness"); + }); + + it("leaves the model absent when nothing names one", () => { + // Not an error and not a guess: every adapter reads an absent model as + // "use your own default", which is the right answer for a bare launch. + expect(resolveTarget({}, {}).model).toBeUndefined(); + }); + + it("does not carry one backend's model across to another", () => { + // The reason `models` is a map. With a single shared default, switching the + // picker to Codex mid-session would send it `claude-default`. + const onCodex = resolveTarget({ agent: "codex" }, OPTS); + + expect(onCodex.model).not.toBe(OPTS.models.claude); + expect(onCodex.model).toBe("codex-default"); + }); + + it("resolves the same values the edit handler validates", () => { + // The guard in `handleMessage` and the send in `startEdit` call this once + // each. If they disagreed, the checked model would stop being the run one. + const request = { agent: "opencode" as const }; + + expect(resolveTarget(request, OPTS)).toEqual(resolveTarget(request, OPTS)); + expect(resolveTarget(request, OPTS).model).toBe( + "anthropic/opencode-default" + ); + }); +}); + +/* + * Which door a bad opencode model came through decides what happens to it. + * + * Three inputs can name one, and they are guarded differently on purpose: + * `--opencode-model` is a hard error at parse time, `--model` warns at launch + * and lets the turn run on opencode's default, and the picker's custom-model box + * — the only one with no guard anywhere — is refused here. + * + * The distinction is the whole finding. A first cut of this guard read the + * *resolved* model, which folded the second case into the third: every turn of + * `airship --agent opencode --model sonnet` would have been refused, a case + * `args.ts` documents in as many words as warn-and-continue. + */ +describe("modelRefusal", () => { + const OPENCODE = { agent: "opencode" as const }; + + it("refuses a bare id the turn asked for", () => { + const refusal = modelRefusal({ ...OPENCODE, model: "sonnet" }, {}); + + expect(refusal).toContain("sonnet"); + expect(refusal).toContain("provider/model"); + expect(refusal).toContain("anthropic/sonnet"); + }); + + it("allows one that names its provider", () => { + expect( + modelRefusal({ ...OPENCODE, model: "anthropic/claude-sonnet-5" }, {}) + ).toBeNull(); + }); + + it("leaves the launch-flag fallback alone", () => { + // `--agent opencode --model sonnet`: `toServeOptions` puts the bare id in + // `models.opencode`, the banner warns about it, and the turn runs on + // opencode's default. Refusing it here would break a documented path. + const opts = { agent: "opencode" as const, models: { opencode: "sonnet" } }; + + expect(modelRefusal({}, opts)).toBeNull(); + }); + + it("leaves the cross-harness fallback alone too", () => { + expect( + modelRefusal({}, { agent: "opencode" as const, model: "sonnet" }) + ).toBeNull(); + }); + + it("says nothing about the backends that take a bare id", () => { + for (const agent of ["claude", "codex"] as const) { + expect(modelRefusal({ agent, model: "sonnet" }, {})).toBeNull(); + } + }); + + it("follows the turn's own backend, not the daemon's", () => { + // Picking Codex in the picker and typing a bare id must not be refused + // just because the daemon launched on opencode. + const opts = { agent: "opencode" as const }; + + expect( + modelRefusal({ agent: "codex", model: "gpt-5.3-codex" }, opts) + ).toBeNull(); + // And the mirror: launched on claude, picker switched to opencode. + expect( + modelRefusal({ agent: "opencode", model: "sonnet" }, { agent: "claude" }) + ).not.toBeNull(); + }); +}); diff --git a/scripts/gen-models.mjs b/scripts/gen-models.mjs new file mode 100644 index 0000000..5ceb13c --- /dev/null +++ b/scripts/gen-models.mjs @@ -0,0 +1,296 @@ +// Generates packages/protocol/src/models.ts — the seed model catalogue — from +// the models.dev registry. +// +// Why a seed exists at all: of the three harnesses, only two can enumerate +// their own models. Claude answers `query.supportedModels()` and OpenCode +// answers `client.config.providers()`, both live and both scoped to what the +// user is actually authenticated for. **Codex can enumerate nothing** — no CLI +// subcommand, no app-server RPC, no config file to read. Without this its list +// would be a constant somebody has to remember to edit on every OpenAI release, +// and the failure mode of forgetting is silent: the picker just stops offering +// the model you wanted. +// +// The seed also does two smaller jobs. It is what the menu paints *before* the +// live probe returns, so opening the picker is never a wait; and it is what the +// menu falls back to when a probe fails or there is no network. +// +// Usage: +// node scripts/gen-models.mjs # fetch and write +// node scripts/gen-models.mjs --check # verify without writing +// +// `--check` exists for local use — `make preflight` deliberately does NOT run +// it. Every other generated file in this repo derives from something committed +// beside it, so a drift gate can only fire when a human changed the input. This +// one derives from a remote file that changes whenever a vendor ships a model, +// so gating on it would make the gate go red on PRs that touched nothing and +// require network to pass. `reference/NEXT-STEPS.md` §7 describes what that +// costs: "the gate will fail on every single PR forever." +// +// Refresh is therefore a deliberate act — `make models:refresh` — reviewed as a +// diff, the way a lockfile bump is. +import { spawnSync } from "node:child_process"; +import { readFileSync, writeFileSync } from "node:fs"; +import { fileURLToPath, pathToFileURL } from "node:url"; + +// `new URL(..., import.meta.url)` throughout, handed straight to the fs calls, +// which take a file URL. Same reasoning as scripts/gen-controls.mjs: nothing +// here converts one to a path string, which sidesteps the `/C:/…` and +// percent-encoding traps packages/overlay/scripts/check-css.mjs documents. +const CURATION = new URL("./models.curation.json", import.meta.url); +const OUT = new URL("../packages/protocol/src/models.ts", import.meta.url); + +// The provider-agnostic file, 279 KB. `api.json` carries the same models keyed +// per provider with pricing attached and is 3.7 MB — thirteen times the bytes +// for a `cost` field no menu row renders. Switch only if a price hint is added. +const SOURCE = "https://models.dev/models.json"; + +/** models.dev id prefix → the harness whose group the model belongs in. */ +const HARNESS = { "anthropic/": "claude", "openai/": "codex" }; + +const BIOME = new URL("../node_modules/.bin/biome", import.meta.url); + +function die(message) { + process.stderr.write(`gen-models: ${message}\n`); + process.exit(1); +} + +/** + * Hand the rendered source to biome before it is written or compared. + * + * Not a nicety — it is what keeps `--check` honest. `make preflight` runs + * `ultracite fix` over the whole tree, so a generated file that is not already + * clean gets rewritten the moment anyone lints, and every subsequent `--check` + * then reports stale against a file nobody touched. Running it here means the + * generator and the linter cannot disagree, and the template is free to emit + * readable one-line rows without predicting where biome wraps. + * + * `check --write` rather than `format`: formatting alone leaves lint rules to + * fire later, which is the same staleness one step removed. Safe fixes only — + * `--unsafe` reflows prose (it "fixes" a JSDoc line starting with an asterisk + * by deleting the space, mangling the sentence), so anything it would touch is + * a template bug to fix here rather than to paper over. + */ +function format(source) { + const out = spawnSync( + fileURLToPath(BIOME), + ["check", "--write", "--stdin-file-path=models.ts"], + { encoding: "utf8", input: source } + ); + if (out.error || out.status !== 0) { + die( + `biome could not format the output: ${out.error?.message ?? out.stderr?.trim() ?? `exit ${out.status}`}` + ); + } + return out.stdout; +} + +/** + * Context window as a menu hint. + * + * `MenuItem.hint` renders right-aligned in a dimmed mono — "a shortcut, a size, + * a unit" — so it wants `1M`, not a sentence. This is the whole reason the seed + * carries metadata rather than bare ids: Claude's own `supportedModels()` + * returns a prose `description`, which is the wrong shape for that slot. + */ +function contextHint(limit) { + const n = limit?.context; + if (!n) { + return; + } + return n >= 1_000_000 + ? `${Math.round(n / 100_000) / 10}M`.replace(".0M", "M") + : `${Math.round(n / 1000)}K`; +} + +/** Strip the provider prefix: `anthropic/claude-opus-5` → `claude-opus-5`. */ +function bareId(id) { + return id.slice(id.indexOf("/") + 1); +} + +/** + * The mechanical half of the filter. + * + * Everything decided here is a property models.dev states outright. Anything + * that needs judgement — that `gpt-realtime-2.1` passes `tool_call` and + * `reasoning` but is not a coding model — belongs in the deny list, where it is + * reviewable, rather than as a special case in this function. + */ +function candidates(models, curation) { + const deny = curation.deny.map((p) => new RegExp(p)); + const out = []; + for (const [id, model] of Object.entries(models)) { + const prefix = Object.keys(HARNESS).find((p) => id.startsWith(p)); + if (!(prefix && model.tool_call && model.reasoning)) { + continue; + } + if (!model.release_date || model.release_date < curation.since) { + continue; + } + if (deny.some((re) => re.test(id))) { + continue; + } + out.push({ + date: model.release_date, + harness: HARNESS[prefix], + hint: contextHint(model.limit), + id, + label: model.name ?? bareId(id), + }); + } + // Newest first, then by id so a same-day pair never reorders between runs. + // Determinism is the property the "run it twice, no diff" check rests on. + out.sort((a, b) => b.date.localeCompare(a.date) || a.id.localeCompare(b.id)); + return out; +} + +/** + * Candidates → the three harness groups. + * + * Claude and Codex take the bare id, which is what `--model` wants for each. + * OpenCode takes the `provider/model` form its adapter's `modelRefFor` splits + * on — a bare id there has no resolvable provider and gets dropped. + * + * OpenCode's group is seeded from the same two providers rather than from the + * whole registry. It resolves models.dev itself at startup and reports back + * only what the user is authenticated for, so anything richer here would be + * both duplicated work and a list of models that cannot be called. This is + * first paint, and the live probe replaces it wholesale. + */ +function group(all, curation) { + const seeded = { claude: [], codex: [], opencode: [] }; + for (const harness of ["claude", "codex"]) { + const extra = curation.extra?.[harness] ?? []; + const derived = all + .filter((m) => m.harness === harness) + .slice(0, curation.limit) + .map((m) => ({ hint: m.hint, id: bareId(m.id), label: m.label })); + seeded[harness] = [...extra, ...derived]; + } + seeded.opencode = all + .slice(0, curation.limit) + .map((m) => ({ hint: m.hint, id: m.id, label: m.label })); + return seeded; +} + +function render(seeded, count) { + const rows = (models) => + models + .map((m) => { + const hint = m.hint ? ` hint: ${JSON.stringify(m.hint)},` : ""; + return ` {${hint} id: ${JSON.stringify(m.id)}, label: ${JSON.stringify(m.label)} },`; + }) + .join("\n"); + + return `// AUTO-GENERATED from ${SOURCE} by scripts/gen-models.mjs. +// Do not edit by hand — edit scripts/models.curation.json and run +// \`make models:refresh\`. ${count} models across three harnesses. + +/** + * The model list the picker paints before anything has been asked, and falls + * back to when a probe fails. + * + * This module is deliberately free of zod and of every other runtime import, + * and reaches the overlay through the \`./models\` export subpath alongside + * \`./tokens\`. That is what lets the browser bundle import the list as a value, + * rather than hand-copying it beside \`AGENTS\` in \`app.ts\`, at no cost — an + * import from the package's main entry would pull the validator in with it. + * + * (The overlay bundle does contain zod today regardless: \`app.ts\` takes + * \`modeToSurface\` and the surface constants from that main entry, which is + * enough to drag it along. This module simply does not add to that, and stays + * correct if those imports are ever moved.) + * + * It is a *seed*, never the truth. Claude and OpenCode both enumerate their own + * models at runtime and their answers win, because only they know what the user + * is signed in to. Codex can enumerate nothing at any layer, so for that + * harness this is the whole list — which is the reason the generator exists. + */ +export interface SeedModel { + /** Right-aligned dimmed hint on the row — a context window. */ + hint?: string; + /** What goes on the wire. Bare for claude and codex; \`provider/model\` for + * opencode, whose \`modelRefFor\` drops an id it cannot attribute. */ + id: string; + label: string; +} + +/* + * Keys spelled out rather than imported as \`AgentKind\`. That type is a + * \`z.infer\`, so naming it here would tie this module to the zod-importing + * barrel for no gain — \`models.test.ts\` asserts these keys against + * \`AGENT_KINDS\` instead, which catches the drift the import would have. + */ +export const SEED_MODELS: Record<"claude" | "codex" | "opencode", SeedModel[]> = + { + claude: [ +${rows(seeded.claude)} + ], + codex: [ +${rows(seeded.codex)} + ], + opencode: [ +${rows(seeded.opencode)} + ], +}; +`; +} + +async function main() { + const curation = JSON.parse(readFileSync(CURATION, "utf8")); + const check = process.argv.includes("--check"); + + let models; + try { + const res = await fetch(SOURCE); + if (!res.ok) { + die(`${SOURCE} returned ${res.status}`); + } + models = await res.json(); + } catch (err) { + die(`could not fetch ${SOURCE}: ${err.message}`); + } + + const all = candidates(models, curation); + if (!all.length) { + // A registry reshuffle that silently emptied the file would take Codex's + // only model list with it, so this fails rather than writing the void. + die( + "no models survived the filter — check `since` and `deny` in models.curation.json" + ); + } + + const seeded = group(all, curation); + const count = Object.values(seeded).reduce((n, g) => n + g.length, 0); + const next = format(render(seeded, count)); + + // `\r\n` normalised on both sides: the repo is checked out with native line + // endings on Windows and this comparison is about content. + const same = (a, b) => a.replace(/\r\n/g, "\n") === b.replace(/\r\n/g, "\n"); + + if (check) { + let current = ""; + try { + current = readFileSync(OUT, "utf8"); + } catch { + die( + "packages/protocol/src/models.ts is missing — run `make models:refresh`." + ); + } + if (!same(current, next)) { + die( + "packages/protocol/src/models.ts is stale — run `make models:refresh` and commit the result." + ); + } + process.stdout.write("packages/protocol/src/models.ts is up to date\n"); + return; + } + + writeFileSync(OUT, next); + process.stdout.write( + `wrote packages/protocol/src/models.ts — ${seeded.claude.length} claude, ${seeded.codex.length} codex, ${seeded.opencode.length} opencode\n` + ); +} + +if (import.meta.url === pathToFileURL(process.argv[1] ?? "").href) { + await main(); +} diff --git a/scripts/models.curation.json b/scripts/models.curation.json new file mode 100644 index 0000000..729d8b3 --- /dev/null +++ b/scripts/models.curation.json @@ -0,0 +1,48 @@ +{ + "$comment": [ + "Hand-maintained curation for scripts/gen-models.mjs. The generator filters", + "models.dev mechanically; everything that needs judgement lives here, so the", + "generated file stays purely derived and this file is what gets reviewed.", + "", + "Run `make models:refresh` after editing." + ], + + "since": "2025-08-01", + "$since": "Release-date floor. Cuts o1/o3/o4, gpt-oss and the Claude 3.x/4.0 era without naming them one by one.", + + "limit": 10, + "$limit": "Rows kept per harness from models.dev, newest first — `extra` below is pinned on top of these rather than counted against them, so claude ships 14. A picker is a menu, not a scroll: `popover-host.ts` caps and scrolls past roughly this many.", + + "deny": [ + "-chat-latest$", + "-instant$", + "^openai/gpt-realtime-", + "^openai/gpt-oss-", + "-deep-research$", + "-\\d{8}$" + ], + "$deny": [ + "Regexes matched against the models.dev id. In order:", + "chat-latest / instant — chat tiers, not agentic; they have no tool loop worth offering here.", + "gpt-realtime — voice; passes the tool_call+reasoning filter but is not a coding model.", + "gpt-oss — open weights, not served on the OpenAI API these harnesses call.", + "deep-research — a hosted research product, not a coding model.", + "-YYYYMMDD — models.dev lists both `claude-opus-4-5` and `claude-opus-4-5-20251101`.", + " The undated id is the better menu row and pins to the same weights." + ], + + "extra": { + "claude": [ + { "id": "opus", "label": "Opus", "hint": "latest" }, + { "id": "sonnet", "label": "Sonnet", "hint": "latest" }, + { "id": "haiku", "label": "Haiku", "hint": "latest" }, + { "id": "fable", "label": "Fable", "hint": "latest" } + ] + }, + "$extra": [ + "Rows models.dev cannot know about, pinned above the derived ones.", + "`claude --model` accepts these aliases and resolves each to the current", + "model in its family, so they stay correct between refreshes in a way a", + "pinned id cannot. `codex -m` and opencode take no aliases, so they get none." + ] +}