From 5690356ccbe7ee9fdda1eea555da036e87fd17b1 Mon Sep 17 00:00:00 2001 From: DingJiaYi Date: Sat, 12 Sep 2026 01:45:02 +0800 Subject: [PATCH] fix(core): honor the host input modalities on the generate transport The generate transport gated image content on the pinned capability snapshot alone. A model published upstream after the last catalog sync is absent from it, so the transport rejected images for a model the host already advertises as vision-capable -- the same model works on the Provider API transport, which resolves image support from the host. Read the host's resolved model.input first and fall back to the snapshot, so models.yml/modelOverrides can unlock (or narrow) image support before the generated catalog catches up. --- CHANGELOG.md | 2 ++ README.md | 2 ++ src/core.ts | 2 +- src/models.ts | 19 ++++++++++++--- src/types.ts | 2 ++ tests/test-models.ts | 18 +++++++++++++++ tests/test-stream.ts | 55 ++++++++++++++++++++++++++++++++++++++++++++ 7 files changed, 96 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 879148f..f5107c4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,8 @@ ## Unreleased +- Honor the host's resolved `model.input` on the legacy generate transport instead of consulting only the pinned capability snapshot, so a vision-capable model that the catalog does not yet list can receive images there too. A host that narrows a catalogued vision model to text is honored as well, keeping the image rejection in place. + ## 0.6.4 - 2026-09-03 - Refresh the generated Command Code capability catalog from `command-code@1.40.1` to `command-code@1.44.0`, adding current image-input, reasoning, effort, and output-limit metadata for newly published models. diff --git a/README.md b/README.md index 909fdbc..7791a64 100644 --- a/README.md +++ b/README.md @@ -143,6 +143,8 @@ The provider advertises image input only for models marked with the `image` inpu For vision-capable models, Pi's native provider adapters forward image blocks from user messages and tool results using the documented OpenAI or Anthropic message schema. Unknown and text-only models remain marked text-only in Pi. +The legacy generate transport resolves image support from the host's `model.input` when the host supplies it, and falls back to the capability snapshot otherwise. Because the snapshot is generated from a single CLI release, a host that marks a model as `["text", "image"]` — for example through a `models.yml` or `models.json` model override — can send images on both transports before the catalog catches up. A host that narrows a catalogued vision model to text is likewise honored. + ## Pricing display The Command Code Provider API does not currently include prices in its model catalog. This extension therefore keeps a static table for models with known prices so pi can display estimated request costs. DeepSeek V4 uses time-dependent rates; pi displays the documented off-peak rate, which applies for 17 hours per day. diff --git a/src/core.ts b/src/core.ts index 0d16e70..43f72d3 100644 --- a/src/core.ts +++ b/src/core.ts @@ -540,7 +540,7 @@ export function createStreamCommandCode(deps: CoreDependencies) { const reasoningEffort = mappedReasoningEffort(model, options) const timeoutMs = options?.timeoutMs - const allowImages = modelSupportsImageInput(model.id) + const allowImages = modelSupportsImageInput(model.id, model.input) if (!allowImages) assertTextOnlyMessages(context.messages) let body: unknown = { diff --git a/src/models.ts b/src/models.ts index 0d8ffe9..603442f 100644 --- a/src/models.ts +++ b/src/models.ts @@ -31,12 +31,25 @@ export type CommandCodeApi = "openai-completions" | "anthropic-messages" const TEXT_INPUT_ONLY = ["text"] as const -export function inputModalitiesForModel(modelId: string): readonly CommandCodeInputType[] { +/** + * Input modalities for a model. + * + * The generated catalog is a snapshot of one Command Code CLI release, so a + * model published afterwards is absent and silently degrades to text-only. + * The host's resolved `model.input` is authoritative when present: it already + * carries `models.yml`/`models.json` overrides and is what decides whether the + * user can attach an image at all. + */ +export function inputModalitiesForModel( + modelId: string, + hostInput?: readonly string[], +): readonly CommandCodeInputType[] { + if (hostInput && hostInput.length > 0) return hostInput as readonly CommandCodeInputType[] return MODEL_INPUT_MODALITIES[modelId] ?? TEXT_INPUT_ONLY } -export function modelSupportsImageInput(modelId: string): boolean { - return inputModalitiesForModel(modelId).includes("image") +export function modelSupportsImageInput(modelId: string, hostInput?: readonly string[]): boolean { + return inputModalitiesForModel(modelId, hostInput).includes("image") } export type PiThinkingLevel = "off" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max" diff --git a/src/types.ts b/src/types.ts index d10f823..351f857 100644 --- a/src/types.ts +++ b/src/types.ts @@ -72,6 +72,8 @@ export interface ModelLike { provider: string maxTokens: number cost: ModelCost + /** Input modalities the host advertises, including `models.yml`/`models.json` overrides. */ + input?: readonly string[] reasoning?: boolean thinkingLevelMap?: Partial> thinking?: { diff --git a/tests/test-models.ts b/tests/test-models.ts index 821a2e6..84121a7 100644 --- a/tests/test-models.ts +++ b/tests/test-models.ts @@ -126,6 +126,24 @@ describe("commandCodeModelsFromApiResponse()", () => { assert.equal(modelSupportsImageInput("unknown-new-model"), false) }) + it("prefers host-resolved input modalities over the catalog snapshot", () => { + // A model published upstream after the pinned CLI release is absent from the + // generated catalog, so the host's resolved modalities must win. + assert.deepEqual(inputModalitiesForModel("unknown-new-model"), ["text"]) + assert.deepEqual(inputModalitiesForModel("unknown-new-model", ["text", "image"]), [ + "text", + "image", + ]) + assert.equal(modelSupportsImageInput("unknown-new-model", ["text", "image"]), true) + + const catalogVision = Object.keys(MODEL_INPUT_MODALITIES)[0] + // A host that narrows a catalogued vision model back to text stays authoritative. + assert.deepEqual(inputModalitiesForModel(catalogVision, ["text"]), ["text"]) + assert.equal(modelSupportsImageInput(catalogVision, ["text"]), false) + // An absent host list falls back to the catalog. + assert.deepEqual(inputModalitiesForModel(catalogVision, []), ["text", "image"]) + }) + it("tracks reasoning independently from selectable effort levels", () => { const reasoningModels = Object.keys(MODEL_REASONING) const effortModels = Object.keys(MODEL_EFFORTS) diff --git a/tests/test-stream.ts b/tests/test-stream.ts index 61f4b13..9a8a3a6 100644 --- a/tests/test-stream.ts +++ b/tests/test-stream.ts @@ -206,6 +206,61 @@ describe("streamCommandCode — successful streams", () => { ) }) + it("forwards images on the generate transport for models the host advertises as vision-capable", async () => { + server.mockResponse({ + type: "success", + events: [JSON.stringify({ type: "finish", finishReason: "stop" })], + }) + const { streamCommandCode } = createTestDeps({ apiBase: server.baseUrl() }) + + // `unknown-new-model` is absent from the pinned capability catalog, so only + // the host's resolved `input` can unlock image content here. + await collectEvents( + streamCommandCode( + makeModel({ id: "unknown-new-model", input: ["text", "image"] }), + makeContext({ + messages: [ + { + role: "user", + content: [ + { type: "text", text: "inspect" }, + { type: "image", data: "aGVsbG8=", mimeType: "image/png" }, + ], + }, + ], + }), + { apiKey: "mock-key" }, + ), + ) + + assert.equal( + objectAt(server.lastRequestBody(), ["params", "messages", "0", "content", "1", "image"]), + "data:image/png;base64,aGVsbG8=", + ) + }) + + it("rejects images when the host narrows a catalogued vision model to text", async () => { + const { streamCommandCode } = createTestDeps({ apiBase: server.baseUrl() }) + + const events = await collectEvents( + streamCommandCode( + makeModel({ id: "gpt-5.6-luna", input: ["text"] }), + makeContext({ + messages: [ + { + role: "user", + content: [{ type: "image", data: "aGVsbG8=", mimeType: "image/png" }], + }, + ], + }), + { apiKey: "mock-key" }, + ), + ) + + assert.equal(events.at(-1)?.type, "error") + assert.equal(server.requestCount(), 0) + }) + it("forwards a tool-result image as a following user image for vision-capable models", async () => { server.mockResponse({ type: "success",