diff --git a/CHANGELOG.md b/CHANGELOG.md index fb10efc..3a7847f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,8 @@ ## Unreleased +- Honor host-resolved image input on the generate transport, including explicit text-only restrictions; fall back to catalog metadata only when host input is absent or empty. Filter unsupported modality strings without unsafe casts. + - Reset the generate transport's idle timeout on every received chunk, allowing active reasoning streams to exceed the timeout overall while still aborting stalled streams (#87). - Batch consecutive tool-result images after all tool results on the generate transport, preventing interleaved user messages from breaking multi-tool turns with "Tool result is missing". diff --git a/README.md b/README.md index 6949106..ec04fca 100644 --- a/README.md +++ b/README.md @@ -143,6 +143,8 @@ The provider advertises image input only for models marked with the `image` inpu For vision-capable models, Pi's native provider adapters forward image blocks from user messages and tool results using the documented OpenAI or Anthropic message schema. Unknown and text-only models remain marked text-only in Pi. +The legacy generate transport resolves image support from the host's `model.input` when the host supplies it, and falls back to the capability snapshot otherwise. Because the snapshot is generated from a single CLI release, a host that marks a model as `["text", "image"]` — for example through a `models.yml` or `models.json` model override — can send images on both transports before the catalog catches up. A host that narrows a catalogued vision model to text is likewise honored. + ## Pricing display The Command Code Provider API does not currently include prices in its model catalog. This extension therefore keeps a static table for models with known prices so pi can display estimated request costs. DeepSeek V4 uses time-dependent rates; pi displays the documented off-peak rate, which applies for 17 hours per day. diff --git a/src/core.ts b/src/core.ts index cc6fe04..51e2d5b 100644 --- a/src/core.ts +++ b/src/core.ts @@ -540,7 +540,7 @@ export function createStreamCommandCode(deps: CoreDependencies) { const reasoningEffort = mappedReasoningEffort(model, options) const timeoutMs = options?.timeoutMs - const allowImages = modelSupportsImageInput(model.id) + const allowImages = modelSupportsImageInput(model.id, model.input) if (!allowImages) assertTextOnlyMessages(context.messages) let body: unknown = { diff --git a/src/models.ts b/src/models.ts index 0d8ffe9..df3dd71 100644 --- a/src/models.ts +++ b/src/models.ts @@ -31,12 +31,29 @@ export type CommandCodeApi = "openai-completions" | "anthropic-messages" const TEXT_INPUT_ONLY = ["text"] as const -export function inputModalitiesForModel(modelId: string): readonly CommandCodeInputType[] { +/** + * Input modalities for a model. + * + * The generated catalog is a snapshot of one Command Code CLI release, so a + * model published afterwards is absent and silently degrades to text-only. + * The host's resolved `model.input` is authoritative when present: it already + * carries `models.yml`/`models.json` overrides and is what decides whether the + * user can attach an image at all. + */ +export function inputModalitiesForModel( + modelId: string, + hostInput?: readonly string[], +): readonly CommandCodeInputType[] { + if (hostInput && hostInput.length > 0) { + return hostInput.filter( + (input): input is CommandCodeInputType => input === "text" || input === "image", + ) + } return MODEL_INPUT_MODALITIES[modelId] ?? TEXT_INPUT_ONLY } -export function modelSupportsImageInput(modelId: string): boolean { - return inputModalitiesForModel(modelId).includes("image") +export function modelSupportsImageInput(modelId: string, hostInput?: readonly string[]): boolean { + return inputModalitiesForModel(modelId, hostInput).includes("image") } export type PiThinkingLevel = "off" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max" diff --git a/src/types.ts b/src/types.ts index d10f823..351f857 100644 --- a/src/types.ts +++ b/src/types.ts @@ -72,6 +72,8 @@ export interface ModelLike { provider: string maxTokens: number cost: ModelCost + /** Input modalities the host advertises, including `models.yml`/`models.json` overrides. */ + input?: readonly string[] reasoning?: boolean thinkingLevelMap?: Partial> thinking?: { diff --git a/tests/test-models.ts b/tests/test-models.ts index b221aa0..6dc1f80 100644 --- a/tests/test-models.ts +++ b/tests/test-models.ts @@ -126,6 +126,27 @@ describe("commandCodeModelsFromApiResponse()", () => { assert.equal(modelSupportsImageInput("unknown-new-model"), false) }) + it("prefers host-resolved input modalities over the catalog snapshot", () => { + // A model published upstream after the pinned CLI release is absent from the + // generated catalog, so the host's resolved modalities must win. + assert.deepEqual(inputModalitiesForModel("unknown-new-model"), ["text"]) + assert.deepEqual(inputModalitiesForModel("unknown-new-model", ["text", "image"]), [ + "text", + "image", + ]) + assert.equal(modelSupportsImageInput("unknown-new-model", ["text", "image"]), true) + + const catalogVision = Object.keys(MODEL_INPUT_MODALITIES)[0] + // A host that narrows a catalogued vision model back to text stays authoritative. + assert.deepEqual(inputModalitiesForModel(catalogVision, ["text"]), ["text"]) + assert.equal(modelSupportsImageInput(catalogVision, ["text"]), false) + assert.deepEqual(inputModalitiesForModel(catalogVision, ["text", "audio"]), ["text"]) + assert.deepEqual(inputModalitiesForModel(catalogVision, ["audio"]), []) + assert.equal(modelSupportsImageInput(catalogVision, ["audio"]), false) + // An absent host list falls back to the catalog. + assert.deepEqual(inputModalitiesForModel(catalogVision, []), ["text", "image"]) + }) + it("tracks reasoning independently from selectable effort levels", () => { const reasoningModels = Object.keys(MODEL_REASONING) const effortModels = Object.keys(MODEL_EFFORTS) diff --git a/tests/test-stream.ts b/tests/test-stream.ts index 61f4b13..9a8a3a6 100644 --- a/tests/test-stream.ts +++ b/tests/test-stream.ts @@ -206,6 +206,61 @@ describe("streamCommandCode — successful streams", () => { ) }) + it("forwards images on the generate transport for models the host advertises as vision-capable", async () => { + server.mockResponse({ + type: "success", + events: [JSON.stringify({ type: "finish", finishReason: "stop" })], + }) + const { streamCommandCode } = createTestDeps({ apiBase: server.baseUrl() }) + + // `unknown-new-model` is absent from the pinned capability catalog, so only + // the host's resolved `input` can unlock image content here. + await collectEvents( + streamCommandCode( + makeModel({ id: "unknown-new-model", input: ["text", "image"] }), + makeContext({ + messages: [ + { + role: "user", + content: [ + { type: "text", text: "inspect" }, + { type: "image", data: "aGVsbG8=", mimeType: "image/png" }, + ], + }, + ], + }), + { apiKey: "mock-key" }, + ), + ) + + assert.equal( + objectAt(server.lastRequestBody(), ["params", "messages", "0", "content", "1", "image"]), + "data:image/png;base64,aGVsbG8=", + ) + }) + + it("rejects images when the host narrows a catalogued vision model to text", async () => { + const { streamCommandCode } = createTestDeps({ apiBase: server.baseUrl() }) + + const events = await collectEvents( + streamCommandCode( + makeModel({ id: "gpt-5.6-luna", input: ["text"] }), + makeContext({ + messages: [ + { + role: "user", + content: [{ type: "image", data: "aGVsbG8=", mimeType: "image/png" }], + }, + ], + }), + { apiKey: "mock-key" }, + ), + ) + + assert.equal(events.at(-1)?.type, "error") + assert.equal(server.requestCount(), 0) + }) + it("forwards a tool-result image as a following user image for vision-capable models", async () => { server.mockResponse({ type: "success",