Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,8 @@

## Unreleased

- Honor host-resolved image input on the generate transport, including explicit text-only restrictions; fall back to catalog metadata only when host input is absent or empty. Filter unsupported modality strings without unsafe casts.

- Reset the generate transport's idle timeout on every received chunk, allowing active reasoning streams to exceed the timeout overall while still aborting stalled streams (#87).

- Batch consecutive tool-result images after all tool results on the generate transport, preventing interleaved user messages from breaking multi-tool turns with "Tool result is missing".
Expand Down
2 changes: 2 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -143,6 +143,8 @@ The provider advertises image input only for models marked with the `image` inpu

For vision-capable models, Pi's native provider adapters forward image blocks from user messages and tool results using the documented OpenAI or Anthropic message schema. Unknown and text-only models remain marked text-only in Pi.

The legacy generate transport resolves image support from the host's `model.input` when the host supplies it, and falls back to the capability snapshot otherwise. Because the snapshot is generated from a single CLI release, a host that marks a model as `["text", "image"]` — for example through a `models.yml` or `models.json` model override — can send images on both transports before the catalog catches up. A host that narrows a catalogued vision model to text is likewise honored.

## Pricing display

The Command Code Provider API does not currently include prices in its model catalog. This extension therefore keeps a static table for models with known prices so pi can display estimated request costs. DeepSeek V4 uses time-dependent rates; pi displays the documented off-peak rate, which applies for 17 hours per day.
Expand Down
2 changes: 1 addition & 1 deletion src/core.ts
Original file line number Diff line number Diff line change
Expand Up @@ -540,7 +540,7 @@ export function createStreamCommandCode(deps: CoreDependencies) {
const reasoningEffort = mappedReasoningEffort(model, options)
const timeoutMs = options?.timeoutMs

const allowImages = modelSupportsImageInput(model.id)
const allowImages = modelSupportsImageInput(model.id, model.input)
if (!allowImages) assertTextOnlyMessages(context.messages)

let body: unknown = {
Expand Down
23 changes: 20 additions & 3 deletions src/models.ts
Original file line number Diff line number Diff line change
Expand Up @@ -31,12 +31,29 @@ export type CommandCodeApi = "openai-completions" | "anthropic-messages"

const TEXT_INPUT_ONLY = ["text"] as const

export function inputModalitiesForModel(modelId: string): readonly CommandCodeInputType[] {
/**
* Input modalities for a model.
*
* The generated catalog is a snapshot of one Command Code CLI release, so a
* model published afterwards is absent and silently degrades to text-only.
* The host's resolved `model.input` is authoritative when present: it already
* carries `models.yml`/`models.json` overrides and is what decides whether the
* user can attach an image at all.
*/
export function inputModalitiesForModel(
modelId: string,
hostInput?: readonly string[],
): readonly CommandCodeInputType[] {
if (hostInput && hostInput.length > 0) {
return hostInput.filter(
(input): input is CommandCodeInputType => input === "text" || input === "image",
)
}
return MODEL_INPUT_MODALITIES[modelId] ?? TEXT_INPUT_ONLY
}

export function modelSupportsImageInput(modelId: string): boolean {
return inputModalitiesForModel(modelId).includes("image")
export function modelSupportsImageInput(modelId: string, hostInput?: readonly string[]): boolean {
return inputModalitiesForModel(modelId, hostInput).includes("image")
}

export type PiThinkingLevel = "off" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max"
Expand Down
2 changes: 2 additions & 0 deletions src/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -72,6 +72,8 @@ export interface ModelLike {
provider: string
maxTokens: number
cost: ModelCost
/** Input modalities the host advertises, including `models.yml`/`models.json` overrides. */
input?: readonly string[]
reasoning?: boolean
thinkingLevelMap?: Partial<Record<string, string | null>>
thinking?: {
Expand Down
21 changes: 21 additions & 0 deletions tests/test-models.ts
Original file line number Diff line number Diff line change
Expand Up @@ -126,6 +126,27 @@ describe("commandCodeModelsFromApiResponse()", () => {
assert.equal(modelSupportsImageInput("unknown-new-model"), false)
})

it("prefers host-resolved input modalities over the catalog snapshot", () => {
// A model published upstream after the pinned CLI release is absent from the
// generated catalog, so the host's resolved modalities must win.
assert.deepEqual(inputModalitiesForModel("unknown-new-model"), ["text"])
assert.deepEqual(inputModalitiesForModel("unknown-new-model", ["text", "image"]), [
"text",
"image",
])
assert.equal(modelSupportsImageInput("unknown-new-model", ["text", "image"]), true)

const catalogVision = Object.keys(MODEL_INPUT_MODALITIES)[0]
// A host that narrows a catalogued vision model back to text stays authoritative.
assert.deepEqual(inputModalitiesForModel(catalogVision, ["text"]), ["text"])
assert.equal(modelSupportsImageInput(catalogVision, ["text"]), false)
assert.deepEqual(inputModalitiesForModel(catalogVision, ["text", "audio"]), ["text"])
assert.deepEqual(inputModalitiesForModel(catalogVision, ["audio"]), [])
assert.equal(modelSupportsImageInput(catalogVision, ["audio"]), false)
// An absent host list falls back to the catalog.
assert.deepEqual(inputModalitiesForModel(catalogVision, []), ["text", "image"])
})

it("tracks reasoning independently from selectable effort levels", () => {
const reasoningModels = Object.keys(MODEL_REASONING)
const effortModels = Object.keys(MODEL_EFFORTS)
Expand Down
55 changes: 55 additions & 0 deletions tests/test-stream.ts
Original file line number Diff line number Diff line change
Expand Up @@ -206,6 +206,61 @@ describe("streamCommandCode — successful streams", () => {
)
})

it("forwards images on the generate transport for models the host advertises as vision-capable", async () => {
server.mockResponse({
type: "success",
events: [JSON.stringify({ type: "finish", finishReason: "stop" })],
})
const { streamCommandCode } = createTestDeps({ apiBase: server.baseUrl() })

// `unknown-new-model` is absent from the pinned capability catalog, so only
// the host's resolved `input` can unlock image content here.
await collectEvents(
streamCommandCode(
makeModel({ id: "unknown-new-model", input: ["text", "image"] }),
makeContext({
messages: [
{
role: "user",
content: [
{ type: "text", text: "inspect" },
{ type: "image", data: "aGVsbG8=", mimeType: "image/png" },
],
},
],
}),
{ apiKey: "mock-key" },
),
)

assert.equal(
objectAt(server.lastRequestBody(), ["params", "messages", "0", "content", "1", "image"]),
"data:image/png;base64,aGVsbG8=",
)
})

it("rejects images when the host narrows a catalogued vision model to text", async () => {
const { streamCommandCode } = createTestDeps({ apiBase: server.baseUrl() })

const events = await collectEvents(
streamCommandCode(
makeModel({ id: "gpt-5.6-luna", input: ["text"] }),
makeContext({
messages: [
{
role: "user",
content: [{ type: "image", data: "aGVsbG8=", mimeType: "image/png" }],
},
],
}),
{ apiKey: "mock-key" },
),
)

assert.equal(events.at(-1)?.type, "error")
assert.equal(server.requestCount(), 0)
})

it("forwards a tool-result image as a following user image for vision-capable models", async () => {
server.mockResponse({
type: "success",
Expand Down
Loading