diff --git a/.github/workflows/model-discovery.yml b/.github/workflows/model-discovery.yml index 48e0a101..95f412a8 100644 --- a/.github/workflows/model-discovery.yml +++ b/.github/workflows/model-discovery.yml @@ -53,6 +53,7 @@ jobs: # Only secrets present in repository settings become configured # provider credentials. Scope them to this step so checkout, setup, # and npm lifecycle scripts never receive provider keys. + AIMLAPI_API_KEY: ${{ secrets.AIMLAPI_API_KEY }} AINETCAFE_API_KEY: ${{ secrets.AINETCAFE_API_KEY }} ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} CEREBRAS_API_KEY: ${{ secrets.CEREBRAS_API_KEY }} diff --git a/README.md b/README.md index a4e412e4..afa59ee1 100644 --- a/README.md +++ b/README.md @@ -255,6 +255,18 @@ Linux installations support the Codex CLI. | Picker label | Model ID | Authentication | | --- | --- | --- | +| GPT-6.1 Sol (AI/ML API) | `aimlapi/gpt-6.1-sol` | AI/ML API key (`AIMLAPI_API_KEY`) | +| GPT-6 Sol (AI/ML API) | `aimlapi/gpt-6-sol` | AI/ML API key (`AIMLAPI_API_KEY`) | +| GPT-6 Luna (AI/ML API) | `aimlapi/gpt-6-luna` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Claude Opus 5.5 (AI/ML API) | `aimlapi/claude-opus-5.5` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Claude Sonnet 5.5 (AI/ML API) | `aimlapi/claude-sonnet-5.5` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Claude Sonnet 5 (AI/ML API) | `aimlapi/claude-sonnet-5` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Gemini 3.8 Flash (AI/ML API) | `aimlapi/gemini-3.8-flash` | AI/ML API key (`AIMLAPI_API_KEY`) | +| DeepSeek V4.1 Flash (AI/ML API) | `aimlapi/deepseek-v4.1-flash` | AI/ML API key (`AIMLAPI_API_KEY`) | +| GLM 5.2 (AI/ML API) | `aimlapi/glm-5.2` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Kimi K3 (AI/ML API) | `aimlapi/kimi-k3` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Grok 4.7 (AI/ML API) | `aimlapi/grok-4.7` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Qwen3.7 Max (AI/ML API) | `aimlapi/qwen3.7-max` | AI/ML API key (`AIMLAPI_API_KEY`) | | K2.7 Coding Highspeed (OAuth) | `kimi-oauth/kimi-for-coding-highspeed` | Existing Kimi Code CLI OAuth session | | K2.7 Coding (OAuth) | `kimi-oauth/kimi-for-coding` | Existing Kimi Code CLI OAuth session | | Kimi K3 (OAuth) | `kimi-oauth/k3` | Existing Kimi Code CLI OAuth session | diff --git a/apps/macos/ModelRouterTray/Resources/PROVIDER-ICON-SOURCES.md b/apps/macos/ModelRouterTray/Resources/PROVIDER-ICON-SOURCES.md index add58a16..09980351 100644 --- a/apps/macos/ModelRouterTray/Resources/PROVIDER-ICON-SOURCES.md +++ b/apps/macos/ModelRouterTray/Resources/PROVIDER-ICON-SOURCES.md @@ -29,6 +29,7 @@ blue accent and DeepSeek's original blue mark are preserved. | NanoGPT | https://nano-gpt.com/ | https://nano-gpt.com/favicon.ico (same official diamond mark as https://nano-gpt.com/logo.png) | | Google Cloud Vertex AI | https://cloud.google.com/vertex-ai/ | https://www.google.com/s2/favicons?domain=cloud.google.com&sz=128 (Google mark, bundled as `google.svg`) | | StepFun | https://www.stepfun.com/ | https://www.stepfun.com/step_favicon.svg | +| AI/ML API | https://aimlapi.com/ | Official hexagon mark from the AI/ML API app (`sidebar-logo.svg`) | The Z.AI, Qwen, Ollama, Cline, MiniMax, and Meta AI marks were fetched on 2026-08-15. The Venice, Nous Research, and OpenRouter marks were fetched on diff --git a/apps/macos/ModelRouterTray/Sources/IslandOverlay.swift b/apps/macos/ModelRouterTray/Sources/IslandOverlay.swift index df12a12e..634fa59c 100644 --- a/apps/macos/ModelRouterTray/Sources/IslandOverlay.swift +++ b/apps/macos/ModelRouterTray/Sources/IslandOverlay.swift @@ -1085,6 +1085,7 @@ struct ProviderIcon: View { if providerID == "venice" { return "venice" } if providerID == "nousresearch" { return "nousresearch" } if providerID == "openrouter" { return "openrouter" } + if providerID == "aimlapi" { return "aimlapi" } if providerID == "nano-gpt" { return "nano-gpt" } // opencode-free plus the opencode-go API/Messages/Responses routes. if providerID.hasPrefix("opencode") { return "opencode-free" } @@ -1104,7 +1105,7 @@ struct ProviderIcon: View { private var assetExtension: String { // Keyed off the asset, not the provider id, so every route sharing a mark // (opencode-go and friends) resolves the same file type. - ["github-copilot", "chutes", "google", "opencode-free", "kilo-free", "nano-gpt", "stepfun"] + ["aimlapi", "github-copilot", "chutes", "google", "opencode-free", "kilo-free", "nano-gpt", "stepfun"] .contains(assetName ?? "") ? "svg" : "png" } @@ -1125,6 +1126,7 @@ struct ProviderIcon: View { if providerID == "venice" { return "Venice" } if providerID == "nousresearch" { return "Nous Research" } if providerID == "openrouter" { return "OpenRouter" } + if providerID == "aimlapi" { return "AI/ML API" } if providerID == "nano-gpt" { return "NanoGPT" } if providerID == "opencode-free" { return "OpenCode Free" } if providerID == "kilo-free" { return "Kilo Free" } diff --git a/apps/macos/ModelRouterTray/Sources/ModelRouterTrayApp.swift b/apps/macos/ModelRouterTray/Sources/ModelRouterTrayApp.swift index feb9636a..2f71dbfe 100644 --- a/apps/macos/ModelRouterTray/Sources/ModelRouterTrayApp.swift +++ b/apps/macos/ModelRouterTray/Sources/ModelRouterTrayApp.swift @@ -1624,6 +1624,7 @@ final class RouterStore: ObservableObject { "orca": "OrcaRouter", "venice": "Venice", "nousresearch": "Nous", + "aimlapi": "AI/ML API", "openrouter": "OpenRouter", ] diff --git a/apps/macos/ModelRouterTray/Sources/Resources/ProviderIcons/aimlapi.svg b/apps/macos/ModelRouterTray/Sources/Resources/ProviderIcons/aimlapi.svg new file mode 100644 index 00000000..39b10a33 --- /dev/null +++ b/apps/macos/ModelRouterTray/Sources/Resources/ProviderIcons/aimlapi.svg @@ -0,0 +1,12 @@ + + + + + + + + diff --git a/config/aimlapi/aimlapi.json b/config/aimlapi/aimlapi.json new file mode 100644 index 00000000..b6b565f7 --- /dev/null +++ b/config/aimlapi/aimlapi.json @@ -0,0 +1,24 @@ +{ + "version": 1, + "providers": [ + { + "id": "aimlapi", + "displayName": "AI/ML API", + "kind": "openai-compatible", + "ownedBy": "aimlapi", + "baseUrl": "https://api.aimlapi.com/v1", + "baseUrlEnv": "AIMLAPI_API_BASE_URL", + "credential": { + "environment": [ + "AIMLAPI_API_KEY" + ], + "file": "aimlapi-api-key.secret", + "legacyFiles": [], + "keychainServices": [ + "codex-router-aimlapi" + ], + "prompt": "AI/ML API key" + } + } + ] +} diff --git a/config/aimlapi/claude-opus-5.5.json b/config/aimlapi/claude-opus-5.5.json new file mode 100644 index 00000000..10fe1316 --- /dev/null +++ b/config/aimlapi/claude-opus-5.5.json @@ -0,0 +1,45 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/claude-opus-5.5", + "gatewayModel": "aimlapi-claude-opus-5-5", + "upstreamModel": "anthropic/claude-opus-5.5", + "provider": "aimlapi", + "listed": true, + "displayName": "Claude Opus 5.5 (AI/ML API)", + "description": "Anthropic Claude Opus 5.5 through AI/ML API's OpenAI-compatible surface; the full effort ladder up to max.", + "priority": 4, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + }, + { + "effort": "xhigh", + "description": "Extra deep reasoning" + }, + { + "effort": "max", + "description": "Maximum reasoning depth" + } + ], + "contextWindow": 1000000, + "autoCompact": 900000, + "inputModalities": [ + "text", + "image" + ], + "compHash": "aimlapi-claude-opus-5-5-v1" + } + ] +} diff --git a/config/aimlapi/claude-sonnet-5.5.json b/config/aimlapi/claude-sonnet-5.5.json new file mode 100644 index 00000000..18f914ab --- /dev/null +++ b/config/aimlapi/claude-sonnet-5.5.json @@ -0,0 +1,45 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/claude-sonnet-5.5", + "gatewayModel": "aimlapi-claude-sonnet-5-5", + "upstreamModel": "anthropic/claude-sonnet-5.5", + "provider": "aimlapi", + "listed": true, + "displayName": "Claude Sonnet 5.5 (AI/ML API)", + "description": "Anthropic Claude Sonnet 5.5 through AI/ML API; the current Sonnet, full effort ladder up to max.", + "priority": 5, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + }, + { + "effort": "xhigh", + "description": "Extra deep reasoning" + }, + { + "effort": "max", + "description": "Maximum reasoning depth" + } + ], + "contextWindow": 1000000, + "autoCompact": 900000, + "inputModalities": [ + "text", + "image" + ], + "compHash": "aimlapi-claude-sonnet-5-5-v1" + } + ] +} diff --git a/config/aimlapi/claude-sonnet-5.json b/config/aimlapi/claude-sonnet-5.json new file mode 100644 index 00000000..3cf2ca70 --- /dev/null +++ b/config/aimlapi/claude-sonnet-5.json @@ -0,0 +1,45 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/claude-sonnet-5", + "gatewayModel": "aimlapi-claude-sonnet-5", + "upstreamModel": "anthropic/claude-sonnet-5", + "provider": "aimlapi", + "listed": true, + "displayName": "Claude Sonnet 5 (AI/ML API)", + "description": "Anthropic Claude Sonnet 5 through AI/ML API; balanced coding model with the full effort ladder.", + "priority": 6, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + }, + { + "effort": "xhigh", + "description": "Extra deep reasoning" + }, + { + "effort": "max", + "description": "Maximum reasoning depth" + } + ], + "contextWindow": 1000000, + "autoCompact": 900000, + "inputModalities": [ + "text", + "image" + ], + "compHash": "aimlapi-claude-sonnet-5-v1" + } + ] +} diff --git a/config/aimlapi/deepseek-v4.1-flash.json b/config/aimlapi/deepseek-v4.1-flash.json new file mode 100644 index 00000000..4bafa304 --- /dev/null +++ b/config/aimlapi/deepseek-v4.1-flash.json @@ -0,0 +1,37 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/deepseek-v4.1-flash", + "gatewayModel": "aimlapi-deepseek-v4-1-flash", + "upstreamModel": "deepseek/deepseek-v4.1-flash", + "provider": "aimlapi", + "listed": true, + "displayName": "DeepSeek V4.1 Flash (AI/ML API)", + "description": "DeepSeek V4.1 Flash through AI/ML API; fast reasoning with tool calls and image understanding.", + "priority": 8, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + } + ], + "contextWindow": 1048576, + "autoCompact": 900000, + "inputModalities": [ + "text", + "image" + ], + "compHash": "aimlapi-deepseek-v4-1-flash-v1" + } + ] +} diff --git a/config/aimlapi/gemini-3.8-flash.json b/config/aimlapi/gemini-3.8-flash.json new file mode 100644 index 00000000..49b1bb8d --- /dev/null +++ b/config/aimlapi/gemini-3.8-flash.json @@ -0,0 +1,41 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/gemini-3.8-flash", + "gatewayModel": "aimlapi-gemini-3-8-flash", + "upstreamModel": "google/gemini-3.8-flash", + "provider": "aimlapi", + "listed": true, + "displayName": "Gemini 3.8 Flash (AI/ML API)", + "description": "Google Gemini 3.8 Flash through AI/ML API; 1M context, text and image input. Rejects xhigh, accepts max.", + "priority": 7, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + }, + { + "effort": "max", + "description": "Maximum reasoning depth" + } + ], + "contextWindow": 1048576, + "autoCompact": 900000, + "inputModalities": [ + "text", + "image" + ], + "compHash": "aimlapi-gemini-3-8-flash-v1" + } + ] +} diff --git a/config/aimlapi/glm-5.2.json b/config/aimlapi/glm-5.2.json new file mode 100644 index 00000000..fcd4d93a --- /dev/null +++ b/config/aimlapi/glm-5.2.json @@ -0,0 +1,48 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/glm-5.2", + "gatewayModel": "aimlapi-glm-5-2", + "upstreamModel": "zhipu/glm-5.2", + "provider": "aimlapi", + "listed": true, + "displayName": "GLM 5.2 (AI/ML API)", + "description": "Zhipu GLM 5.2 through AI/ML API; text only on this route, and the only model here that accepts every effort level.", + "priority": 9, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "minimal", + "description": "Minimal reasoning" + }, + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + }, + { + "effort": "xhigh", + "description": "Extra deep reasoning" + }, + { + "effort": "max", + "description": "Maximum reasoning depth" + } + ], + "contextWindow": 1000000, + "autoCompact": 900000, + "inputModalities": [ + "text" + ], + "compHash": "aimlapi-glm-5-2-v1" + } + ] +} diff --git a/config/aimlapi/gpt-6-luna.json b/config/aimlapi/gpt-6-luna.json new file mode 100644 index 00000000..2bf30f95 --- /dev/null +++ b/config/aimlapi/gpt-6-luna.json @@ -0,0 +1,37 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/gpt-6-luna", + "gatewayModel": "aimlapi-gpt-6-luna", + "upstreamModel": "openai/gpt-6-luna", + "provider": "aimlapi", + "listed": true, + "displayName": "GPT-6 Luna (AI/ML API)", + "description": "OpenAI GPT-6 Luna through AI/ML API; the fast, cost-efficient GPT-6 for high-volume work, 1.05M context.", + "priority": 3, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + } + ], + "contextWindow": 1050000, + "autoCompact": 900000, + "inputModalities": [ + "text", + "image" + ], + "compHash": "aimlapi-gpt-6-luna-v1" + } + ] +} diff --git a/config/aimlapi/gpt-6-sol.json b/config/aimlapi/gpt-6-sol.json new file mode 100644 index 00000000..765bad18 --- /dev/null +++ b/config/aimlapi/gpt-6-sol.json @@ -0,0 +1,37 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/gpt-6-sol", + "gatewayModel": "aimlapi-gpt-6-sol", + "upstreamModel": "openai/gpt-6-sol", + "provider": "aimlapi", + "listed": true, + "displayName": "GPT-6 Sol (AI/ML API)", + "description": "OpenAI GPT-6 Sol through AI/ML API; built for complex coding and agentic work, 1.05M context, text and image input.", + "priority": 2, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + } + ], + "contextWindow": 1050000, + "autoCompact": 900000, + "inputModalities": [ + "text", + "image" + ], + "compHash": "aimlapi-gpt-6-sol-v1" + } + ] +} diff --git a/config/aimlapi/gpt-6.1-sol.json b/config/aimlapi/gpt-6.1-sol.json new file mode 100644 index 00000000..42e593f6 --- /dev/null +++ b/config/aimlapi/gpt-6.1-sol.json @@ -0,0 +1,37 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/gpt-6.1-sol", + "gatewayModel": "aimlapi-gpt-6-1-sol", + "upstreamModel": "openai/gpt-6.1-sol", + "provider": "aimlapi", + "listed": true, + "displayName": "GPT-6.1 Sol (AI/ML API)", + "description": "OpenAI GPT-6.1 Sol through AI/ML API; the current coding and agentic flagship, 1.05M context, text and image input.", + "priority": 1, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + } + ], + "contextWindow": 1050000, + "autoCompact": 900000, + "inputModalities": [ + "text", + "image" + ], + "compHash": "aimlapi-gpt-6-1-sol-v1" + } + ] +} diff --git a/config/aimlapi/grok-4.7.json b/config/aimlapi/grok-4.7.json new file mode 100644 index 00000000..d49f329d --- /dev/null +++ b/config/aimlapi/grok-4.7.json @@ -0,0 +1,48 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/grok-4.7", + "gatewayModel": "aimlapi-grok-4-7", + "upstreamModel": "x-ai/grok-4-7", + "provider": "aimlapi", + "listed": true, + "displayName": "Grok 4.7 (AI/ML API)", + "description": "xAI Grok 4.7 through AI/ML API; text only on this route, full effort ladder.", + "priority": 11, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "minimal", + "description": "Minimal reasoning" + }, + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + }, + { + "effort": "xhigh", + "description": "Extra deep reasoning" + }, + { + "effort": "max", + "description": "Maximum reasoning depth" + } + ], + "contextWindow": 500000, + "autoCompact": 440000, + "inputModalities": [ + "text" + ], + "compHash": "aimlapi-grok-4-7-v1" + } + ] +} diff --git a/config/aimlapi/kimi-k3.json b/config/aimlapi/kimi-k3.json new file mode 100644 index 00000000..c62ed481 --- /dev/null +++ b/config/aimlapi/kimi-k3.json @@ -0,0 +1,37 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/kimi-k3", + "gatewayModel": "aimlapi-kimi-k3", + "upstreamModel": "moonshot/kimi-k3", + "provider": "aimlapi", + "listed": true, + "displayName": "Kimi K3 (AI/ML API)", + "description": "Moonshot Kimi K3 through AI/ML API; 1M context with image input. Accepts low, high and max only.", + "priority": 10, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + }, + { + "effort": "max", + "description": "Maximum reasoning depth" + } + ], + "contextWindow": 1048576, + "autoCompact": 900000, + "inputModalities": [ + "text", + "image" + ], + "compHash": "aimlapi-kimi-k3-v1" + } + ] +} diff --git a/config/aimlapi/qwen3.7-max.json b/config/aimlapi/qwen3.7-max.json new file mode 100644 index 00000000..5fe9ba4a --- /dev/null +++ b/config/aimlapi/qwen3.7-max.json @@ -0,0 +1,36 @@ +{ + "version": 1, + "models": [ + { + "slug": "aimlapi/qwen3.7-max", + "gatewayModel": "aimlapi-qwen3-7-max", + "upstreamModel": "alibaba/qwen3.7-max", + "provider": "aimlapi", + "listed": true, + "displayName": "Qwen3.7 Max (AI/ML API)", + "description": "Alibaba Qwen3.7 Max through AI/ML API; text only. Its thinking budget is large, so medium effort needs an output ceiling above 32k tokens.", + "priority": 12, + "defaultEffort": "high", + "reasoningLevels": [ + { + "effort": "low", + "description": "Faster reasoning" + }, + { + "effort": "medium", + "description": "Balanced reasoning" + }, + { + "effort": "high", + "description": "Deep reasoning" + } + ], + "contextWindow": 1000000, + "autoCompact": 900000, + "inputModalities": [ + "text" + ], + "compHash": "aimlapi-qwen3-7-max-v1" + } + ] +} diff --git a/docs-site/src/content/docs/providers/overview.md b/docs-site/src/content/docs/providers/overview.md index 5aac961c..be9ddd54 100644 --- a/docs-site/src/content/docs/providers/overview.md +++ b/docs-site/src/content/docs/providers/overview.md @@ -4,6 +4,18 @@ description: "Connect provider access without putting secrets in shell history." --- | Picker label | Model ID | Authentication | | --- | --- | --- | +| GPT-6.1 Sol (AI/ML API) | `aimlapi/gpt-6.1-sol` | AI/ML API key (`AIMLAPI_API_KEY`) | +| GPT-6 Sol (AI/ML API) | `aimlapi/gpt-6-sol` | AI/ML API key (`AIMLAPI_API_KEY`) | +| GPT-6 Luna (AI/ML API) | `aimlapi/gpt-6-luna` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Claude Opus 5.5 (AI/ML API) | `aimlapi/claude-opus-5.5` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Claude Sonnet 5.5 (AI/ML API) | `aimlapi/claude-sonnet-5.5` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Claude Sonnet 5 (AI/ML API) | `aimlapi/claude-sonnet-5` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Gemini 3.8 Flash (AI/ML API) | `aimlapi/gemini-3.8-flash` | AI/ML API key (`AIMLAPI_API_KEY`) | +| DeepSeek V4.1 Flash (AI/ML API) | `aimlapi/deepseek-v4.1-flash` | AI/ML API key (`AIMLAPI_API_KEY`) | +| GLM 5.2 (AI/ML API) | `aimlapi/glm-5.2` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Kimi K3 (AI/ML API) | `aimlapi/kimi-k3` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Grok 4.7 (AI/ML API) | `aimlapi/grok-4.7` | AI/ML API key (`AIMLAPI_API_KEY`) | +| Qwen3.7 Max (AI/ML API) | `aimlapi/qwen3.7-max` | AI/ML API key (`AIMLAPI_API_KEY`) | | K2.7 Coding Highspeed (OAuth) | `kimi-oauth/kimi-for-coding-highspeed` | Existing Kimi Code CLI OAuth session | | K2.7 Coding (OAuth) | `kimi-oauth/kimi-for-coding` | Existing Kimi Code CLI OAuth session | | Kimi K3 (OAuth) | `kimi-oauth/k3` | Existing Kimi Code CLI OAuth session | diff --git a/src/aimlapi-attribution.mjs b/src/aimlapi-attribution.mjs new file mode 100644 index 00000000..761340a9 --- /dev/null +++ b/src/aimlapi-attribution.mjs @@ -0,0 +1,68 @@ +// AI/ML API counts a request toward an integration only when it carries these +// headers, so every request the router sends to the gateway carries them -- +// and nothing else does. The gate is the *destination host*, not the provider +// id: `AIMLAPI_API_BASE_URL` can move the aimlapi provider somewhere else, and +// any other provider's baseUrl override can move it here. Attribution follows +// the address, which is the only thing that is true either way. +// +// Matching is exact on the hostname. A suffix test would hand the partner id +// to `api.aimlapi.com.example.net`, which is precisely the kind of host an +// override exists to be careful about. + +import { resolveProviderBaseUrl } from "./model-registry.mjs"; + +export const AIMLAPI_HOST = "api.aimlapi.com"; +export const AIMLAPI_SOURCE_HEADER = "x-aimlapi-source"; +export const AIMLAPI_PARTNER_HEADER = "x-aimlapi-partner-id"; + +export const AIMLAPI_SOURCE = "agent/codex-router"; +// Minted by AI/ML API, not by us. Empty means "not registered yet": an unknown +// partner id is accepted and silently dropped upstream, so sending a +// placeholder would look identical to working and count nothing. +export const AIMLAPI_PARTNER_ID = "part_iHUWvDUArvZvhexX3PzGBwPS"; + +const REFERER = "https://github.com/duolahypercho/codex-router"; +const TITLE = "Codex Router"; + +export function isAimlapiBaseUrl(baseUrl) { + if (typeof baseUrl !== "string" || !baseUrl.trim()) return false; + try { + return new URL(baseUrl).hostname.toLowerCase() === AIMLAPI_HOST; + } catch { + return false; + } +} + +export function aimlapiAttributionHeaders() { + const headers = { + "HTTP-Referer": REFERER, + "X-Title": TITLE, + "X-AIMLAPI-Source": AIMLAPI_SOURCE, + }; + if (AIMLAPI_PARTNER_ID) headers["X-AIMLAPI-Partner-ID"] = AIMLAPI_PARTNER_ID; + return headers; +} + +// Where a request to this endpoint will actually land, honouring the same +// baseUrl override the forwarder honours. A vertex endpoint builds its URL +// elsewhere and never reaches the gateway, so it resolves to nothing. +export function endpointDestination(endpoint, env = process.env) { + if (!endpoint || typeof endpoint !== "object") return ""; + if (endpoint.protocol === "vertex") return ""; + try { + return resolveProviderBaseUrl(endpoint, env).baseUrl || ""; + } catch { + return ""; + } +} + +// Returns whether anything was attached, so a caller (and a test) can tell +// "not our host" from "our host, nothing to send". Pass `endpoint` to have the +// destination resolved (including its env override), or `baseUrl` when the +// caller has already resolved it. +export function applyAimlapiAttributionHeaders(target, { endpoint, baseUrl, env } = {}) { + const destination = baseUrl !== undefined ? baseUrl : endpointDestination(endpoint, env); + if (!isAimlapiBaseUrl(destination)) return false; + Object.assign(target, aimlapiAttributionHeaders()); + return true; +} diff --git a/src/api-forwarder.mjs b/src/api-forwarder.mjs index 857550ba..310f1ffa 100644 --- a/src/api-forwarder.mjs +++ b/src/api-forwarder.mjs @@ -74,6 +74,7 @@ import { normalizeOpenAIRequest, } from "./openai-adapters.mjs"; import { threadIdFromHeaders } from "./codex-session-names.mjs"; +import { applyAimlapiAttributionHeaders } from "./aimlapi-attribution.mjs"; import { applyOpenCodeSessionHeaders, isOpenCodeProvider } from "./opencode-session.mjs"; import { clampOpenCodeMessageContent } from "./opencode-message-compat.mjs"; import { @@ -1582,6 +1583,9 @@ function upstreamHeaders(requestHeaders, body, apiKey, provider, extraHeaders = requestHeaders, body, }); + // Keyed on where the request actually goes, so a baseUrl override that moves + // a provider onto -- or off -- the AI/ML API gateway moves attribution with it. + applyAimlapiAttributionHeaders(headers, { endpoint }); // Content-Length is fetch's to compute. An explicit copy is at best // redundant, and the HTTP/1.1 dispatcher rejects the request outright // (UND_ERR_INVALID_ARG) when a caller-supplied value accompanies a body. diff --git a/src/model-discovery.mjs b/src/model-discovery.mjs index 14d2e2ef..b05567bb 100644 --- a/src/model-discovery.mjs +++ b/src/model-discovery.mjs @@ -22,6 +22,7 @@ import { RUNTIME_PROVIDERS, resolveProviderBaseUrl, } from "./model-registry.mjs"; +import { applyAimlapiAttributionHeaders } from "./aimlapi-attribution.mjs"; import { curatedModelBlockReason } from "./opencode-curation.mjs"; import { OPENCODE_SESSION_FALLBACKS, @@ -334,6 +335,7 @@ async function providerPayload(provider, identity) { fallback: OPENCODE_SESSION_FALLBACKS.discovery, }); } + applyAimlapiAttributionHeaders(headers, { baseUrl }); return fetchUntrustedModelCatalog(`${baseUrl}/models`, { headers, allowPrivate: Boolean(provider.keyless), diff --git a/src/provider-account-usage.mjs b/src/provider-account-usage.mjs index 0c8aca7e..cc209480 100644 --- a/src/provider-account-usage.mjs +++ b/src/provider-account-usage.mjs @@ -780,6 +780,7 @@ const QWEN_PLAN_DASHBOARD_URL = const OLLAMA_DASHBOARD_URL = "https://ollama.com/settings"; const COMMANDCODE_DASHBOARD_URL = "https://commandcode.ai/studio"; const OPENROUTER_DASHBOARD_URL = "https://openrouter.ai/settings/credits"; +const AIMLAPI_DASHBOARD_URL = "https://aimlapi.com/app/billing"; const VENICE_DASHBOARD_URL = "https://venice.ai/settings/api"; // Nous publishes no credits or usage route on the inference API (a 404 on both // /v1/credits and /v1/key), so the portal page is the only honest destination. @@ -967,6 +968,68 @@ async function veniceAccount(fetchImpl) { return account; } +// AI/ML API's GET /v1/key answers the calling key itself: its scopes, whether +// it is disabled, an optional spend `limit`, and month-to-date usage. It is a +// spend report, not a wallet -- there is no remaining balance to read -- so the +// money goes in the message rather than into a `balance` metric that would +// claim to be what is left. A numeric `limit` does make a real quota card. +export function aimlapiKeyMetrics(payload) { + const data = payload?.data; + const spent = numberValue(data?.monthly_usage_usd); + const limit = numberValue(data?.limit); + if (!Number.isFinite(limit) || limit <= 0 || !Number.isFinite(spent)) return []; + const metric = quotaMetric("Monthly spend", { limit, used: spent }, "USD"); + return metric ? [metric] : []; +} + +async function aimlapiAccount(fetchImpl) { + const provider = PROVIDERS.get("aimlapi"); + const credential = resolveProviderCredential(provider); + if (!credential) return { status: "not-configured", source: "official-api", metrics: [] }; + const fallback = (message) => ({ + ...withHeaderQuota("aimlapi", localOnly(message)), + dashboardUrl: AIMLAPI_DASHBOARD_URL, + }); + const baseURL = (process.env[provider.baseUrlEnv] || provider.baseUrl).replace(/\/+$/, ""); + if (new URL(baseURL).origin !== "https://api.aimlapi.com") { + return fallback("Account usage is unavailable for a custom AI/ML API endpoint"); + } + let key; + try { + key = await requestJson(`${baseURL}/key`, credential.value, {}, fetchImpl); + } catch { + return fallback("AI/ML API account usage is unavailable; showing router traffic"); + } + const metrics = aimlapiKeyMetrics(key); + const spent = numberValue(key?.data?.monthly_usage_usd); + const spentText = Number.isFinite(spent) + ? `Spent $${spent.toFixed(2)} this month` + : undefined; + if (key?.data?.disabled === true) { + return { + ...fallback("This AI/ML API key is disabled; re-enable it or create a new one"), + metrics, + ...(metrics.length ? { status: "available", source: "official-api" } : {}), + }; + } + if (!metrics.length) { + // An uncapped key has no percentage to show, so the honest card is router + // traffic plus the month-to-date figure in words. + return fallback( + spentText + ? `${spentText}. This key is uncapped, so there is no limit to show.` + : "AI/ML API reports usage only on the billing page; showing router traffic", + ); + } + return { + status: "available", + source: "official-api", + metrics, + dashboardUrl: AIMLAPI_DASHBOARD_URL, + ...(spentText ? { message: spentText } : {}), + }; +} + async function openRouterAccount(fetchImpl) { const provider = PROVIDERS.get("openrouter"); const credential = resolveProviderCredential(provider); @@ -1072,6 +1135,7 @@ async function accountUsageFor(providerId, fetchImpl) { } if (providerId === "commandcode") return await commandCodeAccount(fetchImpl); if (providerId === "venice") return await veniceAccount(fetchImpl); + if (providerId === "aimlapi") return await aimlapiAccount(fetchImpl); if (providerId === "openrouter") return await openRouterAccount(fetchImpl); if (providerId === "nousresearch") { // Nous Portal shows credits and the subscription tier only in the diff --git a/test/aimlapi-attribution.test.mjs b/test/aimlapi-attribution.test.mjs new file mode 100644 index 00000000..0352ab77 --- /dev/null +++ b/test/aimlapi-attribution.test.mjs @@ -0,0 +1,205 @@ +// AI/ML API counts a routed request toward Codex Router only when it carries +// the attribution headers, and it must never hand them to anyone else. The +// gate is the destination host, so these tests are written against the real +// registry entry and the real baseUrl override, not a hand-made descriptor: +// what is being held is "the shipped aimlapi provider resolves to the gateway, +// and an override moves attribution with it". +import assert from "node:assert/strict"; +import test from "node:test"; + +import { + AIMLAPI_SOURCE, + AIMLAPI_PARTNER_ID, + aimlapiAttributionHeaders, + applyAimlapiAttributionHeaders, + endpointDestination, + isAimlapiBaseUrl, +} from "../src/aimlapi-attribution.mjs"; +import { PROVIDERS } from "../src/model-registry.mjs"; + +test("the host gate matches api.aimlapi.com exactly", () => { + for (const url of [ + "https://api.aimlapi.com/v1", + "https://api.aimlapi.com", + "https://API.AIMLAPI.COM/v1", + ]) assert.equal(isAimlapiBaseUrl(url), true, url); + + for (const url of [ + // The suffix cases are the whole reason this is an equality test. + "https://api.aimlapi.com.example.net/v1", + "https://notapi.aimlapi.com/v1", + "https://aimlapi.com/v1", + "https://openrouter.ai/api/v1", + "http://127.0.0.1:1234/v1", + "api.aimlapi.com/v1", + "", + undefined, + null, + ]) assert.equal(isAimlapiBaseUrl(url), false, String(url)); +}); + +test("the shipped AI/ML API provider attributes its traffic", () => { + const provider = PROVIDERS.get("aimlapi"); + assert.ok(provider, "the registry must ship an aimlapi provider"); + assert.equal(endpointDestination(provider, {}), "https://api.aimlapi.com/v1"); + + const headers = { "User-Agent": "codex-router/test" }; + assert.equal(applyAimlapiAttributionHeaders(headers, { endpoint: provider, env: {} }), true); + assert.equal(headers["X-AIMLAPI-Source"], AIMLAPI_SOURCE); + assert.equal(headers["X-Title"], "Codex Router"); + assert.equal(headers["HTTP-Referer"], "https://github.com/duolahypercho/codex-router"); + // An unregistered partner id is silently ignored upstream, so an empty one is + // sent as nothing at all rather than as a value that looks like it works. + if (AIMLAPI_PARTNER_ID) { + assert.match(AIMLAPI_PARTNER_ID, /^part_[A-Za-z0-9]{1,64}$/); + assert.equal(headers["X-AIMLAPI-Partner-ID"], AIMLAPI_PARTNER_ID); + } else { + assert.equal("X-AIMLAPI-Partner-ID" in headers, false); + } +}); + +test("attribution follows the destination, not the provider id", () => { + const aimlapi = PROVIDERS.get("aimlapi"); + const openrouter = PROVIDERS.get("openrouter"); + + // Moved off the gateway by its own override: no attribution to a host that + // is not us. + const moved = {}; + assert.equal( + applyAimlapiAttributionHeaders(moved, { + endpoint: aimlapi, + env: { AIMLAPI_API_BASE_URL: "https://proxy.example.com/v1" }, + }), + false, + ); + assert.deepEqual(moved, {}); + + // Another provider pointed at the gateway is our traffic and is attributed. + const arrived = {}; + assert.equal( + applyAimlapiAttributionHeaders(arrived, { + endpoint: openrouter, + env: { OPENROUTER_API_BASE_URL: "https://api.aimlapi.com/v1" }, + }), + true, + ); + assert.equal(arrived["X-AIMLAPI-Source"], AIMLAPI_SOURCE); + + // Left alone, OpenRouter gets nothing. + const untouched = {}; + assert.equal( + applyAimlapiAttributionHeaders(untouched, { endpoint: openrouter, env: {} }), + false, + ); + assert.deepEqual(untouched, {}); +}); + +test("no other provider in the registry resolves onto the gateway", () => { + const attributed = [...PROVIDERS.values()] + .filter((provider) => isAimlapiBaseUrl(endpointDestination(provider, {}))) + .map((provider) => provider.id); + assert.deepEqual(attributed, ["aimlapi"]); +}); + +test("the header set is fixed", () => { + const expected = ["HTTP-Referer", "X-AIMLAPI-Source", "X-Title"]; + if (AIMLAPI_PARTNER_ID) expected.push("X-AIMLAPI-Partner-ID"); + assert.deepEqual(Object.keys(aimlapiAttributionHeaders()).sort(), expected.sort()); +}); + +// Everything above tests the decision. This tests the seam: a real forwarder +// process, the shipped registry entry, and a real HTTP request. It is pointed +// at a local upstream because that is the only destination a test can own -- +// which makes it the negative half by construction, and that is the half worth +// automating: a header that leaks to whatever host an override names is the +// failure that would go unnoticed. The positive half is verified live against +// the gateway (see the commit message). +import { spawn } from "node:child_process"; +import http from "node:http"; +import { mkdtempSync } from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +import { openPort } from "./port-pool.mjs"; + +const ROOT = fileURLToPath(new URL("..", import.meta.url)); +const INTERNAL_KEY = "test-internal-service-key-with-sufficient-length"; + +test("a routed AI/ML API model reaches its upstream, and an override takes attribution with it", async () => { + const seen = []; + const upstream = http.createServer(async (request, response) => { + const chunks = []; + for await (const chunk of request) chunks.push(chunk); + seen.push({ + url: request.url, + headers: request.headers, + body: JSON.parse(Buffer.concat(chunks).toString("utf8")), + }); + response.writeHead(200, { "Content-Type": "application/json" }); + response.end(JSON.stringify({ + id: "chatcmpl-test", + choices: [{ index: 0, message: { role: "assistant", content: "ok" }, finish_reason: "stop" }], + })); + }); + await new Promise((resolve) => upstream.listen(0, "127.0.0.1", resolve)); + + const port = await openPort(); + const forwarder = spawn(process.execPath, [path.join(ROOT, "src", "api-forwarder.mjs")], { + cwd: ROOT, + env: { + ...process.env, + CODEX_ROUTER_INTERNAL_KEY: INTERNAL_KEY, + CODEX_ROUTER_API_PORT: String(port), + MODEL_ROUTER_STATE_DIR: mkdtempSync(path.join(os.tmpdir(), "aimlapi-attribution-")), + CODEX_ROUTER_SHOW_ALL_MODELS: "1", + CODEX_ROUTER_QUIET: "1", + AIMLAPI_API_BASE_URL: `http://127.0.0.1:${upstream.address().port}`, + AIMLAPI_API_KEY: "TEST_AIMLAPI_API_KEY", + }, + stdio: ["ignore", "ignore", "pipe"], + }); + forwarder.stderr.setEncoding("utf8"); + let errors = ""; + forwarder.stderr.on("data", (chunk) => { errors += chunk; }); + + try { + const deadline = Date.now() + 10_000; + for (;;) { + if (forwarder.exitCode !== null) throw new Error(`forwarder exited: ${errors}`); + try { + const health = await fetch(`http://127.0.0.1:${port}/health`, { + headers: { Authorization: `Bearer ${INTERNAL_KEY}` }, + }); + if (health.ok) break; + } catch { + // not listening yet + } + if (Date.now() > deadline) throw new Error(`forwarder never came up: ${errors}`); + await new Promise((resolve) => setTimeout(resolve, 50)); + } + + const response = await fetch(`http://127.0.0.1:${port}/v1/chat/completions`, { + method: "POST", + headers: { Authorization: `Bearer ${INTERNAL_KEY}`, "Content-Type": "application/json" }, + body: JSON.stringify({ + model: "aimlapi-gpt-6-sol", + messages: [{ role: "user", content: "hi" }], + }), + }); + assert.equal(response.status, 200, errors); + + assert.equal(seen.length, 1); + // The registry entry routes, and it sends the gateway's own model id. + assert.equal(seen[0].body.model, "openai/gpt-6-sol"); + assert.equal(seen[0].headers.authorization, "Bearer TEST_AIMLAPI_API_KEY"); + // Not the gateway, so not a word about who we are. + for (const name of ["x-aimlapi-source", "x-aimlapi-partner-id", "x-title", "http-referer"]) { + assert.equal(name in seen[0].headers, false, `${name} leaked to a non-gateway host`); + } + } finally { + forwarder.kill("SIGTERM"); + await new Promise((resolve) => forwarder.once("exit", resolve)); + await new Promise((resolve) => upstream.close(resolve)); + } +}); diff --git a/test/provider-catalogs.test.mjs b/test/provider-catalogs.test.mjs index d1d7a432..98036cc4 100644 --- a/test/provider-catalogs.test.mjs +++ b/test/provider-catalogs.test.mjs @@ -11,8 +11,8 @@ import { test("every selectable provider remains a canonical UI family", () => { const canonical = [...PROVIDERS.values()].filter((provider) => !provider.variantOf); - assert.equal(canonical.length, 44); - assert.equal(PROVIDERS.size, 52); + assert.equal(canonical.length, 45); + assert.equal(PROVIDERS.size, 53); }); test("catalog capability comes from backend provider definitions", () => { diff --git a/test/registry.test.mjs b/test/registry.test.mjs index 552f2e33..2792005f 100644 --- a/test/registry.test.mjs +++ b/test/registry.test.mjs @@ -34,6 +34,18 @@ test("provider registry exposes configured API and OAuth model families", () => assert.deepEqual( LISTED_MODELS.map((model) => model.slug), [ + "aimlapi/claude-opus-5.5", + "aimlapi/claude-sonnet-5.5", + "aimlapi/claude-sonnet-5", + "aimlapi/deepseek-v4.1-flash", + "aimlapi/gemini-3.8-flash", + "aimlapi/glm-5.2", + "aimlapi/gpt-6-luna", + "aimlapi/gpt-6-sol", + "aimlapi/gpt-6.1-sol", + "aimlapi/grok-4.7", + "aimlapi/kimi-k3", + "aimlapi/qwen3.7-max", "ainetcafe/kimi-k3", "anthropic-api/claude-opus-4.8", "antigravity-oauth/gemini-3.1-pro",