From 1714693cda0b31846a1c61585066505c871cab3e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:24:41 +0000 Subject: [PATCH 001/392] chore(sync): update OpenRouter model catalog (#7408) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index 07784758cd2..dc34ba0ccf7 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.1485 -output = 0.594 -cache_read = 0.004455 +input = 0.15 +output = 0.6 +cache_read = 0.015 [limit] context = 1_048_576 From 4c2a945bb69df654bfa788994e6d15d85289e4ab Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:24:59 +0000 Subject: [PATCH 002/392] chore(sync): update Kilo model catalog (#7409) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 1a05f7dcedb..673b730eb39 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.1485 -output = 0.594 -cache_read = 0.004455 +input = 0.15 +output = 0.6 +cache_read = 0.015 [limit] context = 1_048_576 From ae065079d839cd1e4d824007aa8acef87ff5465f Mon Sep 17 00:00:00 2001 From: 7Sageer <7sageer@djwcb.cn> Date: Fri, 18 Sep 2026 19:32:13 +0800 Subject: [PATCH 003/392] feat(kimi-for-coding): add kimi-code-plan-global provider, switch both sites to OpenAI-compatible - Add kimi-code-plan-global provider ("Kimi For Coding (kimi.ai)") for the global deployment at https://api.kimi.ai/coding/v1 with the same 4-model catalog (kimi-for-coding, kimi-for-coding-highspeed, k3, k3-256k) - Rename kimi-for-coding to kimi-code-plan-cn ("Kimi For Coding (kimi.com)") - Switch kimi-code-plan-cn (api.kimi.com) from @ai-sdk/anthropic to @ai-sdk/openai-compatible (/v1/chat/completions) - Add interleaved reasoning_content to all models on both providers --- .../logo.svg | 0 .../models/k3-256k.toml | 3 ++ .../models/k3.toml | 3 ++ .../models/kimi-for-coding-highspeed.toml | 3 ++ .../models/kimi-for-coding.toml | 5 ++++ providers/kimi-code-plan-cn/provider.toml | 10 +++++++ providers/kimi-code-plan-global/logo.svg | 3 ++ .../kimi-code-plan-global/models/k3-256k.toml | 29 +++++++++++++++++++ .../kimi-code-plan-global/models/k3.toml | 23 +++++++++++++++ .../models/kimi-for-coding-highspeed.toml | 17 +++++++++++ .../models/kimi-for-coding.toml | 25 ++++++++++++++++ providers/kimi-code-plan-global/provider.toml | 9 ++++++ providers/kimi-for-coding/provider.toml | 10 ------- .../models/kimi-k3.toml | 2 +- 14 files changed, 131 insertions(+), 11 deletions(-) rename providers/{kimi-for-coding => kimi-code-plan-cn}/logo.svg (100%) rename providers/{kimi-for-coding => kimi-code-plan-cn}/models/k3-256k.toml (93%) rename providers/{kimi-for-coding => kimi-code-plan-cn}/models/k3.toml (91%) rename providers/{kimi-for-coding => kimi-code-plan-cn}/models/kimi-for-coding-highspeed.toml (81%) rename providers/{kimi-for-coding => kimi-code-plan-cn}/models/kimi-for-coding.toml (73%) create mode 100644 providers/kimi-code-plan-cn/provider.toml create mode 100644 providers/kimi-code-plan-global/logo.svg create mode 100644 providers/kimi-code-plan-global/models/k3-256k.toml create mode 100644 providers/kimi-code-plan-global/models/k3.toml create mode 100644 providers/kimi-code-plan-global/models/kimi-for-coding-highspeed.toml create mode 100644 providers/kimi-code-plan-global/models/kimi-for-coding.toml create mode 100644 providers/kimi-code-plan-global/provider.toml delete mode 100644 providers/kimi-for-coding/provider.toml diff --git a/providers/kimi-for-coding/logo.svg b/providers/kimi-code-plan-cn/logo.svg similarity index 100% rename from providers/kimi-for-coding/logo.svg rename to providers/kimi-code-plan-cn/logo.svg diff --git a/providers/kimi-for-coding/models/k3-256k.toml b/providers/kimi-code-plan-cn/models/k3-256k.toml similarity index 93% rename from providers/kimi-for-coding/models/k3-256k.toml rename to providers/kimi-code-plan-cn/models/k3-256k.toml index 2e7cd5d2d1f..3d9c4ba810e 100644 --- a/providers/kimi-for-coding/models/k3-256k.toml +++ b/providers/kimi-code-plan-cn/models/k3-256k.toml @@ -22,3 +22,6 @@ context = 262_144 [modalities] input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-for-coding/models/k3.toml b/providers/kimi-code-plan-cn/models/k3.toml similarity index 91% rename from providers/kimi-for-coding/models/k3.toml rename to providers/kimi-code-plan-cn/models/k3.toml index dad0dadecd8..d0dbb42ade9 100644 --- a/providers/kimi-for-coding/models/k3.toml +++ b/providers/kimi-code-plan-cn/models/k3.toml @@ -16,3 +16,6 @@ input = 0 output = 0 cache_read = 0 cache_write = 0 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-for-coding/models/kimi-for-coding-highspeed.toml b/providers/kimi-code-plan-cn/models/kimi-for-coding-highspeed.toml similarity index 81% rename from providers/kimi-for-coding/models/kimi-for-coding-highspeed.toml rename to providers/kimi-code-plan-cn/models/kimi-for-coding-highspeed.toml index 32d5102804a..407bcf710f8 100644 --- a/providers/kimi-for-coding/models/kimi-for-coding-highspeed.toml +++ b/providers/kimi-code-plan-cn/models/kimi-for-coding-highspeed.toml @@ -10,3 +10,6 @@ cache_write = 0 [limit] output = 32_768 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-for-coding/models/kimi-for-coding.toml b/providers/kimi-code-plan-cn/models/kimi-for-coding.toml similarity index 73% rename from providers/kimi-for-coding/models/kimi-for-coding.toml rename to providers/kimi-code-plan-cn/models/kimi-for-coding.toml index 2dc81d0d56d..570fea3fd8a 100644 --- a/providers/kimi-for-coding/models/kimi-for-coding.toml +++ b/providers/kimi-code-plan-cn/models/kimi-for-coding.toml @@ -1,4 +1,6 @@ # https://www.kimi.com/code/docs/en/third-party-tools/opencode.html +# OpenAI-compatible chat completions on api.kimi.com (verified 2026-09-18 with +# kimi-code OAuth credentials; reasoning returned via reasoning_content). # Toggle: thinking.type = "enabled" | "disabled" | "adaptive" # Effort: output_config.effort = "low" | "high" | "max" (default: "max") # Retain the existing output limit; the K2.8 announcement only specifies context. @@ -20,3 +22,6 @@ cache_write = 0 [limit] output = 32_768 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-code-plan-cn/provider.toml b/providers/kimi-code-plan-cn/provider.toml new file mode 100644 index 00000000000..ab7b8962d83 --- /dev/null +++ b/providers/kimi-code-plan-cn/provider.toml @@ -0,0 +1,10 @@ +# NOTE: api.kimi.com/coding serves BOTH Anthropic (/v1/messages) and +# OpenAI-compatible (/v1/chat/completions) protocols. All four models +# verified on /chat/completions 2026-09-18 with kimi-code OAuth credentials +# (reasoning returned via reasoning_content). The global deployment +# (api.kimi.ai) is the separate kimi-code-plan-global provider. +name = "Kimi For Coding (kimi.com)" +env = ["KIMI_API_KEY"] +npm = "@ai-sdk/openai-compatible" +doc = "https://www.kimi.com/code/docs/en/kimi-code/models.html" +api = "https://api.kimi.com/coding/v1" diff --git a/providers/kimi-code-plan-global/logo.svg b/providers/kimi-code-plan-global/logo.svg new file mode 100644 index 00000000000..d41029890d2 --- /dev/null +++ b/providers/kimi-code-plan-global/logo.svg @@ -0,0 +1,3 @@ + + + diff --git a/providers/kimi-code-plan-global/models/k3-256k.toml b/providers/kimi-code-plan-global/models/k3-256k.toml new file mode 100644 index 00000000000..36af21989bd --- /dev/null +++ b/providers/kimi-code-plan-global/models/k3-256k.toml @@ -0,0 +1,29 @@ +# Verified 2026-09-18 against https://api.kimi.ai/coding/v1/chat/completions +# with kimi-code OAuth credentials; reasoning returned via reasoning_content. +# k3-256k is the 256K-context variant of Kimi K3 on the Kimi For Coding +# endpoint. Unlike full K3 it accepts no video input (image only). +# reasoning_options per the Kimi Code docs: +# reasoning_effort = "low" | "high" | "max" (default "high") +# Cost is zeroed per this provider's subscription convention. +base_model = "moonshotai/kimi-k3" +name = "Kimi K3-256K" +description = "256K-context version of Kimi K3, reducing token consumption for shorter coding sessions" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 + +[limit] +context = 262_144 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-code-plan-global/models/k3.toml b/providers/kimi-code-plan-global/models/k3.toml new file mode 100644 index 00000000000..e2cc98df82b --- /dev/null +++ b/providers/kimi-code-plan-global/models/k3.toml @@ -0,0 +1,23 @@ +# Verified 2026-09-18 against https://api.kimi.ai/coding/v1/chat/completions +# with kimi-code OAuth credentials; reasoning returned via reasoning_content. +# reasoning_options mirror the Moonshot AI platform API surface: +# thinking.type = "enabled" | "disabled" | "adaptive" +# output_config.effort = "low" | "high" | "max" +# Cost is zeroed per this provider's subscription convention. +base_model = "moonshotai/kimi-k3" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-code-plan-global/models/kimi-for-coding-highspeed.toml b/providers/kimi-code-plan-global/models/kimi-for-coding-highspeed.toml new file mode 100644 index 00000000000..ff4e80c9104 --- /dev/null +++ b/providers/kimi-code-plan-global/models/kimi-for-coding-highspeed.toml @@ -0,0 +1,17 @@ +# Verified 2026-09-18 against https://api.kimi.ai/coding/v1/chat/completions +# with kimi-code OAuth credentials; reasoning returned via reasoning_content. +base_model = "moonshotai/kimi-k2.7-code-highspeed" +name = "Kimi For Coding HighSpeed" +reasoning_options = [] + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 + +[limit] +output = 32_768 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-code-plan-global/models/kimi-for-coding.toml b/providers/kimi-code-plan-global/models/kimi-for-coding.toml new file mode 100644 index 00000000000..766cc5bfb99 --- /dev/null +++ b/providers/kimi-code-plan-global/models/kimi-for-coding.toml @@ -0,0 +1,25 @@ +# Verified 2026-09-18 against https://api.kimi.ai/coding/v1/chat/completions +# with kimi-code OAuth credentials; reasoning returned via reasoning_content. +# Toggle: thinking.type = "enabled" | "disabled" | "adaptive" +# Effort: output_config.effort = "low" | "high" | "max" (default: "max") +base_model = "moonshotai/kimi-k2.8-preview" +name = "kimi-for-coding" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 + +[limit] +output = 32_768 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-code-plan-global/provider.toml b/providers/kimi-code-plan-global/provider.toml new file mode 100644 index 00000000000..cf4da1d06d8 --- /dev/null +++ b/providers/kimi-code-plan-global/provider.toml @@ -0,0 +1,9 @@ +# Global deployment of Kimi For Coding on api.kimi.ai (separate host and OAuth +# issuer auth.kimi.ai). OpenAI-compatible chat completions; catalog verified +# via GET /models on 2026-09-18 (kimi-for-coding, kimi-for-coding-highspeed, +# k3, k3-256k) with kimi-code OAuth credentials. +name = "Kimi For Coding (kimi.ai)" +env = ["KIMI_API_KEY"] +npm = "@ai-sdk/openai-compatible" +doc = "https://www.kimi.ai/code/docs/en/kimi-code/models.html" +api = "https://api.kimi.ai/coding/v1" diff --git a/providers/kimi-for-coding/provider.toml b/providers/kimi-for-coding/provider.toml deleted file mode 100644 index f8d71f33de3..00000000000 --- a/providers/kimi-for-coding/provider.toml +++ /dev/null @@ -1,10 +0,0 @@ -# NOTE: api.kimi.com/coding serves BOTH Anthropic (/v1/messages) and -# OpenAI-compatible (/v1/chat/completions) protocols (verified 2026-07). -# Registered as @ai-sdk/anthropic: the Messages surface is the officially -# documented one (see doc) and round-trips reasoning natively via -# thinking blocks. -name = "Kimi For Coding" -env = ["KIMI_API_KEY"] -npm = "@ai-sdk/anthropic" -doc = "https://www.kimi.com/code/docs/en/kimi-code/models.html" -api = "https://api.kimi.com/coding/v1" diff --git a/providers/volcengine-coding-plan/models/kimi-k3.toml b/providers/volcengine-coding-plan/models/kimi-k3.toml index 18f097450bd..7b7fb0c9e7e 100644 --- a/providers/volcengine-coding-plan/models/kimi-k3.toml +++ b/providers/volcengine-coding-plan/models/kimi-k3.toml @@ -2,7 +2,7 @@ # (accessed 2026-09-11), listed alongside kimi-k2.7-code. Subscription tier: # no per-token price. # reasoning_options mirror the lab (providers/moonshotai/models/kimi-k3.toml) -# and the kimi-for-coding relay peer. Wire fields on this provider's OpenAI +# and the kimi-code-plan-cn relay peer. Wire fields on this provider's OpenAI # path (provider.toml): POST /api/coding/v3/chat/completions takes # thinking.type = enabled|disabled plus reasoning_effort = low|high|max # (`adaptive` is the Moonshot lab surface, not accepted here); the separate From 37482b15ca21f840961b770c0a6c0317de1e8345 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 13:25:43 +0000 Subject: [PATCH 004/392] chore(sync): update Kilo model catalog (#7411) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 1 + providers/kilo/models/moonshotai/kimi-k3.toml | 6 +++--- providers/kilo/models/z-ai/glm-5.2.toml | 1 - .../kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- .../models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- providers/kilo/models/~z-ai/glm-latest.toml | 10 +++++----- 7 files changed, 17 insertions(+), 17 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index 007b654034e..d4d63d57a55 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -12,3 +12,4 @@ cache_read = 0.044 [limit] context = 1_048_576 +output = 393_216 diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index 3af9cc14392..55b1dc1c908 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 2.0705 +output = 11.5948 +cache_read = 0.240178 [limit] output = 943_718 diff --git a/providers/kilo/models/z-ai/glm-5.2.toml b/providers/kilo/models/z-ai/glm-5.2.toml index f8addeabfc8..bb89edf75cf 100644 --- a/providers/kilo/models/z-ai/glm-5.2.toml +++ b/providers/kilo/models/z-ai/glm-5.2.toml @@ -12,4 +12,3 @@ cache_read = 0.26 [limit] context = 1_048_576 -output = 163_840 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index b14064bf897..957bc0a82fa 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.65868 +output = 1.97604 +cache_read = 0.020958 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 396a029912a..26d07e5604f 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -17,7 +17,7 @@ values = ["none", "low", "high", "max"] [cost] input = 0.0558 output = 0.1767 -cache_read = 0.0088 +cache_read = 0.0102 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 9bbf3c7be77..2fbb91bbcaa 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 2.0705 +output = 11.5948 +cache_read = 0.240178 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 38aa9fed430..2fc8e4f5ea7 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8775 -output = 2.97 -cache_read = 0.1755 +input = 0.9 +output = 3 +cache_read = 0.15 [limit] -context = 262_144 -output = 235_929 +context = 1_000_000 +output = 131_072 [modalities] input = ["text"] From 119bb5a58314f25ba8e1d8bf877cceb4716aaa7d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 13:25:50 +0000 Subject: [PATCH 005/392] chore(sync): update OpenRouter model catalog (#7412) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 ++++--- providers/openrouter/models/moonshotai/kimi-k3.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.2.toml | 7 +++---- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- .../models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- providers/openrouter/models/~z-ai/glm-latest.toml | 8 ++++---- 8 files changed, 25 insertions(+), 25 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 6f817c5a0a0..141d4b41ca6 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.088606 -output = 0.177212 -cache_read = 0.017721 +input = 0.049 +output = 0.098 +cache_read = 0.0098 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..c307f78addc 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.65868 +output = 1.97604 +cache_read = 0.020958 [limit] context = 1_048_576 +output = 393_216 diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index 618c6fb2bbd..4bf49607f67 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 2.0705 +output = 11.5948 +cache_read = 0.240178 [limit] output = 943_718 diff --git a/providers/openrouter/models/z-ai/glm-5.2.toml b/providers/openrouter/models/z-ai/glm-5.2.toml index 426a4d8f486..4fa1fa642a3 100644 --- a/providers/openrouter/models/z-ai/glm-5.2.toml +++ b/providers/openrouter/models/z-ai/glm-5.2.toml @@ -13,10 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.5625 -output = 1.8 -cache_read = 0.105 +input = 0.5614 +output = 1.7644 +cache_read = 0.10426 [limit] context = 1_048_576 -output = 163_840 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index ea09a0d12b6..50d7ce57993 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.65868 +output = 1.97604 +cache_read = 0.020958 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 7356cb84c22..ea3e3a2af96 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -22,7 +22,7 @@ values = ["low", "high", "max"] [cost] input = 0.0558 output = 0.1767 -cache_read = 0.0088 +cache_read = 0.0102 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index d1d254d74de..2ca20af632f 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 2.0705 +output = 11.5948 +cache_read = 0.240178 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 3e3ce108967..0a7b3acfd67 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8775 -output = 2.97 -cache_read = 0.1755 +input = 0.9 +output = 3 +cache_read = 0.15 [limit] context = 1_310_720 -output = 235_929 +output = 131_072 [modalities] input = ["text"] From c7777f749524662fd06478554eb17922535d7c89 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 14:27:20 +0000 Subject: [PATCH 006/392] chore(sync): update OpenRouter model catalog (#7413) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 +++---- providers/openrouter/models/moonshotai/kimi-k3.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.2.toml | 6 +++--- .../models/~deepseek/deepseek-flash-latest.toml | 2 +- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 8 ++++---- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- 8 files changed, 23 insertions(+), 24 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 141d4b41ca6..a9469f150d2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.049 -output = 0.098 -cache_read = 0.0098 +input = 0.04984 +output = 0.09968 +cache_read = 0.009968 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index c307f78addc..7c5828b16e2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,10 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.65868 -output = 1.97604 -cache_read = 0.020958 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index 4bf49607f67..618c6fb2bbd 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.0705 -output = 11.5948 -cache_read = 0.240178 +input = 2.1 +output = 10.95 +cache_read = 0.23 [limit] output = 943_718 diff --git a/providers/openrouter/models/z-ai/glm-5.2.toml b/providers/openrouter/models/z-ai/glm-5.2.toml index 4fa1fa642a3..12ed56b2bd6 100644 --- a/providers/openrouter/models/z-ai/glm-5.2.toml +++ b/providers/openrouter/models/z-ai/glm-5.2.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.5614 -output = 1.7644 -cache_read = 0.10426 +input = 0.5544 +output = 1.7424 +cache_read = 0.10296 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index dc34ba0ccf7..98fd23a037b 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -26,7 +26,7 @@ cache_read = 0.015 [limit] context = 1_048_576 -output = 943_718 +output = 393_216 [modalities] input = ["text", "image"] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 50d7ce57993..e0b3ef90d2e 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.65868 -output = 1.97604 -cache_read = 0.020958 +input = 0.5808 +output = 1.7424 +cache_read = 0.05808 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index ea3e3a2af96..b207a5a59a3 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.0558 -output = 0.1767 -cache_read = 0.0102 +input = 0.055 +output = 0.165 +cache_read = 0.00175 [limit] context = 1_310_720 -output = 943_718 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 2ca20af632f..d1d254d74de 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.0705 -output = 11.5948 -cache_read = 0.240178 +input = 2.1 +output = 10.95 +cache_read = 0.23 [limit] context = 1_048_576 From 268d33960435f9214a292630e926fe3f6a542b75 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 14:27:23 +0000 Subject: [PATCH 007/392] chore(sync): update Kilo model catalog (#7415) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 1 - providers/kilo/models/moonshotai/kimi-k3.toml | 6 +++--- providers/kilo/models/z-ai/glm-5.2.toml | 3 ++- .../kilo/models/~deepseek/deepseek-flash-latest.toml | 4 ++-- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- .../models/~deepseek/deepseek-v4-flash-latest.toml | 10 +++++----- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- 7 files changed, 19 insertions(+), 19 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index d4d63d57a55..007b654034e 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -12,4 +12,3 @@ cache_read = 0.044 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index 55b1dc1c908..3af9cc14392 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.0705 -output = 11.5948 -cache_read = 0.240178 +input = 2.1 +output = 10.95 +cache_read = 0.23 [limit] output = 943_718 diff --git a/providers/kilo/models/z-ai/glm-5.2.toml b/providers/kilo/models/z-ai/glm-5.2.toml index bb89edf75cf..dcb3cc80d7f 100644 --- a/providers/kilo/models/z-ai/glm-5.2.toml +++ b/providers/kilo/models/z-ai/glm-5.2.toml @@ -11,4 +11,5 @@ output = 4.4 cache_read = 0.26 [limit] -context = 1_048_576 +context = 1_024_000 +output = 128_000 diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 673b730eb39..f491c2853c8 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -20,8 +20,8 @@ output = 0.6 cache_read = 0.015 [limit] -context = 1_048_576 -output = 943_718 +context = 1_000_000 +output = 393_216 [modalities] input = ["text", "image"] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 957bc0a82fa..bd4537b7471 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,12 +15,12 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.65868 -output = 1.97604 -cache_read = 0.020958 +input = 0.5808 +output = 1.7424 +cache_read = 0.05808 [limit] -context = 1_048_576 +context = 1_000_000 output = 393_216 [modalities] diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 26d07e5604f..844f1914d6e 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.0558 -output = 0.1767 -cache_read = 0.0102 +input = 0.055 +output = 0.165 +cache_read = 0.00175 [limit] -context = 1_048_576 -output = 943_718 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 2fbb91bbcaa..9bbf3c7be77 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 2.0705 -output = 11.5948 -cache_read = 0.240178 +input = 2.1 +output = 10.95 +cache_read = 0.23 [limit] context = 1_048_576 From 19a6cfba6cc3ef772309e09a80e6c20d1a2606ec Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 14:27:27 +0000 Subject: [PATCH 008/392] chore(sync): update OVHcloud AI Endpoints model catalog (#7414) Co-authored-by: opencode-agent[bot] --- providers/ovhcloud/models/qwen3-32b.toml | 26 ------------------------ 1 file changed, 26 deletions(-) delete mode 100644 providers/ovhcloud/models/qwen3-32b.toml diff --git a/providers/ovhcloud/models/qwen3-32b.toml b/providers/ovhcloud/models/qwen3-32b.toml deleted file mode 100644 index b117c1bb3b6..00000000000 --- a/providers/ovhcloud/models/qwen3-32b.toml +++ /dev/null @@ -1,26 +0,0 @@ -# Put `/no_think` in prompt content to disable the model's default reasoning. -# https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/qwen-3-32b/ - -name = "Qwen3-32B" -description = "Reasoning model for deliberate analysis, multi-step problem solving, and tool use" -release_date = "2025-07-16" -last_updated = "2025-07-16" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -open_weights = true - -[cost] -input = 0.09 -output = 0.25 - -[limit] -context = 32_768 -output = 32_768 - -[modalities] -input = ["text"] -output = ["text"] From 511b7612e9b764df50381aff89c239800b6421ad Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Fri, 18 Sep 2026 23:09:47 +0800 Subject: [PATCH 009/392] feat(aihubmix): add claude-fable-5-1 (#7416) * feat(aihubmix): add claude-fable-5-1 AIHubMix serves Claude Fable 5.1 but the catalog only had Fable 5. The numbers come from the provider's public model listing; limit, modalities and the rest inherit from the Anthropic base entry. Co-Authored-By: Claude Opus 5 * docs(aihubmix): cite the toggle and effort wire paths AGENTS.md requires a leading top-of-file comment with the exact request field for every toggle; the header only cited pricing. Co-Authored-By: Claude Opus 5 --------- Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- .../aihubmix/models/claude-fable-5-1.toml | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) create mode 100644 providers/aihubmix/models/claude-fable-5-1.toml diff --git a/providers/aihubmix/models/claude-fable-5-1.toml b/providers/aihubmix/models/claude-fable-5-1.toml new file mode 100644 index 00000000000..b29ef102523 --- /dev/null +++ b/providers/aihubmix/models/claude-fable-5-1.toml @@ -0,0 +1,18 @@ +# Toggle: $.thinking.type = "enabled"|"disabled"|"adaptive" on the Anthropic-compatible +# /v1/messages path; $.enable_thinking = true|false on /v1/chat/completions. +# Effort: $.output_config.effort on /v1/messages; $.reasoning_effort on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message +# Pricing, context window, max output and the effort levels come from +# https://aihubmix.com/api/v1/models?type=llm (accessed 2026-09-18), which lists this +# model on both faces. Cache read is a quarter of Claude Fable 5's rate; input, output +# and cache write are unchanged from it, matching the sibling entry in this directory. +base_model = "anthropic/claude-fable-5-1" +structured_output = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +interleaved = true + +[cost] +input = 11 +output = 55 +cache_read = 0.275 +cache_write = 13.75 From be7ef0c6b57f85e67f68be9a720b23997afa5622 Mon Sep 17 00:00:00 2001 From: "C.C." Date: Fri, 18 Sep 2026 23:10:46 +0800 Subject: [PATCH 010/392] feat(vivgrid): add claude-opus-5, claude-sonnet-5, jev; use anthropic sdk for claude models (#7405) * feat(vivgrid): add claude-opus-5, claude-sonnet-5; use anthropic sdk for claude models Co-Authored-By: Claude Opus 5 * feat(vivgrid): add jev; add doc source links to model files Co-Authored-By: Claude Opus 5 --------- Co-authored-by: Claude Opus 5 --- providers/vivgrid/models/claude-fable-5-1.toml | 5 ++--- providers/vivgrid/models/claude-fable-5.toml | 5 ++--- providers/vivgrid/models/claude-opus-5.toml | 14 ++++++++++++++ providers/vivgrid/models/claude-sonnet-5.toml | 16 ++++++++++++++++ providers/vivgrid/models/deepseek-v4-flash.toml | 2 ++ providers/vivgrid/models/glm-5.2.toml | 2 ++ providers/vivgrid/models/glm-5.3.toml | 2 ++ providers/vivgrid/models/gpt-5.1-codex-max.toml | 2 ++ providers/vivgrid/models/gpt-5.1-codex.toml | 2 ++ providers/vivgrid/models/gpt-5.2-codex.toml | 2 ++ providers/vivgrid/models/gpt-5.3-codex.toml | 2 ++ providers/vivgrid/models/gpt-5.6-luna.toml | 2 ++ providers/vivgrid/models/gpt-5.6-sol.toml | 2 ++ providers/vivgrid/models/gpt-5.6-terra.toml | 2 ++ providers/vivgrid/models/gpt-6-astra.toml | 2 ++ providers/vivgrid/models/jev.toml | 7 +++++++ providers/vivgrid/models/kimi-k3.toml | 2 ++ 17 files changed, 65 insertions(+), 6 deletions(-) create mode 100644 providers/vivgrid/models/claude-opus-5.toml create mode 100644 providers/vivgrid/models/claude-sonnet-5.toml create mode 100644 providers/vivgrid/models/jev.toml diff --git a/providers/vivgrid/models/claude-fable-5-1.toml b/providers/vivgrid/models/claude-fable-5-1.toml index 471e3cb571d..756c74d20a3 100644 --- a/providers/vivgrid/models/claude-fable-5-1.toml +++ b/providers/vivgrid/models/claude-fable-5-1.toml @@ -1,7 +1,6 @@ # Pricing, context window and max output: # https://docs.vivgrid.com/models/claude-fable-5-1 (accessed 2026-09-07) -# Cached input is $0.50/MTok. Vivgrid publishes no cache-write rate and states -# pricing matches the original provider, so Anthropic's $12.50/MTok applies. +# Effort: output_config.effort = low|medium|high|xhigh|max base_model = "anthropic/claude-fable-5-1" structured_output = true reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] @@ -13,4 +12,4 @@ cache_read = 0.5 cache_write = 12.5 [provider] -npm = "@ai-sdk/openai-compatible" \ No newline at end of file +npm = "@ai-sdk/anthropic" diff --git a/providers/vivgrid/models/claude-fable-5.toml b/providers/vivgrid/models/claude-fable-5.toml index 912e2ff029a..31f1362efa1 100644 --- a/providers/vivgrid/models/claude-fable-5.toml +++ b/providers/vivgrid/models/claude-fable-5.toml @@ -1,7 +1,6 @@ # Pricing, context window and max output: # https://docs.vivgrid.com/models/claude-fable-5 (accessed 2026-09-07) -# Cached input is $1.25/MTok. Vivgrid publishes no cache-write rate and states -# pricing matches the original provider, so Anthropic's $12.50/MTok applies. +# Effort: output_config.effort = low|medium|high|xhigh|max base_model = "anthropic/claude-fable-5" structured_output = true reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] @@ -13,4 +12,4 @@ cache_read = 1.25 cache_write = 12.5 [provider] -npm = "@ai-sdk/openai-compatible" \ No newline at end of file +npm = "@ai-sdk/anthropic" diff --git a/providers/vivgrid/models/claude-opus-5.toml b/providers/vivgrid/models/claude-opus-5.toml new file mode 100644 index 00000000000..fbc0e693efc --- /dev/null +++ b/providers/vivgrid/models/claude-opus-5.toml @@ -0,0 +1,14 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/claude-opus-5 (accessed 2026-09-07) +base_model = "anthropic/claude-opus-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 5.0 +output = 25.0 +cache_read = 0.50 +cache_write = 6.25 + +[provider] +npm = "@ai-sdk/anthropic" diff --git a/providers/vivgrid/models/claude-sonnet-5.toml b/providers/vivgrid/models/claude-sonnet-5.toml new file mode 100644 index 00000000000..44da13b30e5 --- /dev/null +++ b/providers/vivgrid/models/claude-sonnet-5.toml @@ -0,0 +1,16 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/claude-sonnet-5 (accessed 2026-09-07) +# Toggle: thinking.type = adaptive|disabled +# Effort: output_config.effort = low|medium|high|xhigh|max +base_model = "anthropic/claude-sonnet-5" +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +structured_output = true + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[provider] +npm = "@ai-sdk/anthropic" \ No newline at end of file diff --git a/providers/vivgrid/models/deepseek-v4-flash.toml b/providers/vivgrid/models/deepseek-v4-flash.toml index 38f0598e8fe..464e7b70d2d 100644 --- a/providers/vivgrid/models/deepseek-v4-flash.toml +++ b/providers/vivgrid/models/deepseek-v4-flash.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/deepseek-v4-flash (accessed 2026-09-18) base_model = "deepseek/deepseek-v4-flash-0731" name = "DeepSeek V4 Flash" diff --git a/providers/vivgrid/models/glm-5.2.toml b/providers/vivgrid/models/glm-5.2.toml index ad841133eeb..2ba2cd2bf76 100644 --- a/providers/vivgrid/models/glm-5.2.toml +++ b/providers/vivgrid/models/glm-5.2.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/glm-5.2 (accessed 2026-09-18) base_model = "zhipuai/glm-5.2" [[reasoning_options]] diff --git a/providers/vivgrid/models/glm-5.3.toml b/providers/vivgrid/models/glm-5.3.toml index 704b2199e03..da960ec71fa 100644 --- a/providers/vivgrid/models/glm-5.3.toml +++ b/providers/vivgrid/models/glm-5.3.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/glm-5.3 (accessed 2026-09-18) base_model = "zhipuai/glm-5.3" reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] diff --git a/providers/vivgrid/models/gpt-5.1-codex-max.toml b/providers/vivgrid/models/gpt-5.1-codex-max.toml index b96035a695e..68ec50cbee8 100644 --- a/providers/vivgrid/models/gpt-5.1-codex-max.toml +++ b/providers/vivgrid/models/gpt-5.1-codex-max.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.1-codex-max (accessed 2026-09-18) name = "GPT-5.1 Codex Max" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" family = "gpt-codex" diff --git a/providers/vivgrid/models/gpt-5.1-codex.toml b/providers/vivgrid/models/gpt-5.1-codex.toml index c8b1158f20c..f6e6f2fa8ee 100644 --- a/providers/vivgrid/models/gpt-5.1-codex.toml +++ b/providers/vivgrid/models/gpt-5.1-codex.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.1-codex (accessed 2026-09-18) name = "GPT-5.1 Codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" family = "gpt-codex" diff --git a/providers/vivgrid/models/gpt-5.2-codex.toml b/providers/vivgrid/models/gpt-5.2-codex.toml index e28e266ee10..7319407808d 100644 --- a/providers/vivgrid/models/gpt-5.2-codex.toml +++ b/providers/vivgrid/models/gpt-5.2-codex.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.2-codex (accessed 2026-09-18) name = "GPT-5.2 Codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" family = "gpt-codex" diff --git a/providers/vivgrid/models/gpt-5.3-codex.toml b/providers/vivgrid/models/gpt-5.3-codex.toml index 715972403fe..454bc79f662 100644 --- a/providers/vivgrid/models/gpt-5.3-codex.toml +++ b/providers/vivgrid/models/gpt-5.3-codex.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.3-codex (accessed 2026-09-18) name = "GPT-5.3 Codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" family = "gpt-codex" diff --git a/providers/vivgrid/models/gpt-5.6-luna.toml b/providers/vivgrid/models/gpt-5.6-luna.toml index fb81c98d331..9c4852173d2 100644 --- a/providers/vivgrid/models/gpt-5.6-luna.toml +++ b/providers/vivgrid/models/gpt-5.6-luna.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.6-luna (accessed 2026-09-18) base_model = "openai/gpt-5.6-luna" name = "GPT 5.6 Luna" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" diff --git a/providers/vivgrid/models/gpt-5.6-sol.toml b/providers/vivgrid/models/gpt-5.6-sol.toml index df3250cf9b6..65eb349ce72 100644 --- a/providers/vivgrid/models/gpt-5.6-sol.toml +++ b/providers/vivgrid/models/gpt-5.6-sol.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.6-sol (accessed 2026-09-18) base_model = "openai/gpt-5.6-sol" name = "GPT 5.6 Sol" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" diff --git a/providers/vivgrid/models/gpt-5.6-terra.toml b/providers/vivgrid/models/gpt-5.6-terra.toml index 385029c25bd..371995729d7 100644 --- a/providers/vivgrid/models/gpt-5.6-terra.toml +++ b/providers/vivgrid/models/gpt-5.6-terra.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.6-terra (accessed 2026-09-18) base_model = "openai/gpt-5.6-terra" name = "GPT 5.6 Terra" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" diff --git a/providers/vivgrid/models/gpt-6-astra.toml b/providers/vivgrid/models/gpt-6-astra.toml index afdc38e7f28..2a246582bfd 100644 --- a/providers/vivgrid/models/gpt-6-astra.toml +++ b/providers/vivgrid/models/gpt-6-astra.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-6-astra (accessed 2026-09-18) base_model = "openai/gpt-6-astra" reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] diff --git a/providers/vivgrid/models/jev.toml b/providers/vivgrid/models/jev.toml new file mode 100644 index 00000000000..f499745e785 --- /dev/null +++ b/providers/vivgrid/models/jev.toml @@ -0,0 +1,7 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/jev +base_model = "typesafe/jev-latest" + +[cost] +input = 0.042 +output = 0 diff --git a/providers/vivgrid/models/kimi-k3.toml b/providers/vivgrid/models/kimi-k3.toml index 7e11b4c39c2..1ca4fad3636 100644 --- a/providers/vivgrid/models/kimi-k3.toml +++ b/providers/vivgrid/models/kimi-k3.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/kimi-k3 (accessed 2026-09-18) base_model = "moonshotai/kimi-k3" description = "Kimi multimodal agent model for visual understanding, coding, and planning" reasoning_options = [] From a72e9666472c91973fcdc97d808029c2a60f4274 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 10:10:56 -0500 Subject: [PATCH 011/392] fix: [DeepSeek Lab] The weights of DeepSeek V4 Flash Vision Exp and DeepSeek V4.1 Flash are available on Hugging Face (#7407) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- models/deepseek/deepseek-v4-flash-vision-exp.toml | 10 ++++++++-- models/deepseek/deepseek-v4.1-flash.toml | 7 ++++++- 2 files changed, 14 insertions(+), 3 deletions(-) diff --git a/models/deepseek/deepseek-v4-flash-vision-exp.toml b/models/deepseek/deepseek-v4-flash-vision-exp.toml index ef5426b119c..18d7744f7c8 100644 --- a/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -1,14 +1,16 @@ +# Open weights (MIT): https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp name = "DeepSeek V4 Flash Vision Exp" description = "Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work" family = "deepseek-flash" release_date = "2026-08-21" -last_updated = "2026-08-21" +last_updated = "2026-09-01" attachment = true reasoning = true temperature = true tool_call = true structured_output = true -open_weights = false +open_weights = true +license = "MIT" [limit] context = 1_000_000 @@ -17,3 +19,7 @@ output = 384_000 [modalities] input = ["text", "image"] output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp" diff --git a/models/deepseek/deepseek-v4.1-flash.toml b/models/deepseek/deepseek-v4.1-flash.toml index aac72588623..68b2de09bb4 100644 --- a/models/deepseek/deepseek-v4.1-flash.toml +++ b/models/deepseek/deepseek-v4.1-flash.toml @@ -1,3 +1,4 @@ +# Open weights (MIT): https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash name = "DeepSeek V4.1 Flash" description = "DeepSeek V4.1 Flash model for reasoning and agentic coding" family = "deepseek-flash" @@ -18,4 +19,8 @@ output = 384_000 [modalities] input = ["text", "image"] -output = ["text"] \ No newline at end of file +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash" From 4056f866d2777df0e5d278740f8faf3f1ae9d8b7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=BC=A0=E5=9C=A3=E5=A5=87?= Date: Fri, 18 Sep 2026 23:11:11 +0800 Subject: [PATCH 012/392] feat(alibaba-cn): add deepseek-v4.1-flash (#7396) --- .../models/deepseek-v4.1-flash.toml | 26 +++++++++++++++++++ 1 file changed, 26 insertions(+) create mode 100644 providers/alibaba-cn/models/deepseek-v4.1-flash.toml diff --git a/providers/alibaba-cn/models/deepseek-v4.1-flash.toml b/providers/alibaba-cn/models/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..933dcb373c8 --- /dev/null +++ b/providers/alibaba-cn/models/deepseek-v4.1-flash.toml @@ -0,0 +1,26 @@ +# Sources (accessed 2026-09-18): +# https://bailian.console.aliyun.com/cn-beijing?tab=model#/model-market/detail/deepseek-v4.1-flash?serviceSite=asia-pacific-china +# https://help.aliyun.com/zh/model-studio/deepseek-api — capability table, +# reasoning_effort ladder, max_tokens default (393_216, shared with thinking) +# https://help.aliyun.com/zh/model-studio/billing-for-model-studio — Beijing +# peak/off-peak pricing: CNY 2/1 input, 8/4 output per 1M tokens with context +# cache discount; peak price converted to USD following the existing +# deepseek-v4-flash.toml convention on this host (off-peak is half). +# Hybrid reasoning: enable_thinking = true|false toggles thinking; +# reasoning_effort = low|high|max (default high) — low is only supported on +# v4.1-flash / v4-flash-0731 / v4-pro-0813, and v4.1-flash is the current +# low-tier bearer. reasoning_content streams in deltas. Responses API +# supported (Beijing and Singapore only). Limits and modalities are identical +# to the deepseek lab entry (1M context, 384_000 output, text+image input), +# so they are inherited and not restated. +base_model = "deepseek/deepseek-v4.1-flash" + +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "high", "max"] }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.278 +output = 1.111 +cache_read = 0.014 From b17f6fcd1bde3fb35f28213e2847f00e72b71049 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 10:14:46 -0500 Subject: [PATCH 013/392] chore(sync): update Vercel AI Gateway model catalog (#7382) * chore(sync): update Vercel AI Gateway model catalog * fix(vercel): correct GLM 5.3 FlashX metadata --------- Co-authored-by: opencode-agent[bot] Co-authored-by: Aiden Cline --- .../vercel/models/zai/glm-5.3-flashx.toml | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) create mode 100644 providers/vercel/models/zai/glm-5.3-flashx.toml diff --git a/providers/vercel/models/zai/glm-5.3-flashx.toml b/providers/vercel/models/zai/glm-5.3-flashx.toml new file mode 100644 index 00000000000..9cf4fc74365 --- /dev/null +++ b/providers/vercel/models/zai/glm-5.3-flashx.toml @@ -0,0 +1,18 @@ +# GLM-5.3-FlashX uses the same model and forced-thinking effort levels as Flash. +# https://docs.z.ai/guides/vlm/glm-5.3-flash +base_model = "zhipuai/glm-5.3-flash" +name = "GLM 5.3 FlashX" +release_date = "2026-09-18" +last_updated = "2026-09-18" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.37 +output = 1.25 +cache_read = 0.075 + +[modalities] +input = ["text", "image"] From 79bdd0da1da7bab187c850ff45dff0d834b365fe Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:25:57 +0000 Subject: [PATCH 014/392] chore(sync): update Kilo model catalog (#7420) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml | 1 + providers/kilo/models/z-ai/glm-5.2.toml | 3 +-- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index 007b654034e..d4d63d57a55 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -12,3 +12,4 @@ cache_read = 0.044 [limit] context = 1_048_576 +output = 393_216 diff --git a/providers/kilo/models/z-ai/glm-5.2.toml b/providers/kilo/models/z-ai/glm-5.2.toml index dcb3cc80d7f..bb89edf75cf 100644 --- a/providers/kilo/models/z-ai/glm-5.2.toml +++ b/providers/kilo/models/z-ai/glm-5.2.toml @@ -11,5 +11,4 @@ output = 4.4 cache_read = 0.26 [limit] -context = 1_024_000 -output = 128_000 +context = 1_048_576 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index bd4537b7471..522fff2ce46 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,12 +15,12 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.5808 -output = 1.7424 -cache_read = 0.05808 +input = 0.57816 +output = 1.73448 +cache_read = 0.018396 [limit] -context = 1_000_000 +context = 1_048_576 output = 393_216 [modalities] From 127aff4232309717d4198d5cc4087928a6da3af0 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:26:00 +0000 Subject: [PATCH 015/392] chore(sync): update Eden AI model catalog (#7421) Co-authored-by: opencode-agent[bot] --- providers/edenai/models/flexai/DeepSeek-V4-Flash-0731.toml | 2 +- .../infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml | 4 ++-- .../models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml | 4 ++-- providers/edenai/models/ionos/openai/gpt-oss-120b.toml | 4 ++-- providers/edenai/models/mistral/mistral-large-2512.toml | 6 +++--- .../edenai/models/scaleway/deepseek-v4-flash-0731.toml | 4 ++-- providers/edenai/models/scaleway/gpt-oss-120b.toml | 4 ++-- .../edenai/models/scaleway/llama-3.3-70b-instruct.toml | 4 ++-- providers/edenai/models/tensorx/moonshotai/kimi-k2.5.toml | 1 - 9 files changed, 16 insertions(+), 17 deletions(-) diff --git a/providers/edenai/models/flexai/DeepSeek-V4-Flash-0731.toml b/providers/edenai/models/flexai/DeepSeek-V4-Flash-0731.toml index 8a3564408ab..1e1e8029798 100644 --- a/providers/edenai/models/flexai/DeepSeek-V4-Flash-0731.toml +++ b/providers/edenai/models/flexai/DeepSeek-V4-Flash-0731.toml @@ -12,4 +12,4 @@ input = 0.065 output = 0.18 [limit] -context = 786_432 +context = 1_048_576 diff --git a/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml b/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml index d5d18a5d5f0..9b2ca243154 100644 --- a/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml +++ b/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml @@ -5,8 +5,8 @@ tool_call = false structured_output = false [cost] -input = 0.34443 -output = 0.45924 +input = 0.3438 +output = 0.4584 [limit] context = 100_000 diff --git a/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml b/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml index fc91f75adc5..b5b5ca537eb 100644 --- a/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml +++ b/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml @@ -4,5 +4,5 @@ tool_call = false structured_output = false [cost] -input = 0.746265 -output = 0.746265 +input = 0.7449 +output = 0.7449 diff --git a/providers/edenai/models/ionos/openai/gpt-oss-120b.toml b/providers/edenai/models/ionos/openai/gpt-oss-120b.toml index 488fd48dab9..fb25f513d77 100644 --- a/providers/edenai/models/ionos/openai/gpt-oss-120b.toml +++ b/providers/edenai/models/ionos/openai/gpt-oss-120b.toml @@ -8,5 +8,5 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.172215 -output = 0.746265 +input = 0.1719 +output = 0.7449 diff --git a/providers/edenai/models/mistral/mistral-large-2512.toml b/providers/edenai/models/mistral/mistral-large-2512.toml index 96902738ba4..6944f065750 100644 --- a/providers/edenai/models/mistral/mistral-large-2512.toml +++ b/providers/edenai/models/mistral/mistral-large-2512.toml @@ -2,9 +2,9 @@ base_model = "mistral/mistral-large-2512" structured_output = true [cost] -input = 0.55 -output = 1.65 -cache_read = 0.055 +input = 0.5 +output = 1.5 +cache_read = 0.05 [modalities] input = ["text", "image", "pdf"] diff --git a/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml b/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml index 2236e082a15..ed1db7ea77d 100644 --- a/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml +++ b/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml @@ -7,8 +7,8 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.45924 -output = 0.91848 +input = 0.4584 +output = 0.9168 [limit] context = 256_000 diff --git a/providers/edenai/models/scaleway/gpt-oss-120b.toml b/providers/edenai/models/scaleway/gpt-oss-120b.toml index c78acf705d7..df8da020b1b 100644 --- a/providers/edenai/models/scaleway/gpt-oss-120b.toml +++ b/providers/edenai/models/scaleway/gpt-oss-120b.toml @@ -7,8 +7,8 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.172215 -output = 0.68886 +input = 0.1719 +output = 0.6876 [limit] context = 128_000 diff --git a/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml b/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml index b6133a15c15..da84cd257a7 100644 --- a/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml +++ b/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml @@ -3,5 +3,5 @@ name = "Llama-3.3-70B-Instruct (Scaleway)" structured_output = false [cost] -input = 1.03329 -output = 1.03329 +input = 1.0314 +output = 1.0314 diff --git a/providers/edenai/models/tensorx/moonshotai/kimi-k2.5.toml b/providers/edenai/models/tensorx/moonshotai/kimi-k2.5.toml index 87d3cc945d7..fc327494824 100644 --- a/providers/edenai/models/tensorx/moonshotai/kimi-k2.5.toml +++ b/providers/edenai/models/tensorx/moonshotai/kimi-k2.5.toml @@ -1,6 +1,5 @@ base_model = "moonshotai/kimi-k2.5" name = "Kimi K2.5 (TensorX)" -structured_output = false reasoning_options = [] [cost] From 68147ad3c36ba3dd8abe24fa0cd8b9185492bf9d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:26:04 +0000 Subject: [PATCH 016/392] chore(sync): update OpenRouter model catalog (#7419) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 ++++--- providers/openrouter/models/meta/muse-glimmer-30b.toml | 4 ++-- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 3 files changed, 9 insertions(+), 8 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..7c435d69ac6 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.57816 +output = 1.73448 +cache_read = 0.018396 [limit] context = 1_048_576 +output = 393_216 diff --git a/providers/openrouter/models/meta/muse-glimmer-30b.toml b/providers/openrouter/models/meta/muse-glimmer-30b.toml index 12578d9072c..3b46ee2987a 100644 --- a/providers/openrouter/models/meta/muse-glimmer-30b.toml +++ b/providers/openrouter/models/meta/muse-glimmer-30b.toml @@ -10,8 +10,8 @@ type = "effort" values = ["low", "medium", "high", "xhigh"] [cost] -input = 0.3 -output = 1.1 +input = 0.35 +output = 1.5 cache_read = 0.04 [limit] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index e0b3ef90d2e..42e4245ad89 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5808 -output = 1.7424 -cache_read = 0.05808 +input = 0.57816 +output = 1.73448 +cache_read = 0.018396 [limit] context = 1_048_576 From b1be347e4a1464ee069ae9a383b4236e1227d286 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:26:14 +0000 Subject: [PATCH 017/392] chore(sync): update Vercel AI Gateway model catalog (#7418) Co-authored-by: opencode-agent[bot] --- providers/vercel/models/zai/glm-5.3-flashx.toml | 2 -- 1 file changed, 2 deletions(-) diff --git a/providers/vercel/models/zai/glm-5.3-flashx.toml b/providers/vercel/models/zai/glm-5.3-flashx.toml index 9cf4fc74365..c7fc6f5e326 100644 --- a/providers/vercel/models/zai/glm-5.3-flashx.toml +++ b/providers/vercel/models/zai/glm-5.3-flashx.toml @@ -2,8 +2,6 @@ # https://docs.z.ai/guides/vlm/glm-5.3-flash base_model = "zhipuai/glm-5.3-flash" name = "GLM 5.3 FlashX" -release_date = "2026-09-18" -last_updated = "2026-09-18" [[reasoning_options]] type = "effort" From 58f89895180de2cd5b56110a8da1dc6dfa1a7732 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 10:31:18 -0500 Subject: [PATCH 018/392] fix(alibaba-cn): document DeepSeek exchange rate (#7417) Co-authored-by: rekram1-node --- providers/alibaba-cn/models/deepseek-v4.1-flash.toml | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/providers/alibaba-cn/models/deepseek-v4.1-flash.toml b/providers/alibaba-cn/models/deepseek-v4.1-flash.toml index 933dcb373c8..6fe0ab81592 100644 --- a/providers/alibaba-cn/models/deepseek-v4.1-flash.toml +++ b/providers/alibaba-cn/models/deepseek-v4.1-flash.toml @@ -1,11 +1,12 @@ +# CNY→USD rate: 6.721845, 2026-09-18, source: https://open.er-api.com/v6/latest/USD # Sources (accessed 2026-09-18): # https://bailian.console.aliyun.com/cn-beijing?tab=model#/model-market/detail/deepseek-v4.1-flash?serviceSite=asia-pacific-china # https://help.aliyun.com/zh/model-studio/deepseek-api — capability table, # reasoning_effort ladder, max_tokens default (393_216, shared with thinking) # https://help.aliyun.com/zh/model-studio/billing-for-model-studio — Beijing # peak/off-peak pricing: CNY 2/1 input, 8/4 output per 1M tokens with context -# cache discount; peak price converted to USD following the existing -# deepseek-v4-flash.toml convention on this host (off-peak is half). +# cache discount; peak price converted to USD using the rate above (off-peak +# is half). # Hybrid reasoning: enable_thinking = true|false toggles thinking; # reasoning_effort = low|high|max (default high) — low is only supported on # v4.1-flash / v4-flash-0731 / v4-pro-0813, and v4.1-flash is the current @@ -21,6 +22,6 @@ reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "h field = "reasoning_content" [cost] -input = 0.278 -output = 1.111 -cache_read = 0.014 +input = 0.29754 +output = 1.19015 +cache_read = 0.01488 From 0204777328cb663967b9b52093d88c62b4730854 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 16:26:49 +0000 Subject: [PATCH 019/392] chore(sync): update OpenRouter model catalog (#7431) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/tencent/hy3.toml | 6 +++--- providers/openrouter/models/upstage/solar-pro-3.toml | 3 ++- providers/openrouter/models/upstage/solar-pro4.toml | 3 ++- 3 files changed, 7 insertions(+), 5 deletions(-) diff --git a/providers/openrouter/models/tencent/hy3.toml b/providers/openrouter/models/tencent/hy3.toml index 80ecfdd3aa8..f61fa3175e9 100644 --- a/providers/openrouter/models/tencent/hy3.toml +++ b/providers/openrouter/models/tencent/hy3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.132 -output = 0.528 -cache_read = 0.033 +input = 0.0825 +output = 0.33 +cache_read = 0.020625 [limit] context = 262_144 diff --git a/providers/openrouter/models/upstage/solar-pro-3.toml b/providers/openrouter/models/upstage/solar-pro-3.toml index 146505f4863..c9eda434a24 100644 --- a/providers/openrouter/models/upstage/solar-pro-3.toml +++ b/providers/openrouter/models/upstage/solar-pro-3.toml @@ -13,7 +13,8 @@ structured_output = true open_weights = false [[reasoning_options]] -type = "toggle" +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] [cost] input = 0.15 diff --git a/providers/openrouter/models/upstage/solar-pro4.toml b/providers/openrouter/models/upstage/solar-pro4.toml index adc9761fb07..7e3d19352a7 100644 --- a/providers/openrouter/models/upstage/solar-pro4.toml +++ b/providers/openrouter/models/upstage/solar-pro4.toml @@ -13,7 +13,8 @@ structured_output = true open_weights = false [[reasoning_options]] -type = "toggle" +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 0.09 From 7a433d1ea8dd9c508d7f0ca44cb6197bf7d41bab Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 16:26:59 +0000 Subject: [PATCH 020/392] chore(sync): update Kilo model catalog (#7432) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/tencent/hy3.toml | 6 +++--- providers/kilo/models/upstage/solar-pro-3.toml | 2 +- providers/kilo/models/upstage/solar-pro4.toml | 2 +- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index 8ea8b76bcff..e96e3110ab1 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.14 -output = 0.58 -cache_read = 0.035 +input = 0.0825 +output = 0.33 +cache_read = 0.020625 [limit] context = 262_144 diff --git a/providers/kilo/models/upstage/solar-pro-3.toml b/providers/kilo/models/upstage/solar-pro-3.toml index 54205b2bb45..ad7adb31fe8 100644 --- a/providers/kilo/models/upstage/solar-pro-3.toml +++ b/providers/kilo/models/upstage/solar-pro-3.toml @@ -12,7 +12,7 @@ open_weights = false [[reasoning_options]] type = "effort" -values = ["none", "high"] +values = ["none", "minimal", "low", "medium", "high"] [cost] input = 0.15 diff --git a/providers/kilo/models/upstage/solar-pro4.toml b/providers/kilo/models/upstage/solar-pro4.toml index 4b483744118..d3486cf5460 100644 --- a/providers/kilo/models/upstage/solar-pro4.toml +++ b/providers/kilo/models/upstage/solar-pro4.toml @@ -12,7 +12,7 @@ open_weights = false [[reasoning_options]] type = "effort" -values = ["none", "high"] +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 0.3 From 0efba743ed6ff791cc5aa52643e4f20dd46583b4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:22:35 +0000 Subject: [PATCH 021/392] chore(sync): update Kilo model catalog (#7436) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/z-ai/glm-5.3.toml | 3 +-- providers/kilo/models/~z-ai/glm-latest.toml | 8 ++++---- 2 files changed, 5 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/z-ai/glm-5.3.toml b/providers/kilo/models/z-ai/glm-5.3.toml index 70fec865ada..e9d5bdad891 100644 --- a/providers/kilo/models/z-ai/glm-5.3.toml +++ b/providers/kilo/models/z-ai/glm-5.3.toml @@ -11,5 +11,4 @@ output = 4.4 cache_read = 0.26 [limit] -context = 1_048_575 -output = 943_717 +context = 1_048_576 diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 2fc8e4f5ea7..c6f7c87e901 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,12 +15,12 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.9 -output = 3 -cache_read = 0.15 +input = 0.8988 +output = 2.8248 +cache_read = 0.16692 [limit] -context = 1_000_000 +context = 1_048_576 output = 131_072 [modalities] From 1dc425d366c3c0ab4511b9504f6fbf01f78f5e9c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:22:50 +0000 Subject: [PATCH 022/392] chore(sync): update Merge Gateway model catalog (#7435) Co-authored-by: opencode-agent[bot] --- providers/merge-gateway/models/qwen/qwen3.5-122b-a10b.toml | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/providers/merge-gateway/models/qwen/qwen3.5-122b-a10b.toml b/providers/merge-gateway/models/qwen/qwen3.5-122b-a10b.toml index 5ef98e7a3a0..b86a0e589cd 100644 --- a/providers/merge-gateway/models/qwen/qwen3.5-122b-a10b.toml +++ b/providers/merge-gateway/models/qwen/qwen3.5-122b-a10b.toml @@ -1,7 +1,6 @@ # Source: https://api-gateway.merge.dev/v1/models?model=qwen%2Fqwen3.5-122b-a10b (accessed 2026-07-21) base_model = "alibaba/qwen3.5-122b-a10b" name = "Qwen3.5 122B A10B" -attachment = false [[reasoning_options]] type = "toggle" @@ -12,8 +11,8 @@ output = 0.917 cache_read = 0.023 [limit] -context = 131_072 -output = 32_768 +context = 256_000 +output = 64_000 [modalities] -input = ["text"] +input = ["text", "image"] From c78add9b32185e8282abaa042b9fdd8b88f73574 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:23:34 +0000 Subject: [PATCH 023/392] chore(sync): update OpenRouter model catalog (#7437) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.3.toml | 7 +++---- providers/openrouter/models/~z-ai/glm-latest.toml | 6 +++--- 3 files changed, 9 insertions(+), 10 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index a9469f150d2..887f8fb33c5 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04984 -output = 0.09968 -cache_read = 0.009968 +input = 0.04956 +output = 0.09912 +cache_read = 0.009912 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index d077461d393..6ca3ad0c4f6 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,10 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.4 -output = 4.4 -cache_read = 0.26 +input = 0.91 +output = 2.86 +cache_read = 0.169 [limit] context = 1_310_720 -output = 943_717 diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 0a7b3acfd67..1aaa44a3087 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.9 -output = 3 -cache_read = 0.15 +input = 0.8988 +output = 2.8248 +cache_read = 0.16692 [limit] context = 1_310_720 From 76bb6adbd1f734d4473f3fb789c89141a79139f9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 18:29:52 +0000 Subject: [PATCH 024/392] chore(sync): update Kilo model catalog (#7439) Co-authored-by: opencode-agent[bot] --- .../kilo/models/~deepseek/deepseek-flash-latest.toml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index f491c2853c8..69e45ef88a2 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.15 -output = 0.6 -cache_read = 0.015 +input = 0.14 +output = 0.42 +cache_read = 0.0042 [limit] -context = 1_000_000 -output = 393_216 +context = 1_048_576 +output = 131_072 [modalities] input = ["text", "image"] From bcf2e8b580e7b0721d29767ecb0551080b22168e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 18:29:54 +0000 Subject: [PATCH 025/392] chore(sync): update OpenRouter model catalog (#7438) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../models/~deepseek/deepseek-flash-latest.toml | 8 ++++---- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 887f8fb33c5..58e87cdc226 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04956 -output = 0.09912 -cache_read = 0.009912 +input = 0.04928 +output = 0.09856 +cache_read = 0.009856 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index 98fd23a037b..5e8eab978e2 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.15 -output = 0.6 -cache_read = 0.015 +input = 0.14 +output = 0.42 +cache_read = 0.0042 [limit] context = 1_048_576 -output = 393_216 +output = 131_072 [modalities] input = ["text", "image"] From b9e66aafa07959402c64d998930d8909de9762bd Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:23:07 +0000 Subject: [PATCH 026/392] chore(sync): update Kilo model catalog (#7441) Co-authored-by: opencode-agent[bot] --- .../models/prism-ml/ternary-bonsai-2-27b.toml | 26 +++++++++++++++++++ providers/kilo/models/~z-ai/glm-latest.toml | 8 +++--- 2 files changed, 30 insertions(+), 4 deletions(-) create mode 100644 providers/kilo/models/prism-ml/ternary-bonsai-2-27b.toml diff --git a/providers/kilo/models/prism-ml/ternary-bonsai-2-27b.toml b/providers/kilo/models/prism-ml/ternary-bonsai-2-27b.toml new file mode 100644 index 00000000000..a1eb487d1bb --- /dev/null +++ b/providers/kilo/models/prism-ml/ternary-bonsai-2-27b.toml @@ -0,0 +1,26 @@ +name = "PrismML: Ternary Bonsai 2 27B" +description = "Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window. Ternary compression shrinks..." +release_date = "2026-09-18" +last_updated = "2026-09-18" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "medium", "xhigh"] + +[cost] +input = 0.075 +output = 0.5 + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index c6f7c87e901..da0527bbfae 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8988 -output = 2.8248 -cache_read = 0.16692 +input = 0.8925 +output = 2.805 +cache_read = 0.146625 [limit] context = 1_048_576 -output = 131_072 +output = 943_718 [modalities] input = ["text"] From 743e022ceb024827c6aa737a4ff4883161c98489 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:23:15 +0000 Subject: [PATCH 027/392] chore(sync): update OpenRouter model catalog (#7440) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash.toml | 6 ++-- .../models/prism-ml/ternary-bonsai-2-27b.toml | 31 +++++++++++++++++++ .../openrouter/models/~z-ai/glm-latest.toml | 8 ++--- 3 files changed, 38 insertions(+), 7 deletions(-) create mode 100644 providers/openrouter/models/prism-ml/ternary-bonsai-2-27b.toml diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 58e87cdc226..141d4b41ca6 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04928 -output = 0.09856 -cache_read = 0.009856 +input = 0.049 +output = 0.098 +cache_read = 0.0098 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/prism-ml/ternary-bonsai-2-27b.toml b/providers/openrouter/models/prism-ml/ternary-bonsai-2-27b.toml new file mode 100644 index 00000000000..316e8fde320 --- /dev/null +++ b/providers/openrouter/models/prism-ml/ternary-bonsai-2-27b.toml @@ -0,0 +1,31 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +name = "Ternary Bonsai 2 27B" +description = "Multimodal reasoning model for visual analysis, planning, and tool use" +release_date = "2026-09-18" +last_updated = "2026-09-18" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["medium", "xhigh"] + +[cost] +input = 0.075 +output = 0.5 + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 1aaa44a3087..0d21da4aeaf 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8988 -output = 2.8248 -cache_read = 0.16692 +input = 0.8925 +output = 2.805 +cache_read = 0.146625 [limit] context = 1_310_720 -output = 131_072 +output = 943_718 [modalities] input = ["text"] From 5ecf83932e114a43cd4c31a2f5d7e137ffd23fd4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 20:25:52 +0000 Subject: [PATCH 028/392] chore(sync): update OpenRouter model catalog (#7443) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/moonshotai/kimi-k3.toml | 6 +++--- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- providers/openrouter/models/~z-ai/glm-latest.toml | 8 ++++---- 4 files changed, 13 insertions(+), 13 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 141d4b41ca6..2309a32472b 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.049 -output = 0.098 -cache_read = 0.0098 +input = 0.04872 +output = 0.09744 +cache_read = 0.009744 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index 618c6fb2bbd..4e8299c6b8d 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 1.95 +output = 10.92 +cache_read = 0.2262 [limit] output = 943_718 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index d1d254d74de..56f8b023ae1 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 1.95 +output = 10.92 +cache_read = 0.2262 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 0d21da4aeaf..d3b500c9777 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8925 -output = 2.805 -cache_read = 0.146625 +input = 0.8918 +output = 2.8028 +cache_read = 0.16562 [limit] context = 1_310_720 -output = 943_718 +output = 131_072 [modalities] input = ["text"] From fe33af6dcb907ac1ccf1b5fedb5d3d6dbacafccd Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 20:26:10 +0000 Subject: [PATCH 029/392] chore(sync): update Kilo model catalog (#7442) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/moonshotai/kimi-k3.toml | 6 +++--- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- providers/kilo/models/~z-ai/glm-latest.toml | 8 ++++---- 3 files changed, 10 insertions(+), 10 deletions(-) diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index 3af9cc14392..9fe8a4fd3f8 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 1.95 +output = 10.92 +cache_read = 0.2262 [limit] output = 943_718 diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 9bbf3c7be77..fcb5b91d30e 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 1.95 +output = 10.92 +cache_read = 0.2262 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index da0527bbfae..73bfc195431 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8925 -output = 2.805 -cache_read = 0.146625 +input = 0.8918 +output = 2.8028 +cache_read = 0.16562 [limit] context = 1_048_576 -output = 943_718 +output = 131_072 [modalities] input = ["text"] From 115bd0c7a86ad1a2635fbd7c3aa268ce10490de4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 21:23:41 +0000 Subject: [PATCH 030/392] chore(sync): update Kilo model catalog (#7445) Co-authored-by: opencode-agent[bot] --- .../kilo/models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 844f1914d6e..b96573b6b7a 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.055 -output = 0.165 -cache_read = 0.00175 +input = 0.05412 +output = 0.16236 +cache_read = 0.001722 [limit] context = 1_024_000 From 709448a4719255680e5727052faafd3661e4206d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 21:23:45 +0000 Subject: [PATCH 031/392] chore(sync): update Vercel AI Gateway model catalog (#7446) Co-authored-by: opencode-agent[bot] --- providers/vercel/models/fish-audio/s1-free.toml | 16 ---------------- .../vercel/models/fish-audio/s2-pro-free.toml | 16 ---------------- .../vercel/models/fish-audio/s2.1-pro-free.toml | 16 ---------------- .../models/fish-audio/transcribe-1-free.toml | 16 ---------------- 4 files changed, 64 deletions(-) delete mode 100644 providers/vercel/models/fish-audio/s1-free.toml delete mode 100644 providers/vercel/models/fish-audio/s2-pro-free.toml delete mode 100644 providers/vercel/models/fish-audio/s2.1-pro-free.toml delete mode 100644 providers/vercel/models/fish-audio/transcribe-1-free.toml diff --git a/providers/vercel/models/fish-audio/s1-free.toml b/providers/vercel/models/fish-audio/s1-free.toml deleted file mode 100644 index 2eb0f456def..00000000000 --- a/providers/vercel/models/fish-audio/s1-free.toml +++ /dev/null @@ -1,16 +0,0 @@ -name = "S1 (Free)" -description = "Speech generation model for controllable voice, narration, and audio delivery" -release_date = "2025-10-20" -last_updated = "2025-10-20" -attachment = false -reasoning = false -tool_call = false -open_weights = false - -[limit] -context = 0 -output = 0 - -[modalities] -input = ["text"] -output = ["audio"] diff --git a/providers/vercel/models/fish-audio/s2-pro-free.toml b/providers/vercel/models/fish-audio/s2-pro-free.toml deleted file mode 100644 index c2bbbb724c8..00000000000 --- a/providers/vercel/models/fish-audio/s2-pro-free.toml +++ /dev/null @@ -1,16 +0,0 @@ -name = "S2 Pro (Free)" -description = "Speech generation model for controllable voice, narration, and audio delivery" -release_date = "2026-03-09" -last_updated = "2026-03-09" -attachment = false -reasoning = false -tool_call = false -open_weights = false - -[limit] -context = 0 -output = 0 - -[modalities] -input = ["text"] -output = ["audio"] diff --git a/providers/vercel/models/fish-audio/s2.1-pro-free.toml b/providers/vercel/models/fish-audio/s2.1-pro-free.toml deleted file mode 100644 index c580e92cd34..00000000000 --- a/providers/vercel/models/fish-audio/s2.1-pro-free.toml +++ /dev/null @@ -1,16 +0,0 @@ -name = "S2.1 Pro (Free)" -description = "Speech generation model for controllable voice, narration, and audio delivery" -release_date = "2026-07-28" -last_updated = "2026-07-28" -attachment = false -reasoning = false -tool_call = false -open_weights = false - -[limit] -context = 0 -output = 0 - -[modalities] -input = ["text"] -output = ["audio"] diff --git a/providers/vercel/models/fish-audio/transcribe-1-free.toml b/providers/vercel/models/fish-audio/transcribe-1-free.toml deleted file mode 100644 index e21df2298c0..00000000000 --- a/providers/vercel/models/fish-audio/transcribe-1-free.toml +++ /dev/null @@ -1,16 +0,0 @@ -name = "Transcribe-1 (Free)" -description = "Speech transcription model for accurate audio-to-text and captioning workflows" -release_date = "2026-03-01" -last_updated = "2026-03-01" -attachment = false -reasoning = false -tool_call = false -open_weights = false - -[limit] -context = 0 -output = 0 - -[modalities] -input = ["audio"] -output = ["text"] From 6a9e725a50369370c91d12c06cb8d5dc67e3f059 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 21:23:52 +0000 Subject: [PATCH 032/392] chore(sync): update OpenRouter model catalog (#7444) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/google/gemma-4-26b-a4b-it.toml | 1 + .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- 3 files changed, 7 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 2309a32472b..814013f36dd 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04872 -output = 0.09744 -cache_read = 0.009744 +input = 0.04844 +output = 0.09688 +cache_read = 0.009688 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml b/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml index 7d230a2be6f..872203b2862 100644 --- a/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml +++ b/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml @@ -11,6 +11,7 @@ output = 0.3 cache_read = 0.05 [limit] +context = 1_000_000 output = 235_929 [modalities] diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index b207a5a59a3..9f76b8f4c29 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.055 -output = 0.165 -cache_read = 0.00175 +input = 0.05412 +output = 0.16236 +cache_read = 0.001722 [limit] context = 1_310_720 From 9f4dbb29e1960215c03d76714c6675910e0fb9bb Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 22:24:16 +0000 Subject: [PATCH 033/392] chore(sync): update OpenRouter model catalog (#7449) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/google/gemma-4-26b-a4b-it.toml | 1 - .../models/~deepseek/deepseek-flash-latest.toml | 8 ++++---- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- 3 files changed, 7 insertions(+), 8 deletions(-) diff --git a/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml b/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml index 872203b2862..7d230a2be6f 100644 --- a/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml +++ b/providers/openrouter/models/google/gemma-4-26b-a4b-it.toml @@ -11,7 +11,6 @@ output = 0.3 cache_read = 0.05 [limit] -context = 1_000_000 output = 235_929 [modalities] diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index 5e8eab978e2..bdcd94e72e3 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.14 -output = 0.42 -cache_read = 0.0042 +input = 0.135 +output = 0.54 +cache_read = 0.00405 [limit] context = 1_048_576 -output = 131_072 +output = 943_718 [modalities] input = ["text", "image"] diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 9f76b8f4c29..f00eacc90b8 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.05412 -output = 0.16236 -cache_read = 0.001722 +input = 0.05324 +output = 0.15972 +cache_read = 0.001694 [limit] context = 1_310_720 From ff883d057d726877568b1d89cec733ea95df9ef0 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 22:24:20 +0000 Subject: [PATCH 034/392] chore(sync): update Kilo model catalog (#7450) Co-authored-by: opencode-agent[bot] --- .../kilo/models/~deepseek/deepseek-flash-latest.toml | 8 ++++---- .../kilo/models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 69e45ef88a2..0b2d5078274 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.14 -output = 0.42 -cache_read = 0.0042 +input = 0.135 +output = 0.54 +cache_read = 0.00405 [limit] context = 1_048_576 -output = 131_072 +output = 943_718 [modalities] input = ["text", "image"] diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index b96573b6b7a..cdd99b9392d 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.05412 -output = 0.16236 -cache_read = 0.001722 +input = 0.05324 +output = 0.15972 +cache_read = 0.001694 [limit] context = 1_024_000 From 91086d2c8d667ca5058a2bc77f8f8684293c871e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 23:24:12 +0000 Subject: [PATCH 035/392] chore(sync): update OpenRouter model catalog (#7455) Co-authored-by: opencode-agent[bot] --- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index f00eacc90b8..83799fb5d7c 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.05324 -output = 0.15972 -cache_read = 0.001694 +input = 0.05236 +output = 0.15708 +cache_read = 0.001666 [limit] context = 1_310_720 From 6b1fd113c90ca967bd9e4d8d3a529cceb9a226a3 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 23:24:16 +0000 Subject: [PATCH 036/392] chore(sync): update Kilo model catalog (#7456) Co-authored-by: opencode-agent[bot] --- .../kilo/models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index cdd99b9392d..64711cafb8f 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.05324 -output = 0.15972 -cache_read = 0.001694 +input = 0.05236 +output = 0.15708 +cache_read = 0.001666 [limit] context = 1_024_000 From 6b5d1eaa97c36e23adaa4584640d60651f5fce94 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 23:24:46 +0000 Subject: [PATCH 037/392] chore(sync): update NanoGPT model catalog (#7457) Co-authored-by: opencode-agent[bot] --- .../mistral-large-3-675b-instruct-2512.toml | 24 ----------------- .../models/prism-ml/ternary-bonsai-2-27b.toml | 27 +++++++++++++++++++ 2 files changed, 27 insertions(+), 24 deletions(-) delete mode 100644 providers/nano-gpt/models/mistralai/mistral-large-3-675b-instruct-2512.toml create mode 100644 providers/nano-gpt/models/prism-ml/ternary-bonsai-2-27b.toml diff --git a/providers/nano-gpt/models/mistralai/mistral-large-3-675b-instruct-2512.toml b/providers/nano-gpt/models/mistralai/mistral-large-3-675b-instruct-2512.toml deleted file mode 100644 index 0a0650a3987..00000000000 --- a/providers/nano-gpt/models/mistralai/mistral-large-3-675b-instruct-2512.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Mistral Large 3 675B" -description = "Flagship Mistral model for advanced reasoning, coding, and multilingual work" -family = "mistral-large" -release_date = "2025-12-25" -last_updated = "2025-12-02" -attachment = true -reasoning = false -tool_call = false -structured_output = false -open_weights = true - -[cost] -input = 1 -output = 3 -cache_read = 0.5 - -[limit] -context = 262_144 -input = 262_144 -output = 256_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/nano-gpt/models/prism-ml/ternary-bonsai-2-27b.toml b/providers/nano-gpt/models/prism-ml/ternary-bonsai-2-27b.toml new file mode 100644 index 00000000000..32eb4c91ae5 --- /dev/null +++ b/providers/nano-gpt/models/prism-ml/ternary-bonsai-2-27b.toml @@ -0,0 +1,27 @@ +name = "Ternary Bonsai 2 27B" +description = "Ternary Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window." +release_date = "2026-09-18" +last_updated = "2026-09-18" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = true + +[[reasoning_options]] +type = "effort" +values = ["medium", "xhigh"] + +[cost] +input = 0.075 +output = 0.5 +cache_read = 0.0375 + +[limit] +context = 262_144 +input = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] From dcd5fc29801fdfa285c8cd0553bfbca3bb7652a4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 00:57:46 +0000 Subject: [PATCH 038/392] chore(sync): update OpenRouter model catalog (#7460) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 7 +++---- .../models/nvidia/nemotron-3-ultra-550b-a55b.toml | 8 ++++---- .../openrouter/models/nvidia/nemotron-3.5-lightning.toml | 4 ++-- .../openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml | 4 ++-- providers/openrouter/models/tencent/hy3.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- 6 files changed, 17 insertions(+), 18 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 95460b6f5f3..4ce5d404923 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,10 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 1.6 -output = 3.2 -cache_read = 0.135 +input = 0.675816 +output = 1.351632 +cache_read = 0.056318 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/openrouter/models/nvidia/nemotron-3-ultra-550b-a55b.toml b/providers/openrouter/models/nvidia/nemotron-3-ultra-550b-a55b.toml index f471e71439f..5dcd117448e 100644 --- a/providers/openrouter/models/nvidia/nemotron-3-ultra-550b-a55b.toml +++ b/providers/openrouter/models/nvidia/nemotron-3-ultra-550b-a55b.toml @@ -14,10 +14,10 @@ values = ["medium", "high"] type = "budget_tokens" [cost] -input = 0.625 -output = 3.125 -cache_read = 0.1875 +input = 0.6 +output = 2.4 +cache_read = 0.12 [limit] context = 262_144 -output = 32_768 +output = 182_520 diff --git a/providers/openrouter/models/nvidia/nemotron-3.5-lightning.toml b/providers/openrouter/models/nvidia/nemotron-3.5-lightning.toml index 3b251679339..842e5779095 100644 --- a/providers/openrouter/models/nvidia/nemotron-3.5-lightning.toml +++ b/providers/openrouter/models/nvidia/nemotron-3.5-lightning.toml @@ -7,9 +7,9 @@ description = "Nemotron model for efficient reasoning, coding, and specialized A type = "toggle" [cost] -input = 0.08 +input = 0.07 output = 0.2 cache_read = 0.04 [limit] -output = 131_072 +output = 235_929 diff --git a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml index d857fa53dab..91402212177 100644 --- a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml +++ b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml @@ -12,8 +12,8 @@ knowledge = "2025-03-31" open_weights = true [cost] -input = 0.13 -output = 0.52 +input = 0.2 +output = 0.7 [limit] context = 262_144 diff --git a/providers/openrouter/models/tencent/hy3.toml b/providers/openrouter/models/tencent/hy3.toml index f61fa3175e9..80ecfdd3aa8 100644 --- a/providers/openrouter/models/tencent/hy3.toml +++ b/providers/openrouter/models/tencent/hy3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.0825 -output = 0.33 -cache_read = 0.020625 +input = 0.132 +output = 0.528 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 83799fb5d7c..f51872c2fb8 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.05236 -output = 0.15708 -cache_read = 0.001666 +input = 0.05148 +output = 0.15444 +cache_read = 0.001638 [limit] context = 1_310_720 From 255919f646ed4f9382b0b5fbe7eb23fe18be1126 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 00:57:52 +0000 Subject: [PATCH 039/392] chore(sync): update Kilo model catalog (#7462) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/deepseek/deepseek-v4-pro.toml | 3 +-- .../kilo/models/nvidia/nemotron-3-ultra-550b-a55b.toml | 4 ++-- .../kilo/models/nvidia/nemotron-3.5-lightning.toml | 2 +- providers/kilo/models/tencent/hy3.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 10 +++++----- 5 files changed, 12 insertions(+), 13 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro.toml b/providers/kilo/models/deepseek/deepseek-v4-pro.toml index e58e24281d4..88500bd7dd8 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro.toml @@ -11,5 +11,4 @@ output = 3.2 cache_read = 0.135 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 diff --git a/providers/kilo/models/nvidia/nemotron-3-ultra-550b-a55b.toml b/providers/kilo/models/nvidia/nemotron-3-ultra-550b-a55b.toml index 212e280b8a9..25c72a0b0fb 100644 --- a/providers/kilo/models/nvidia/nemotron-3-ultra-550b-a55b.toml +++ b/providers/kilo/models/nvidia/nemotron-3-ultra-550b-a55b.toml @@ -12,5 +12,5 @@ output = 2.2 cache_read = 0.1 [limit] -context = 256_000 -output = 32_768 +context = 202_800 +output = 182_520 diff --git a/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml b/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml index 78104d7319f..5244dc1ac02 100644 --- a/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml +++ b/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml @@ -10,4 +10,4 @@ input = 0.065 output = 0.18 [limit] -output = 131_072 +output = 235_929 diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index e96e3110ab1..2bdfa3cbd7b 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.0825 -output = 0.33 -cache_read = 0.020625 +input = 0.132 +output = 0.528 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 64711cafb8f..dcbc3433279 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.05236 -output = 0.15708 -cache_read = 0.001666 +input = 0.05104 +output = 0.15312 +cache_read = 0.001624 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 131_072 [modalities] input = ["text"] From fead8587cea02673f3a10c6be712ae83851f8c17 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 00:57:54 +0000 Subject: [PATCH 040/392] chore(sync): update NanoGPT model catalog (#7461) Co-authored-by: opencode-agent[bot] --- .../nano-gpt/models/deepseek/deepseek-v4-flash-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-latest.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-latest.toml index abb7e798efd..fb7f9db2975 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-latest.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-latest.toml @@ -19,9 +19,9 @@ output = 0.16 cache_read = 0.013 [limit] -context = 1_048_576 -input = 1_048_576 -output = 384_000 +context = 1_000_000 +input = 1_000_000 +output = 131_072 [modalities] input = ["text"] From dff014f67f04d7f17b6a7024510c77dd4a5b47e5 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 00:57:57 +0000 Subject: [PATCH 041/392] chore(sync): update Venice model catalog (#7463) Co-authored-by: opencode-agent[bot] --- .../venice/models/openai-gpt-56-sol-pro.toml | 16 ++++++++-------- providers/venice/models/openai-gpt-56-sol.toml | 16 ++++++++-------- 2 files changed, 16 insertions(+), 16 deletions(-) diff --git a/providers/venice/models/openai-gpt-56-sol-pro.toml b/providers/venice/models/openai-gpt-56-sol-pro.toml index 36093d01914..1b837b0d610 100644 --- a/providers/venice/models/openai-gpt-56-sol-pro.toml +++ b/providers/venice/models/openai-gpt-56-sol-pro.toml @@ -8,18 +8,18 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 2.5 -output = 12.5 -cache_read = 0.25 -cache_write = 3.125 - -[[cost.tiers]] -tier = { type = "context", size = 272_000 } input = 5 -output = 18.75 +output = 25 cache_read = 0.5 cache_write = 6.25 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 10 +output = 37.5 +cache_read = 1 +cache_write = 12.5 + [limit] context = 1_000_000 diff --git a/providers/venice/models/openai-gpt-56-sol.toml b/providers/venice/models/openai-gpt-56-sol.toml index dedfc528097..a0a89340131 100644 --- a/providers/venice/models/openai-gpt-56-sol.toml +++ b/providers/venice/models/openai-gpt-56-sol.toml @@ -7,18 +7,18 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 2.5 -output = 12.5 -cache_read = 0.25 -cache_write = 3.125 - -[[cost.tiers]] -tier = { type = "context", size = 272_000 } input = 5 -output = 18.75 +output = 25 cache_read = 0.5 cache_write = 6.25 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 10 +output = 37.5 +cache_read = 1 +cache_write = 12.5 + [limit] context = 1_000_000 From 248a44ccbcf6c2bc38fe8212720b5050cff01d76 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 01:34:31 +0000 Subject: [PATCH 042/392] chore(sync): update Kilo model catalog (#7465) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/moonshotai/kimi-k3.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 10 +++++----- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- 3 files changed, 11 insertions(+), 11 deletions(-) diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index 9fe8a4fd3f8..3af9cc14392 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.95 -output = 10.92 -cache_read = 0.2262 +input = 2.1 +output = 10.95 +cache_read = 0.23 [limit] output = 943_718 diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index dcbc3433279..84e8197e6b4 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.05104 -output = 0.15312 -cache_read = 0.001624 +input = 0.0506 +output = 0.1518 +cache_read = 0.00161 [limit] -context = 1_048_576 -output = 131_072 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index fcb5b91d30e..9bbf3c7be77 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.95 -output = 10.92 -cache_read = 0.2262 +input = 2.1 +output = 10.95 +cache_read = 0.23 [limit] context = 1_048_576 From 6303a062a3782391f3147e9db1100ee93139abbb Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 01:34:47 +0000 Subject: [PATCH 043/392] chore(sync): update OpenRouter model catalog (#7466) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/moonshotai/kimi-k3.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- 4 files changed, 12 insertions(+), 12 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 4ce5d404923..62b0bbc3ecf 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.675816 -output = 1.351632 -cache_read = 0.056318 +input = 0.66555 +output = 1.3311 +cache_read = 0.055463 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index 4e8299c6b8d..618c6fb2bbd 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.95 -output = 10.92 -cache_read = 0.2262 +input = 2.1 +output = 10.95 +cache_read = 0.23 [limit] output = 943_718 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index f51872c2fb8..7c73e02e9d6 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.05148 -output = 0.15444 -cache_read = 0.001638 +input = 0.0506 +output = 0.1518 +cache_read = 0.00161 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 56f8b023ae1..d1d254d74de 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.95 -output = 10.92 -cache_read = 0.2262 +input = 2.1 +output = 10.95 +cache_read = 0.23 [limit] context = 1_048_576 From 63ad21b91c8a4dd1be7331a44313ca8dcdeefce5 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Sat, 19 Sep 2026 10:21:04 +0800 Subject: [PATCH 044/392] feat(aihubmix): add ox-alpha (#7428) AIHubMix serves this route but the catalog carried no aihubmix entry for it. Cost comes from the provider's public model listing and the reasoning controls from its published model-data index; limit, modalities, attachment and tool_call inherit from models/zhipuai/glm-5.3-flash.toml rather than being repeated here. Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- providers/aihubmix/models/ox-alpha.toml | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) create mode 100644 providers/aihubmix/models/ox-alpha.toml diff --git a/providers/aihubmix/models/ox-alpha.toml b/providers/aihubmix/models/ox-alpha.toml new file mode 100644 index 00000000000..800e20046f6 --- /dev/null +++ b/providers/aihubmix/models/ox-alpha.toml @@ -0,0 +1,23 @@ +# Toggle: +# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11); +# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path. +# Effort: low|high|max +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "zhipuai/glm-5.3-flash" +name = "Ox Alpha" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 From af57606c96af3f5960fe7a12f5e6cca77a8518d2 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Sat, 19 Sep 2026 10:21:13 +0800 Subject: [PATCH 045/392] feat(aihubmix): add deepseek-v4-flash-vision-exp (#7427) AIHubMix serves this route but the catalog carried no aihubmix entry for it. Cost comes from the provider's public model listing and the reasoning controls from its published model-data index; limit, modalities, attachment and tool_call inherit from models/deepseek/deepseek-v4-flash-vision-exp.toml rather than being repeated here. Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- .../models/deepseek-v4-flash-vision-exp.toml | 23 +++++++++++++++++++ 1 file changed, 23 insertions(+) create mode 100644 providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml diff --git a/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml b/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml new file mode 100644 index 00000000000..3cd40d48314 --- /dev/null +++ b/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml @@ -0,0 +1,23 @@ +# Toggle: +# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11); +# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path. +# Effort: low|high|max +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "deepseek/deepseek-v4-flash-vision-exp" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.155 +output = 0.62 +cache_read = 0.0031 From d809a3f9909fb8320cd17acf4c6ab6aeb0eb8077 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Sat, 19 Sep 2026 10:21:21 +0800 Subject: [PATCH 046/392] feat(aihubmix): add qwen3.8-flash (#7426) AIHubMix serves this route but the catalog carried no aihubmix entry for it. Cost comes from the provider's public model listing and the reasoning controls from its published model-data index; limit, modalities, attachment and tool_call inherit from models/alibaba/qwen3.8-flash.toml rather than being repeated here. Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- providers/aihubmix/models/qwen3.8-flash.toml | 31 ++++++++++++++++++++ 1 file changed, 31 insertions(+) create mode 100644 providers/aihubmix/models/qwen3.8-flash.toml diff --git a/providers/aihubmix/models/qwen3.8-flash.toml b/providers/aihubmix/models/qwen3.8-flash.toml new file mode 100644 index 00000000000..69e8766006b --- /dev/null +++ b/providers/aihubmix/models/qwen3.8-flash.toml @@ -0,0 +1,31 @@ +# Toggle: +# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11); +# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path. +# Effort: low|medium|xhigh +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# Budget: +# integer $.reasoning.max_tokens on /v1/chat/completions; $.thinking.budget_tokens >= 1024 on /v1/messages; +# integer $.generationConfig.thinkingConfig.thinkingBudget on the Gemini path (-1 dynamic, 0 off where supported); +# the Responses path carries effort but has no reasoning-token budget field. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "alibaba/qwen3.8-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1126 +output = 0.380025 +cache_read = 0.014075 +cache_write = 0.175937 From f14883a075e95ed2dad7565a276c90dfa30d1481 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 02:33:13 +0000 Subject: [PATCH 047/392] chore(sync): update OpenRouter model catalog (#7469) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-0731.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 8 ++++---- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml index df31c0bbb32..1273df92943 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.06 -output = 0.12 -cache_read = 0.012 +input = 0.05 +output = 0.1 +cache_read = 0.01 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 7c73e02e9d6..5a559b9f1f5 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.0506 -output = 0.1518 -cache_read = 0.00161 +input = 0.04972 +output = 0.14916 +cache_read = 0.001582 [limit] context = 1_310_720 -output = 384_000 +output = 131_072 [modalities] input = ["text"] From bc21426e3a59ca8b9b1afd8d24973bdc16e02217 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 02:33:15 +0000 Subject: [PATCH 048/392] chore(sync): update Kilo model catalog (#7468) Co-authored-by: opencode-agent[bot] --- .../models/~deepseek/deepseek-v4-flash-latest.toml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 84e8197e6b4..857a18e8b48 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.0506 -output = 0.1518 -cache_read = 0.00161 +input = 0.04972 +output = 0.14916 +cache_read = 0.001582 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 131_072 [modalities] input = ["text"] From e775635942b83c102ebde5b772e1ab8c7663339e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 03:29:08 +0000 Subject: [PATCH 049/392] chore(sync): update Kilo model catalog (#7471) Co-authored-by: opencode-agent[bot] --- .../models/nvidia/nemotron-3.5-lightning.toml | 2 +- providers/kilo/models/openai/gpt-oss-20b.toml | 4 +-- .../kilo/models/z-ai/glm-5.3-flashx.toml | 28 +++++++++++++++++++ .../~deepseek/deepseek-v4-flash-latest.toml | 10 +++---- 4 files changed, 36 insertions(+), 8 deletions(-) create mode 100644 providers/kilo/models/z-ai/glm-5.3-flashx.toml diff --git a/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml b/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml index 5244dc1ac02..6759948a30e 100644 --- a/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml +++ b/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml @@ -6,7 +6,7 @@ type = "effort" values = ["none", "high"] [cost] -input = 0.065 +input = 0.04 output = 0.18 [limit] diff --git a/providers/kilo/models/openai/gpt-oss-20b.toml b/providers/kilo/models/openai/gpt-oss-20b.toml index 895f7fb4602..fe3d2508db6 100644 --- a/providers/kilo/models/openai/gpt-oss-20b.toml +++ b/providers/kilo/models/openai/gpt-oss-20b.toml @@ -6,8 +6,8 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.02 -output = 0.1 +input = 0.018 +output = 0.09 [limit] output = 117_964 diff --git a/providers/kilo/models/z-ai/glm-5.3-flashx.toml b/providers/kilo/models/z-ai/glm-5.3-flashx.toml new file mode 100644 index 00000000000..a730a63619c --- /dev/null +++ b/providers/kilo/models/z-ai/glm-5.3-flashx.toml @@ -0,0 +1,28 @@ +name = "Z.ai: GLM 5.3 FlashX" +description = "GLM-5.3-FlashX is the high-speed variant of Z.ai's GLM-5.3-Flash, a native multimodal model delivering inference speeds of up to 200 tokens/s. Built on the same hybrid sparse and linear attention architecture..." +family = "glm" +release_date = "2026-09-18" +last_updated = "2026-09-18" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.37 +output = 1.25 +cache_read = 0.075 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 857a18e8b48..50af69be57a 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.04972 -output = 0.14916 -cache_read = 0.001582 +input = 0.04928 +output = 0.14784 +cache_read = 0.001568 [limit] -context = 1_048_576 -output = 131_072 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] From 4676efad5e24df24ef63f4624696df368c22f06f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 03:29:25 +0000 Subject: [PATCH 050/392] chore(sync): update OpenRouter model catalog (#7470) Co-authored-by: opencode-agent[bot] --- .../deepseek/deepseek-v4-flash-0731.toml | 6 ++-- .../models/deepseek/deepseek-v4-pro.toml | 6 ++-- .../models/z-ai/glm-5.3-flashx.toml | 28 +++++++++++++++++++ .../~deepseek/deepseek-v4-flash-latest.toml | 8 +++--- 4 files changed, 38 insertions(+), 10 deletions(-) create mode 100644 providers/openrouter/models/z-ai/glm-5.3-flashx.toml diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml index 1273df92943..d06dd2cd8d1 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.05 -output = 0.1 -cache_read = 0.01 +input = 0.065 +output = 0.18 +cache_read = 0.016 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 62b0bbc3ecf..a3a0173e68a 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.66555 -output = 1.3311 -cache_read = 0.055463 +input = 0.645366 +output = 1.290732 +cache_read = 0.053781 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-5.3-flashx.toml b/providers/openrouter/models/z-ai/glm-5.3-flashx.toml new file mode 100644 index 00000000000..ea5b2ee8747 --- /dev/null +++ b/providers/openrouter/models/z-ai/glm-5.3-flashx.toml @@ -0,0 +1,28 @@ +name = "GLM 5.3 FlashX" +description = "GLM vision model for visual reasoning, documents, and multimodal agents" +family = "glm" +release_date = "2026-09-18" +last_updated = "2026-09-18" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.37 +output = 1.25 +cache_read = 0.075 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 5a559b9f1f5..189296e3dde 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.04972 -output = 0.14916 -cache_read = 0.001582 +input = 0.04928 +output = 0.14784 +cache_read = 0.001568 [limit] context = 1_310_720 -output = 131_072 +output = 384_000 [modalities] input = ["text"] From 227516b6e6f96960c5b952e7c0dfaef00a91b54a Mon Sep 17 00:00:00 2001 From: Jack Date: Sat, 19 Sep 2026 11:45:24 +0800 Subject: [PATCH 051/392] feat(opencode): add qwen3.8 flash to zen --- providers/opencode/models/qwen3.8-flash.toml | 25 ++++++++++++++++++++ 1 file changed, 25 insertions(+) create mode 100644 providers/opencode/models/qwen3.8-flash.toml diff --git a/providers/opencode/models/qwen3.8-flash.toml b/providers/opencode/models/qwen3.8-flash.toml new file mode 100644 index 00000000000..1cd079b5cb3 --- /dev/null +++ b/providers/opencode/models/qwen3.8-flash.toml @@ -0,0 +1,25 @@ +# https://opencode.ai/docs/go/#endpoints +# https://help.aliyun.com/en/model-studio/anthropic-api-messages +# Toggle (Messages): thinking.type = enabled|disabled +# Effort (Messages): output_config.effort = low|medium|xhigh +base_model = "alibaba/qwen3.8-flash" +temperature = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.15 +output = 0.47 +cache_read = 0.016 +cache_write = 0.2 + +[provider] +npm = "@ai-sdk/anthropic" \ No newline at end of file From 27c0683399bdf23e3bfecc716cba524c8ad5afb6 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 04:28:48 +0000 Subject: [PATCH 052/392] chore(sync): update OpenRouter model catalog (#7474) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/moonshotai/kimi-k3.toml | 6 +++--- .../openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml | 4 ++-- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- 6 files changed, 17 insertions(+), 17 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 814013f36dd..b5c7d255938 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04844 -output = 0.09688 -cache_read = 0.009688 +input = 0.04816 +output = 0.09632 +cache_read = 0.009632 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index a3a0173e68a..ff6ad910cd9 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.645366 -output = 1.290732 -cache_read = 0.053781 +input = 0.614916 +output = 1.229832 +cache_read = 0.051243 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index 618c6fb2bbd..d374007e872 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 2.0695 +output = 11.5892 +cache_read = 0.240062 [limit] output = 943_718 diff --git a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml index 91402212177..d857fa53dab 100644 --- a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml +++ b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml @@ -12,8 +12,8 @@ knowledge = "2025-03-31" open_weights = true [cost] -input = 0.2 -output = 0.7 +input = 0.13 +output = 0.52 [limit] context = 262_144 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 189296e3dde..5aff2dc4d72 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.04928 -output = 0.14784 -cache_read = 0.001568 +input = 0.0484 +output = 0.1452 +cache_read = 0.00154 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index d1d254d74de..b15245b9227 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 2.0695 +output = 11.5892 +cache_read = 0.240062 [limit] context = 1_048_576 From 7e6dd18cd010b678157324056be2fe1be53ac0b3 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 04:28:53 +0000 Subject: [PATCH 053/392] chore(sync): update Kilo model catalog (#7473) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/moonshotai/kimi-k3.toml | 6 +++--- .../kilo/models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index 3af9cc14392..7965ee661dd 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 2.0695 +output = 11.5892 +cache_read = 0.240062 [limit] output = 943_718 diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 50af69be57a..4cea316de5e 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.04928 -output = 0.14784 -cache_read = 0.001568 +input = 0.0484 +output = 0.1452 +cache_read = 0.00154 [limit] context = 1_024_000 diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 9bbf3c7be77..86ff6cbcca0 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 2.0695 +output = 11.5892 +cache_read = 0.240062 [limit] context = 1_048_576 From 043d5793090f6a78503f36abf97214782f10281a Mon Sep 17 00:00:00 2001 From: Daniel Chen Date: Fri, 18 Sep 2026 21:30:58 -0700 Subject: [PATCH 054/392] feat(opencode):added deepseek-v4.1-flash to zen --- providers/opencode/models/deepseek-v4.1-flash.toml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 providers/opencode/models/deepseek-v4.1-flash.toml diff --git a/providers/opencode/models/deepseek-v4.1-flash.toml b/providers/opencode/models/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..20a4928d010 --- /dev/null +++ b/providers/opencode/models/deepseek-v4.1-flash.toml @@ -0,0 +1,14 @@ +base_model = "deepseek/deepseek-v4.1-flash" +name = "DeepSeek V4.1 Flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.006 From ebfeed5f5eb53e48f6fdf659cf9907ee0172ec03 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 00:17:06 -0500 Subject: [PATCH 055/392] feat: add GLM-5.3-FlashX to Z.AI providers (#7475) Co-authored-by: Aiden Cline --- providers/zai/models/glm-5.3-flash.toml | 7 ++++--- providers/zai/models/glm-5.3-flashx.toml | 21 ++++++++++++++++++++ providers/zhipuai/models/glm-5.3-flash.toml | 9 ++++----- providers/zhipuai/models/glm-5.3-flashx.toml | 21 ++++++++++++++++++++ 4 files changed, 50 insertions(+), 8 deletions(-) create mode 100644 providers/zai/models/glm-5.3-flashx.toml create mode 100644 providers/zhipuai/models/glm-5.3-flashx.toml diff --git a/providers/zai/models/glm-5.3-flash.toml b/providers/zai/models/glm-5.3-flash.toml index 59029d6336a..5850a4727dc 100644 --- a/providers/zai/models/glm-5.3-flash.toml +++ b/providers/zai/models/glm-5.3-flash.toml @@ -2,6 +2,7 @@ base_model = "zhipuai/glm-5.3-flash" # GLM-5.3-Flash reasons by default; effort levels low|high|max, and thinking can # be suppressed with {"thinking": {"type": "disabled"}}. # https://z.ai/blog/glm-5.3-flash +# Pricing: https://docs.z.ai/guides/overview/pricing (accessed 2026-09-19) # Served by https://api.z.ai/api/paas/v4 (verified against GET /models 2026-08-26). [[reasoning_options]] @@ -12,7 +13,7 @@ values = ["low", "high", "max"] field = "reasoning_content" [cost] -input = 0.075 -output = 0.25 -cache_read = 0.015 +input = 0.15 +output = 0.50 +cache_read = 0.03 cache_write = 0 diff --git a/providers/zai/models/glm-5.3-flashx.toml b/providers/zai/models/glm-5.3-flashx.toml new file mode 100644 index 00000000000..059cb836933 --- /dev/null +++ b/providers/zai/models/glm-5.3-flashx.toml @@ -0,0 +1,21 @@ +# GLM-5.3-FlashX is the high-speed serving option for GLM-5.3-Flash. +# https://docs.z.ai/guides/vlm/glm-5.3-flash +# Pricing: https://docs.z.ai/guides/overview/pricing (accessed 2026-09-19) +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3-FlashX" +description = "High-speed GLM-5.3-Flash serving option for coding and agent workflows" +release_date = "2026-09-18" +last_updated = "2026-09-18" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.37 +output = 1.25 +cache_read = 0.075 +cache_write = 0 diff --git a/providers/zhipuai/models/glm-5.3-flash.toml b/providers/zhipuai/models/glm-5.3-flash.toml index 84efd9a9cef..d2ff44820e7 100644 --- a/providers/zhipuai/models/glm-5.3-flash.toml +++ b/providers/zhipuai/models/glm-5.3-flash.toml @@ -1,8 +1,7 @@ # GLM-5.3-Flash always reasons (thinking cannot be disabled); text params # match GLM-5.3 — effort low|high|max with default max. # https://docs.bigmodel.cn/cn/guide/models/vlm/glm-5.3-flash -# Cost: https://docs.z.ai/guides/overview/pricing (accessed 2026-08-26) -# GLM-5.3-Flash 50% promo ends 2026-09-09 24:00 UTC+8; list $0.15/$0.03/$0.50. +# Pricing: https://docs.z.ai/guides/overview/pricing (accessed 2026-09-19) base_model = "zhipuai/glm-5.3-flash" [[reasoning_options]] @@ -13,7 +12,7 @@ values = ["low", "high", "max"] field = "reasoning_content" [cost] -input = 0.075 -output = 0.25 -cache_read = 0.015 +input = 0.15 +output = 0.50 +cache_read = 0.03 cache_write = 0 diff --git a/providers/zhipuai/models/glm-5.3-flashx.toml b/providers/zhipuai/models/glm-5.3-flashx.toml new file mode 100644 index 00000000000..851a6969e41 --- /dev/null +++ b/providers/zhipuai/models/glm-5.3-flashx.toml @@ -0,0 +1,21 @@ +# GLM-5.3-FlashX is the high-speed serving option for GLM-5.3-Flash. +# https://docs.bigmodel.cn/cn/guide/models/vlm/glm-5.3-flash +# Pricing: https://docs.z.ai/guides/overview/pricing (accessed 2026-09-19) +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3-FlashX" +description = "High-speed GLM-5.3-Flash serving option for coding and agent workflows" +release_date = "2026-09-18" +last_updated = "2026-09-18" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.37 +output = 1.25 +cache_read = 0.075 +cache_write = 0 From a25d952334e2331e6649b1e054b9a7b4579dd978 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 05:25:15 +0000 Subject: [PATCH 056/392] chore(sync): update OpenRouter model catalog (#7476) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml | 4 ++-- 3 files changed, 8 insertions(+), 8 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index b5c7d255938..cc0ba12eb64 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04816 -output = 0.09632 -cache_read = 0.009632 +input = 0.0476 +output = 0.0952 +cache_read = 0.00952 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index ff6ad910cd9..b670ae33837 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.614916 -output = 1.229832 -cache_read = 0.051243 +input = 0.594558 +output = 1.189116 +cache_read = 0.049547 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml index d857fa53dab..91402212177 100644 --- a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml +++ b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml @@ -12,8 +12,8 @@ knowledge = "2025-03-31" open_weights = true [cost] -input = 0.13 -output = 0.52 +input = 0.2 +output = 0.7 [limit] context = 262_144 From b979ab5354dd2b222740c3cce5eda0db8d281394 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 06:37:07 +0000 Subject: [PATCH 057/392] chore(sync): update OpenRouter model catalog (#7478) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/moonshotai/kimi-k3.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- 5 files changed, 15 insertions(+), 15 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index cc0ba12eb64..c5c46f3604a 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.0476 -output = 0.0952 -cache_read = 0.00952 +input = 0.04732 +output = 0.09464 +cache_read = 0.009464 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index b670ae33837..f453f0bbdbd 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.594558 -output = 1.189116 -cache_read = 0.049547 +input = 0.564282 +output = 1.128564 +cache_read = 0.047024 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index d374007e872..c1033f6ad0b 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.0695 -output = 11.5892 -cache_read = 0.240062 +input = 1.875 +output = 10.5 +cache_read = 0.2175 [limit] output = 943_718 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 5aff2dc4d72..8324901e3db 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.0484 -output = 0.1452 -cache_read = 0.00154 +input = 0.04752 +output = 0.14256 +cache_read = 0.001512 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index b15245b9227..ec32a69e6bd 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.0695 -output = 11.5892 -cache_read = 0.240062 +input = 1.875 +output = 10.5 +cache_read = 0.2175 [limit] context = 1_048_576 From d25c8085b265e08910be5c834098c7afa65b58ba Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 06:37:12 +0000 Subject: [PATCH 058/392] chore(sync): update Kilo model catalog (#7480) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/moonshotai/kimi-k3.toml | 6 ++-- .../models/openai/gpt-5.6-sol-discounted.toml | 30 ------------------- .../~deepseek/deepseek-v4-flash-latest.toml | 6 ++-- .../kilo/models/~moonshotai/kimi-latest.toml | 6 ++-- 4 files changed, 9 insertions(+), 39 deletions(-) delete mode 100644 providers/kilo/models/openai/gpt-5.6-sol-discounted.toml diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index 7965ee661dd..90c5d22728f 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.0695 -output = 11.5892 -cache_read = 0.240062 +input = 1.875 +output = 10.5 +cache_read = 0.2175 [limit] output = 943_718 diff --git a/providers/kilo/models/openai/gpt-5.6-sol-discounted.toml b/providers/kilo/models/openai/gpt-5.6-sol-discounted.toml deleted file mode 100644 index 6008af017c3..00000000000 --- a/providers/kilo/models/openai/gpt-5.6-sol-discounted.toml +++ /dev/null @@ -1,30 +0,0 @@ -name = "OpenAI: GPT-5.6 Sol (50% off)" -description = "GPT-5.6 Sol served by OpenAI through Vercel AI Gateway at 50% lower cost than other available inference providers. This promotion runs through September 18, 2026." -family = "gpt" -release_date = "2025-08-26" -last_updated = "2025-08-26" -attachment = true -reasoning = true -temperature = true -tool_call = true -structured_output = false -open_weights = false - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 2 -output = 10 -reasoning = 0 -cache_read = 0.2 -cache_write = 2.5 - -[limit] -context = 1_050_000 -output = 128_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 4cea316de5e..87db0089581 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.0484 -output = 0.1452 -cache_read = 0.00154 +input = 0.04752 +output = 0.14256 +cache_read = 0.001512 [limit] context = 1_024_000 diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 86ff6cbcca0..f09092fb568 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 2.0695 -output = 11.5892 -cache_read = 0.240062 +input = 1.875 +output = 10.5 +cache_read = 0.2175 [limit] context = 1_048_576 From 97816c1e80054356e29db7c4542a257248d0594e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 06:37:15 +0000 Subject: [PATCH 059/392] chore(sync): update Vercel AI Gateway model catalog (#7479) Co-authored-by: opencode-agent[bot] --- providers/vercel/models/openai/gpt-5.6-sol-fast.toml | 8 ++++---- providers/vercel/models/openai/gpt-5.6-sol.toml | 8 ++++---- 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/providers/vercel/models/openai/gpt-5.6-sol-fast.toml b/providers/vercel/models/openai/gpt-5.6-sol-fast.toml index adf0da2fcb0..007699dc75e 100644 --- a/providers/vercel/models/openai/gpt-5.6-sol-fast.toml +++ b/providers/vercel/models/openai/gpt-5.6-sol-fast.toml @@ -6,7 +6,7 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 4 -output = 20 -cache_read = 0.4 -cache_write = 5 +input = 8 +output = 40 +cache_read = 0.8 +cache_write = 10 diff --git a/providers/vercel/models/openai/gpt-5.6-sol.toml b/providers/vercel/models/openai/gpt-5.6-sol.toml index 8b88a743a65..b07d7577a9a 100644 --- a/providers/vercel/models/openai/gpt-5.6-sol.toml +++ b/providers/vercel/models/openai/gpt-5.6-sol.toml @@ -8,7 +8,7 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 2 -output = 10 -cache_read = 0.2 -cache_write = 2.5 +input = 4 +output = 20 +cache_read = 0.4 +cache_write = 5 From 1337abc7d3d91826557f6cf4e87c45ccc5424f60 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 07:25:51 +0000 Subject: [PATCH 060/392] chore(sync): update OpenRouter model catalog (#7488) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-0731.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/moonshotai/kimi-k3.toml | 6 +++--- .../models/~deepseek/deepseek-flash-latest.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 8 ++++---- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- 6 files changed, 19 insertions(+), 19 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml index d06dd2cd8d1..8dce0b17a1e 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.065 -output = 0.18 -cache_read = 0.016 +input = 0.04 +output = 0.08 +cache_read = 0.008 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index c5c46f3604a..664a084d4ad 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04732 -output = 0.09464 -cache_read = 0.009464 +input = 0.04704 +output = 0.09408 +cache_read = 0.009408 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index c1033f6ad0b..22b1407542a 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.875 -output = 10.5 -cache_read = 0.2175 +input = 1.7 +output = 8.5 +cache_read = 0.17 [limit] output = 943_718 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index bdcd94e72e3..643ffd542be 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.135 -output = 0.54 -cache_read = 0.00405 +input = 0.13 +output = 0.52 +cache_read = 0.0026 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 8324901e3db..a8bf0f0a408 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.04752 -output = 0.14256 -cache_read = 0.001512 +input = 0.04 +output = 0.08 +cache_read = 0.008 [limit] context = 1_310_720 -output = 384_000 +output = 943_718 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index ec32a69e6bd..7031b748f98 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.875 -output = 10.5 -cache_read = 0.2175 +input = 1.7 +output = 8.5 +cache_read = 0.17 [limit] context = 1_048_576 From 8fcdf73a0977a6e0b7b222c24556ac19a347f0da Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 07:26:02 +0000 Subject: [PATCH 061/392] chore(sync): update Kilo model catalog (#7487) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/moonshotai/kimi-k3.toml | 6 +++--- .../kilo/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 10 +++++----- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- 4 files changed, 14 insertions(+), 14 deletions(-) diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index 90c5d22728f..fd81e087ca6 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.875 -output = 10.5 -cache_read = 0.2175 +input = 1.7 +output = 8.5 +cache_read = 0.17 [limit] output = 943_718 diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 0b2d5078274..9e0f74cdd43 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.135 -output = 0.54 -cache_read = 0.00405 +input = 0.13 +output = 0.52 +cache_read = 0.0026 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 87db0089581..c34bd2f2e08 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.04752 -output = 0.14256 -cache_read = 0.001512 +input = 0.04 +output = 0.08 +cache_read = 0.008 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 943_718 [modalities] input = ["text"] diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index f09092fb568..631c1758bcd 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.875 -output = 10.5 -cache_read = 0.2175 +input = 1.7 +output = 8.5 +cache_read = 0.17 [limit] context = 1_048_576 From bfc571555f1f22a908b248c30e2be9a248eddaba Mon Sep 17 00:00:00 2001 From: Frank Date: Sat, 19 Sep 2026 04:07:21 -0400 Subject: [PATCH 062/392] update zen models --- providers/opencode/models/jev-1.13-free.toml | 7 +++++++ providers/opencode/models/jev-1.13.toml | 8 ++++++++ 2 files changed, 15 insertions(+) create mode 100644 providers/opencode/models/jev-1.13-free.toml create mode 100644 providers/opencode/models/jev-1.13.toml diff --git a/providers/opencode/models/jev-1.13-free.toml b/providers/opencode/models/jev-1.13-free.toml new file mode 100644 index 00000000000..3c2a23a4f39 --- /dev/null +++ b/providers/opencode/models/jev-1.13-free.toml @@ -0,0 +1,7 @@ +# Availability: https://opencode.ai/zen/v1/models +base_model = "typesafe/jev-latest" +name = "Jev 1.13 Free" + +[cost] +input = 0 +output = 0 diff --git a/providers/opencode/models/jev-1.13.toml b/providers/opencode/models/jev-1.13.toml new file mode 100644 index 00000000000..c31caeddd67 --- /dev/null +++ b/providers/opencode/models/jev-1.13.toml @@ -0,0 +1,8 @@ +# Availability: https://opencode.ai/zen/v1/models +# Pricing: https://docs.typesafe.ai/models.md +base_model = "typesafe/jev-latest" +name = "Jev 1.13" + +[cost] +input = 0.042 +output = 0 From 0cc3d9e4b02208767a532fd4c43a2a72892c77d9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 08:30:20 +0000 Subject: [PATCH 063/392] chore(sync): update OpenRouter model catalog (#7490) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 664a084d4ad..f3c7c5a82dc 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04704 -output = 0.09408 -cache_read = 0.009408 +input = 0.04592 +output = 0.09184 +cache_read = 0.009184 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index f453f0bbdbd..b7faf4d187f 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.564282 -output = 1.128564 -cache_read = 0.047024 +input = 0.533832 +output = 1.067664 +cache_read = 0.044486 [limit] context = 1_048_576 From 66563f8f71f4077633ceaa18ec4196f48df43188 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 09:25:43 +0000 Subject: [PATCH 064/392] chore(sync): update OpenRouter model catalog (#7492) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash-0731.toml | 2 +- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- 4 files changed, 8 insertions(+), 8 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml index 8dce0b17a1e..5e9065ff206 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml @@ -12,7 +12,7 @@ values = ["low", "high", "max"] [cost] input = 0.04 output = 0.08 -cache_read = 0.008 +cache_read = 0.016 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index f3c7c5a82dc..a93a1a5d132 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04592 -output = 0.09184 -cache_read = 0.009184 +input = 0.04536 +output = 0.09072 +cache_read = 0.009072 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index b7faf4d187f..d8e373b8232 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.533832 -output = 1.067664 -cache_read = 0.044486 +input = 0.503382 +output = 1.006764 +cache_read = 0.041949 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index a8bf0f0a408..cd84b32afe3 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -22,7 +22,7 @@ values = ["low", "high", "max"] [cost] input = 0.04 output = 0.08 -cache_read = 0.008 +cache_read = 0.016 [limit] context = 1_310_720 From a8604c4080b59f43b16bb85f31f70591a90f57d6 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 09:25:48 +0000 Subject: [PATCH 065/392] chore(sync): update Kilo model catalog (#7491) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index c34bd2f2e08..2c01541878c 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -17,7 +17,7 @@ values = ["none", "low", "high", "max"] [cost] input = 0.04 output = 0.08 -cache_read = 0.008 +cache_read = 0.016 [limit] context = 1_048_576 From b507a904ffd5408f3f207e9c073d9bc0e1f6c8a5 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 09:26:21 +0000 Subject: [PATCH 066/392] chore(sync): update NanoGPT model catalog (#7493) Co-authored-by: opencode-agent[bot] --- .../models/{slowburn => google}/gemma4-31b-splituntied.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) rename providers/nano-gpt/models/{slowburn => google}/gemma4-31b-splituntied.toml (65%) diff --git a/providers/nano-gpt/models/slowburn/gemma4-31b-splituntied.toml b/providers/nano-gpt/models/google/gemma4-31b-splituntied.toml similarity index 65% rename from providers/nano-gpt/models/slowburn/gemma4-31b-splituntied.toml rename to providers/nano-gpt/models/google/gemma4-31b-splituntied.toml index 60615a27959..2a2440d5a87 100644 --- a/providers/nano-gpt/models/slowburn/gemma4-31b-splituntied.toml +++ b/providers/nano-gpt/models/google/gemma4-31b-splituntied.toml @@ -1,5 +1,5 @@ name = "Gemma 4 31B Split-Untied" -description = "Slowburn's Split-Untied is a text-only Gemma 4 31B community finetune with an untied BF16 output head, built for creative writing, roleplay, expressive dialogue, and tool use." +description = "Blazed-Forge's Split-Untied is a text-only Gemma 4 31B community finetune with an untied BF16 output head, built for creative writing, roleplay, expressive dialogue, and tool use." family = "gemma" release_date = "2026-09-17" last_updated = "2026-09-17" From 66b55e0e71c11c6cc3426c0d344f7293fbc8d14c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 10:24:15 +0000 Subject: [PATCH 067/392] chore(sync): update OpenRouter model catalog (#7496) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index a93a1a5d132..be38a00704b 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04536 -output = 0.09072 -cache_read = 0.009072 +input = 0.04508 +output = 0.09016 +cache_read = 0.009016 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index d8e373b8232..cf0b7555b6f 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.503382 -output = 1.006764 -cache_read = 0.041949 +input = 0.49329 +output = 0.98658 +cache_read = 0.041108 [limit] context = 1_048_576 From 19203ba786113081f7f6eb76720915f89bbc017a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 11:22:24 +0000 Subject: [PATCH 068/392] chore(sync): update OpenRouter model catalog (#7497) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index be38a00704b..99fb509a97e 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04508 -output = 0.09016 -cache_read = 0.009016 +input = 0.04452 +output = 0.08904 +cache_read = 0.008904 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index cf0b7555b6f..8f63f2cd80d 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.49329 -output = 0.98658 -cache_read = 0.041108 +input = 0.46284 +output = 0.92568 +cache_read = 0.03857 [limit] context = 1_048_576 From 34885ce1a71bdd23359484a4cf1a40d6e2c143f6 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 12:32:48 +0000 Subject: [PATCH 069/392] chore(sync): update OpenRouter model catalog (#7499) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 99fb509a97e..638aaa8a0cd 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04452 -output = 0.08904 -cache_read = 0.008904 +input = 0.04368 +output = 0.08736 +cache_read = 0.008736 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 8f63f2cd80d..5bc792c3ae0 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.46284 -output = 0.92568 -cache_read = 0.03857 +input = 0.422298 +output = 0.844596 +cache_read = 0.035192 [limit] context = 1_048_576 From 2111a4c4ecb43abb4d7bb823e6b4fec2147731b1 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 12:32:57 +0000 Subject: [PATCH 070/392] chore(sync): update NanoGPT model catalog (#7498) Co-authored-by: opencode-agent[bot] --- .../gemma-4-26b-a4b-it-cybersecurity.toml | 25 +++++++++++++++++ .../nvidia/nemotron-3.5-content-safety.toml | 17 +++++++++++ .../qwen/qwen3.8-27b-cybersecurity.toml | 25 +++++++++++++++++ .../models/qwen/qwen3.8-27b-uncensored.toml | 6 ++-- .../qwen/qwen3.8-27b-uncensored:thinking.toml | 6 ++-- .../z-ai/glm-5.3-flash-cybersecurity.toml | 28 +++++++++++++++++++ 6 files changed, 101 insertions(+), 6 deletions(-) create mode 100644 providers/nano-gpt/models/google/gemma-4-26b-a4b-it-cybersecurity.toml create mode 100644 providers/nano-gpt/models/nvidia/nemotron-3.5-content-safety.toml create mode 100644 providers/nano-gpt/models/qwen/qwen3.8-27b-cybersecurity.toml create mode 100644 providers/nano-gpt/models/z-ai/glm-5.3-flash-cybersecurity.toml diff --git a/providers/nano-gpt/models/google/gemma-4-26b-a4b-it-cybersecurity.toml b/providers/nano-gpt/models/google/gemma-4-26b-a4b-it-cybersecurity.toml new file mode 100644 index 00000000000..502d73081ea --- /dev/null +++ b/providers/nano-gpt/models/google/gemma-4-26b-a4b-it-cybersecurity.toml @@ -0,0 +1,25 @@ +name = "Gemma 4 26B A4B Cybersecurity" +description = "Gemma 4 26B A4B Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports optional reasoning, image understanding, tool calling, and a 262,144-token context window." +family = "gemma" +release_date = "2026-09-19" +last_updated = "2026-09-19" +attachment = true +reasoning = true +tool_call = true +structured_output = false +open_weights = true +reasoning_options = [] + +[cost] +input = 0.1056 +output = 0.3344 +cache_read = 0.0528 + +[limit] +context = 262_144 +input = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/nano-gpt/models/nvidia/nemotron-3.5-content-safety.toml b/providers/nano-gpt/models/nvidia/nemotron-3.5-content-safety.toml new file mode 100644 index 00000000000..8564f16dea6 --- /dev/null +++ b/providers/nano-gpt/models/nvidia/nemotron-3.5-content-safety.toml @@ -0,0 +1,17 @@ +base_model = "nvidia/nemotron-3.5-content-safety" +attachment = false +structured_output = false +reasoning_options = [] + +[cost] +input = 0.05 +output = 0.15 +cache_read = 0.025 + +[limit] +context = 131_072 +input = 131_072 +output = 32_768 + +[modalities] +input = ["text"] diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-cybersecurity.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-cybersecurity.toml new file mode 100644 index 00000000000..fcae44481c3 --- /dev/null +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-cybersecurity.toml @@ -0,0 +1,25 @@ +name = "Qwen 3.8 27B Cybersecurity" +description = "Qwen 3.8 27B Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports optional reasoning, image understanding, tool calling, and a 262,144-token context window." +family = "qwen" +release_date = "2026-09-19" +last_updated = "2026-09-19" +attachment = true +reasoning = true +tool_call = true +structured_output = false +open_weights = true +reasoning_options = [] + +[cost] +input = 0.1 +output = 0.6 +cache_read = 0.05 + +[limit] +context = 262_144 +input = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml index 516d35fc982..e34ca7cb8aa 100644 --- a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml @@ -11,9 +11,9 @@ open_weights = true reasoning_options = [] [cost] -input = 0.25 -output = 1.5 -cache_read = 0.125 +input = 0.2 +output = 1.7 +cache_read = 0.2 [limit] context = 524_288 diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml index 169511f2187..17f4a01df20 100644 --- a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml @@ -11,9 +11,9 @@ open_weights = true reasoning_options = [] [cost] -input = 0.25 -output = 1.5 -cache_read = 0.125 +input = 0.2 +output = 1.7 +cache_read = 0.2 [limit] context = 524_288 diff --git a/providers/nano-gpt/models/z-ai/glm-5.3-flash-cybersecurity.toml b/providers/nano-gpt/models/z-ai/glm-5.3-flash-cybersecurity.toml new file mode 100644 index 00000000000..b55aaf3a6a5 --- /dev/null +++ b/providers/nano-gpt/models/z-ai/glm-5.3-flash-cybersecurity.toml @@ -0,0 +1,28 @@ +name = "GLM 5.3 Flash Cybersecurity" +description = "GLM 5.3 Flash Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports always-on reasoning, image understanding, tool calling, and a 1,048,576-token context window." +family = "glm" +release_date = "2026-09-19" +last_updated = "2026-09-19" +attachment = true +reasoning = true +tool_call = true +structured_output = false +open_weights = true + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.075 + +[limit] +context = 1_048_576 +input = 1_048_576 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] From 797f773e891df6e8c800464d20a01bc694150e25 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 13:23:47 +0000 Subject: [PATCH 071/392] chore(sync): update OpenRouter model catalog (#7502) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/z-ai/glm-4.6.toml | 9 +++------ 2 files changed, 6 insertions(+), 9 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 638aaa8a0cd..ac1de47f47f 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04368 -output = 0.08736 -cache_read = 0.008736 +input = 0.04312 +output = 0.08624 +cache_read = 0.008624 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-4.6.toml b/providers/openrouter/models/z-ai/glm-4.6.toml index 9a813d7b44e..e562f006f6c 100644 --- a/providers/openrouter/models/z-ai/glm-4.6.toml +++ b/providers/openrouter/models/z-ai/glm-4.6.toml @@ -7,9 +7,6 @@ structured_output = true type = "toggle" [cost] -input = 0.43 -output = 1.75 -cache_read = 0.08 - -[limit] -output = 16_384 +input = 0.5 +output = 2 +cache_read = 0.1 From 29e117b6ea7fa6dc95ec3cdc4a270ddaab8f973a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 13:23:59 +0000 Subject: [PATCH 072/392] chore(sync): update Kilo model catalog (#7503) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/z-ai/glm-4.6.toml | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/providers/kilo/models/z-ai/glm-4.6.toml b/providers/kilo/models/z-ai/glm-4.6.toml index 2a611ba49e7..fd68fe0c68b 100644 --- a/providers/kilo/models/z-ai/glm-4.6.toml +++ b/providers/kilo/models/z-ai/glm-4.6.toml @@ -7,10 +7,9 @@ type = "effort" values = ["none", "high"] [cost] -input = 0.43 -output = 1.75 -cache_read = 0.08 +input = 0.5 +output = 2 +cache_read = 0.1 [limit] -context = 198_000 -output = 16_384 +context = 202_752 From add971a83271a43e7487626eec10131f14763ed7 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 14:23:46 +0000 Subject: [PATCH 073/392] chore(sync): update NanoGPT model catalog (#7505) Co-authored-by: opencode-agent[bot] --- providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml | 2 +- .../nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml index e34ca7cb8aa..612900717c2 100644 --- a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml @@ -13,7 +13,7 @@ reasoning_options = [] [cost] input = 0.2 output = 1.7 -cache_read = 0.2 +cache_read = 0.18 [limit] context = 524_288 diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml index 17f4a01df20..46cc8830338 100644 --- a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml @@ -13,7 +13,7 @@ reasoning_options = [] [cost] input = 0.2 output = 1.7 -cache_read = 0.2 +cache_read = 0.18 [limit] context = 524_288 From d3acf770c2294aeda45e8892205dc624b378dd3f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 14:23:48 +0000 Subject: [PATCH 074/392] chore(sync): update OpenRouter model catalog (#7504) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index ac1de47f47f..b4277c2cb62 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04312 -output = 0.08624 -cache_read = 0.008624 +input = 0.04284 +output = 0.08568 +cache_read = 0.008568 [limit] context = 1_048_576 From 7d315b77805625381ea9f5c663fdcc019c99dcd9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 15:23:21 +0000 Subject: [PATCH 075/392] chore(sync): update OpenRouter model catalog (#7506) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index b4277c2cb62..37587a85fb8 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04284 -output = 0.08568 -cache_read = 0.008568 +input = 0.04228 +output = 0.08456 +cache_read = 0.008456 [limit] context = 1_048_576 From b638683cd779bde5e76708f96b895075dc1a024d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 16:25:26 +0000 Subject: [PATCH 076/392] chore(sync): update LLM Gateway model catalog (#7508) Co-authored-by: opencode-agent[bot] --- .../models/vichar-ai/glm-5.3-flash.toml | 20 ------------------- .../models/vichar-ai/glm-5.3.toml | 16 --------------- 2 files changed, 36 deletions(-) delete mode 100644 providers/llmgateway-providers/models/vichar-ai/glm-5.3-flash.toml delete mode 100644 providers/llmgateway-providers/models/vichar-ai/glm-5.3.toml diff --git a/providers/llmgateway-providers/models/vichar-ai/glm-5.3-flash.toml b/providers/llmgateway-providers/models/vichar-ai/glm-5.3-flash.toml deleted file mode 100644 index a8d0eb4cba1..00000000000 --- a/providers/llmgateway-providers/models/vichar-ai/glm-5.3-flash.toml +++ /dev/null @@ -1,20 +0,0 @@ -base_model = "zhipuai/glm-5.3-flash" -name = "GLM-5.3 Flash (vichar-ai)" -attachment = false -structured_output = false - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0.15 -output = 0.5 -cache_read = 0.03 - -[limit] -context = 1_048_000 -output = 128_000 - -[modalities] -input = ["text"] diff --git a/providers/llmgateway-providers/models/vichar-ai/glm-5.3.toml b/providers/llmgateway-providers/models/vichar-ai/glm-5.3.toml deleted file mode 100644 index 561369dc9e2..00000000000 --- a/providers/llmgateway-providers/models/vichar-ai/glm-5.3.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "zhipuai/glm-5.3" -name = "GLM-5.3 (vichar-ai)" -structured_output = false - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 1.4 -output = 4.4 -cache_read = 0.26 - -[limit] -context = 1_048_000 -output = 128_000 From b55fb9f10b8499c4d7bdfd63b100bce976d634b9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 16:25:32 +0000 Subject: [PATCH 077/392] chore(sync): update OpenRouter model catalog (#7507) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml | 4 ++-- providers/openrouter/models/tencent/hy3.toml | 6 +++--- 3 files changed, 8 insertions(+), 8 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 37587a85fb8..ff66884b828 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04228 -output = 0.08456 -cache_read = 0.008456 +input = 0.04172 +output = 0.08344 +cache_read = 0.008344 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml index 91402212177..d857fa53dab 100644 --- a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml +++ b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml @@ -12,8 +12,8 @@ knowledge = "2025-03-31" open_weights = true [cost] -input = 0.2 -output = 0.7 +input = 0.13 +output = 0.52 [limit] context = 262_144 diff --git a/providers/openrouter/models/tencent/hy3.toml b/providers/openrouter/models/tencent/hy3.toml index 80ecfdd3aa8..f61fa3175e9 100644 --- a/providers/openrouter/models/tencent/hy3.toml +++ b/providers/openrouter/models/tencent/hy3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.132 -output = 0.528 -cache_read = 0.033 +input = 0.0825 +output = 0.33 +cache_read = 0.020625 [limit] context = 262_144 From 6b9d0a0d2de2753979c3c503baae4f9d35026493 Mon Sep 17 00:00:00 2001 From: bpnrc Date: Sat, 19 Sep 2026 19:11:25 +0200 Subject: [PATCH 078/392] chore: update TensorX model catalog (add 6, remove 7 retired) (#7486) * chore: update TensorX model catalog (add 6, remove 7 retired) Sync TensorX provider catalog against the live GET /v1/models endpoint. Add (6): - z-ai/glm-5.3 (effort low|high|max) - z-ai/glm-5.3-flash (effort low|high|max) - qwen/qwen3.8-2.4t-a95b (effort low|medium|xhigh) - qwen/qwen3.8-27b (effort low|medium|xhigh) - qwen/qwen3.8-flash-next (effort low|medium|xhigh) - deepseek/deepseek-v4-pro-0813 (toggle thinking) Remove (7, no longer served): deepseek-chat-v3.1, deepseek-v4-flash, nemotron-3-super-120b-a12b, gpt-oss-120b, qwen3-coder-30b-a3b-instruct, qwen3-vl-235b-a22b-instruct, glm-4.7. Pricing, context/output limits and reasoning_options verified against the live TensorX API (2026-09-19). * fix: doc-accurate reasoning_options and override-only limits for new TensorX models Align reasoning controls with docs.tensorx.ai/api-reference/reasoning: - GLM 5.3 / 5.3-Flash: toggle enable_thinking + effort low|high - DeepSeek V4 Pro 0813: toggle thinking + effort high|max - Qwen 3.8: effort low|medium|xhigh (always-on) Drop redundant [limit] overrides that merely restate the lab (Qwen 3.8 context/output) and keep only host deltas (output = 64000). * fix: GLM-5.3 effort includes max; exact toggle wire paths - GLM 5.3 / 5.3-Flash: effort low|high|max (docs + live probe confirm max is accepted; max is the default depth). Note that TensorX allows disabling thinking (enable_thinking) unlike first-party Z.AI. - DeepSeek V4 Pro 0813: document the exact toggle wire path chat_template_kwargs.thinking in a leading header. --------- Co-authored-by: bpnrc --- .../models/deepseek/deepseek-chat-v3.1.toml | 30 ------------------- .../models/deepseek/deepseek-v4-flash.toml | 13 -------- .../models/deepseek/deepseek-v4-pro-0813.toml | 19 ++++++++++++ .../nvidia/nemotron-3-super-120b-a12b.toml | 11 ------- .../tensorx/models/openai/gpt-oss-120b.toml | 12 -------- .../qwen/qwen3-coder-30b-a3b-instruct.toml | 10 ------- .../qwen/qwen3-vl-235b-a22b-instruct.toml | 25 ---------------- .../models/qwen/qwen3.8-2.4t-a95b.toml | 15 ++++++++++ .../tensorx/models/qwen/qwen3.8-27b.toml | 11 +++++++ .../models/qwen/qwen3.8-flash-next.toml | 15 ++++++++++ providers/tensorx/models/z-ai/glm-4.7.toml | 19 ------------ .../tensorx/models/z-ai/glm-5.3-flash.toml | 20 +++++++++++++ providers/tensorx/models/z-ai/glm-5.3.toml | 20 +++++++++++++ 13 files changed, 100 insertions(+), 120 deletions(-) delete mode 100644 providers/tensorx/models/deepseek/deepseek-chat-v3.1.toml delete mode 100644 providers/tensorx/models/deepseek/deepseek-v4-flash.toml create mode 100644 providers/tensorx/models/deepseek/deepseek-v4-pro-0813.toml delete mode 100644 providers/tensorx/models/nvidia/nemotron-3-super-120b-a12b.toml delete mode 100644 providers/tensorx/models/openai/gpt-oss-120b.toml delete mode 100644 providers/tensorx/models/qwen/qwen3-coder-30b-a3b-instruct.toml delete mode 100644 providers/tensorx/models/qwen/qwen3-vl-235b-a22b-instruct.toml create mode 100644 providers/tensorx/models/qwen/qwen3.8-2.4t-a95b.toml create mode 100644 providers/tensorx/models/qwen/qwen3.8-27b.toml create mode 100644 providers/tensorx/models/qwen/qwen3.8-flash-next.toml delete mode 100644 providers/tensorx/models/z-ai/glm-4.7.toml create mode 100644 providers/tensorx/models/z-ai/glm-5.3-flash.toml create mode 100644 providers/tensorx/models/z-ai/glm-5.3.toml diff --git a/providers/tensorx/models/deepseek/deepseek-chat-v3.1.toml b/providers/tensorx/models/deepseek/deepseek-chat-v3.1.toml deleted file mode 100644 index 93143b88c68..00000000000 --- a/providers/tensorx/models/deepseek/deepseek-chat-v3.1.toml +++ /dev/null @@ -1,30 +0,0 @@ -name = "DeepSeek Chat V3.1" -description = "DeepSeek chat model for instruction following, coding, and analysis" -family = "deepseek" -release_date = "2025-08-21" -last_updated = "2025-08-21" -attachment = false -reasoning = true -temperature = true -tool_call = true -knowledge = "2024-11" -open_weights = true - - -[[reasoning_options]] -type = "effort" # API: {"reasoning_effort": }; "none" disables reasoning -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.2 -output = 0.8 -cache_read = 0.05 -cache_write = 0.25 - -[limit] -context = 164000 -output = 163840 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/tensorx/models/deepseek/deepseek-v4-flash.toml b/providers/tensorx/models/deepseek/deepseek-v4-flash.toml deleted file mode 100644 index fdea8b9a03d..00000000000 --- a/providers/tensorx/models/deepseek/deepseek-v4-flash.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "deepseek/deepseek-v4-flash" - -[[reasoning_options]] -type = "toggle" # API: {"chat_template_kwargs": {"thinking": true}} (default off) - -[cost] -input = 0.15 -output = 0.3 -cache_read = 0.0375 -cache_write = 0.1875 - -[limit] -context = 1048576 diff --git a/providers/tensorx/models/deepseek/deepseek-v4-pro-0813.toml b/providers/tensorx/models/deepseek/deepseek-v4-pro-0813.toml new file mode 100644 index 00000000000..82492863b2c --- /dev/null +++ b/providers/tensorx/models/deepseek/deepseek-v4-pro-0813.toml @@ -0,0 +1,19 @@ +base_model = "deepseek/deepseek-v4-pro-0813" +# DeepSeek V4 reasons off by default; toggle chat_template_kwargs.thinking, effort high|max when on. +# https://docs.tensorx.ai/api-reference/reasoning + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["high", "max"] + +[cost] +input = 2.0 +output = 4.0 +cache_read = 0.5 + +[limit] +context = 1048576 +output = 64000 \ No newline at end of file diff --git a/providers/tensorx/models/nvidia/nemotron-3-super-120b-a12b.toml b/providers/tensorx/models/nvidia/nemotron-3-super-120b-a12b.toml deleted file mode 100644 index 279eb86ea6a..00000000000 --- a/providers/tensorx/models/nvidia/nemotron-3-super-120b-a12b.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "nvidia/nemotron-3-super-120b-a12b" - -[[reasoning_options]] -type = "effort" # API: {"reasoning_effort": }; "none" disables reasoning -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.3 -output = 0.9 -cache_read = 0.075 -cache_write = 0.375 diff --git a/providers/tensorx/models/openai/gpt-oss-120b.toml b/providers/tensorx/models/openai/gpt-oss-120b.toml deleted file mode 100644 index b34a24534e3..00000000000 --- a/providers/tensorx/models/openai/gpt-oss-120b.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/gpt-oss-120b" -knowledge = "2024-10" - -[[reasoning_options]] -type = "effort" # API: {"reasoning_effort": }; reasoning is mandatory, "none" is rejected -values = ["minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.04 -output = 0.2 -cache_read = 0.01 -cache_write = 0.05 diff --git a/providers/tensorx/models/qwen/qwen3-coder-30b-a3b-instruct.toml b/providers/tensorx/models/qwen/qwen3-coder-30b-a3b-instruct.toml deleted file mode 100644 index bf2ca2550b6..00000000000 --- a/providers/tensorx/models/qwen/qwen3-coder-30b-a3b-instruct.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "alibaba/qwen3-coder-30b-a3b-instruct" - -[cost] -input = 0.06 -output = 0.25 -cache_read = 0.015 -cache_write = 0.075 - -[limit] -context = 262000 diff --git a/providers/tensorx/models/qwen/qwen3-vl-235b-a22b-instruct.toml b/providers/tensorx/models/qwen/qwen3-vl-235b-a22b-instruct.toml deleted file mode 100644 index 63f03d54a44..00000000000 --- a/providers/tensorx/models/qwen/qwen3-vl-235b-a22b-instruct.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Qwen3 VL 235B-A22B Instruct" -description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" -family = "qwen" -release_date = "2025-09-23" -last_updated = "2025-09-23" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-03-31" -open_weights = true - -[cost] -input = 0.21 -output = 1.9 -cache_read = 0.0525 -cache_write = 0.2625 - -[limit] -context = 131000 -output = 131072 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/tensorx/models/qwen/qwen3.8-2.4t-a95b.toml b/providers/tensorx/models/qwen/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..70a027487ce --- /dev/null +++ b/providers/tensorx/models/qwen/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,15 @@ +base_model = "alibaba/qwen3.8-2.4t-a95b" +# Qwen3.8 always reasons; effort xhigh (default) | medium | low. +# https://docs.tensorx.ai/api-reference/reasoning + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 2.5 +output = 6.0 +cache_read = 0.63 + +[limit] +output = 64000 \ No newline at end of file diff --git a/providers/tensorx/models/qwen/qwen3.8-27b.toml b/providers/tensorx/models/qwen/qwen3.8-27b.toml new file mode 100644 index 00000000000..552bc1978f7 --- /dev/null +++ b/providers/tensorx/models/qwen/qwen3.8-27b.toml @@ -0,0 +1,11 @@ +base_model = "alibaba/qwen3.8-27b" +# Thinking is always on; reasoning_effort low|medium|xhigh (default xhigh). + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 0.4 +output = 2.4 +cache_read = 0.1 \ No newline at end of file diff --git a/providers/tensorx/models/qwen/qwen3.8-flash-next.toml b/providers/tensorx/models/qwen/qwen3.8-flash-next.toml new file mode 100644 index 00000000000..7df5eb110ec --- /dev/null +++ b/providers/tensorx/models/qwen/qwen3.8-flash-next.toml @@ -0,0 +1,15 @@ +base_model = "alibaba/qwen3.8-flash-next" +# Qwen3.8 always reasons; effort xhigh (default) | medium | low. +# https://docs.tensorx.ai/api-reference/reasoning + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 0.2 +output = 0.5 +cache_read = 0.05 + +[limit] +output = 64000 \ No newline at end of file diff --git a/providers/tensorx/models/z-ai/glm-4.7.toml b/providers/tensorx/models/z-ai/glm-4.7.toml deleted file mode 100644 index 931c963e6a7..00000000000 --- a/providers/tensorx/models/z-ai/glm-4.7.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "zhipuai/glm-4.7" - -[[reasoning_options]] -type = "effort" # API: {"reasoning_effort": }; "none" disables reasoning -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.6 -output = 2.2 -cache_read = 0.15 -cache_write = 0.75 - -[limit] -context = 200000 -output = 200000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/tensorx/models/z-ai/glm-5.3-flash.toml b/providers/tensorx/models/z-ai/glm-5.3-flash.toml new file mode 100644 index 00000000000..ad6a7edac8e --- /dev/null +++ b/providers/tensorx/models/z-ai/glm-5.3-flash.toml @@ -0,0 +1,20 @@ +base_model = "zhipuai/glm-5.3-flash" +# GLM-5.3-Flash reasons by default; toggle enable_thinking, effort low|high|max (default max). +# Unlike first-party Z.AI, TensorX allows thinking to be disabled (enable_thinking). +# https://docs.tensorx.ai/api-reference/reasoning + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.2 +output = 0.5 +cache_read = 0.05 + +[limit] +context = 1048576 +output = 64000 \ No newline at end of file diff --git a/providers/tensorx/models/z-ai/glm-5.3.toml b/providers/tensorx/models/z-ai/glm-5.3.toml new file mode 100644 index 00000000000..85587ef550d --- /dev/null +++ b/providers/tensorx/models/z-ai/glm-5.3.toml @@ -0,0 +1,20 @@ +base_model = "zhipuai/glm-5.3" +# GLM-5.3 reasons by default; toggle enable_thinking, effort low|high|max (default max). +# Unlike first-party Z.AI, TensorX allows thinking to be disabled (enable_thinking). +# https://docs.tensorx.ai/api-reference/reasoning + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 1.75 +output = 4.5 +cache_read = 0.44 + +[limit] +context = 1048576 +output = 64000 \ No newline at end of file From bc1e1a83ac95e718b48e4122d553ef0e706ffd82 Mon Sep 17 00:00:00 2001 From: Alcatraz-Zhang Date: Sun, 20 Sep 2026 01:11:44 +0800 Subject: [PATCH 079/392] chore(zenmux): refresh Z.AI and xAI families (#7484) * chore(zenmux): refresh Z.AI and xAI families * fix(zenmux): document Z.AI host deltas --- models/xai/grok-voice-stt-1.0.toml | 17 ++++++++ models/xai/grok-voice-tts-1.0.toml | 17 ++++++++ models/zhipuai/glm-image.toml | 22 ++++++++++ providers/zenmux/models/x-ai/grok-4-fast.toml | 25 ----------- .../x-ai/grok-4.1-fast-non-reasoning.toml | 24 ----------- .../zenmux/models/x-ai/grok-4.1-fast.toml | 25 ----------- .../x-ai/grok-4.2-fast-non-reasoning.toml | 26 ++++++------ .../zenmux/models/x-ai/grok-4.2-fast.toml | 32 +++++++------- providers/zenmux/models/x-ai/grok-4.3.toml | 23 +++++----- providers/zenmux/models/x-ai/grok-4.6.toml | 19 +++++++++ providers/zenmux/models/x-ai/grok-4.toml | 25 ----------- .../zenmux/models/x-ai/grok-build-0.1.toml | 14 +++++-- .../zenmux/models/x-ai/grok-code-fast-1.toml | 25 ----------- .../models/x-ai/grok-imagine-image-2.0.toml | 7 ++++ .../models/x-ai/grok-voice-stt-1.0.toml | 3 ++ .../models/x-ai/grok-voice-tts-1.0.toml | 3 ++ providers/zenmux/models/z-ai/glm-4.5-air.toml | 35 ++++++++-------- providers/zenmux/models/z-ai/glm-4.5.toml | 36 ++++++++-------- providers/zenmux/models/z-ai/glm-4.6.toml | 36 ++++++++-------- .../models/z-ai/glm-4.6v-flash-free.toml | 34 ++++++++------- .../zenmux/models/z-ai/glm-4.6v-flash.toml | 36 ++++++++-------- providers/zenmux/models/z-ai/glm-4.6v.toml | 35 ++++++++-------- .../models/z-ai/glm-4.7-flash-free.toml | 35 +++++++--------- .../zenmux/models/z-ai/glm-4.7-flashx.toml | 36 +++++++--------- providers/zenmux/models/z-ai/glm-4.7.toml | 42 +++++++++---------- providers/zenmux/models/z-ai/glm-5-turbo.toml | 36 ++++++++-------- providers/zenmux/models/z-ai/glm-5.1.toml | 32 +++++++------- .../zenmux/models/z-ai/glm-5.2-free.toml | 11 ----- providers/zenmux/models/z-ai/glm-5.2.toml | 17 ++++++-- .../zenmux/models/z-ai/glm-5.3-flash.toml | 20 +++++++++ .../zenmux/models/z-ai/glm-5.3-flashx.toml | 20 +++++++++ providers/zenmux/models/z-ai/glm-5.3.toml | 20 +++++++++ providers/zenmux/models/z-ai/glm-5.toml | 28 ++++++------- .../zenmux/models/z-ai/glm-5v-turbo.toml | 27 ++++++------ providers/zenmux/models/z-ai/glm-image.toml | 9 ++++ 35 files changed, 439 insertions(+), 413 deletions(-) create mode 100644 models/xai/grok-voice-stt-1.0.toml create mode 100644 models/xai/grok-voice-tts-1.0.toml create mode 100644 models/zhipuai/glm-image.toml delete mode 100644 providers/zenmux/models/x-ai/grok-4-fast.toml delete mode 100644 providers/zenmux/models/x-ai/grok-4.1-fast-non-reasoning.toml delete mode 100644 providers/zenmux/models/x-ai/grok-4.1-fast.toml create mode 100644 providers/zenmux/models/x-ai/grok-4.6.toml delete mode 100644 providers/zenmux/models/x-ai/grok-4.toml delete mode 100644 providers/zenmux/models/x-ai/grok-code-fast-1.toml create mode 100644 providers/zenmux/models/x-ai/grok-imagine-image-2.0.toml create mode 100644 providers/zenmux/models/x-ai/grok-voice-stt-1.0.toml create mode 100644 providers/zenmux/models/x-ai/grok-voice-tts-1.0.toml delete mode 100644 providers/zenmux/models/z-ai/glm-5.2-free.toml create mode 100644 providers/zenmux/models/z-ai/glm-5.3-flash.toml create mode 100644 providers/zenmux/models/z-ai/glm-5.3-flashx.toml create mode 100644 providers/zenmux/models/z-ai/glm-5.3.toml create mode 100644 providers/zenmux/models/z-ai/glm-image.toml diff --git a/models/xai/grok-voice-stt-1.0.toml b/models/xai/grok-voice-stt-1.0.toml new file mode 100644 index 00000000000..725c96c69d7 --- /dev/null +++ b/models/xai/grok-voice-stt-1.0.toml @@ -0,0 +1,17 @@ +name = "Grok Voice STT 1.0" +description = "Grok Voice STT 1.0 is xAI's speech-to-text model. It supports transcription with word-level timestamps, optional speaker diarization, and multichannel audio." +release_date = "2026-08-04" +last_updated = "2026-08-04" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 15_000 +output = 15_000 + +[modalities] +input = ["audio"] +output = ["text"] diff --git a/models/xai/grok-voice-tts-1.0.toml b/models/xai/grok-voice-tts-1.0.toml new file mode 100644 index 00000000000..b519e016a19 --- /dev/null +++ b/models/xai/grok-voice-tts-1.0.toml @@ -0,0 +1,17 @@ +name = "Grok Voice TTS 1.0" +description = "Convert text into spoken audio with a single API call. The API supports a rich set of expressive voices, inline speech tags for fine-grained delivery control, and output formats from high-fidelity MP3 to telephony-optimized μ-law." +release_date = "2026-07-31" +last_updated = "2026-07-31" +attachment = false +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 15_000 +output = 15_000 + +[modalities] +input = ["text"] +output = ["audio"] diff --git a/models/zhipuai/glm-image.toml b/models/zhipuai/glm-image.toml new file mode 100644 index 00000000000..b3f3746cb2e --- /dev/null +++ b/models/zhipuai/glm-image.toml @@ -0,0 +1,22 @@ +name = "GLM-Image" +description = "GLM-Image is an image generation model adopts a hybrid autoregressive + diffusion decoder architecture. In general image generation quality, GLM‑Image aligns with mainstream latent diffusion approaches, but it shows significant advantages in text-rendering and knowledge‑intensive generation scenarios. It performs especially well in tasks requiring precise semantic understanding and complex information expression, while maintaining strong capabilities in high‑fidelity and fine‑grained detail generation. In addition to text‑to‑image generation, GLM‑Image also supports a rich set of image‑to‑image tasks including image editing, style transfer, identity‑preserving generation, and multi‑subject consistency." +release_date = "2026-01-19" +last_updated = "2026-01-19" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = true +license = "MIT" + +[limit] +context = 10_240 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["image"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/zai-org/GLM-Image" diff --git a/providers/zenmux/models/x-ai/grok-4-fast.toml b/providers/zenmux/models/x-ai/grok-4-fast.toml deleted file mode 100644 index f722ac7ac71..00000000000 --- a/providers/zenmux/models/x-ai/grok-4-fast.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Grok 4 Fast" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2025-09-19" -last_updated = "2025-09-19" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.20 -output = 0.50 -cache_read = 0.05 - -[limit] -context = 2000_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/zenmux/models/x-ai/grok-4.1-fast-non-reasoning.toml b/providers/zenmux/models/x-ai/grok-4.1-fast-non-reasoning.toml deleted file mode 100644 index 997935a9745..00000000000 --- a/providers/zenmux/models/x-ai/grok-4.1-fast-non-reasoning.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Grok 4.1 Fast Non Reasoning" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2025-11-20" -last_updated = "2025-11-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.20 -output = 0.50 -cache_read = 0.05 - -[limit] -context = 2000_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/zenmux/models/x-ai/grok-4.1-fast.toml b/providers/zenmux/models/x-ai/grok-4.1-fast.toml deleted file mode 100644 index 6ddced52aef..00000000000 --- a/providers/zenmux/models/x-ai/grok-4.1-fast.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Grok 4.1 Fast" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2025-11-20" -last_updated = "2025-11-20" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.20 -output = 0.50 -cache_read = 0.05 - -[limit] -context = 2000_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/zenmux/models/x-ai/grok-4.2-fast-non-reasoning.toml b/providers/zenmux/models/x-ai/grok-4.2-fast-non-reasoning.toml index 5fd0f203453..b521181860f 100644 --- a/providers/zenmux/models/x-ai/grok-4.2-fast-non-reasoning.toml +++ b/providers/zenmux/models/x-ai/grok-4.2-fast-non-reasoning.toml @@ -1,22 +1,20 @@ +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "xai/grok-4.20-0309-non-reasoning" name = "Grok 4.2 Fast Non Reasoning" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false [cost] -input = 3.00 -output = 9.00 +input = 2 +output = 6 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 4 +output = 12 +cache_read = 0.2 [limit] context = 2_000_000 -output = 30_000 [modalities] -input = ["text", "image", "video"] -output = ["text"] +input = ["text", "image"] diff --git a/providers/zenmux/models/x-ai/grok-4.2-fast.toml b/providers/zenmux/models/x-ai/grok-4.2-fast.toml index c768fe47135..fc524864b65 100644 --- a/providers/zenmux/models/x-ai/grok-4.2-fast.toml +++ b/providers/zenmux/models/x-ai/grok-4.2-fast.toml @@ -1,23 +1,25 @@ +# ZenMux's 2026-09-19 page has supports_reasoning=0 and both live OpenAI- and +# Anthropic-compatible entries have capabilities.reasoning=false for this id. +# This route therefore uses the reasoning checkpoint identity but disables the +# reasoning surface as an explicit host delta. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "xai/grok-4.20-0309-reasoning" name = "Grok 4.2 Fast" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false +reasoning = false [cost] -input = 3.00 -output = 9.00 +input = 2 +output = 6 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 4 +output = 12 +cache_read = 0.2 [limit] context = 2_000_000 -output = 30_000 [modalities] -input = ["text", "image", "video"] -output = ["text"] +input = ["text", "image"] diff --git a/providers/zenmux/models/x-ai/grok-4.3.toml b/providers/zenmux/models/x-ai/grok-4.3.toml index 455779665fb..ab8ed82e9a7 100644 --- a/providers/zenmux/models/x-ai/grok-4.3.toml +++ b/providers/zenmux/models/x-ai/grok-4.3.toml @@ -1,19 +1,22 @@ +# Effort: reasoning_effort = none|low|medium|high +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "xai/grok-4.3" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] [cost] input = 1.25 -output = 2.50 -cache_read = 0.20 -cache_write = 0 +output = 2.5 +cache_read = 0.2 [[cost.tiers]] -tier = { size = 200_000 } -input = 2.50 -output = 5.00 -cache_read = 0.40 -cache_write = 0 +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 5 +cache_read = 0.4 [limit] -context = 1_000_000 output = 1_000_000 diff --git a/providers/zenmux/models/x-ai/grok-4.6.toml b/providers/zenmux/models/x-ai/grok-4.6.toml new file mode 100644 index 00000000000..7b40b61706e --- /dev/null +++ b/providers/zenmux/models/x-ai/grok-4.6.toml @@ -0,0 +1,19 @@ +# Effort: reasoning_effort = low|medium|high|xhigh +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "xai/grok-4.6" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 diff --git a/providers/zenmux/models/x-ai/grok-4.toml b/providers/zenmux/models/x-ai/grok-4.toml deleted file mode 100644 index 9a3bd9acea0..00000000000 --- a/providers/zenmux/models/x-ai/grok-4.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Grok 4" -description = "Grok model for agentic tool use, reasoning, coding, and live assistance" -release_date = "2025-07-09" -last_updated = "2025-07-09" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 3.00 -output = 15.00 -cache_read = 0.75 - -[limit] -context = 256_000 -output = 64_000 - -[modalities] -input = ["image", "text"] -output = ["text"] diff --git a/providers/zenmux/models/x-ai/grok-build-0.1.toml b/providers/zenmux/models/x-ai/grok-build-0.1.toml index 59480b618c9..f00530ac65f 100644 --- a/providers/zenmux/models/x-ai/grok-build-0.1.toml +++ b/providers/zenmux/models/x-ai/grok-build-0.1.toml @@ -1,7 +1,13 @@ +# ZenMux's 2026-09-19 page has supports_reasoning=0 and both live OpenAI- and +# Anthropic-compatible entries have capabilities.reasoning=false for this id. +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "xai/grok-build-0.1" -reasoning_options = [] +reasoning = false [cost] -input = 1.00 -output = 2.00 -cache_read = 0.20 +input = 1 +output = 2 +cache_read = 0.2 + +[modalities] +input = ["text", "image"] diff --git a/providers/zenmux/models/x-ai/grok-code-fast-1.toml b/providers/zenmux/models/x-ai/grok-code-fast-1.toml deleted file mode 100644 index 5a692a5e5d9..00000000000 --- a/providers/zenmux/models/x-ai/grok-code-fast-1.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Grok Code Fast 1" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2025-08-26" -last_updated = "2025-08-26" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.20 -output = 1.50 -cache_read = 0.02 - -[limit] -context = 256_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/x-ai/grok-imagine-image-2.0.toml b/providers/zenmux/models/x-ai/grok-imagine-image-2.0.toml new file mode 100644 index 00000000000..05abc6d0374 --- /dev/null +++ b/providers/zenmux/models/x-ai/grok-imagine-image-2.0.toml @@ -0,0 +1,7 @@ +# Host context: page and OpenAI list both report 66000 tokens. +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "xai/grok-imagine-image-2.0" + +[limit] +context = 66_000 diff --git a/providers/zenmux/models/x-ai/grok-voice-stt-1.0.toml b/providers/zenmux/models/x-ai/grok-voice-stt-1.0.toml new file mode 100644 index 00000000000..c4bf10390e9 --- /dev/null +++ b/providers/zenmux/models/x-ai/grok-voice-stt-1.0.toml @@ -0,0 +1,3 @@ +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "xai/grok-voice-stt-1.0" diff --git a/providers/zenmux/models/x-ai/grok-voice-tts-1.0.toml b/providers/zenmux/models/x-ai/grok-voice-tts-1.0.toml new file mode 100644 index 00000000000..27f8f70b699 --- /dev/null +++ b/providers/zenmux/models/x-ai/grok-voice-tts-1.0.toml @@ -0,0 +1,3 @@ +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "xai/grok-voice-tts-1.0" diff --git a/providers/zenmux/models/z-ai/glm-4.5-air.toml b/providers/zenmux/models/z-ai/glm-4.5-air.toml index 756ca62b321..3797e830f0e 100644 --- a/providers/zenmux/models/z-ai/glm-4.5-air.toml +++ b/providers/zenmux/models/z-ai/glm-4.5-air.toml @@ -1,24 +1,23 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.5-air" name = "GLM 4.5 Air" -description = "Efficient GLM model for fast reasoning, coding, and agent workflows" -release_date = "2025-07-25" -last_updated = "2025-07-25" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.11 -output = 0.56 -cache_read = 0.02 +input = 0.1165 +output = 0.2911 +cache_read = 0.0233 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.1747 +output = 1.1645 +cache_read = 0.0349 [limit] context = 128_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 96_000 diff --git a/providers/zenmux/models/z-ai/glm-4.5.toml b/providers/zenmux/models/z-ai/glm-4.5.toml index b2f7197b924..fffd19cacc1 100644 --- a/providers/zenmux/models/z-ai/glm-4.5.toml +++ b/providers/zenmux/models/z-ai/glm-4.5.toml @@ -1,24 +1,24 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.5" name = "GLM 4.5" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -release_date = "2025-07-25" -last_updated = "2025-07-25" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.35 -output = 1.54 -cache_read = 0.07 +input = 0.2911 +output = 1.1645 +cache_read = 0.0582 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.5823 +output = 2.3291 +cache_read = 0.1165 [limit] context = 128_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 96_000 diff --git a/providers/zenmux/models/z-ai/glm-4.6.toml b/providers/zenmux/models/z-ai/glm-4.6.toml index e5f9d2ca1a4..d4d4456a446 100644 --- a/providers/zenmux/models/z-ai/glm-4.6.toml +++ b/providers/zenmux/models/z-ai/glm-4.6.toml @@ -1,24 +1,24 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.6" name = "GLM 4.6" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -release_date = "2025-09-30" -last_updated = "2025-09-30" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.35 -output = 1.54 -cache_read = 0.07 +input = 0.2911 +output = 1.1645 +cache_read = 0.0582 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.5823 +output = 2.3291 +cache_read = 0.1165 [limit] context = 200_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-4.6v-flash-free.toml b/providers/zenmux/models/z-ai/glm-4.6v-flash-free.toml index 33b4db29d44..18b34502274 100644 --- a/providers/zenmux/models/z-ai/glm-4.6v-flash-free.toml +++ b/providers/zenmux/models/z-ai/glm-4.6v-flash-free.toml @@ -1,23 +1,27 @@ +# Toggle: reasoning.enabled = true|false +# Host modalities: the live free route includes file/PDF input; the paid route omits it. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.6v-flash" name = "GLM 4.6V Flash (Free)" -description = "GLM vision model for visual reasoning, documents, and multimodal agents" -release_date = "2025-12-08" -last_updated = "2025-12-08" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 +cache_read = 0 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0 +output = 0 +cache_read = 0 [limit] context = 200_000 -output = 64_000 +output = 128_000 [modalities] -input = ["text", "image", "video"] -output = ["text"] +input = ["text", "image", "video", "pdf"] diff --git a/providers/zenmux/models/z-ai/glm-4.6v-flash.toml b/providers/zenmux/models/z-ai/glm-4.6v-flash.toml index 00a8fa34b67..6c53afd51b0 100644 --- a/providers/zenmux/models/z-ai/glm-4.6v-flash.toml +++ b/providers/zenmux/models/z-ai/glm-4.6v-flash.toml @@ -1,24 +1,24 @@ +# Toggle: reasoning.enabled = true|false +# Host naming: the 2026-09-19 live route z-ai/glm-4.6v-flash is labeled "GLM 4.6V FlashX". +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.6v-flash" name = "GLM 4.6V FlashX" -description = "GLM vision model for visual reasoning, documents, and multimodal agents" -release_date = "2025-12-08" -last_updated = "2025-12-08" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.02 -output = 0.21 -cache_read = 0.0043 +input = 0.0218 +output = 0.2184 +cache_read = 0.0044 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.0437 +output = 0.4367 +cache_read = 0.0044 [limit] context = 200_000 -output = 64_000 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-4.6v.toml b/providers/zenmux/models/z-ai/glm-4.6v.toml index 73690ef9e4f..e8c345954c8 100644 --- a/providers/zenmux/models/z-ai/glm-4.6v.toml +++ b/providers/zenmux/models/z-ai/glm-4.6v.toml @@ -1,24 +1,23 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.6v" name = "GLM 4.6V" -description = "GLM vision model for visual reasoning, documents, and multimodal agents" -release_date = "2025-12-08" -last_updated = "2025-12-08" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.14 -output = 0.42 -cache_read = 0.03 +input = 0.1456 +output = 0.4367 +cache_read = 0.0291 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.2911 +output = 0.8734 +cache_read = 0.0582 [limit] context = 200_000 -output = 64_000 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-4.7-flash-free.toml b/providers/zenmux/models/z-ai/glm-4.7-flash-free.toml index b2ddbce96ec..df0e92ac02a 100644 --- a/providers/zenmux/models/z-ai/glm-4.7-flash-free.toml +++ b/providers/zenmux/models/z-ai/glm-4.7-flash-free.toml @@ -1,26 +1,19 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.7-flash" name = "GLM 4.7 Flash (Free)" -description = "Efficient GLM model for fast reasoning, coding, and agent workflows" -release_date = "2026-01-19" -last_updated = "2026-01-19" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 200_000 -output = 64_000 [interleaved] field = "reasoning_content" -[modalities] -input = ["text"] -output = ["text"] +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 +cache_read = 0 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-4.7-flashx.toml b/providers/zenmux/models/z-ai/glm-4.7-flashx.toml index eba199ff720..b131a75aa2a 100644 --- a/providers/zenmux/models/z-ai/glm-4.7-flashx.toml +++ b/providers/zenmux/models/z-ai/glm-4.7-flashx.toml @@ -1,27 +1,19 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.7-flashx" name = "GLM 4.7 FlashX" -description = "Efficient GLM model for fast reasoning, coding, and agent workflows" -release_date = "2026-01-19" -last_updated = "2026-01-19" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -[cost] -input = 0.07 -output = 0.42 -cache_read = 0.01 - -[limit] -context = 200_000 -output = 64_000 [interleaved] field = "reasoning_content" -[modalities] -input = ["text"] -output = ["text"] +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.0728 +output = 0.4367 +cache_read = 0.0146 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-4.7.toml b/providers/zenmux/models/z-ai/glm-4.7.toml index 266005dda7e..c32d76030a6 100644 --- a/providers/zenmux/models/z-ai/glm-4.7.toml +++ b/providers/zenmux/models/z-ai/glm-4.7.toml @@ -1,27 +1,27 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.7" name = "GLM 4.7" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -release_date = "2025-12-23" -last_updated = "2025-12-23" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.28 -output = 1.14 -cache_read = 0.06 +input = 0.2911 +output = 1.1645 +cache_read = 0.0582 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.5823 +output = 2.3291 +cache_read = 0.1165 [limit] context = 200_000 -output = 64_000 - -[interleaved] -field = "reasoning_content" - -[modalities] -input = ["text"] -output = ["text"] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5-turbo.toml b/providers/zenmux/models/z-ai/glm-5-turbo.toml index ef6438e58ac..ecc0fbae95d 100644 --- a/providers/zenmux/models/z-ai/glm-5-turbo.toml +++ b/providers/zenmux/models/z-ai/glm-5-turbo.toml @@ -1,23 +1,25 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5-turbo" name = "GLM 5 Turbo" -description = "Efficient GLM model for fast reasoning, coding, and agent workflows" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.88 -output = 3.48 +input = 0.73 +output = 3.19 +cache_read = 0.174 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 1.02 +output = 3.77 +cache_read = 0.261 [limit] -context = 200_000 output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/z-ai/glm-5.1.toml b/providers/zenmux/models/z-ai/glm-5.1.toml index 3b4c0bec049..3a64eedea4a 100644 --- a/providers/zenmux/models/z-ai/glm-5.1.toml +++ b/providers/zenmux/models/z-ai/glm-5.1.toml @@ -1,27 +1,25 @@ -name = "GLM-5.1" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -release_date = "2026-04-03" -last_updated = "2026-04-03" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -open_weights = false +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5.1" +name = "GLM 5.1" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.8781 output = 3.5126 cache_read = 0.1903 -[limit] -context = 200000 -output = 131072 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 1.1709 +output = 4.098 +cache_read = 0.2927 -[modalities] -input = ["text"] -output = ["text"] +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5.2-free.toml b/providers/zenmux/models/z-ai/glm-5.2-free.toml deleted file mode 100644 index 561b079445e..00000000000 --- a/providers/zenmux/models/z-ai/glm-5.2-free.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "zhipuai/glm-5.2" -name = "GLM 5.2 (Free)" - -[cost] -input = 0 -output = 0 -cache_read = 0 - -[[reasoning_options]] -type = "effort" -values = ["high", "max"] diff --git a/providers/zenmux/models/z-ai/glm-5.2.toml b/providers/zenmux/models/z-ai/glm-5.2.toml index 7d3e1a5ebf4..09a38c80a35 100644 --- a/providers/zenmux/models/z-ai/glm-5.2.toml +++ b/providers/zenmux/models/z-ai/glm-5.2.toml @@ -1,11 +1,20 @@ +# Effort: reasoning_effort = high|max +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "zhipuai/glm-5.2" name = "GLM 5.2" -[cost] -input = 1.40 -output = 4.50 -cache_read = 0.26 +[interleaved] +field = "reasoning_content" [[reasoning_options]] type = "effort" values = ["high", "max"] + +[cost] +input = 0.98 +output = 3.08 +cache_read = 0.182 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5.3-flash.toml b/providers/zenmux/models/z-ai/glm-5.3-flash.toml new file mode 100644 index 00000000000..3bd77e65bd9 --- /dev/null +++ b/providers/zenmux/models/z-ai/glm-5.3-flash.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = low|high|max +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5.3-flash" +name = "GLM 5.3 Flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.03 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5.3-flashx.toml b/providers/zenmux/models/z-ai/glm-5.3-flashx.toml new file mode 100644 index 00000000000..50ad6eb5ea5 --- /dev/null +++ b/providers/zenmux/models/z-ai/glm-5.3-flashx.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = low|high|max +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5.3-flash" +name = "GLM 5.3 FlashX" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.375 +output = 1.25 +cache_read = 0.075 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5.3.toml b/providers/zenmux/models/z-ai/glm-5.3.toml new file mode 100644 index 00000000000..18b45ff89ad --- /dev/null +++ b/providers/zenmux/models/z-ai/glm-5.3.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = low|high|max +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5.3" +name = "GLM 5.3" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 1.4 +output = 4.4 +cache_read = 0.26 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5.toml b/providers/zenmux/models/z-ai/glm-5.toml index 3be16bf629a..f2ee297f1d0 100644 --- a/providers/zenmux/models/z-ai/glm-5.toml +++ b/providers/zenmux/models/z-ai/glm-5.toml @@ -1,27 +1,27 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5" name = "GLM 5" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -release_date = "2026-02-12" -last_updated = "2026-02-12" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = true +structured_output = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.58 output = 2.6 cache_read = 0.14 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.87 +output = 3.18 +cache_read = 0.22 + [limit] context = 200_000 output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/z-ai/glm-5v-turbo.toml b/providers/zenmux/models/z-ai/glm-5v-turbo.toml index 00fc2f71901..0e75e57e194 100644 --- a/providers/zenmux/models/z-ai/glm-5v-turbo.toml +++ b/providers/zenmux/models/z-ai/glm-5v-turbo.toml @@ -1,26 +1,25 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5v-turbo" name = "GLM 5V Turbo" -description = "GLM vision model for visual reasoning, documents, and multimodal agents" -release_date = "2026-04-01" -last_updated = "2026-04-01" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.726 output = 3.1946 cache_read = 0.1743 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 1.0165 +output = 3.7754 +cache_read = 0.2614 + [limit] -context = 200_000 output = 128_000 - -[modalities] -input = ["text", "image", "video", "pdf"] -output = ["text"] diff --git a/providers/zenmux/models/z-ai/glm-image.toml b/providers/zenmux/models/z-ai/glm-image.toml new file mode 100644 index 00000000000..79c6282b394 --- /dev/null +++ b/providers/zenmux/models/z-ai/glm-image.toml @@ -0,0 +1,9 @@ +# ZenMux's 2026-09-19 page and live Google-compatible list expose text input only; +# image editing exists in the lab model but is unavailable on this hosted route. +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "zhipuai/glm-image" +attachment = false + +[modalities] +input = ["text"] From 55a53aa2df138ce8a26af0e198025fa41185c219 Mon Sep 17 00:00:00 2001 From: Sujee Maniyam Date: Sat, 19 Sep 2026 10:12:03 -0700 Subject: [PATCH 080/392] feat(nebius): add DeepSeek V4.1 Flash, V4 Pro 0813, GLM-5.3 (#7477) --- .../deepseek-ai/DeepSeek-V4-Pro-0813.toml | 28 +++++++++++++++++ .../deepseek-ai/DeepSeek-V4.1-Flash.toml | 30 +++++++++++++++++++ providers/nebius/models/zai-org/GLM-5.3.toml | 28 +++++++++++++++++ 3 files changed, 86 insertions(+) create mode 100644 providers/nebius/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml create mode 100644 providers/nebius/models/deepseek-ai/DeepSeek-V4.1-Flash.toml create mode 100644 providers/nebius/models/zai-org/GLM-5.3.toml diff --git a/providers/nebius/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml b/providers/nebius/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml new file mode 100644 index 00000000000..b439ca31a66 --- /dev/null +++ b/providers/nebius/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml @@ -0,0 +1,28 @@ +# Sources: +# - https://tokenfactory.nebius.com/ (model catalog) +# - https://tokenfactory.nebius.com/api/public/models_info (pricing, context, max length) +# Accessed 2026-09-18. +# Effort: reasoning_effort = none|low|high|max (none = thinking off). Same gateway +# behavior as the other DeepSeek V4 entries on this host (DeepSeek-V4-Flash-0731, +# DeepSeek-V4-Pro): the generic schema accepts none/minimal/low/medium/high/xhigh/max, +# but the mid tiers collapse (minimal=low=medium, high=xhigh) and none disables +# thinking, leaving none/low/high/max as the distinct levels. +# Served window: models_info max_model_len = 979_000 for this snapshot with no separate +# output cap, so context and output both take the full served window. +# No discounted prompt-cache tier is published in the public catalog: cached and fresh +# input are billed identically, so cache_read = input (same rationale as +# DeepSeek-V4-Flash-0731 / Kimi-K3 / GLM-5.3-Flash). +base_model = "deepseek/deepseek-v4-pro-0813" +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1.32 +output = 3.96 +cache_read = 1.32 + +[limit] +context = 979_000 +output = 979_000 diff --git a/providers/nebius/models/deepseek-ai/DeepSeek-V4.1-Flash.toml b/providers/nebius/models/deepseek-ai/DeepSeek-V4.1-Flash.toml new file mode 100644 index 00000000000..cc631b9cfee --- /dev/null +++ b/providers/nebius/models/deepseek-ai/DeepSeek-V4.1-Flash.toml @@ -0,0 +1,30 @@ +# Sources: +# - https://tokenfactory.nebius.com/ (model catalog) +# - https://tokenfactory.nebius.com/api/public/models_info (pricing, context, max length) +# Accessed 2026-09-18. +# Effort: reasoning_effort = none|low|high|max (none = thinking off). Same gateway +# behavior as the other DeepSeek V4 entries on this host (DeepSeek-V4-Flash-0731, +# DeepSeek-V4-Pro): the generic schema accepts none/minimal/low/medium/high/xhigh/max, +# but the mid tiers collapse (minimal=low=medium, high=xhigh) and none disables +# thinking, leaving none/low/high/max as the distinct levels. +# Server reports model type image2text, so attachment/modalities are inherited from the +# lab entry. +# Served window: models_info max_model_len = 1_048_000 for this model with no separate +# output cap, so context and output both take the full served window. +# No discounted prompt-cache tier is published in the public catalog: cached and fresh +# input are billed identically, so cache_read = input (same rationale as +# DeepSeek-V4-Flash-0731 / Kimi-K3 / GLM-5.3-Flash). +base_model = "deepseek/deepseek-v4.1-flash" +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.3 + +[limit] +context = 1_048_000 +output = 1_048_000 diff --git a/providers/nebius/models/zai-org/GLM-5.3.toml b/providers/nebius/models/zai-org/GLM-5.3.toml new file mode 100644 index 00000000000..be8b6208f43 --- /dev/null +++ b/providers/nebius/models/zai-org/GLM-5.3.toml @@ -0,0 +1,28 @@ +# Sources: +# - https://tokenfactory.nebius.com/ (model catalog) +# - https://tokenfactory.nebius.com/api/public/models_info (pricing, context, max length) +# - https://docs.bigmodel.cn/cn/guide/models/text/glm-5.3 (effort set; thinking cannot +# be disabled, so there is no toggle — effort per lab: low|high|max, default max) +# Accessed 2026-09-18. +# Served window: models_info max_model_len = 1_024_000 with no separate output cap, so +# context and output both take the full served window. +# No discounted prompt-cache tier is published in the public catalog: cached and fresh +# input are billed identically, so cache_read = input (same rationale as +# GLM-5.3-Flash / DeepSeek-V4-Flash-0731 / Kimi-K3). +base_model = "zhipuai/glm-5.3" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1.4 +output = 4.4 +cache_read = 1.4 + +[limit] +context = 1_024_000 +output = 1_024_000 From a55d9f67599f0dd5d974913d68622777122b787b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 12:13:22 -0500 Subject: [PATCH 081/392] chore(sync): update Vercel AI Gateway model catalog (#7509) * chore(sync): update Vercel AI Gateway model catalog * fix(vercel): normalize synced model metadata --------- Co-authored-by: opencode-agent[bot] Co-authored-by: rekram1-node --- models/mixedbread/toast-1.toml | 19 +++++++++++++++++++ models/quiverai/arrow-2-telos.toml | 19 +++++++++++++++++++ models/quiverai/arrow-2.toml | 19 +++++++++++++++++++ .../vercel/models/mixedbread/toast-1.toml | 6 ++++++ .../vercel/models/quiverai/arrow-2-telos.toml | 13 +++++++++++++ providers/vercel/models/quiverai/arrow-2.toml | 12 ++++++++++++ providers/vercel/models/typesafe-ai/jev.toml | 3 +++ 7 files changed, 91 insertions(+) create mode 100644 models/mixedbread/toast-1.toml create mode 100644 models/quiverai/arrow-2-telos.toml create mode 100644 models/quiverai/arrow-2.toml create mode 100644 providers/vercel/models/mixedbread/toast-1.toml create mode 100644 providers/vercel/models/quiverai/arrow-2-telos.toml create mode 100644 providers/vercel/models/quiverai/arrow-2.toml diff --git a/models/mixedbread/toast-1.toml b/models/mixedbread/toast-1.toml new file mode 100644 index 00000000000..775acb667f6 --- /dev/null +++ b/models/mixedbread/toast-1.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-19): +# https://www.mixedbread.com/docs/agent/models +# https://www.mixedbread.com/docs/agent/chat-completions +name = "Toast 1" +description = "Specialized search model for knowledge-intensive questions, multi-step retrieval, and evidence synthesis" +release_date = "2026-08-13" +last_updated = "2026-08-13" +attachment = false +reasoning = false +tool_call = true +open_weights = false + +[limit] +context = 131_000 +output = 4_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/models/quiverai/arrow-2-telos.toml b/models/quiverai/arrow-2-telos.toml new file mode 100644 index 00000000000..ae0e17b52d6 --- /dev/null +++ b/models/quiverai/arrow-2-telos.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-19): +# https://docs.quiver.ai/developers/models +# https://docs.quiver.ai/developers/models/arrow-2-telos +name = "Arrow 2 Telos" +description = "High-fidelity SVG generation model for complex vector work and long-context refinement" +release_date = "2026-09-16" +last_updated = "2026-09-16" +attachment = true +reasoning = true +tool_call = true +open_weights = false + +[limit] +context = 1_050_000 +output = 65_536 + +[modalities] +input = ["text", "image"] +output = ["text", "image"] diff --git a/models/quiverai/arrow-2.toml b/models/quiverai/arrow-2.toml new file mode 100644 index 00000000000..5d73a411794 --- /dev/null +++ b/models/quiverai/arrow-2.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-19): +# https://docs.quiver.ai/developers/models +# https://docs.quiver.ai/developers/models/arrow-2 +name = "Arrow 2" +description = "Fast SVG generation model for creation, vectorization, editing, and animation" +release_date = "2026-09-16" +last_updated = "2026-09-16" +attachment = true +reasoning = true +tool_call = true +open_weights = false + +[limit] +context = 131_072 +output = 65_536 + +[modalities] +input = ["text", "image"] +output = ["text", "image"] diff --git a/providers/vercel/models/mixedbread/toast-1.toml b/providers/vercel/models/mixedbread/toast-1.toml new file mode 100644 index 00000000000..b50f8e55367 --- /dev/null +++ b/providers/vercel/models/mixedbread/toast-1.toml @@ -0,0 +1,6 @@ +base_model = "mixedbread/toast-1" + +[cost] +input = 0.3 +output = 0.72 +cache_read = 0.036 diff --git a/providers/vercel/models/quiverai/arrow-2-telos.toml b/providers/vercel/models/quiverai/arrow-2-telos.toml new file mode 100644 index 00000000000..59dae7b0880 --- /dev/null +++ b/providers/vercel/models/quiverai/arrow-2-telos.toml @@ -0,0 +1,13 @@ +# https://docs.quiver.ai/api-reference/openresponses/createopenresponse +base_model = "quiverai/arrow-2-telos" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] + +[cost] +input = 6 +output = 30 +cache_read = 0.6 +cache_write = 7.5 + +[limit] +context = 131_072 +output = 131_072 diff --git a/providers/vercel/models/quiverai/arrow-2.toml b/providers/vercel/models/quiverai/arrow-2.toml new file mode 100644 index 00000000000..2b7145c9081 --- /dev/null +++ b/providers/vercel/models/quiverai/arrow-2.toml @@ -0,0 +1,12 @@ +# https://docs.quiver.ai/api-reference/openresponses/createopenresponse +base_model = "quiverai/arrow-2" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.4 +cache_write = 5 + +[limit] +output = 131_072 diff --git a/providers/vercel/models/typesafe-ai/jev.toml b/providers/vercel/models/typesafe-ai/jev.toml index e3c1d0652ad..f18f60fb34b 100644 --- a/providers/vercel/models/typesafe-ai/jev.toml +++ b/providers/vercel/models/typesafe-ai/jev.toml @@ -3,3 +3,6 @@ base_model = "typesafe/jev-latest" [cost] input = 0.042 output = 0 + +[limit] +context = 32_000 From a685c6900d143d3dd381be35f04bf5feb7f045ab Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 12:15:59 -0500 Subject: [PATCH 082/392] =?UTF-8?q?fix:=20MiniMax=20China=20API=20endpoint?= =?UTF-8?q?=20migrated=20from=20api.minimaxi.com=20to=20api.minimax.cn=20?= =?UTF-8?q?=E2=80=94=20update=20minimax-cn=20providers=20(#7501)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- providers/minimax-cn-coding-plan/provider.toml | 9 +++++++-- providers/minimax-cn/provider.toml | 9 +++++++-- 2 files changed, 14 insertions(+), 4 deletions(-) diff --git a/providers/minimax-cn-coding-plan/provider.toml b/providers/minimax-cn-coding-plan/provider.toml index b72447a7d59..28bd4e0b23d 100644 --- a/providers/minimax-cn-coding-plan/provider.toml +++ b/providers/minimax-cn-coding-plan/provider.toml @@ -1,5 +1,10 @@ -name = "MiniMax Token Plan (minimaxi.com)" +# China API host migrated from api.minimaxi.com → api.minimax.cn +# Sources: +# - https://platform.minimaxi.com/docs/guides/quickstart-preparation +# - https://platform.minimaxi.com/docs/token-plan/other-tools +# Anthropic-compatible base: https://api.minimax.cn/anthropic +name = "MiniMax Token Plan (minimax.cn)" env = ["MINIMAX_API_KEY"] npm = "@ai-sdk/anthropic" doc = "https://platform.minimaxi.com/docs/token-plan/intro" -api = "https://api.minimaxi.com/anthropic/v1" +api = "https://api.minimax.cn/anthropic/v1" diff --git a/providers/minimax-cn/provider.toml b/providers/minimax-cn/provider.toml index a72226e7345..2a75ce8de3a 100644 --- a/providers/minimax-cn/provider.toml +++ b/providers/minimax-cn/provider.toml @@ -1,5 +1,10 @@ -name = "MiniMax (minimaxi.com)" +# China API host migrated from api.minimaxi.com → api.minimax.cn +# Sources: +# - https://platform.minimaxi.com/docs/guides/quickstart-preparation +# - https://platform.minimaxi.com/docs/api-reference/text-anthropic-api +# Anthropic-compatible base: https://api.minimax.cn/anthropic +name = "MiniMax (minimax.cn)" env = ["MINIMAX_API_KEY"] npm = "@ai-sdk/anthropic" doc = "https://platform.minimaxi.com/docs/guides/quickstart" -api = "https://api.minimaxi.com/anthropic/v1" +api = "https://api.minimax.cn/anthropic/v1" From ceebaa8234220ca37e83517dc58a10c95c15d3b1 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 17:21:30 +0000 Subject: [PATCH 083/392] chore(sync): update NanoGPT model catalog (#7510) Co-authored-by: opencode-agent[bot] --- .../models/google/diffusiongemma.toml | 27 +++++++++++++++++++ 1 file changed, 27 insertions(+) create mode 100644 providers/nano-gpt/models/google/diffusiongemma.toml diff --git a/providers/nano-gpt/models/google/diffusiongemma.toml b/providers/nano-gpt/models/google/diffusiongemma.toml new file mode 100644 index 00000000000..244dd04defb --- /dev/null +++ b/providers/nano-gpt/models/google/diffusiongemma.toml @@ -0,0 +1,27 @@ +name = "DiffusionGemma" +description = "DiffusionGemma is a high-speed diffusion-based version of Gemma 4 26B A4B. It supports optional reasoning and a 262,144-token context window." +release_date = "2026-09-19" +last_updated = "2026-09-19" +attachment = false +reasoning = true +tool_call = false +structured_output = false +open_weights = true + +[[reasoning_options]] +type = "effort" +values = ["none", "xhigh"] + +[cost] +input = 0.05 +output = 0.15 +cache_read = 0.025 + +[limit] +context = 262_144 +input = 262_144 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] From 9f4c5f6929250e13482037e71d5ff0546c67d0c2 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 17:21:35 +0000 Subject: [PATCH 084/392] chore(sync): update OpenRouter model catalog (#7511) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/z-ai/glm-4.6.toml | 9 ++++++--- 2 files changed, 9 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index ff66884b828..6d4eb7a48b8 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04172 -output = 0.08344 -cache_read = 0.008344 +input = 0.04116 +output = 0.08232 +cache_read = 0.008232 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-4.6.toml b/providers/openrouter/models/z-ai/glm-4.6.toml index e562f006f6c..9a813d7b44e 100644 --- a/providers/openrouter/models/z-ai/glm-4.6.toml +++ b/providers/openrouter/models/z-ai/glm-4.6.toml @@ -7,6 +7,9 @@ structured_output = true type = "toggle" [cost] -input = 0.5 -output = 2 -cache_read = 0.1 +input = 0.43 +output = 1.75 +cache_read = 0.08 + +[limit] +output = 16_384 From e68a3aad18efe08a2bf44401eefedd2a9df0f9f5 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 17:22:09 +0000 Subject: [PATCH 085/392] chore(sync): update Kilo model catalog (#7512) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/tencent/hy3.toml | 6 +++--- providers/kilo/models/z-ai/glm-4.6.toml | 9 +++++---- 2 files changed, 8 insertions(+), 7 deletions(-) diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index 2bdfa3cbd7b..e96e3110ab1 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.132 -output = 0.528 -cache_read = 0.033 +input = 0.0825 +output = 0.33 +cache_read = 0.020625 [limit] context = 262_144 diff --git a/providers/kilo/models/z-ai/glm-4.6.toml b/providers/kilo/models/z-ai/glm-4.6.toml index fd68fe0c68b..2a611ba49e7 100644 --- a/providers/kilo/models/z-ai/glm-4.6.toml +++ b/providers/kilo/models/z-ai/glm-4.6.toml @@ -7,9 +7,10 @@ type = "effort" values = ["none", "high"] [cost] -input = 0.5 -output = 2 -cache_read = 0.1 +input = 0.43 +output = 1.75 +cache_read = 0.08 [limit] -context = 202_752 +context = 198_000 +output = 16_384 From b6e376eda74813bbfe3bf907b0ad99c8198017d2 Mon Sep 17 00:00:00 2001 From: Aiden Cline <63023139+rekram1-node@users.noreply.github.com> Date: Sat, 19 Sep 2026 12:36:09 -0500 Subject: [PATCH 086/392] Update reasoning_options in muse-spark-1.3.toml Removed 'max' from reasoning_options values. --- providers/opencode/models/muse-spark-1.3.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/providers/opencode/models/muse-spark-1.3.toml b/providers/opencode/models/muse-spark-1.3.toml index cfaa3f534d3..80f0b886543 100644 --- a/providers/opencode/models/muse-spark-1.3.toml +++ b/providers/opencode/models/muse-spark-1.3.toml @@ -1,5 +1,5 @@ base_model = "meta/muse-spark-1.3" -reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high", "xhigh", "max"] }] +reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high", "xhigh"] }] [cost] input = 1.25 From d68788d330a64dc6a3edf0c165db9d65747a2314 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:27:34 +0000 Subject: [PATCH 087/392] chore(sync): update Kilo model catalog (#7513) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~z-ai/glm-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 73bfc195431..c5af7ff24a4 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8918 -output = 2.8028 -cache_read = 0.16562 +input = 0.8442 +output = 2.6532 +cache_read = 0.15678 [limit] context = 1_048_576 From c820ca27ddc04b04bacc60df09fa33b2e9fd986c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:27:45 +0000 Subject: [PATCH 088/392] chore(sync): update OpenRouter model catalog (#7514) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/~z-ai/glm-latest.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 6d4eb7a48b8..22943b94574 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04116 -output = 0.08232 -cache_read = 0.008232 +input = 0.0406 +output = 0.0812 +cache_read = 0.00812 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index d3b500c9777..f42b1197f93 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8918 -output = 2.8028 -cache_read = 0.16562 +input = 0.8442 +output = 2.6532 +cache_read = 0.15678 [limit] context = 1_310_720 From 3baa7cd3ba4a2269ee94399e6eacb32a1eeecd38 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 19:20:27 +0000 Subject: [PATCH 089/392] chore(sync): update OpenRouter model catalog (#7516) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 22943b94574..463a135fc41 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.0406 -output = 0.0812 -cache_read = 0.00812 +input = 0.04004 +output = 0.08008 +cache_read = 0.008008 [limit] context = 1_048_576 From 8e6b6b80eb3a3cdbead0e94d46c61a23e747cbc2 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 20:24:31 +0000 Subject: [PATCH 090/392] chore(sync): update OpenRouter model catalog (#7517) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 463a135fc41..f8d9e7ed131 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.04004 -output = 0.08008 -cache_read = 0.008008 +input = 0.03976 +output = 0.07952 +cache_read = 0.007952 [limit] context = 1_048_576 From d748926e1ab74e8da86e3b9f2fa942118f609112 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 21:22:23 +0000 Subject: [PATCH 091/392] chore(sync): update OpenRouter model catalog (#7519) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index f8d9e7ed131..21112c8f9b1 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.03976 -output = 0.07952 -cache_read = 0.007952 +input = 0.03948 +output = 0.07896 +cache_read = 0.007896 [limit] context = 1_048_576 From 36bf49018bd02e4987f6cfc890a1fb1f479ebb5d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 22:23:26 +0000 Subject: [PATCH 092/392] chore(sync): update OpenRouter model catalog (#7520) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 21112c8f9b1..87ebe0279bb 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.03948 -output = 0.07896 -cache_read = 0.007896 +input = 0.03864 +output = 0.07728 +cache_read = 0.007728 [limit] context = 1_048_576 From f85553799b09968cd6e0d4b94a65e771d8eaf397 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sat, 19 Sep 2026 23:21:56 +0000 Subject: [PATCH 093/392] chore(sync): update OpenRouter model catalog (#7521) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.2.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 87ebe0279bb..c1ff87abe3f 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.03864 -output = 0.07728 -cache_read = 0.007728 +input = 0.03808 +output = 0.07616 +cache_read = 0.007616 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-5.2.toml b/providers/openrouter/models/z-ai/glm-5.2.toml index 12ed56b2bd6..e1246ed1b0a 100644 --- a/providers/openrouter/models/z-ai/glm-5.2.toml +++ b/providers/openrouter/models/z-ai/glm-5.2.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.5544 -output = 1.7424 -cache_read = 0.10296 +input = 0.6496 +output = 2.0416 +cache_read = 0.12064 [limit] context = 1_048_576 From 132454ed014ae43c9199b232a2ace3bd37031dae Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 01:05:27 +0000 Subject: [PATCH 094/392] chore(sync): update OpenRouter model catalog (#7522) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml | 8 ++++++-- providers/openrouter/models/qwen/qwen3.5-9b.toml | 4 +++- providers/openrouter/models/qwen/qwen3.8-27b.toml | 6 +++--- providers/openrouter/models/tencent/hy3.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.2.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.3.toml | 6 +++--- 7 files changed, 24 insertions(+), 18 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index c1ff87abe3f..d635c80b599 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.03808 -output = 0.07616 -cache_read = 0.007616 +input = 0.03724 +output = 0.07448 +cache_read = 0.007448 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml b/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml index 34aa7a00097..8257248c271 100644 --- a/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml +++ b/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml @@ -6,8 +6,12 @@ base_model = "alibaba/qwen3.5-35b-a3b" type = "toggle" [cost] -input = 0.1625 -output = 1.3 +input = 0.3125 +output = 1.25 +cache_read = 0.15625 + +[limit] +output = 16_384 [modalities] input = ["text", "image", "video"] diff --git a/providers/openrouter/models/qwen/qwen3.5-9b.toml b/providers/openrouter/models/qwen/qwen3.5-9b.toml index 34660f4a8ed..8f34db0bc11 100644 --- a/providers/openrouter/models/qwen/qwen3.5-9b.toml +++ b/providers/openrouter/models/qwen/qwen3.5-9b.toml @@ -1,3 +1,5 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens base_model = "alibaba/qwen3.5-9b" [[reasoning_options]] @@ -8,4 +10,4 @@ input = 0.1 output = 0.15 [limit] -output = 235_929 +output = 32_768 diff --git a/providers/openrouter/models/qwen/qwen3.8-27b.toml b/providers/openrouter/models/qwen/qwen3.8-27b.toml index 488923c2abf..d1ddcffa2ed 100644 --- a/providers/openrouter/models/qwen/qwen3.8-27b.toml +++ b/providers/openrouter/models/qwen/qwen3.8-27b.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "medium", "xhigh"] [cost] -input = 0.214 -output = 2.55 -cache_read = 0.15 +input = 0.42 +output = 3 +cache_read = 0.085 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/tencent/hy3.toml b/providers/openrouter/models/tencent/hy3.toml index f61fa3175e9..80ecfdd3aa8 100644 --- a/providers/openrouter/models/tencent/hy3.toml +++ b/providers/openrouter/models/tencent/hy3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.0825 -output = 0.33 -cache_read = 0.020625 +input = 0.132 +output = 0.528 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/openrouter/models/z-ai/glm-5.2.toml b/providers/openrouter/models/z-ai/glm-5.2.toml index e1246ed1b0a..12ed56b2bd6 100644 --- a/providers/openrouter/models/z-ai/glm-5.2.toml +++ b/providers/openrouter/models/z-ai/glm-5.2.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.6496 -output = 2.0416 -cache_read = 0.12064 +input = 0.5544 +output = 1.7424 +cache_read = 0.10296 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index 6ca3ad0c4f6..7e314ece82e 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.91 -output = 2.86 -cache_read = 0.169 +input = 0.896 +output = 2.816 +cache_read = 0.1664 [limit] context = 1_310_720 From 198dec37e53fa369af5943183ebcd3ae75cea55a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 01:05:37 +0000 Subject: [PATCH 095/392] chore(sync): update Kilo model catalog (#7523) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/qwen/qwen3.5-35b-a3b.toml | 4 ++++ providers/kilo/models/qwen/qwen3.5-9b.toml | 3 ++- providers/kilo/models/qwen/qwen3.8-27b.toml | 1 + providers/kilo/models/tencent/hy3.toml | 6 +++--- 4 files changed, 10 insertions(+), 4 deletions(-) diff --git a/providers/kilo/models/qwen/qwen3.5-35b-a3b.toml b/providers/kilo/models/qwen/qwen3.5-35b-a3b.toml index 96076003f63..bdebd95519c 100644 --- a/providers/kilo/models/qwen/qwen3.5-35b-a3b.toml +++ b/providers/kilo/models/qwen/qwen3.5-35b-a3b.toml @@ -9,5 +9,9 @@ values = ["none", "high"] input = 0.1625 output = 1.3 +[limit] +context = 256_000 +output = 16_384 + [modalities] input = ["text", "image", "video"] diff --git a/providers/kilo/models/qwen/qwen3.5-9b.toml b/providers/kilo/models/qwen/qwen3.5-9b.toml index bc4a13e3e11..7c1d5856e31 100644 --- a/providers/kilo/models/qwen/qwen3.5-9b.toml +++ b/providers/kilo/models/qwen/qwen3.5-9b.toml @@ -10,4 +10,5 @@ input = 0.1 output = 0.15 [limit] -output = 235_929 +context = 256_000 +output = 32_768 diff --git a/providers/kilo/models/qwen/qwen3.8-27b.toml b/providers/kilo/models/qwen/qwen3.8-27b.toml index 9b18fb98b33..820f69fb194 100644 --- a/providers/kilo/models/qwen/qwen3.8-27b.toml +++ b/providers/kilo/models/qwen/qwen3.8-27b.toml @@ -12,4 +12,5 @@ cache_read = 0.085 cache_write = 0.53125 [limit] +context = 1_000_000 output = 131_072 diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index e96e3110ab1..2bdfa3cbd7b 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.0825 -output = 0.33 -cache_read = 0.020625 +input = 0.132 +output = 0.528 +cache_read = 0.033 [limit] context = 262_144 From 1dfbdde36eec9ea3c13ebeebea7fcddbb2e0a20b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 02:34:41 +0000 Subject: [PATCH 096/392] chore(sync): update Kilo model catalog (#7524) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 1 - .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 2 files changed, 5 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index d4d63d57a55..007b654034e 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -12,4 +12,3 @@ cache_read = 0.044 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 522fff2ce46..384684585d6 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.57816 -output = 1.73448 -cache_read = 0.018396 +input = 0.57684 +output = 1.73052 +cache_read = 0.019228 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] From 57567944cfd29c30605831a52ad72bbdb09cea46 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 02:34:45 +0000 Subject: [PATCH 097/392] chore(sync): update OpenRouter model catalog (#7525) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 +++---- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 2 files changed, 7 insertions(+), 8 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c435d69ac6..7c5828b16e2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,10 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57816 -output = 1.73448 -cache_read = 0.018396 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 42e4245ad89..af6134d3fef 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57816 -output = 1.73448 -cache_read = 0.018396 +input = 0.57684 +output = 1.73052 +cache_read = 0.019228 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] From a680f9b5bf4021da44987fc5080f840a5982323b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 03:30:48 +0000 Subject: [PATCH 098/392] chore(sync): update OpenRouter model catalog (#7526) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index d635c80b599..0d037ac480a 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.03724 -output = 0.07448 -cache_read = 0.007448 +input = 0.03696 +output = 0.07392 +cache_read = 0.007392 [limit] context = 1_048_576 From fb1319b931674274adef728d69ed497edcf4f052 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 03:31:07 +0000 Subject: [PATCH 099/392] chore(sync): update Kilo model catalog (#7527) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/baidu/ernie-4.5-vl-424b-a47b.toml | 2 +- .../kilo/models/deepseek/deepseek-r1-distill-llama-70b.toml | 2 +- providers/kilo/models/deepseek/deepseek-v3.1-terminus.toml | 2 +- providers/kilo/models/deepseek/deepseek-v3.2-exp.toml | 2 +- 4 files changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/kilo/models/baidu/ernie-4.5-vl-424b-a47b.toml b/providers/kilo/models/baidu/ernie-4.5-vl-424b-a47b.toml index a962a373363..ef65a04722c 100644 --- a/providers/kilo/models/baidu/ernie-4.5-vl-424b-a47b.toml +++ b/providers/kilo/models/baidu/ernie-4.5-vl-424b-a47b.toml @@ -1,4 +1,4 @@ -name = "Baidu: ERNIE 4.5 VL 424B A47B " +name = "Baidu: ERNIE 4.5 VL 424B A47B (retires Oct 8)" description = "Multimodal reasoning model for visual analysis, planning, and tool use" family = "ernie" release_date = "2025-06-30" diff --git a/providers/kilo/models/deepseek/deepseek-r1-distill-llama-70b.toml b/providers/kilo/models/deepseek/deepseek-r1-distill-llama-70b.toml index 3635f13d091..18d5e9a099b 100644 --- a/providers/kilo/models/deepseek/deepseek-r1-distill-llama-70b.toml +++ b/providers/kilo/models/deepseek/deepseek-r1-distill-llama-70b.toml @@ -1,4 +1,4 @@ -name = "DeepSeek: R1 Distill Llama 70B" +name = "DeepSeek: R1 Distill Llama 70B (retires Sep 28)" description = "DeepSeek reasoning model for multi-step analysis, math, coding, and tools" family = "deepseek" release_date = "2025-01-23" diff --git a/providers/kilo/models/deepseek/deepseek-v3.1-terminus.toml b/providers/kilo/models/deepseek/deepseek-v3.1-terminus.toml index 79663dcfa0f..0715bd02e00 100644 --- a/providers/kilo/models/deepseek/deepseek-v3.1-terminus.toml +++ b/providers/kilo/models/deepseek/deepseek-v3.1-terminus.toml @@ -1,4 +1,4 @@ -name = "DeepSeek: DeepSeek V3.1 Terminus" +name = "DeepSeek: DeepSeek V3.1 Terminus (retires Sep 28)" description = "DeepSeek chat model for instruction following, coding, and analysis" family = "deepseek" release_date = "2025-09-22" diff --git a/providers/kilo/models/deepseek/deepseek-v3.2-exp.toml b/providers/kilo/models/deepseek/deepseek-v3.2-exp.toml index 7e9c21db938..0ac45019600 100644 --- a/providers/kilo/models/deepseek/deepseek-v3.2-exp.toml +++ b/providers/kilo/models/deepseek/deepseek-v3.2-exp.toml @@ -1,4 +1,4 @@ -name = "DeepSeek: DeepSeek V3.2 Exp" +name = "DeepSeek: DeepSeek V3.2 Exp (retires Sep 28)" description = "DeepSeek chat model for instruction following, coding, and analysis" family = "deepseek" release_date = "2025-09-29" From e582fd85652879f0222385b6e994cd2d46e9037d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 04:29:15 +0000 Subject: [PATCH 100/392] chore(sync): update Kilo model catalog (#7528) Co-authored-by: opencode-agent[bot] --- .../deepseek/deepseek-v4-flash-0731:free.toml | 15 --------------- 1 file changed, 15 deletions(-) delete mode 100644 providers/kilo/models/deepseek/deepseek-v4-flash-0731:free.toml diff --git a/providers/kilo/models/deepseek/deepseek-v4-flash-0731:free.toml b/providers/kilo/models/deepseek/deepseek-v4-flash-0731:free.toml deleted file mode 100644 index 73a49138fbe..00000000000 --- a/providers/kilo/models/deepseek/deepseek-v4-flash-0731:free.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "deepseek/deepseek-v4-flash-0731" -name = "DeepSeek: DeepSeek V4 Flash 0731 (free)" -description = "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows...." - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "high", "max"] - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 -output = 393_216 From 9fc144896fb2b84a5c19dc36d6442b5de1c16c55 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 04:29:25 +0000 Subject: [PATCH 101/392] chore(sync): update OpenRouter model catalog (#7529) Co-authored-by: opencode-agent[bot] --- .../deepseek/deepseek-v4-flash-0731:free.toml | 20 ------------------- .../models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.2.toml | 6 +++--- 3 files changed, 6 insertions(+), 26 deletions(-) delete mode 100644 providers/openrouter/models/deepseek/deepseek-v4-flash-0731:free.toml diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731:free.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731:free.toml deleted file mode 100644 index 1f431a902ab..00000000000 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731:free.toml +++ /dev/null @@ -1,20 +0,0 @@ -# Toggle: reasoning.enabled = true|false -# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens -base_model = "deepseek/deepseek-v4-flash-0731" -name = "DeepSeek V4 Flash 0731 (free)" -description = "Fast DeepSeek model for efficient chat, coding help, and agent loops" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 -output = 393_216 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 0d037ac480a..bd86f5a8f95 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.03696 -output = 0.07392 -cache_read = 0.007392 +input = 0.03668 +output = 0.07336 +cache_read = 0.007336 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-5.2.toml b/providers/openrouter/models/z-ai/glm-5.2.toml index 12ed56b2bd6..e1246ed1b0a 100644 --- a/providers/openrouter/models/z-ai/glm-5.2.toml +++ b/providers/openrouter/models/z-ai/glm-5.2.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.5544 -output = 1.7424 -cache_read = 0.10296 +input = 0.6496 +output = 2.0416 +cache_read = 0.12064 [limit] context = 1_048_576 From 4a018cd92f582e01dfcd86fbdc5178d58907f0c4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 05:25:15 +0000 Subject: [PATCH 102/392] chore(sync): update Vercel AI Gateway model catalog (#7530) Co-authored-by: opencode-agent[bot] --- .../vercel/models/stepfun/step-5-preview.toml | 22 +++++++++++++++++++ 1 file changed, 22 insertions(+) create mode 100644 providers/vercel/models/stepfun/step-5-preview.toml diff --git a/providers/vercel/models/stepfun/step-5-preview.toml b/providers/vercel/models/stepfun/step-5-preview.toml new file mode 100644 index 00000000000..8db9a1d1622 --- /dev/null +++ b/providers/vercel/models/stepfun/step-5-preview.toml @@ -0,0 +1,22 @@ +name = "Step 5 Preview" +description = "StepFun flash model for efficient multimodal reasoning, coding, and tool use" +family = "step" +release_date = "2026-09-20" +last_updated = "2026-09-20" +attachment = true +reasoning = false +tool_call = false +open_weights = false + +[cost] +input = 1 +output = 2.7 +cache_read = 0.05 + +[limit] +context = 1_000_000 +output = 1_000_000 + +[modalities] +input = ["text", "image"] +output = ["text"] From 65fe1856216db31941d962ca5895a047d0d2aca4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 06:37:31 +0000 Subject: [PATCH 103/392] chore(sync): update Kilo model catalog (#7533) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/meta/muse-glimmer-30b.toml | 2 +- .../models/mistralai/mistral-small-3.1-24b-instruct.toml | 2 +- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/providers/kilo/models/meta/muse-glimmer-30b.toml b/providers/kilo/models/meta/muse-glimmer-30b.toml index 2c352c0d845..aa01201020a 100644 --- a/providers/kilo/models/meta/muse-glimmer-30b.toml +++ b/providers/kilo/models/meta/muse-glimmer-30b.toml @@ -11,7 +11,7 @@ output = 1.1 cache_read = 0.04 [limit] -output = 117_964 +output = 16_384 [modalities] input = ["text", "image", "pdf"] diff --git a/providers/kilo/models/mistralai/mistral-small-3.1-24b-instruct.toml b/providers/kilo/models/mistralai/mistral-small-3.1-24b-instruct.toml index e5a92d90a1c..cddc0eb9495 100644 --- a/providers/kilo/models/mistralai/mistral-small-3.1-24b-instruct.toml +++ b/providers/kilo/models/mistralai/mistral-small-3.1-24b-instruct.toml @@ -6,7 +6,7 @@ last_updated = "2025-03-17" attachment = true reasoning = false temperature = true -tool_call = false +tool_call = true structured_output = false open_weights = false diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 384684585d6..5dd589c4cbf 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.57684 -output = 1.73052 -cache_read = 0.019228 +input = 0.5742 +output = 1.7226 +cache_read = 0.01914 [limit] context = 1_024_000 From aea958a0052aa3480685b872fea93bded65bbff7 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 06:38:05 +0000 Subject: [PATCH 104/392] chore(sync): update OpenRouter model catalog (#7532) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/meta/muse-glimmer-30b.toml | 6 +++--- .../models/mistralai/mistral-small-3.1-24b-instruct.toml | 2 +- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 3 files changed, 7 insertions(+), 7 deletions(-) diff --git a/providers/openrouter/models/meta/muse-glimmer-30b.toml b/providers/openrouter/models/meta/muse-glimmer-30b.toml index 3b46ee2987a..d4077f81891 100644 --- a/providers/openrouter/models/meta/muse-glimmer-30b.toml +++ b/providers/openrouter/models/meta/muse-glimmer-30b.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "medium", "high", "xhigh"] [cost] -input = 0.35 -output = 1.5 +input = 0.3 +output = 1.2 cache_read = 0.04 [limit] -output = 117_964 +output = 16_384 diff --git a/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml b/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml index 45893fba68d..bc32aea52eb 100644 --- a/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml +++ b/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml @@ -6,7 +6,7 @@ last_updated = "2025-03-17" attachment = true reasoning = false temperature = true -tool_call = false +tool_call = true structured_output = false knowledge = "2023-10-31" open_weights = true diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index af6134d3fef..d54f33b7b90 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57684 -output = 1.73052 -cache_read = 0.019228 +input = 0.5742 +output = 1.7226 +cache_read = 0.01914 [limit] context = 1_048_576 From a24c1c5bd9edd972bd942e83fd64cb4444289d60 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 07:26:49 +0000 Subject: [PATCH 105/392] chore(sync): update Kilo model catalog (#7536) Co-authored-by: opencode-agent[bot] --- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 5dd589c4cbf..2b08b36b516 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.5742 -output = 1.7226 -cache_read = 0.01914 +input = 0.57288 +output = 1.71864 +cache_read = 0.018228 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 393_216 [modalities] input = ["text"] From 9c86b872b37895567458adabb0298ec512c8b4fa Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 07:26:54 +0000 Subject: [PATCH 106/392] chore(sync): update OpenRouter model catalog (#7537) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index d54f33b7b90..6f7ab19b7ee 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5742 -output = 1.7226 -cache_read = 0.01914 +input = 0.57288 +output = 1.71864 +cache_read = 0.018228 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] From e82469bebab52d539d292fe1aab14f3a54b3fc00 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 08:31:07 +0000 Subject: [PATCH 107/392] chore(sync): update OpenRouter model catalog (#7538) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- providers/openrouter/models/~z-ai/glm-flash-latest.toml | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 6f7ab19b7ee..cbf8d0161f9 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57288 -output = 1.71864 -cache_read = 0.018228 +input = 0.57024 +output = 1.71072 +cache_read = 0.018144 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~z-ai/glm-flash-latest.toml b/providers/openrouter/models/~z-ai/glm-flash-latest.toml index c32fca67d4b..8f90bbaaf93 100644 --- a/providers/openrouter/models/~z-ai/glm-flash-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-flash-latest.toml @@ -21,7 +21,7 @@ cache_read = 0.015 [limit] context = 1_310_720 -output = 131_072 +output = 943_718 [modalities] input = ["text", "image", "video"] From bd94b6ab639374b576d40c6ce78774b7742101ef Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 08:31:17 +0000 Subject: [PATCH 108/392] chore(sync): update Kilo model catalog (#7539) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- providers/kilo/models/~z-ai/glm-flash-latest.toml | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 2b08b36b516..b30e8d73cb1 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.57288 -output = 1.71864 -cache_read = 0.018228 +input = 0.57024 +output = 1.71072 +cache_read = 0.018144 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~z-ai/glm-flash-latest.toml b/providers/kilo/models/~z-ai/glm-flash-latest.toml index e4ce4e5dfe6..aee70deebb0 100644 --- a/providers/kilo/models/~z-ai/glm-flash-latest.toml +++ b/providers/kilo/models/~z-ai/glm-flash-latest.toml @@ -21,7 +21,7 @@ cache_read = 0.015 [limit] context = 1_048_576 -output = 131_072 +output = 943_718 [modalities] input = ["text", "image", "video"] From 688ef859c6005b6c46c527b3ef7dc2b70c09dd0d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 09:25:40 +0000 Subject: [PATCH 109/392] chore(sync): update Kilo model catalog (#7540) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index b30e8d73cb1..493f2b1a807 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.57024 -output = 1.71072 -cache_read = 0.018144 +input = 0.5676 +output = 1.7028 +cache_read = 0.01806 [limit] context = 1_048_576 From ff0fd62ef4dab82f27ea16545974a944d54a2864 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 09:25:44 +0000 Subject: [PATCH 110/392] chore(sync): update NanoGPT model catalog (#7541) Co-authored-by: opencode-agent[bot] --- .../nano-gpt/models/TEE/deepseek-v4-flash.toml | 13 ------------- 1 file changed, 13 deletions(-) delete mode 100644 providers/nano-gpt/models/TEE/deepseek-v4-flash.toml diff --git a/providers/nano-gpt/models/TEE/deepseek-v4-flash.toml b/providers/nano-gpt/models/TEE/deepseek-v4-flash.toml deleted file mode 100644 index e9d130676e5..00000000000 --- a/providers/nano-gpt/models/TEE/deepseek-v4-flash.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "deepseek/deepseek-v4-flash" -name = "DeepSeek V4 Flash TEE" -reasoning_options = [] - -[cost] -input = 0.2 -output = 0.4 -cache_read = 0.04 - -[limit] -context = 1_048_576 -input = 1_048_576 -output = 393_216 From 5e6d15218298092d8310dff46544c2e72f3641c5 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 09:26:11 +0000 Subject: [PATCH 111/392] chore(sync): update OpenRouter model catalog (#7542) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index bd86f5a8f95..a3b41c6546f 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.03668 -output = 0.07336 -cache_read = 0.007336 +input = 0.0364 +output = 0.0728 +cache_read = 0.00728 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index cbf8d0161f9..2b68d4267da 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57024 -output = 1.71072 -cache_read = 0.018144 +input = 0.5676 +output = 1.7028 +cache_read = 0.01806 [limit] context = 1_048_576 From 18dd8a6b1fd7820a785ba26e498aab90c7422372 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 10:24:23 +0000 Subject: [PATCH 112/392] chore(sync): update Kilo model catalog (#7547) Co-authored-by: opencode-agent[bot] --- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 493f2b1a807..0afbdafc83a 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.5676 -output = 1.7028 -cache_read = 0.01806 +input = 0.56628 +output = 1.69884 +cache_read = 0.018876 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] From 551c5ef299e818b0f2dc9de5177eac4426a7f8f5 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 10:24:27 +0000 Subject: [PATCH 113/392] chore(sync): update NanoGPT model catalog (#7545) Co-authored-by: opencode-agent[bot] --- .../nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml | 6 +++--- .../models/qwen/qwen3.8-27b-uncensored:thinking.toml | 6 +++--- providers/nano-gpt/models/typesafe/jev-latest.toml | 12 ++++++++++++ 3 files changed, 18 insertions(+), 6 deletions(-) create mode 100644 providers/nano-gpt/models/typesafe/jev-latest.toml diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml index 612900717c2..3dc414edb50 100644 --- a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml @@ -11,9 +11,9 @@ open_weights = true reasoning_options = [] [cost] -input = 0.2 -output = 1.7 -cache_read = 0.18 +input = 0.15 +output = 1.2 +cache_read = 0.125 [limit] context = 524_288 diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml index 46cc8830338..d4ee170d540 100644 --- a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml @@ -11,9 +11,9 @@ open_weights = true reasoning_options = [] [cost] -input = 0.2 -output = 1.7 -cache_read = 0.18 +input = 0.15 +output = 1.2 +cache_read = 0.125 [limit] context = 524_288 diff --git a/providers/nano-gpt/models/typesafe/jev-latest.toml b/providers/nano-gpt/models/typesafe/jev-latest.toml new file mode 100644 index 00000000000..49c0acf91f0 --- /dev/null +++ b/providers/nano-gpt/models/typesafe/jev-latest.toml @@ -0,0 +1,12 @@ +base_model = "typesafe/jev-latest" +name = "Jev Latest" +structured_output = false + +[cost] +input = 0.042 +output = 0 +cache_read = 0.021 + +[limit] +context = 32_000 +input = 32_000 From ea28e08347e6d191b2596c285a5494fb6398bad4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 10:24:30 +0000 Subject: [PATCH 114/392] chore(sync): update CrossModel model catalog (#7548) Co-authored-by: opencode-agent[bot] --- .../models/qwen/qwen3.8-omni-flash.toml | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 providers/crossmodel/models/qwen/qwen3.8-omni-flash.toml diff --git a/providers/crossmodel/models/qwen/qwen3.8-omni-flash.toml b/providers/crossmodel/models/qwen/qwen3.8-omni-flash.toml new file mode 100644 index 00000000000..b70235cd8f0 --- /dev/null +++ b/providers/crossmodel/models/qwen/qwen3.8-omni-flash.toml @@ -0,0 +1,17 @@ +base_model = "alibaba/qwen3.8-omni-flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.13 +output = 0.43 +cache_read = 0.016 +cache_write = 0.13 From 085f4d737d63f33e68f36a281a38b961c80b0e9f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 10:24:38 +0000 Subject: [PATCH 115/392] chore(sync): update OpenRouter model catalog (#7546) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml | 4 ++-- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml index d857fa53dab..91402212177 100644 --- a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml +++ b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml @@ -12,8 +12,8 @@ knowledge = "2025-03-31" open_weights = true [cost] -input = 0.13 -output = 0.52 +input = 0.2 +output = 0.7 [limit] context = 262_144 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 2b68d4267da..b85f4a83354 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5676 -output = 1.7028 -cache_read = 0.01806 +input = 0.56628 +output = 1.69884 +cache_read = 0.018876 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] From ea88df45a8e84fb7e67023a7458b14fad41c7a2e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:23:08 +0000 Subject: [PATCH 116/392] chore(sync): update Kilo model catalog (#7560) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 0afbdafc83a..6565f5a8e9b 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.56628 -output = 1.69884 -cache_read = 0.018876 +input = 0.56364 +output = 1.69092 +cache_read = 0.018788 [limit] context = 1_024_000 From 38ef8bdc93357492d429dcbab47058e66eb6365c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:23:12 +0000 Subject: [PATCH 117/392] chore(sync): update OpenRouter model catalog (#7561) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index b85f4a83354..ff538327fb0 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.56628 -output = 1.69884 -cache_read = 0.018876 +input = 0.56364 +output = 1.69092 +cache_read = 0.018788 [limit] context = 1_048_576 From 9062eaf44335489e1a31c53ea90d1f7de214010b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 11:23:15 +0000 Subject: [PATCH 118/392] chore(sync): update NanoGPT model catalog (#7563) Co-authored-by: opencode-agent[bot] --- providers/nano-gpt/models/qwen/qwen3-14b.toml | 2 +- providers/nano-gpt/models/qwen/qwen3-30b-a3b.toml | 1 - providers/nano-gpt/models/qwen/qwen3-8b.toml | 2 +- 3 files changed, 2 insertions(+), 3 deletions(-) diff --git a/providers/nano-gpt/models/qwen/qwen3-14b.toml b/providers/nano-gpt/models/qwen/qwen3-14b.toml index 21b6ecf5bc8..0ec7e2615e0 100644 --- a/providers/nano-gpt/models/qwen/qwen3-14b.toml +++ b/providers/nano-gpt/models/qwen/qwen3-14b.toml @@ -6,7 +6,7 @@ release_date = "2024-01-01" last_updated = "2024-01-01" attachment = false reasoning = false -tool_call = false +tool_call = true structured_output = false open_weights = true diff --git a/providers/nano-gpt/models/qwen/qwen3-30b-a3b.toml b/providers/nano-gpt/models/qwen/qwen3-30b-a3b.toml index ef583b4ad7d..3fe31cbc4c1 100644 --- a/providers/nano-gpt/models/qwen/qwen3-30b-a3b.toml +++ b/providers/nano-gpt/models/qwen/qwen3-30b-a3b.toml @@ -1,7 +1,6 @@ # Included in subscription base_model = "alibaba/qwen3-30b-a3b" reasoning = false -tool_call = false structured_output = false [cost] diff --git a/providers/nano-gpt/models/qwen/qwen3-8b.toml b/providers/nano-gpt/models/qwen/qwen3-8b.toml index 5dd994f1ea3..96cac74c86b 100644 --- a/providers/nano-gpt/models/qwen/qwen3-8b.toml +++ b/providers/nano-gpt/models/qwen/qwen3-8b.toml @@ -5,7 +5,7 @@ release_date = "2024-01-01" last_updated = "2024-01-01" attachment = false reasoning = false -tool_call = false +tool_call = true structured_output = false open_weights = true From 5e24bc6d4dcbc10d6c6702070349cb64c5208677 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 12:33:21 +0000 Subject: [PATCH 119/392] chore(sync): update OpenRouter model catalog (#7565) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index ff538327fb0..e56641ce63e 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.56364 -output = 1.69092 -cache_read = 0.018788 +input = 0.56232 +output = 1.68696 +cache_read = 0.017892 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] From f717efa8a69901691a5590d774ad5f43aab14754 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 12:33:59 +0000 Subject: [PATCH 120/392] chore(sync): update Kilo model catalog (#7566) Co-authored-by: opencode-agent[bot] --- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 6565f5a8e9b..c178a343e37 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.56364 -output = 1.69092 -cache_read = 0.018788 +input = 0.56232 +output = 1.68696 +cache_read = 0.017892 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 393_216 [modalities] input = ["text"] From b8fb6496f17b8058260ac12d67ca16af8e835b41 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:23:36 +0000 Subject: [PATCH 121/392] chore(sync): update NanoGPT model catalog (#7569) Co-authored-by: opencode-agent[bot] --- .../models/stepfun/step-5-preview.toml | 28 +++++++++++++++++++ 1 file changed, 28 insertions(+) create mode 100644 providers/nano-gpt/models/stepfun/step-5-preview.toml diff --git a/providers/nano-gpt/models/stepfun/step-5-preview.toml b/providers/nano-gpt/models/stepfun/step-5-preview.toml new file mode 100644 index 00000000000..e6b122e57eb --- /dev/null +++ b/providers/nano-gpt/models/stepfun/step-5-preview.toml @@ -0,0 +1,28 @@ +name = "Step 5 Preview" +description = "Step 5 Preview is StepFun's 600B sparse MoE frontier model for production-scale agents, activating 27B parameters per token. It is built for software engineering, long-horizon tool use, research, professional knowledge work, and finance, with native text, image, and video understanding and a 1M-token context window. ⚠️ Note: This model routes through StepFun, so privacy and logging guarantees may be limited." +family = "step" +release_date = "2026-09-20" +last_updated = "2026-09-20" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1 +output = 2.7 +cache_read = 0.05 + +[limit] +context = 1_000_000 +input = 1_000_000 +output = 1_000_000 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] From 3fdcb5a3f3d3180e76f7a105da5d2602a710f522 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:23:41 +0000 Subject: [PATCH 122/392] chore(sync): update Kilo model catalog (#7568) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/qwen/qwen3.8-27b.toml | 1 - .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 2 files changed, 5 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/qwen/qwen3.8-27b.toml b/providers/kilo/models/qwen/qwen3.8-27b.toml index 820f69fb194..9b18fb98b33 100644 --- a/providers/kilo/models/qwen/qwen3.8-27b.toml +++ b/providers/kilo/models/qwen/qwen3.8-27b.toml @@ -12,5 +12,4 @@ cache_read = 0.085 cache_write = 0.53125 [limit] -context = 1_000_000 output = 131_072 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index c178a343e37..0a7f6cc12c4 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.56232 -output = 1.68696 -cache_read = 0.017892 +input = 0.55836 +output = 1.67508 +cache_read = 0.018612 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] From 5f6b993ea8252aabb2a6fd298bd198ff5240081a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 13:23:48 +0000 Subject: [PATCH 123/392] chore(sync): update OpenRouter model catalog (#7570) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/qwen/qwen3.8-27b.toml | 4 ++-- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index a3b41c6546f..0ab3f377da2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.0364 -output = 0.0728 -cache_read = 0.00728 +input = 0.03612 +output = 0.07224 +cache_read = 0.007224 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3.8-27b.toml b/providers/openrouter/models/qwen/qwen3.8-27b.toml index d1ddcffa2ed..87b348fc2c9 100644 --- a/providers/openrouter/models/qwen/qwen3.8-27b.toml +++ b/providers/openrouter/models/qwen/qwen3.8-27b.toml @@ -11,8 +11,8 @@ type = "effort" values = ["low", "medium", "xhigh"] [cost] -input = 0.42 -output = 3 +input = 0.2 +output = 2.55 cache_read = 0.085 [limit] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index e56641ce63e..b7cb08f5927 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.56232 -output = 1.68696 -cache_read = 0.017892 +input = 0.55836 +output = 1.67508 +cache_read = 0.018612 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] From a85f89fbf19e18a9fa6121f77a90dcea5510104a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 14:23:34 +0000 Subject: [PATCH 124/392] chore(sync): update Kilo model catalog (#7572) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml | 2 +- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index 007b654034e..01800e18158 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -11,4 +11,4 @@ output = 3.96 cache_read = 0.044 [limit] -context = 1_048_576 +context = 1_024_000 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 0a7f6cc12c4..42c36b3bc07 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.55836 -output = 1.67508 -cache_read = 0.018612 +input = 0.55044 +output = 1.65132 +cache_read = 0.018348 [limit] context = 1_024_000 From 70277a8a17439d36460a6d12741021e771910aa5 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 14:23:42 +0000 Subject: [PATCH 125/392] chore(sync): update OpenRouter model catalog (#7573) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.3.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 4 files changed, 12 insertions(+), 12 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 0ab3f377da2..af806ba67d3 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.03612 -output = 0.07224 -cache_read = 0.007224 +input = 0.03584 +output = 0.07168 +cache_read = 0.007168 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..679f96c65ab 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.55044 +output = 1.65132 +cache_read = 0.018348 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index 7e314ece82e..452bead80a9 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.896 -output = 2.816 -cache_read = 0.1664 +input = 0.882 +output = 2.772 +cache_read = 0.1638 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index b7cb08f5927..21b1ebe291f 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.55836 -output = 1.67508 -cache_read = 0.018612 +input = 0.55044 +output = 1.65132 +cache_read = 0.018348 [limit] context = 1_048_576 From ad95e11fcdf4df38c5f7336b925e4d99baacb541 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 14:23:43 +0000 Subject: [PATCH 126/392] chore(sync): update NanoGPT model catalog (#7571) Co-authored-by: opencode-agent[bot] --- providers/nano-gpt/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- .../models/deepseek/deepseek-v4.1-flash:thinking.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash.toml b/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash.toml index 8e61a5cbf70..c7745ff077e 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.1 -output = 0.4 -cache_read = 0.003 +input = 0.13 +output = 0.52 +cache_read = 0.006 [limit] input = 1_000_000 diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash:thinking.toml b/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash:thinking.toml index 90c16dc648c..c26ac296879 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash:thinking.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash:thinking.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.1 -output = 0.4 -cache_read = 0.003 +input = 0.13 +output = 0.52 +cache_read = 0.006 [limit] input = 1_000_000 From e35cc0fb47862c2c383b47fdc9feb1e22459dca3 Mon Sep 17 00:00:00 2001 From: Pedro Santos <18473317+pc-gs@users.noreply.github.com> Date: Sun, 20 Sep 2026 15:04:58 +0000 Subject: [PATCH 127/392] fix(opencode-go): correct Go docs URL and MiniMax cache pricing Fixes #2097 --- providers/opencode-go/models/minimax-m2.5.toml | 3 ++- providers/opencode-go/models/minimax-m2.7.toml | 1 + providers/opencode-go/provider.toml | 2 +- 3 files changed, 4 insertions(+), 2 deletions(-) diff --git a/providers/opencode-go/models/minimax-m2.5.toml b/providers/opencode-go/models/minimax-m2.5.toml index 81fcd79363a..f6973ebf170 100644 --- a/providers/opencode-go/models/minimax-m2.5.toml +++ b/providers/opencode-go/models/minimax-m2.5.toml @@ -15,7 +15,8 @@ status = "deprecated" [cost] input = 0.3 output = 1.2 -cache_read = 0.03 +cache_read = 0.06 +cache_write = 0.375 [limit] context = 204_800 diff --git a/providers/opencode-go/models/minimax-m2.7.toml b/providers/opencode-go/models/minimax-m2.7.toml index 311a9c4af6f..33faddac7d9 100644 --- a/providers/opencode-go/models/minimax-m2.7.toml +++ b/providers/opencode-go/models/minimax-m2.7.toml @@ -15,6 +15,7 @@ open_weights = true input = 0.3 output = 1.2 cache_read = 0.06 +cache_write = 0.375 [limit] context = 204_800 diff --git a/providers/opencode-go/provider.toml b/providers/opencode-go/provider.toml index 833a7cb0002..7ddf38e1f5d 100644 --- a/providers/opencode-go/provider.toml +++ b/providers/opencode-go/provider.toml @@ -8,4 +8,4 @@ npm = "@ai-sdk/openai-compatible" # by the public Zen page. # https://opencode.ai/docs/zen#endpoints api = "https://opencode.ai/zen/go/v1" -doc = "https://opencode.ai/docs/zen" +doc = "https://opencode.ai/docs/go" From dc365bb0098e0fb32f8c12012c8a0a7e99ecd9b7 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 15:23:28 +0000 Subject: [PATCH 128/392] chore(sync): update LLM Gateway model catalog (#7578) Co-authored-by: opencode-agent[bot] --- .../models/baidu/deepseek-v4-flash.toml | 2 +- .../models/baidu/deepseek-v4-pro.toml | 2 +- .../models/baidu/deepseek-v4.1-flash.toml | 15 +++++++++++++++ 3 files changed, 17 insertions(+), 2 deletions(-) create mode 100644 providers/llmgateway-providers/models/baidu/deepseek-v4.1-flash.toml diff --git a/providers/llmgateway-providers/models/baidu/deepseek-v4-flash.toml b/providers/llmgateway-providers/models/baidu/deepseek-v4-flash.toml index 95bedd8498e..40383b0961a 100644 --- a/providers/llmgateway-providers/models/baidu/deepseek-v4-flash.toml +++ b/providers/llmgateway-providers/models/baidu/deepseek-v4-flash.toml @@ -12,7 +12,7 @@ values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 0.44 output = 1.32 -cache_read = 0.044 +cache_read = 0.014 [limit] context = 1_048_576 diff --git a/providers/llmgateway-providers/models/baidu/deepseek-v4-pro.toml b/providers/llmgateway-providers/models/baidu/deepseek-v4-pro.toml index bab1be34eeb..e9463e0fe2a 100644 --- a/providers/llmgateway-providers/models/baidu/deepseek-v4-pro.toml +++ b/providers/llmgateway-providers/models/baidu/deepseek-v4-pro.toml @@ -12,7 +12,7 @@ values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 1.32 output = 3.96 -cache_read = 0.132 +cache_read = 0.042 [limit] context = 1_048_576 diff --git a/providers/llmgateway-providers/models/baidu/deepseek-v4.1-flash.toml b/providers/llmgateway-providers/models/baidu/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..a22c803e3f5 --- /dev/null +++ b/providers/llmgateway-providers/models/baidu/deepseek-v4.1-flash.toml @@ -0,0 +1,15 @@ +base_model = "deepseek/deepseek-v4.1-flash" +name = "DeepSeek V4.1 Flash (Baidu)" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.006 + +[limit] +context = 1_048_576 +output = 393_216 From 1801d72511dc39c04b33b77ef1315643938b433d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 15:23:38 +0000 Subject: [PATCH 129/392] chore(sync): update Kilo model catalog (#7577) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 42c36b3bc07..e4d3acd84bf 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.55044 -output = 1.65132 -cache_read = 0.018348 +input = 0.5478 +output = 1.6434 +cache_read = 0.01826 [limit] context = 1_024_000 From e20b0c573ab134b6c55ec001f06463342f40d4e8 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 15:23:41 +0000 Subject: [PATCH 130/392] chore(sync): update OpenRouter model catalog (#7576) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.3.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 4 files changed, 12 insertions(+), 12 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index af806ba67d3..98ee0445042 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.03584 -output = 0.07168 -cache_read = 0.007168 +input = 0.03556 +output = 0.07112 +cache_read = 0.007112 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 679f96c65ab..7191760d710 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.55044 -output = 1.65132 -cache_read = 0.018348 +input = 0.5478 +output = 1.6434 +cache_read = 0.01826 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index 452bead80a9..6ca3ad0c4f6 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.882 -output = 2.772 -cache_read = 0.1638 +input = 0.91 +output = 2.86 +cache_read = 0.169 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 21b1ebe291f..78806039bab 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.55044 -output = 1.65132 -cache_read = 0.018348 +input = 0.5478 +output = 1.6434 +cache_read = 0.01826 [limit] context = 1_048_576 From 6947a51e1b8198ae8bedb4b13ca839433fe14b63 Mon Sep 17 00:00:00 2001 From: Aiden Cline <63023139+rekram1-node@users.noreply.github.com> Date: Sun, 20 Sep 2026 10:59:09 -0500 Subject: [PATCH 131/392] feat: add model type filtering (#7579) * feat: add model type filtering * fix: preserve model types across catalog paths --------- Co-authored-by: opencode-agent --- README.md | 12 +++ bun.lock | 3 + models/typesafe/jev-latest.toml | 1 + packages/core/src/filter.ts | 83 ++++++++++++++++ packages/core/src/index.ts | 1 + packages/core/src/schema.ts | 5 + packages/core/src/sync/index.ts | 1 + packages/core/test/filter.test.ts | 103 +++++++++++++++++++ packages/core/test/schema.test.ts | 6 ++ packages/core/test/sync.test.ts | 22 ++++ packages/function/package.json | 3 + packages/function/src/worker.ts | 98 ++++++++++++++++-- packages/function/test/worker.test.ts | 138 ++++++++++++++++++++++++++ packages/sdk/src/client.ts | 14 ++- packages/sdk/src/effect.ts | 2 +- packages/sdk/src/effect/client.ts | 28 ++++-- packages/sdk/src/types.ts | 5 + packages/sdk/test/client.test.ts | 11 ++ packages/sdk/test/effect.test.ts | 14 +++ packages/web/script/build.ts | 28 ++++-- packages/web/src/render.tsx | 9 +- packages/web/src/server.ts | 56 +++++++---- 22 files changed, 590 insertions(+), 53 deletions(-) create mode 100644 packages/core/src/filter.ts create mode 100644 packages/core/test/filter.test.ts create mode 100644 packages/function/test/worker.test.ts diff --git a/README.md b/README.md index f26ba19ab83..8a6c27e736a 100644 --- a/README.md +++ b/README.md @@ -22,6 +22,17 @@ You can access this data through an API. curl https://models.dev/api.json ``` +Specialized model types are omitted by default. Filter by one or more +comma-separated model types, or use `all` for the complete catalog: + +```bash +curl "https://models.dev/api.json?type=decision" +curl "https://models.dev/api.json?type=all" +``` + +The currently supported model type is `decision`. The `type` parameter is also +available on `models.json`, `catalog.json`, and `model-schema.json`. + Use the **Model ID** field to do a lookup on any model; it's the identifier used by [AI SDK](https://ai-sdk.dev/). Provider-agnostic model metadata is available separately: @@ -270,6 +281,7 @@ Models must conform to the following schema, as defined in `packages/core/src/sc **Model Schema:** - `name`: String — Display name of the model +- `type` _(optional)_: String — Specialized model behavior; currently supports `decision` - `attachment`: Boolean — Supports file attachments - `reasoning`: Boolean — Supports reasoning / chain-of-thought - `tool_call`: Boolean - Supports tool calling diff --git a/bun.lock b/bun.lock index 2ba689570d3..2e4ee0e36cd 100644 --- a/bun.lock +++ b/bun.lock @@ -24,6 +24,9 @@ }, "packages/function": { "name": "@models.dev/function", + "dependencies": { + "@models.dev/core": "workspace:*", + }, "devDependencies": { "@cloudflare/workers-types": "4.20250522.0", "@tsconfig/bun": "catalog:", diff --git a/models/typesafe/jev-latest.toml b/models/typesafe/jev-latest.toml index 9b2f3a75896..6b54d1bd25f 100644 --- a/models/typesafe/jev-latest.toml +++ b/models/typesafe/jev-latest.toml @@ -4,6 +4,7 @@ # - https://typesafe.ai/blog/introducing-system-one-models-and-jev # The 64k context limit covers the state plus all questions in one request. name = "Jev" +type = "decision" description = "System One model for fast, typed probabilistic decisions over text or structured state" release_date = "2026-09-15" last_updated = "2026-09-15" diff --git a/packages/core/src/filter.ts b/packages/core/src/filter.ts new file mode 100644 index 00000000000..3cba07d6cae --- /dev/null +++ b/packages/core/src/filter.ts @@ -0,0 +1,83 @@ +export const MODEL_TYPES = ["decision"] as const; + +export type ModelTypeValue = (typeof MODEL_TYPES)[number]; +export type ModelTypeFilter = "default" | "all" | ModelTypeValue[]; + +interface TypedModel { + type?: ModelTypeValue; +} + +interface TypedProvider { + models: Record; +} + +export class InvalidModelTypeError extends Error { + constructor(value: string) { + super(`Invalid type value: ${value}`); + this.name = "InvalidModelTypeError"; + } +} + +export function parseModelTypes(value: string | null): ModelTypeFilter { + if (value === null || value === "") return "default"; + if (value === "all") return "all"; + + const types = value.split(","); + if ( + types.length === 0 || + types.includes("all") || + types.some((type) => !MODEL_TYPES.includes(type as ModelTypeValue)) + ) { + throw new InvalidModelTypeError(value); + } + + return [...new Set(types)] as ModelTypeValue[]; +} + +function includesModel(model: TypedModel, filter: ModelTypeFilter) { + if (filter === "all") return true; + if (filter === "default") return model.type === undefined; + return model.type !== undefined && filter.includes(model.type); +} + +export function filterProvidersByModelType< + T extends Record, +>(providers: T, filter: ModelTypeFilter): T { + if (filter === "all") return providers; + + return Object.fromEntries( + Object.entries(providers).flatMap(([providerID, provider]) => { + const models = Object.fromEntries( + Object.entries(provider.models).filter(([, model]) => + includesModel(model, filter), + ), + ); + return Object.keys(models).length === 0 + ? [] + : [[providerID, { ...provider, models }]]; + }), + ) as T; +} + +export function filterModelsByModelType>( + models: T, + filter: ModelTypeFilter, +): T { + if (filter === "all") return models; + return Object.fromEntries( + Object.entries(models).filter(([, model]) => includesModel(model, filter)), + ) as T; +} + +export function filterCatalogByModelType< + TProviders extends Record, + TModels extends Record, +>( + catalog: { providers: TProviders; models: TModels }, + filter: ModelTypeFilter, +): { providers: TProviders; models: TModels } { + return { + providers: filterProvidersByModelType(catalog.providers, filter), + models: filterModelsByModelType(catalog.models, filter), + }; +} diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 69f4e893dcb..b8be1ed0f80 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -2,3 +2,4 @@ export * from "./schema.js"; export * from "./generate.js"; export * from "./describe.js"; export * from "./family.js"; +export * from "./filter.js"; diff --git a/packages/core/src/schema.ts b/packages/core/src/schema.ts index c6c8b8f38b2..5c2d7b06681 100644 --- a/packages/core/src/schema.ts +++ b/packages/core/src/schema.ts @@ -1,6 +1,7 @@ import { z } from "zod"; import { ModelFamily } from "./family"; +import { MODEL_TYPES } from "./filter"; type JsonValue = | string @@ -148,6 +149,8 @@ const DateString = z const Modality = z.enum(["text", "audio", "image", "video", "pdf"]); +export const ModelType = z.enum(MODEL_TYPES); + const Modalities = z .object({ input: z.array(Modality), @@ -219,6 +222,7 @@ export const BenchmarkResult = z const ModelMetadataBase = z.object({ id: z.string(), + type: ModelType.optional(), name: z.string().min(1, "Model name cannot be empty"), description: z.string().min(1, "Model description cannot be empty"), family: ModelFamily.optional(), @@ -245,6 +249,7 @@ export type ModelMetadata = z.infer; const ModelBase = z.object({ id: z.string(), + type: ModelType.optional(), name: z.string().min(1, "Model name cannot be empty"), description: z.string().min(1, "Model description cannot be empty"), family: ModelFamily.optional(), diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index 5095f4124ef..031341b56cd 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -1011,6 +1011,7 @@ export function formatToml(model: z.infer) { if ("base_model_omit" in model && model.base_model_omit !== undefined) { lines.push(`base_model_omit = [${model.base_model_omit.map(quote).join(", ")}]`); } + if (model.type !== undefined) lines.push(`type = ${quote(model.type)}`); if (model.name !== undefined) lines.push(`name = ${quote(model.name)}`); if (model.description !== undefined) lines.push(`description = ${quote(model.description)}`); if (model.family !== undefined) lines.push(`family = ${quote(model.family)}`); diff --git a/packages/core/test/filter.test.ts b/packages/core/test/filter.test.ts new file mode 100644 index 00000000000..ea14f1588fb --- /dev/null +++ b/packages/core/test/filter.test.ts @@ -0,0 +1,103 @@ +import { describe, expect, test } from "bun:test"; + +import { + filterCatalogByModelType, + generateCatalog, + InvalidModelTypeError, + parseModelTypes, +} from "../src/index.js"; +import type { ModelMetadata, Provider } from "../src/index.js"; +import path from "node:path"; + +describe("model type filtering", () => { + test("defaults to untyped models and supports specific types and all", () => { + expect(parseModelTypes(null)).toBe("default"); + expect(parseModelTypes("")).toBe("default"); + expect(parseModelTypes("decision")).toEqual(["decision"]); + expect(parseModelTypes("all")).toBe("all"); + }); + + test("rejects unknown types and combining all with a type", () => { + expect(() => parseModelTypes("unknown")).toThrow(InvalidModelTypeError); + expect(() => parseModelTypes("all,decision")).toThrow( + InvalidModelTypeError, + ); + }); + + test("omits typed models by default", () => { + const catalog = fixture(); + const filtered = filterCatalogByModelType(catalog, "default"); + + expect(Object.keys(filtered.models)).toEqual(["standard"]); + expect(Object.keys(filtered.providers.example!.models)).toEqual([ + "standard", + ]); + expect(filtered.providers.decisionOnly).toBeUndefined(); + }); + + test("filters canonical and provider models by requested type", () => { + const catalog = fixture(); + const filtered = filterCatalogByModelType(catalog, ["decision"]); + + expect(Object.keys(filtered.models)).toEqual(["decision"]); + expect(Object.keys(filtered.providers.example!.models)).toEqual([ + "decision", + ]); + expect(Object.keys(filtered.providers.decisionOnly!.models)).toEqual([ + "decision", + ]); + expect(filterCatalogByModelType(catalog, "all")).toEqual(catalog); + }); + + test("every repository Jev model inherits decision and is omitted by default", async () => { + const root = path.join(import.meta.dir, "..", "..", ".."); + const catalog = await generateCatalog(root); + const jevModels = Object.values(catalog.providers).flatMap((provider) => + Object.values(provider.models).filter((model) => + model.id.toLowerCase().includes("jev"), + ), + ); + + expect(jevModels.length).toBeGreaterThan(0); + expect(jevModels.every((model) => model.type === "decision")).toBe(true); + + const defaults = filterCatalogByModelType(catalog, "default"); + expect( + Object.values(defaults.providers).some((provider) => + Object.values(provider.models).some((model) => model.type !== undefined), + ), + ).toBe(false); + expect(defaults.models["typesafe/jev-latest"]).toBeUndefined(); + expect(defaults.providers.vivgrid?.models.jev).toBeUndefined(); + + const decisions = filterCatalogByModelType(catalog, ["decision"]); + expect(decisions.models["typesafe/jev-latest"]?.type).toBe("decision"); + expect( + Object.values(decisions.providers).flatMap((provider) => + Object.values(provider.models), + ).length, + ).toBe(jevModels.length); + }); +}); + +function fixture() { + const standard = model("standard-model"); + const decision = model("decision-model", "decision"); + return { + models: { standard, decision }, + providers: { + example: { + id: "example", + models: { standard, decision }, + } as unknown as Provider, + decisionOnly: { + id: "decision-only", + models: { decision }, + } as unknown as Provider, + }, + }; +} + +function model(id: string, type?: "decision") { + return { id, type } as unknown as ModelMetadata; +} diff --git a/packages/core/test/schema.test.ts b/packages/core/test/schema.test.ts index f343c50b615..97a6d602032 100644 --- a/packages/core/test/schema.test.ts +++ b/packages/core/test/schema.test.ts @@ -70,6 +70,12 @@ describe("model schema", () => { expect(AuthoredModel.safeParse(model).success).toBe(false); }); + test("accepts decision model types", () => { + const model = baseModel({ type: "decision" }); + + expect(AuthoredModel.safeParse(model).success).toBe(true); + }); + test("accepts calendar-valid model dates", () => { for (const field of dateFields) { for (const value of [ diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 3557ba66127..d0609631969 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -2806,6 +2806,28 @@ test("resolves Eden AI aliases to the model they point at", () => { ).toBe("anthropic/claude-opus-5"); }); +test("preserves model type when formatting synced TOML", () => { + const content = formatToml({ + id: "typesafe/jev-latest", + type: "decision", + name: "Jev", + description: "System One model for typed decisions", + release_date: "2026-09-15", + last_updated: "2026-09-15", + attachment: false, + reasoning: false, + tool_call: false, + open_weights: false, + limit: { context: 64_000, output: 0 }, + modalities: { input: ["text"], output: ["text"] }, + }); + + expect(Bun.TOML.parse(content)).toMatchObject({ + type: "decision", + name: "Jev", + }); +}); + test("formats interleaved as a root field before reasoning option tables", () => { const content = formatToml({ id: "example/model", diff --git a/packages/function/package.json b/packages/function/package.json index 1d1c7c9f5c1..e1fd3d8260a 100644 --- a/packages/function/package.json +++ b/packages/function/package.json @@ -3,6 +3,9 @@ "name": "@models.dev/function", "private": true, "type": "module", + "dependencies": { + "@models.dev/core": "workspace:*" + }, "devDependencies": { "@cloudflare/workers-types": "4.20250522.0", "@tsconfig/bun": "catalog:" diff --git a/packages/function/src/worker.ts b/packages/function/src/worker.ts index 9ccccdd42e4..d616805d8f9 100644 --- a/packages/function/src/worker.ts +++ b/packages/function/src/worker.ts @@ -1,3 +1,13 @@ +import { + filterCatalogByModelType, + filterModelsByModelType, + filterProvidersByModelType, + InvalidModelTypeError, + MODEL_TYPES, + parseModelTypes, +} from "@models.dev/core/src/filter.js"; +import type { ModelTypeValue } from "@models.dev/core/src/filter.js"; + export interface Env { ASSETS: any; PosthogToken: string; @@ -64,11 +74,8 @@ export default { } if (url.pathname === "/model-schema.json") { - const apiUrl = new URL(url); - apiUrl.pathname = "/_api.json"; - const apiResponse = await env.ASSETS.fetch( - new Request(apiUrl.toString(), request), - ); + const apiResponse = await catalogResponse(url, request, env, "api"); + if (!apiResponse.ok) return apiResponse; const providers = (await apiResponse.json()) as Record< string, { models: Record } @@ -102,11 +109,11 @@ export default { } if (url.pathname === "/api.json") { - url.pathname = "/_api.json"; + return catalogResponse(url, request, env, "api"); } else if (url.pathname === "/models.json") { - url.pathname = "/_models.json"; + return catalogResponse(url, request, env, "models"); } else if (url.pathname === "/catalog.json") { - url.pathname = "/_catalog.json"; + return catalogResponse(url, request, env, "catalog"); } else if ( url.pathname === "/" || url.pathname === "/index.html" || @@ -143,6 +150,81 @@ export default { }, }; +type CatalogEndpoint = "api" | "models" | "catalog"; + +async function catalogResponse( + url: URL, + request: Request, + env: Env, + endpoint: CatalogEndpoint, +) { + let filter; + try { + filter = parseModelTypes(url.searchParams.get("type")); + } catch (error) { + if (!(error instanceof InvalidModelTypeError)) throw error; + return Response.json( + { + error: error.message, + allowed: [...MODEL_TYPES, "all"], + }, + { + status: 400, + headers: { "Access-Control-Allow-Origin": "*" }, + }, + ); + } + + const assetUrl = new URL(url); + const suffix = filter === "default" + ? "" + : filter === "all" + ? "-all" + : filter.length === 1 + ? `-${filter[0]}` + : undefined; + assetUrl.pathname = `/_${endpoint}${suffix ?? "-all"}.json`; + assetUrl.search = ""; + const assetResponse = await env.ASSETS.fetch( + new Request(assetUrl.toString(), request), + ); + if (!assetResponse.ok || suffix !== undefined) return assetResponse; + + const value = await assetResponse.json(); + const filtered = endpoint === "api" + ? filterProvidersByModelType( + value as Record, + filter, + ) + : endpoint === "models" + ? filterModelsByModelType( + value as Record, + filter, + ) + : filterCatalogByModelType( + value as { + providers: Record; + models: Record; + }, + filter, + ); + + const headers = new Headers(assetResponse.headers); + headers.delete("Content-Length"); + headers.delete("ETag"); + headers.set("Content-Type", "application/json"); + headers.set("Cache-Control", "public, max-age=3600"); + return new Response(JSON.stringify(filtered), { headers }); +} + +interface CatalogModel { + type?: ModelTypeValue; +} + +interface CatalogProvider { + models: Record; +} + function isHtmlRoute(pathname: string) { return ( pathname === "/models" || diff --git a/packages/function/test/worker.test.ts b/packages/function/test/worker.test.ts new file mode 100644 index 00000000000..26d9e0fb04a --- /dev/null +++ b/packages/function/test/worker.test.ts @@ -0,0 +1,138 @@ +import { describe, expect, test } from "bun:test"; + +import worker, { type Env } from "../src/worker.js"; + +const textModel = { + id: "text-model", + modalities: { input: ["text"], output: ["text"] }, +}; +const decisionModel = { + id: "decision-model", + type: "decision", + modalities: { input: ["text"], output: ["text"] }, +}; +const providers = { + example: { + id: "example", + models: { text: textModel, decision: decisionModel }, + }, +}; +const models = { text: textModel, decision: decisionModel }; + +describe("catalog API model type filtering", () => { + test("omits typed models from api.json by default", async () => { + const response = await request("/api.json"); + const body = await response.json(); + + expect(Object.keys(body.example.models)).toEqual(["text"]); + }); + + test("omits typed models from models.json by default", async () => { + const response = await request("/models.json"); + const body = await response.json(); + + expect(Object.keys(body)).toEqual(["text"]); + }); + + test("omits typed models from catalog.json by default", async () => { + const response = await request("/catalog.json"); + const body = await response.json(); + + expect(Object.keys(body.models)).toEqual(["text"]); + expect(Object.keys(body.providers.example.models)).toEqual(["text"]); + }); + + test("returns explicitly requested decision models", async () => { + const response = await request("/catalog.json?type=decision"); + const body = await response.json(); + + expect(Object.keys(body.models)).toEqual(["decision"]); + expect(Object.keys(body.providers.example.models)).toEqual(["decision"]); + }); + + test("returns the complete static catalog for all", async () => { + const response = await request("/models.json?type=all"); + const body = await response.json(); + + expect(Object.keys(body)).toEqual(["text", "decision"]); + }); + + test("omits typed models from model-schema.json by default", async () => { + const response = await request("/model-schema.json"); + const body = await response.json(); + + expect(body.$defs.Model.enum).toEqual(["example/text"]); + }); + + test("includes typed models in model-schema.json when requested", async () => { + const response = await request("/model-schema.json?type=all"); + const body = await response.json(); + + expect(body.$defs.Model.enum).toEqual([ + "example/decision", + "example/text", + ]); + }); + + test("rejects unknown model types", async () => { + const response = await request("/api.json?type=unknown"); + + expect(response.status).toBe(400); + }); +}); + +async function request(path: string) { + const env = { + ASSETS: { + fetch(input: Request) { + const pathname = new URL(input.url).pathname; + if (pathname === "/_api.json") { + return Response.json({ + example: { ...providers.example, models: { text: textModel } }, + }); + } + if (pathname === "/_api-all.json") return Response.json(providers); + if (pathname === "/_api-decision.json") { + return Response.json({ + example: { ...providers.example, models: { decision: decisionModel } }, + }); + } + if (pathname === "/_models.json") { + return Response.json({ text: textModel }); + } + if (pathname === "/_models-all.json") return Response.json(models); + if (pathname === "/_models-decision.json") { + return Response.json({ decision: decisionModel }); + } + if (pathname === "/_catalog.json") { + return Response.json({ + providers: { + example: { ...providers.example, models: { text: textModel } }, + }, + models: { text: textModel }, + }); + } + if (pathname === "/_catalog-all.json") { + return Response.json({ providers, models }); + } + if (pathname === "/_catalog-decision.json") { + return Response.json({ + providers: { + example: { ...providers.example, models: { decision: decisionModel } }, + }, + models: { decision: decisionModel }, + }); + } + return new Response(null, { status: 404 }); + }, + }, + } as unknown as Env; + + return worker.fetch( + new Request(`https://models.dev${path}`, { + headers: { "user-agent": "test" }, + }), + env, + { waitUntil() {} } as unknown as ExecutionContext, + ); +} diff --git a/packages/sdk/src/client.ts b/packages/sdk/src/client.ts index f63aa5c3660..ee3848e4e3f 100644 --- a/packages/sdk/src/client.ts +++ b/packages/sdk/src/client.ts @@ -1,5 +1,5 @@ import { ModelsDevError } from "./error.js" -import type { Catalog, ModelMetadataMap, ProviderMap } from "./types.js" +import type { Catalog, ModelMetadataMap, ModelType, ProviderMap } from "./types.js" /** Accepted anywhere headers can be passed. Same shapes as the standard `HeadersInit`. */ export type HeadersInput = Headers | Record | Array<[string, string]> @@ -21,6 +21,8 @@ export interface RequestOptions { readonly signal?: AbortSignal /** Extra headers for this request. Overrides client-level headers. */ readonly headers?: HeadersInput + /** Specialized model types to include. Omit for standard models; use `"all"` for the complete catalog. */ + readonly modelTypes?: "all" | readonly ModelType[] } /** @@ -41,7 +43,15 @@ export function make(options: ClientOptions = {}) { let response: Response try { - response = await fetch(new URL(path, base), { + const url = new URL(path, base) + const modelTypes = requestOptions?.modelTypes + if (modelTypes === "all") { + url.searchParams.set("type", "all") + } else if (modelTypes && modelTypes.length > 0) { + url.searchParams.set("type", modelTypes.join(",")) + } + + response = await fetch(url, { method: "GET", headers, signal: requestOptions?.signal, diff --git a/packages/sdk/src/effect.ts b/packages/sdk/src/effect.ts index cd1780d182e..24827173e48 100644 --- a/packages/sdk/src/effect.ts +++ b/packages/sdk/src/effect.ts @@ -1,4 +1,4 @@ // Effect-native client. Requires the optional peer dependency `effect`. export * as Models from "./effect/client.js" -export { ModelsDevError, type ClientOptions, type ModelsClient } from "./effect/client.js" +export { ModelsDevError, type ClientOptions, type ModelsClient, type RequestOptions } from "./effect/client.js" export type * from "./types.js" diff --git a/packages/sdk/src/effect/client.ts b/packages/sdk/src/effect/client.ts index 0eb98835d55..ea005cf8f82 100644 --- a/packages/sdk/src/effect/client.ts +++ b/packages/sdk/src/effect/client.ts @@ -1,6 +1,6 @@ import { Context, Effect, Layer, Schema } from "effect" import { HttpClient, HttpClientResponse } from "effect/unstable/http" -import type { Catalog, ModelMetadataMap, ProviderMap } from "../types.js" +import type { Catalog, ModelMetadataMap, ModelType, ProviderMap } from "../types.js" /** The only error in the failure channel of client methods. Wraps the underlying `HttpClientError` as `cause`. */ export class ModelsDevError extends Schema.TaggedErrorClass()("ModelsDevError", { @@ -14,6 +14,11 @@ export interface ClientOptions { readonly headers?: Record } +export interface RequestOptions { + /** Specialized model types to include. Omit for standard models; use `"all"` for the complete catalog. */ + readonly modelTypes?: "all" | readonly ModelType[] +} + /** * Creates a stateless models.dev client on top of the `HttpClient` service * from the environment (`FetchHttpClient.layer`, `NodeHttpClient.layer`, or a @@ -26,9 +31,17 @@ export const make = (options?: ClientOptions) => const baseUrl = options?.baseUrl ?? "https://models.dev" const base = baseUrl.endsWith("/") ? baseUrl : baseUrl + "/" - const get = (path: string): Effect.Effect => - http - .get(new URL(path, base), { + const get = (path: string, requestOptions?: RequestOptions): Effect.Effect => { + const url = new URL(path, base) + const modelTypes = requestOptions?.modelTypes + if (modelTypes === "all") { + url.searchParams.set("type", "all") + } else if (modelTypes && modelTypes.length > 0) { + url.searchParams.set("type", modelTypes.join(",")) + } + + return http + .get(url, { headers: options?.headers, }) .pipe( @@ -37,14 +50,15 @@ export const make = (options?: ClientOptions) => Effect.map((data) => data as A), Effect.mapError((cause) => new ModelsDevError({ cause })), ) + } return { /** All providers with their models, pricing, and limits (`/api.json`). */ - providers: () => get("api.json"), + providers: (options?: RequestOptions) => get("api.json", options), /** Provider-agnostic model metadata (`/models.json`). */ - models: () => get("models.json"), + models: (options?: RequestOptions) => get("models.json", options), /** Providers and model metadata in a single request (`/catalog.json`). */ - catalog: () => get("catalog.json"), + catalog: (options?: RequestOptions) => get("catalog.json", options), } }) diff --git a/packages/sdk/src/types.ts b/packages/sdk/src/types.ts index 92999e2dea7..250551d5e7a 100644 --- a/packages/sdk/src/types.ts +++ b/packages/sdk/src/types.ts @@ -80,6 +80,9 @@ export interface ModelCost extends Cost { /** Input/output data types a model supports. */ export type Modality = "text" | "audio" | "image" | "video" | "pdf" +/** A model's specialized behavioral contract. Omitted for standard generative models. */ +export type ModelType = "decision" + export interface Modalities { input: Modality[] output: Modality[] @@ -143,6 +146,7 @@ export interface BenchmarkResult { export interface ModelMetadata { /** Canonical model ID, e.g. "anthropic/claude-opus-4-6". */ id: string + type?: ModelType name: string description: string family?: ModelFamily @@ -208,6 +212,7 @@ export interface ModelProviderConfig { export interface Model { /** Provider-scoped model ID, e.g. "claude-opus-4-6". */ id: string + type?: ModelType name: string description: string family?: ModelFamily diff --git a/packages/sdk/test/client.test.ts b/packages/sdk/test/client.test.ts index 3f898ef9157..61bc8fda104 100644 --- a/packages/sdk/test/client.test.ts +++ b/packages/sdk/test/client.test.ts @@ -40,6 +40,17 @@ test("models() and catalog() hit their endpoints", async () => { expect(calls.map((call) => call.url.href)).toEqual(["https://models.dev/models.json", "https://models.dev/catalog.json"]) }) +test("catalog endpoints encode model type filters", async () => { + const { calls, fetch } = stub({}) + const client = Models.make({ fetch }) + await client.providers({ modelTypes: ["decision"] }) + await client.models({ modelTypes: "all" }) + expect(calls.map((call) => call.url.href)).toEqual([ + "https://models.dev/api.json?type=decision", + "https://models.dev/models.json?type=all", + ]) +}) + test("baseUrl with subpath is preserved, with or without trailing slash", async () => { const { calls, fetch } = stub({}) await Models.make({ fetch, baseUrl: "https://example.com/mirror" }).providers() diff --git a/packages/sdk/test/effect.test.ts b/packages/sdk/test/effect.test.ts index 943acedbeb7..025e4a190e3 100644 --- a/packages/sdk/test/effect.test.ts +++ b/packages/sdk/test/effect.test.ts @@ -42,6 +42,20 @@ test("models() and catalog() hit their endpoints, baseUrl subpath preserved", as ]) }) +test("catalog endpoints encode model type filters", async () => { + const { requests, layer } = stub({}) + const program = Effect.gen(function* () { + const client = yield* Models.make() + yield* client.providers({ modelTypes: ["decision"] }) + yield* client.catalog({ modelTypes: "all" }) + }) + await program.pipe(Effect.provide(layer), Effect.runPromise) + expect(requests.map((request) => request.url)).toEqual([ + "https://models.dev/api.json?type=decision", + "https://models.dev/catalog.json?type=all", + ]) +}) + test("custom headers are sent", async () => { const { requests, layer } = stub({}) const program = Effect.gen(function* () { diff --git a/packages/web/script/build.ts b/packages/web/script/build.ts index f387bccd848..8866aa8e435 100755 --- a/packages/web/script/build.ts +++ b/packages/web/script/build.ts @@ -1,6 +1,11 @@ #!/usr/bin/env bun import { RenderedPages, Providers, Models, renderDocument } from "../src/render"; +import { + filterCatalogByModelType, + MODEL_TYPES, + type ModelTypeFilter, +} from "@models.dev/core"; import fs from "fs/promises"; import path from "path"; @@ -74,15 +79,18 @@ for (const [route, rendered] of RenderedPages) { await Bun.write(filePath, renderDocument(template, rendered)); } -await Bun.write("./dist/api.json", JSON.stringify(Providers)); -await Bun.write( - "./dist/catalog.json", - JSON.stringify({ models: Models, providers: Providers }), -); -await Bun.write("./dist/models.json", JSON.stringify(Models)); - -await fs.rename("./dist/api.json", "./dist/_api.json"); -await fs.rename("./dist/catalog.json", "./dist/_catalog.json"); -await fs.rename("./dist/models.json", "./dist/_models.json"); +const catalog = { models: Models, providers: Providers }; +const variants: Array<[suffix: string, filter: ModelTypeFilter]> = [ + ["", "default"], + ["-all", "all"], + ...MODEL_TYPES.map((type) => [`-${type}`, [type]] as const), +]; + +for (const [suffix, filter] of variants) { + const filtered = filterCatalogByModelType(catalog, filter); + await Bun.write(`./dist/_api${suffix}.json`, JSON.stringify(filtered.providers)); + await Bun.write(`./dist/_models${suffix}.json`, JSON.stringify(filtered.models)); + await Bun.write(`./dist/_catalog${suffix}.json`, JSON.stringify(filtered)); +} await fs.rm("./dist/index.html", { force: true }); diff --git a/packages/web/src/render.tsx b/packages/web/src/render.tsx index 93c73c6e66e..34632f57b1b 100644 --- a/packages/web/src/render.tsx +++ b/packages/web/src/render.tsx @@ -1497,21 +1497,22 @@ function HelpDialog() {

API

You can access provider data, provider-agnostic model metadata, or the - combined catalog through JSON endpoints. + combined catalog through JSON endpoints. Specialized model types are + omitted by default; the site includes all model types.

Logos

diff --git a/packages/web/src/server.ts b/packages/web/src/server.ts index 5adb9e9fc60..f43e6ce830a 100644 --- a/packages/web/src/server.ts +++ b/packages/web/src/server.ts @@ -1,5 +1,12 @@ import Index from "../index.html"; import { getRenderedPage, Models, Providers, renderDocument } from "./render"; +import { + filterCatalogByModelType, + filterModelsByModelType, + filterProvidersByModelType, + InvalidModelTypeError, + parseModelTypes, +} from "@models.dev/core"; import path from "path"; const assetPort = Number(Bun.env.ASSET_PORT ?? 16000); @@ -99,30 +106,37 @@ Bun.serve({ }, }); }, - "/api.json": () => - Response.json(Providers, { - headers: { - "Cache-Control": "public, max-age=3600", - }, - }), - "/models.json": () => - Response.json(Models, { - headers: { - "Cache-Control": "public, max-age=3600", - }, - }), - "/catalog.json": () => - Response.json( - { models: Models, providers: Providers }, - { - headers: { - "Cache-Control": "public, max-age=3600", - }, - }, - ), + "/api.json": (req) => catalogResponse(req, "api"), + "/models.json": (req) => catalogResponse(req, "models"), + "/catalog.json": (req) => catalogResponse(req, "catalog"), }, }); +function catalogResponse(req: Request, endpoint: "api" | "models" | "catalog") { + let filter; + try { + filter = parseModelTypes(new URL(req.url).searchParams.get("type")); + } catch (error) { + if (!(error instanceof InvalidModelTypeError)) throw error; + return Response.json({ error: error.message }, { status: 400 }); + } + + const value = endpoint === "api" + ? filterProvidersByModelType(Providers, filter) + : endpoint === "models" + ? filterModelsByModelType(Models, filter) + : filterCatalogByModelType( + { models: Models, providers: Providers }, + filter, + ); + + return Response.json(value, { + headers: { + "Cache-Control": "public, max-age=3600", + }, + }); +} + const server = Bun.serve({ development: true, hostname: "0.0.0.0", From 79ed8a751dc10c55f982ab1d233c59a48abc755c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 16:25:49 +0000 Subject: [PATCH 132/392] chore(sync): update Kilo model catalog (#7580) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/tencent/hy3.toml | 6 +++--- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index 2bdfa3cbd7b..e96e3110ab1 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.132 -output = 0.528 -cache_read = 0.033 +input = 0.0825 +output = 0.33 +cache_read = 0.020625 [limit] context = 262_144 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index e4d3acd84bf..4a153e157f4 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.5478 -output = 1.6434 -cache_read = 0.01826 +input = 0.54252 +output = 1.62756 +cache_read = 0.018084 [limit] context = 1_024_000 From 8540388b7fff1a0ae57f3de053ff7befb1f2125e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 16:26:06 +0000 Subject: [PATCH 133/392] chore(sync): update OpenRouter model catalog (#7581) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml | 4 ++-- providers/openrouter/models/tencent/hy3.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 4 files changed, 11 insertions(+), 11 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7191760d710..9b249382289 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5478 -output = 1.6434 -cache_read = 0.01826 +input = 0.54252 +output = 1.62756 +cache_read = 0.018084 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml index 91402212177..d857fa53dab 100644 --- a/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml +++ b/providers/openrouter/models/qwen/qwen3-vl-30b-a3b-instruct.toml @@ -12,8 +12,8 @@ knowledge = "2025-03-31" open_weights = true [cost] -input = 0.2 -output = 0.7 +input = 0.13 +output = 0.52 [limit] context = 262_144 diff --git a/providers/openrouter/models/tencent/hy3.toml b/providers/openrouter/models/tencent/hy3.toml index 80ecfdd3aa8..f61fa3175e9 100644 --- a/providers/openrouter/models/tencent/hy3.toml +++ b/providers/openrouter/models/tencent/hy3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.132 -output = 0.528 -cache_read = 0.033 +input = 0.0825 +output = 0.33 +cache_read = 0.020625 [limit] context = 262_144 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 78806039bab..70af1d85157 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5478 -output = 1.6434 -cache_read = 0.01826 +input = 0.54252 +output = 1.62756 +cache_read = 0.018084 [limit] context = 1_048_576 From bd3424ccb06184edcfc14356cbd460200b2dc1f6 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 17:22:05 +0000 Subject: [PATCH 134/392] chore(sync): update OpenRouter model catalog (#7583) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 ++++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- providers/openrouter/models/~z-ai/glm-latest.toml | 6 +++--- 3 files changed, 11 insertions(+), 10 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 9b249382289..fe38fdd1760 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.54252 -output = 1.62756 -cache_read = 0.018084 +input = 0.53856 +output = 1.61568 +cache_read = 0.017136 [limit] context = 1_048_576 +output = 393_216 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 70af1d85157..809399be81d 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.54252 -output = 1.62756 -cache_read = 0.018084 +input = 0.53856 +output = 1.61568 +cache_read = 0.017136 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index f42b1197f93..c7cc1981beb 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8442 -output = 2.6532 -cache_read = 0.15678 +input = 0.8316 +output = 2.6136 +cache_read = 0.15444 [limit] context = 1_310_720 From 2d25c71b1ab3000b8492798e0108cfbe8e2a4c10 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 17:22:20 +0000 Subject: [PATCH 135/392] chore(sync): update Kilo model catalog (#7582) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 3 ++- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- providers/kilo/models/~z-ai/glm-latest.toml | 6 +++--- 3 files changed, 10 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index 01800e18158..d4d63d57a55 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -11,4 +11,5 @@ output = 3.96 cache_read = 0.044 [limit] -context = 1_024_000 +context = 1_048_576 +output = 393_216 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 4a153e157f4..f1d4ad4c3b8 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.54252 -output = 1.62756 -cache_read = 0.018084 +input = 0.53856 +output = 1.61568 +cache_read = 0.017136 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index c5af7ff24a4..4331c62d136 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8442 -output = 2.6532 -cache_read = 0.15678 +input = 0.8316 +output = 2.6136 +cache_read = 0.15444 [limit] context = 1_048_576 From ba313562117db54326bcc8b3933024f92aa61fc9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 18:28:15 +0000 Subject: [PATCH 136/392] chore(sync): update Kilo model catalog (#7585) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/nvidia/nemotron-3.5-lightning.toml | 2 +- providers/kilo/models/qwen/qwen3.6-35b-a3b.toml | 4 ++-- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml b/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml index 6759948a30e..5244dc1ac02 100644 --- a/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml +++ b/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml @@ -6,7 +6,7 @@ type = "effort" values = ["none", "high"] [cost] -input = 0.04 +input = 0.065 output = 0.18 [limit] diff --git a/providers/kilo/models/qwen/qwen3.6-35b-a3b.toml b/providers/kilo/models/qwen/qwen3.6-35b-a3b.toml index 1385363812a..21f9408b4a8 100644 --- a/providers/kilo/models/qwen/qwen3.6-35b-a3b.toml +++ b/providers/kilo/models/qwen/qwen3.6-35b-a3b.toml @@ -6,8 +6,8 @@ type = "effort" values = ["none", "high"] [cost] -input = 0.1 -output = 0.9 +input = 0.15 +output = 1 cache_read = 0.05 [limit] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index f1d4ad4c3b8..c995568e34b 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.53856 -output = 1.61568 -cache_read = 0.017136 +input = 0.53328 +output = 1.59984 +cache_read = 0.016968 [limit] context = 1_048_576 From e755d2ad97947686bb6b8221db177a53132fe435 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 18:28:19 +0000 Subject: [PATCH 137/392] chore(sync): update OpenRouter model catalog (#7584) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml | 4 ++-- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 3 files changed, 8 insertions(+), 8 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index fe38fdd1760..e3772508842 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.53856 -output = 1.61568 -cache_read = 0.017136 +input = 0.53328 +output = 1.59984 +cache_read = 0.016968 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml b/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml index ac95b282785..a223e74df58 100644 --- a/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml +++ b/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml @@ -6,8 +6,8 @@ base_model = "alibaba/qwen3.6-35b-a3b" type = "toggle" [cost] -input = 0.1 -output = 0.9 +input = 0.15 +output = 1 cache_read = 0.05 [limit] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 809399be81d..45e0c0ee8cc 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.53856 -output = 1.61568 -cache_read = 0.017136 +input = 0.53328 +output = 1.59984 +cache_read = 0.016968 [limit] context = 1_048_576 From cf992a84f6132bdfc2ad6b8f408c3e1b36536857 Mon Sep 17 00:00:00 2001 From: Frank Date: Sun, 20 Sep 2026 14:59:35 -0400 Subject: [PATCH 138/392] Delete jev-latest.toml --- providers/opencode/models/jev-latest.toml | 6 ------ 1 file changed, 6 deletions(-) delete mode 100644 providers/opencode/models/jev-latest.toml diff --git a/providers/opencode/models/jev-latest.toml b/providers/opencode/models/jev-latest.toml deleted file mode 100644 index 409cf1a8931..00000000000 --- a/providers/opencode/models/jev-latest.toml +++ /dev/null @@ -1,6 +0,0 @@ -# Pricing source: https://typesafe.ai/blog/introducing-system-one-models-and-jev -base_model = "typesafe/jev-latest" - -[cost] -input = 0.042 -output = 0 From 546157ba4e6e59956ed56ef16545e86709a76055 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 19:20:59 +0000 Subject: [PATCH 139/392] chore(sync): update OpenRouter model catalog (#7588) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index e3772508842..0db1bf12100 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.53328 -output = 1.59984 -cache_read = 0.016968 +input = 0.528 +output = 1.584 +cache_read = 0.0168 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 45e0c0ee8cc..813379229ea 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.53328 -output = 1.59984 -cache_read = 0.016968 +input = 0.528 +output = 1.584 +cache_read = 0.0168 [limit] context = 1_048_576 From f9bd1ec77589641514cdf17a4f358f3f15d842f6 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 19:21:07 +0000 Subject: [PATCH 140/392] chore(sync): update Kilo model catalog (#7589) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index c995568e34b..7c557b67a56 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.53328 -output = 1.59984 -cache_read = 0.016968 +input = 0.528 +output = 1.584 +cache_read = 0.0168 [limit] context = 1_048_576 From 4a583e6170b8dabd19cac0ab602a91920ef47b5b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 20:24:51 +0000 Subject: [PATCH 141/392] chore(sync): update OpenRouter model catalog (#7590) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-0731.toml | 2 +- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 8 ++++---- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 +++---- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- .../models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- 5 files changed, 13 insertions(+), 14 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml index 5e9065ff206..a96d85eba1b 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml @@ -11,7 +11,7 @@ values = ["low", "high", "max"] [cost] input = 0.04 -output = 0.08 +output = 0.12 cache_read = 0.016 [limit] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-vision-exp.toml index 0417e3027d5..2186c60b6fb 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -11,10 +11,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.2156 -output = 0.6468 -cache_read = 0.00686 +input = 0.22 +output = 0.66 +cache_read = 0.007 [limit] context = 1_048_576 -output = 262_144 +output = 943_718 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 0db1bf12100..6d3a571d213 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,10 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.528 -output = 1.584 -cache_read = 0.0168 +input = 0.52668 +output = 1.58004 +cache_read = 0.017556 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 813379229ea..6f16213302e 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.528 -output = 1.584 -cache_read = 0.0168 +input = 0.52668 +output = 1.58004 +cache_read = 0.017556 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index cd84b32afe3..f83290ea281 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -21,7 +21,7 @@ values = ["low", "high", "max"] [cost] input = 0.04 -output = 0.08 +output = 0.12 cache_read = 0.016 [limit] From 3118014fb461b48f37d26b0bd7ab5a908408d98d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 20:24:58 +0000 Subject: [PATCH 142/392] chore(sync): update Kilo model catalog (#7591) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 2 +- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 3 +-- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- .../models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- 4 files changed, 8 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/kilo/models/deepseek/deepseek-v4-flash-vision-exp.toml index 41e3e4d74b4..a280435f49f 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -12,4 +12,4 @@ cache_read = 0.028 [limit] context = 1_048_576 -output = 262_144 +output = 943_718 diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index d4d63d57a55..01800e18158 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -11,5 +11,4 @@ output = 3.96 cache_read = 0.044 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 7c557b67a56..aac53c4f859 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.528 -output = 1.584 -cache_read = 0.0168 +input = 0.52668 +output = 1.58004 +cache_read = 0.017556 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 2c01541878c..0662392bcff 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -16,7 +16,7 @@ values = ["none", "low", "high", "max"] [cost] input = 0.04 -output = 0.08 +output = 0.12 cache_read = 0.016 [limit] From 2b269a637f367f958b0788ad6cca8f75b4fe68da Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 21:23:14 +0000 Subject: [PATCH 143/392] chore(sync): update OpenRouter model catalog (#7593) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash-0731.toml | 2 +- .../openrouter/models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml index a96d85eba1b..be2f907d844 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml @@ -11,7 +11,7 @@ values = ["low", "high", "max"] [cost] input = 0.04 -output = 0.12 +output = 0.16 cache_read = 0.016 [limit] diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index f83290ea281..c98a00af617 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -21,7 +21,7 @@ values = ["low", "high", "max"] [cost] input = 0.04 -output = 0.12 +output = 0.16 cache_read = 0.016 [limit] From e2b20e7430e1d42fbda3dab37e72fa1d4a5182e6 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 21:23:21 +0000 Subject: [PATCH 144/392] chore(sync): update NanoGPT model catalog (#7594) Co-authored-by: opencode-agent[bot] --- .../models/z-ai/glm-4.7-flash-original.toml | 25 ------------------- .../z-ai/glm-4.7-flash-original:thinking.toml | 25 ------------------- 2 files changed, 50 deletions(-) delete mode 100644 providers/nano-gpt/models/z-ai/glm-4.7-flash-original.toml delete mode 100644 providers/nano-gpt/models/z-ai/glm-4.7-flash-original:thinking.toml diff --git a/providers/nano-gpt/models/z-ai/glm-4.7-flash-original.toml b/providers/nano-gpt/models/z-ai/glm-4.7-flash-original.toml deleted file mode 100644 index 3491643442e..00000000000 --- a/providers/nano-gpt/models/z-ai/glm-4.7-flash-original.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "GLM 4.7 Flash Original" -description = "GLM-4.7-Flash is a lightweight 30B model optimized for coding and agentic tasks. Balances high performance with efficiency, perfect for local deployment." -family = "glm-flash" -release_date = "2026-01-19" -last_updated = "2026-01-19" -attachment = false -reasoning = true -tool_call = true -structured_output = true -open_weights = true -reasoning_options = [] - -[cost] -input = 0.07 -output = 0.4 -cache_read = 0.035 - -[limit] -context = 200_000 -input = 200_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/nano-gpt/models/z-ai/glm-4.7-flash-original:thinking.toml b/providers/nano-gpt/models/z-ai/glm-4.7-flash-original:thinking.toml deleted file mode 100644 index 65480647fea..00000000000 --- a/providers/nano-gpt/models/z-ai/glm-4.7-flash-original:thinking.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "GLM 4.7 Flash Original Thinking" -description = "GLM-4.7-Flash with extended thinking capabilities for complex reasoning. Lightweight 30B model optimized for coding and agentic tasks." -family = "glm-flash" -release_date = "2026-01-19" -last_updated = "2026-01-19" -attachment = false -reasoning = true -tool_call = true -structured_output = false -open_weights = true -reasoning_options = [] - -[cost] -input = 0.07 -output = 0.4 -cache_read = 0.035 - -[limit] -context = 200_000 -input = 200_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] From 3c1aed35ba863dde45f0233a6d78a48cc8fb9913 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 21:23:25 +0000 Subject: [PATCH 145/392] chore(sync): update Kilo model catalog (#7595) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 0662392bcff..7ead984259a 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -16,7 +16,7 @@ values = ["none", "low", "high", "max"] [cost] input = 0.04 -output = 0.12 +output = 0.16 cache_read = 0.016 [limit] From 34293334cff48dca9f9b914243b8a51aa9ba0efa Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 22:24:00 +0000 Subject: [PATCH 146/392] chore(sync): update Kilo model catalog (#7597) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- .../kilo/models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- providers/kilo/models/~z-ai/glm-latest.toml | 6 +++--- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index aac53c4f859..02af10ccee9 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.52668 -output = 1.58004 -cache_read = 0.017556 +input = 0.52404 +output = 1.57212 +cache_read = 0.017468 [limit] context = 1_024_000 diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 7ead984259a..b586ef3f991 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -16,12 +16,12 @@ values = ["none", "low", "high", "max"] [cost] input = 0.04 -output = 0.16 -cache_read = 0.016 +output = 0.1 +cache_read = 0.01 [limit] context = 1_048_576 -output = 943_718 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 4331c62d136..2237628e2fc 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8316 -output = 2.6136 -cache_read = 0.15444 +input = 0.7728 +output = 2.4288 +cache_read = 0.14352 [limit] context = 1_048_576 From 28edbfdc1a374a69095558a98cd03ba9fe1a4c22 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 22:24:07 +0000 Subject: [PATCH 147/392] chore(sync): update OpenRouter model catalog (#7598) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- providers/openrouter/models/~z-ai/glm-latest.toml | 6 +++--- 4 files changed, 12 insertions(+), 12 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 6d3a571d213..46ca79925cf 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.52668 -output = 1.58004 -cache_read = 0.017556 +input = 0.52404 +output = 1.57212 +cache_read = 0.017468 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 6f16213302e..b3b53f82fcd 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.52668 -output = 1.58004 -cache_read = 0.017556 +input = 0.52404 +output = 1.57212 +cache_read = 0.017468 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index c98a00af617..12c99d28b38 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -21,12 +21,12 @@ values = ["low", "high", "max"] [cost] input = 0.04 -output = 0.16 -cache_read = 0.016 +output = 0.1 +cache_read = 0.01 [limit] context = 1_310_720 -output = 943_718 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index c7cc1981beb..709b8881ffa 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8316 -output = 2.6136 -cache_read = 0.15444 +input = 0.7728 +output = 2.4288 +cache_read = 0.14352 [limit] context = 1_310_720 From 27efbe920eea10c2c411f2620f58f2170f3b8a5f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 23:23:01 +0000 Subject: [PATCH 148/392] chore(sync): update OpenRouter model catalog (#7600) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 ++++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- 3 files changed, 11 insertions(+), 10 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 46ca79925cf..f4c4b33d3cc 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.52404 -output = 1.57212 -cache_read = 0.017468 +input = 0.52272 +output = 1.56816 +cache_read = 0.016632 [limit] context = 1_048_576 +output = 393_216 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index b3b53f82fcd..23ba1c5d915 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.52404 -output = 1.57212 -cache_read = 0.017468 +input = 0.52272 +output = 1.56816 +cache_read = 0.016632 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 12c99d28b38..c98a00af617 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -21,12 +21,12 @@ values = ["low", "high", "max"] [cost] input = 0.04 -output = 0.1 -cache_read = 0.01 +output = 0.16 +cache_read = 0.016 [limit] context = 1_310_720 -output = 393_216 +output = 943_718 [modalities] input = ["text"] From 9c20af0e283a3dea2136291dbc7b059acbd820d4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Sun, 20 Sep 2026 23:23:05 +0000 Subject: [PATCH 149/392] chore(sync): update Kilo model catalog (#7601) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 3 ++- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- 3 files changed, 10 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index 01800e18158..d4d63d57a55 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -11,4 +11,5 @@ output = 3.96 cache_read = 0.044 [limit] -context = 1_024_000 +context = 1_048_576 +output = 393_216 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 02af10ccee9..6bd317a45ab 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.52404 -output = 1.57212 -cache_read = 0.017468 +input = 0.52272 +output = 1.56816 +cache_read = 0.016632 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index b586ef3f991..7ead984259a 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -16,12 +16,12 @@ values = ["none", "low", "high", "max"] [cost] input = 0.04 -output = 0.1 -cache_read = 0.01 +output = 0.16 +cache_read = 0.016 [limit] context = 1_048_576 -output = 393_216 +output = 943_718 [modalities] input = ["text"] From 91c2e781204dc9fb57c5c50a086b7c941748af2c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:04:42 +0000 Subject: [PATCH 150/392] chore(sync): update Kilo model catalog (#7603) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml | 1 - .../kilo/models/meta-llama/llama-3.1-70b-instruct.toml | 2 +- providers/kilo/models/meta-llama/llama-4-maverick.toml | 2 +- providers/kilo/models/tencent/hy3.toml | 6 +++--- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- providers/kilo/models/~z-ai/glm-latest.toml | 8 ++++---- 6 files changed, 13 insertions(+), 14 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index d4d63d57a55..007b654034e 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -12,4 +12,3 @@ cache_read = 0.044 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/kilo/models/meta-llama/llama-3.1-70b-instruct.toml b/providers/kilo/models/meta-llama/llama-3.1-70b-instruct.toml index 881a90e0d15..6fc67901197 100644 --- a/providers/kilo/models/meta-llama/llama-3.1-70b-instruct.toml +++ b/providers/kilo/models/meta-llama/llama-3.1-70b-instruct.toml @@ -8,4 +8,4 @@ output = 0.4 [limit] context = 131_072 -output = 16_384 +output = 8_192 diff --git a/providers/kilo/models/meta-llama/llama-4-maverick.toml b/providers/kilo/models/meta-llama/llama-4-maverick.toml index fab1dd0fc03..537ab8cd5fd 100644 --- a/providers/kilo/models/meta-llama/llama-4-maverick.toml +++ b/providers/kilo/models/meta-llama/llama-4-maverick.toml @@ -15,7 +15,7 @@ input = 0.1875 output = 0.6525 [limit] -context = 128_000 +context = 1_048_576 output = 16_384 [modalities] diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index e96e3110ab1..2bdfa3cbd7b 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.0825 -output = 0.33 -cache_read = 0.020625 +input = 0.132 +output = 0.528 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 6bd317a45ab..b14064bf897 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.52272 -output = 1.56816 -cache_read = 0.016632 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 2237628e2fc..b6ea0fea56d 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7728 -output = 2.4288 -cache_read = 0.14352 +input = 0.7735 +output = 2.431 +cache_read = 0.127075 [limit] context = 1_048_576 -output = 131_072 +output = 943_718 [modalities] input = ["text"] From c44d7ad670bb33f55d4e8955049112ae30c5b69f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:04:47 +0000 Subject: [PATCH 151/392] chore(sync): update OpenRouter model catalog (#7604) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 +++---- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/ibm-granite/granite-4.2-8b.toml | 6 +++--- .../models/meta-llama/llama-3.1-70b-instruct.toml | 6 +++--- .../openrouter/models/meta-llama/llama-4-maverick.toml | 4 ++-- providers/openrouter/models/tencent/hy3.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- providers/openrouter/models/~z-ai/glm-latest.toml | 8 ++++---- 9 files changed, 28 insertions(+), 29 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 98ee0445042..57d98266566 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.03556 -output = 0.07112 -cache_read = 0.007112 +input = 0.089866 +output = 0.179732 +cache_read = 0.017973 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index f4c4b33d3cc..7c5828b16e2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,10 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.52272 -output = 1.56816 -cache_read = 0.016632 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 5bc792c3ae0..9be6b267131 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.422298 -output = 0.844596 -cache_read = 0.035192 +input = 0.9483 +output = 1.8966 +cache_read = 0.079025 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/ibm-granite/granite-4.2-8b.toml b/providers/openrouter/models/ibm-granite/granite-4.2-8b.toml index 62943cfaadb..1b683dd536e 100644 --- a/providers/openrouter/models/ibm-granite/granite-4.2-8b.toml +++ b/providers/openrouter/models/ibm-granite/granite-4.2-8b.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.06 -output = 0.25 -cache_read = 0.015 +input = 0.1 +output = 0.15 +cache_read = 0.05 [limit] context = 131_072 diff --git a/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml b/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml index 881a90e0d15..1352c1ee599 100644 --- a/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml +++ b/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml @@ -3,9 +3,9 @@ description = "Open Llama instruction model for multilingual chat, reasoning, an structured_output = true [cost] -input = 0.4 -output = 0.4 +input = 0.72 +output = 0.72 [limit] context = 131_072 -output = 16_384 +output = 8_192 diff --git a/providers/openrouter/models/meta-llama/llama-4-maverick.toml b/providers/openrouter/models/meta-llama/llama-4-maverick.toml index b9b42749c83..de6e5240ff9 100644 --- a/providers/openrouter/models/meta-llama/llama-4-maverick.toml +++ b/providers/openrouter/models/meta-llama/llama-4-maverick.toml @@ -12,8 +12,8 @@ knowledge = "2024-08-31" open_weights = true [cost] -input = 0.1875 -output = 0.6525 +input = 0.2 +output = 0.8 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/tencent/hy3.toml b/providers/openrouter/models/tencent/hy3.toml index f61fa3175e9..80ecfdd3aa8 100644 --- a/providers/openrouter/models/tencent/hy3.toml +++ b/providers/openrouter/models/tencent/hy3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.0825 -output = 0.33 -cache_read = 0.020625 +input = 0.132 +output = 0.528 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 23ba1c5d915..ea09a0d12b6 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.52272 -output = 1.56816 -cache_read = 0.016632 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 709b8881ffa..25e0814a971 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7728 -output = 2.4288 -cache_read = 0.14352 +input = 0.7735 +output = 2.431 +cache_read = 0.127075 [limit] context = 1_310_720 -output = 131_072 +output = 943_718 [modalities] input = ["text"] From 34e14cb7160470457fec09c934a2497ee4f4e337 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:04:54 +0000 Subject: [PATCH 152/392] chore(sync): update NanoGPT model catalog (#7602) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index b4cce19afb3..a62995b0beb 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.22 -output = 0.66 -cache_read = 0.007 +input = 0.44 +output = 1.32 +cache_read = 0.014 [limit] context = 1_048_576 From fdee7705c5d1254d4f169c3a27585d07264c163f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:41:41 +0000 Subject: [PATCH 153/392] chore(sync): update OpenRouter model catalog (#7606) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- providers/openrouter/models/qwen/qwen3.8-27b.toml | 4 ++-- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 4 files changed, 12 insertions(+), 12 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..b8d53b0809c 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 1.32 +output = 3.96 +cache_read = 0.044 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 16f058ed841..854b27b6a72 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.15 -output = 0.6 -cache_read = 0.003 +input = 0.3 +output = 1.2 +cache_read = 0.006 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3.8-27b.toml b/providers/openrouter/models/qwen/qwen3.8-27b.toml index 87b348fc2c9..64a9a2f6128 100644 --- a/providers/openrouter/models/qwen/qwen3.8-27b.toml +++ b/providers/openrouter/models/qwen/qwen3.8-27b.toml @@ -12,8 +12,8 @@ values = ["low", "medium", "xhigh"] [cost] input = 0.2 -output = 2.55 -cache_read = 0.085 +output = 2.5 +cache_read = 0.05 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index ea09a0d12b6..dfb8b85c369 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.7 +output = 2.88 +cache_read = 0.088 [limit] context = 1_048_576 -output = 384_000 +output = 943_718 [modalities] input = ["text"] From 76488ffa1fb42e92850d7bba321815492fefb462 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:41:49 +0000 Subject: [PATCH 154/392] chore(sync): update Kilo model catalog (#7605) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index b14064bf897..46e3c60f672 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.7 +output = 2.88 +cache_read = 0.088 [limit] context = 1_048_576 -output = 384_000 +output = 943_718 [modalities] input = ["text"] From 62c3e418a832af3bfe6d721be8b8014feef550aa Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 02:38:35 +0000 Subject: [PATCH 155/392] chore(sync): update Kilo model catalog (#7608) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/moonshotai/kimi-k3.toml | 6 +++--- providers/kilo/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index fd81e087ca6..31838f83a2c 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.7 -output = 8.5 -cache_read = 0.17 +input = 1.675 +output = 9.38 +cache_read = 0.1943 [limit] output = 943_718 diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 9e0f74cdd43..2193a017b17 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.13 -output = 0.52 -cache_read = 0.0026 +input = 0.12 +output = 0.48 +cache_read = 0.0036 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 631c1758bcd..71dba53f553 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.7 -output = 8.5 -cache_read = 0.17 +input = 1.675 +output = 9.38 +cache_read = 0.1943 [limit] context = 1_048_576 From 52eff77bbe3c9b75a83adf769120305150c359c3 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 02:38:39 +0000 Subject: [PATCH 156/392] chore(sync): update OpenRouter model catalog (#7609) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/moonshotai/kimi-k3.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- 5 files changed, 15 insertions(+), 15 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 57d98266566..6f817c5a0a0 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.089866 -output = 0.179732 -cache_read = 0.017973 +input = 0.088606 +output = 0.177212 +cache_read = 0.017721 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 9be6b267131..e7683af4617 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.9483 -output = 1.8966 -cache_read = 0.079025 +input = 0.95526 +output = 1.91052 +cache_read = 0.079605 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index 22b1407542a..3b2924c9c2d 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.7 -output = 8.5 -cache_read = 0.17 +input = 1.675 +output = 9.38 +cache_read = 0.1943 [limit] output = 943_718 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index 643ffd542be..56426182daf 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.13 -output = 0.52 -cache_read = 0.0026 +input = 0.12 +output = 0.48 +cache_read = 0.0036 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 7031b748f98..731915d3eeb 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.7 -output = 8.5 -cache_read = 0.17 +input = 1.675 +output = 9.38 +cache_read = 0.1943 [limit] context = 1_048_576 From b44a3432f47d0ec1d137eaa26bad06ee99897539 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 03:33:34 +0000 Subject: [PATCH 157/392] chore(sync): update OpenRouter model catalog (#7611) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/moonshotai/kimi-k3.toml | 6 +++--- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index 3b2924c9c2d..22b1407542a 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.675 -output = 9.38 -cache_read = 0.1943 +input = 1.7 +output = 8.5 +cache_read = 0.17 [limit] output = 943_718 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 731915d3eeb..7031b748f98 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.675 -output = 9.38 -cache_read = 0.1943 +input = 1.7 +output = 8.5 +cache_read = 0.17 [limit] context = 1_048_576 From 3764983dc09fe350e6e768a2867a531476ff7154 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 03:33:41 +0000 Subject: [PATCH 158/392] chore(sync): update Kilo model catalog (#7610) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/moonshotai/kimi-k3.toml | 6 +++--- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index 31838f83a2c..fd81e087ca6 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.675 -output = 9.38 -cache_read = 0.1943 +input = 1.7 +output = 8.5 +cache_read = 0.17 [limit] output = 943_718 diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 71dba53f553..631c1758bcd 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.675 -output = 9.38 -cache_read = 0.1943 +input = 1.7 +output = 8.5 +cache_read = 0.17 [limit] context = 1_048_576 From 1c43b109c5d2d1f538069005bd8f6e6f4939bb40 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 04:32:58 +0000 Subject: [PATCH 159/392] chore(sync): update NanoGPT model catalog (#7613) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index a62995b0beb..b4cce19afb3 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.44 -output = 1.32 -cache_read = 0.014 +input = 0.22 +output = 0.66 +cache_read = 0.007 [limit] context = 1_048_576 From 4e14674202195bc855e571cebc4d1829c9e54e93 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 04:33:04 +0000 Subject: [PATCH 160/392] chore(sync): update Kilo model catalog (#7612) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 46e3c60f672..b14064bf897 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.7 -output = 2.88 -cache_read = 0.088 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 943_718 +output = 384_000 [modalities] input = ["text"] From 2a91d70292280963d53f3df71fd48611a031e8a8 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 04:33:08 +0000 Subject: [PATCH 161/392] chore(sync): update OpenRouter model catalog (#7614) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 3 files changed, 10 insertions(+), 10 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index b8d53b0809c..7c5828b16e2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.32 -output = 3.96 -cache_read = 0.044 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 854b27b6a72..16f058ed841 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.3 -output = 1.2 -cache_read = 0.006 +input = 0.15 +output = 0.6 +cache_read = 0.003 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index dfb8b85c369..ea09a0d12b6 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7 -output = 2.88 -cache_read = 0.088 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 943_718 +output = 384_000 [modalities] input = ["text"] From cfc5669256db6c46f77060f8b0f56b53e4d8beb9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 05:32:43 +0000 Subject: [PATCH 162/392] chore(sync): update Eden AI model catalog (#7616) Co-authored-by: opencode-agent[bot] --- .../models/cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/providers/edenai/models/cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct.toml b/providers/edenai/models/cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct.toml index 33e52d4c7a4..be84a18d071 100644 --- a/providers/edenai/models/cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct.toml +++ b/providers/edenai/models/cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct.toml @@ -1,7 +1,7 @@ base_model = "alibaba/qwen2.5-coder-32b-instruct" name = "Qwen2.5-Coder-32B-Instruct (Cloudflare)" tool_call = false -structured_output = false +structured_output = true [cost] input = 0.66 From 5ec54bd903bfe2dc13e1a654dd450a68463e652d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 06:52:10 +0000 Subject: [PATCH 163/392] chore(sync): update OpenRouter model catalog (#7618) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 3 files changed, 10 insertions(+), 10 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..b8d53b0809c 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 1.32 +output = 3.96 +cache_read = 0.044 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 16f058ed841..854b27b6a72 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.15 -output = 0.6 -cache_read = 0.003 +input = 0.3 +output = 1.2 +cache_read = 0.006 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index ea09a0d12b6..dfb8b85c369 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.7 +output = 2.88 +cache_read = 0.088 [limit] context = 1_048_576 -output = 384_000 +output = 943_718 [modalities] input = ["text"] From b8647dc5b85a2bd941c20340a5c90f0a56b7a522 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 06:52:13 +0000 Subject: [PATCH 164/392] chore(sync): update NanoGPT model catalog (#7621) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index b4cce19afb3..a62995b0beb 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.22 -output = 0.66 -cache_read = 0.007 +input = 0.44 +output = 1.32 +cache_read = 0.014 [limit] context = 1_048_576 From 5cd0647e7143f4716f1c3363f4685e5aea599b24 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 06:52:18 +0000 Subject: [PATCH 165/392] chore(sync): update LLM Gateway model catalog (#7619) Co-authored-by: opencode-agent[bot] --- .../models/scx-ai-gp/deepseek-v4.1-flash.toml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 providers/llmgateway-providers/models/scx-ai-gp/deepseek-v4.1-flash.toml diff --git a/providers/llmgateway-providers/models/scx-ai-gp/deepseek-v4.1-flash.toml b/providers/llmgateway-providers/models/scx-ai-gp/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..9ea85540b29 --- /dev/null +++ b/providers/llmgateway-providers/models/scx-ai-gp/deepseek-v4.1-flash.toml @@ -0,0 +1,14 @@ +base_model = "deepseek/deepseek-v4.1-flash" +name = "DeepSeek V4.1 Flash (SCX.ai)" +attachment = false +reasoning = false +tool_call = false +structured_output = false + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.01 + +[modalities] +input = ["text"] From d8d98a4ab0ed3f25d24c50e7c28e06217379ffe1 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 06:52:24 +0000 Subject: [PATCH 166/392] chore(sync): update Kilo model catalog (#7620) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index b14064bf897..46e3c60f672 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.7 +output = 2.88 +cache_read = 0.088 [limit] context = 1_048_576 -output = 384_000 +output = 943_718 [modalities] input = ["text"] From 962a7ceec798b906e2fb3a027b207cb3581ae42b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 08:37:44 +0000 Subject: [PATCH 167/392] chore(sync): update NanoGPT model catalog (#7623) Co-authored-by: opencode-agent[bot] --- .../nano-gpt/models/qwen/qwen3-30b-a3b.toml | 1 + .../models/qwen/qwen3.8-27b-hemmingway.toml | 25 +++++++++++++++++++ 2 files changed, 26 insertions(+) create mode 100644 providers/nano-gpt/models/qwen/qwen3.8-27b-hemmingway.toml diff --git a/providers/nano-gpt/models/qwen/qwen3-30b-a3b.toml b/providers/nano-gpt/models/qwen/qwen3-30b-a3b.toml index 3fe31cbc4c1..ef583b4ad7d 100644 --- a/providers/nano-gpt/models/qwen/qwen3-30b-a3b.toml +++ b/providers/nano-gpt/models/qwen/qwen3-30b-a3b.toml @@ -1,6 +1,7 @@ # Included in subscription base_model = "alibaba/qwen3-30b-a3b" reasoning = false +tool_call = false structured_output = false [cost] diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-hemmingway.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-hemmingway.toml new file mode 100644 index 00000000000..e8e4e1ce33f --- /dev/null +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-hemmingway.toml @@ -0,0 +1,25 @@ +name = "Qwen 3.8 27B Hemingway" +description = "Qwen 3.8 27B Hemingway is an open-weight NVFP4 multimodal creative finetune for long-form prose, character dialogue, storytelling, and roleplay." +family = "qwen" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +tool_call = true +structured_output = false +open_weights = true +reasoning_options = [] + +[cost] +input = 0.25 +output = 1.5 +cache_read = 0.125 + +[limit] +context = 262_144 +input = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] From 25c48bf4e195c5b7da7466969976df06c0e3c5c9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 08:38:17 +0000 Subject: [PATCH 168/392] chore(sync): update Cortecs model catalog (#7624) Co-authored-by: opencode-agent[bot] --- providers/cortecs/models/qwen3-32b.toml | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/providers/cortecs/models/qwen3-32b.toml b/providers/cortecs/models/qwen3-32b.toml index d6ad27de61b..32225e598f0 100644 --- a/providers/cortecs/models/qwen3-32b.toml +++ b/providers/cortecs/models/qwen3-32b.toml @@ -4,9 +4,8 @@ structured_output = true reasoning_options = [] [cost] -input = 0.089 -output = 0.312 +input = 0.179 +output = 0.697 [limit] -context = 32_000 -output = 32_000 +context = 16_384 From 60efdd3429eb7d396331e8988c327216c1e52393 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 10:28:21 +0000 Subject: [PATCH 169/392] chore(sync): update NanoGPT model catalog (#7628) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index a62995b0beb..b4cce19afb3 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.44 -output = 1.32 -cache_read = 0.014 +input = 0.22 +output = 0.66 +cache_read = 0.007 [limit] context = 1_048_576 From 5f15e38c6e8221f63afacb019ea7bfe7b84d4d27 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 10:28:24 +0000 Subject: [PATCH 170/392] chore(sync): update OpenRouter model catalog (#7629) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 3 files changed, 10 insertions(+), 10 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index b8d53b0809c..7c5828b16e2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.32 -output = 3.96 -cache_read = 0.044 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 854b27b6a72..16f058ed841 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.3 -output = 1.2 -cache_read = 0.006 +input = 0.15 +output = 0.6 +cache_read = 0.003 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index dfb8b85c369..ea09a0d12b6 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7 -output = 2.88 -cache_read = 0.088 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 943_718 +output = 384_000 [modalities] input = ["text"] From b3c181144b4157b1e68b1bb8d8d547267f0ca9ac Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 10:28:32 +0000 Subject: [PATCH 171/392] chore(sync): update Kilo model catalog (#7627) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 46e3c60f672..b14064bf897 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.7 -output = 2.88 -cache_read = 0.088 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 943_718 +output = 384_000 [modalities] input = ["text"] From 1c56caef93a8e78727d6e859a1ee115cf133e3c8 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 11:27:19 +0000 Subject: [PATCH 172/392] chore(sync): update NanoGPT model catalog (#7630) Co-authored-by: opencode-agent[bot] --- .../nano-gpt/models/qwen/qwen3-vl-235b-a22b-instruct.toml | 2 +- .../models/qwen3-vl-235b-a22b-instruct-original.toml | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/nano-gpt/models/qwen/qwen3-vl-235b-a22b-instruct.toml b/providers/nano-gpt/models/qwen/qwen3-vl-235b-a22b-instruct.toml index a648f070f2e..8b97c180ac4 100644 --- a/providers/nano-gpt/models/qwen/qwen3-vl-235b-a22b-instruct.toml +++ b/providers/nano-gpt/models/qwen/qwen3-vl-235b-a22b-instruct.toml @@ -4,7 +4,7 @@ structured_output = false [cost] input = 0.3 -output = 1.2 +output = 1.9 cache_read = 0.15 [limit] diff --git a/providers/nano-gpt/models/qwen3-vl-235b-a22b-instruct-original.toml b/providers/nano-gpt/models/qwen3-vl-235b-a22b-instruct-original.toml index 2f39269ea3f..851b51ce84f 100644 --- a/providers/nano-gpt/models/qwen3-vl-235b-a22b-instruct-original.toml +++ b/providers/nano-gpt/models/qwen3-vl-235b-a22b-instruct-original.toml @@ -9,9 +9,9 @@ structured_output = false open_weights = false [cost] -input = 0.5 -output = 1.2 -cache_read = 0.25 +input = 0.3 +output = 1.9 +cache_read = 0.15 [limit] context = 32_768 From 48e0aaa37303ccd9da5b9c13cf2e7e3a6e039b13 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:27:49 +0000 Subject: [PATCH 173/392] chore(sync): update Kilo model catalog (#7632) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index b14064bf897..653d45fd9d1 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,12 +15,12 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.65472 +output = 1.96416 +cache_read = 0.021824 [limit] -context = 1_048_576 +context = 1_024_000 output = 384_000 [modalities] From cfcc2393f3eebc260859697f7a5addc3f2f07e56 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:28:04 +0000 Subject: [PATCH 174/392] chore(sync): update OpenRouter model catalog (#7631) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 6f817c5a0a0..b8c1d29ced8 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.088606 -output = 0.177212 -cache_read = 0.017721 +input = 0.05852 +output = 0.11704 +cache_read = 0.011704 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index e7683af4617..2a431cd5648 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.95526 -output = 1.91052 -cache_read = 0.079605 +input = 0.951432 +output = 1.902864 +cache_read = 0.079286 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index ea09a0d12b6..dd01ad1ecf9 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.65472 +output = 1.96416 +cache_read = 0.021824 [limit] context = 1_048_576 From e379bc9252c87075160ecd47fc894bd4e74f098a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 14:29:39 +0000 Subject: [PATCH 175/392] chore(sync): update OpenRouter model catalog (#7634) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 ++++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../models/meta-llama/llama-3.1-70b-instruct.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- providers/openrouter/models/~z-ai/glm-latest.toml | 8 ++++---- 6 files changed, 21 insertions(+), 20 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index b8c1d29ced8..b044540752e 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.05852 -output = 0.11704 -cache_read = 0.011704 +input = 0.05698 +output = 0.11396 +cache_read = 0.011396 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..c6c87510247 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.57948 +output = 1.73844 +cache_read = 0.018438 [limit] context = 1_048_576 +output = 393_216 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 2a431cd5648..ae6a4dbd88c 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.951432 -output = 1.902864 -cache_read = 0.079286 +input = 0.948126 +output = 1.896252 +cache_read = 0.079011 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml b/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml index 1352c1ee599..881a90e0d15 100644 --- a/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml +++ b/providers/openrouter/models/meta-llama/llama-3.1-70b-instruct.toml @@ -3,9 +3,9 @@ description = "Open Llama instruction model for multilingual chat, reasoning, an structured_output = true [cost] -input = 0.72 -output = 0.72 +input = 0.4 +output = 0.4 [limit] context = 131_072 -output = 8_192 +output = 16_384 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index dd01ad1ecf9..d4c1d5ba628 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.65472 -output = 1.96416 -cache_read = 0.021824 +input = 0.57948 +output = 1.73844 +cache_read = 0.018438 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 25e0814a971..709b8881ffa 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7735 -output = 2.431 -cache_read = 0.127075 +input = 0.7728 +output = 2.4288 +cache_read = 0.14352 [limit] context = 1_310_720 -output = 943_718 +output = 131_072 [modalities] input = ["text"] From 074421e6fe8dbe32f450c3c27f1933ae181ffde0 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 14:29:55 +0000 Subject: [PATCH 176/392] chore(sync): update Kilo model catalog (#7633) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 1 + .../kilo/models/meta-llama/llama-3.1-70b-instruct.toml | 2 +- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- providers/kilo/models/~z-ai/glm-latest.toml | 8 ++++---- 4 files changed, 11 insertions(+), 10 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index 007b654034e..d4d63d57a55 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -12,3 +12,4 @@ cache_read = 0.044 [limit] context = 1_048_576 +output = 393_216 diff --git a/providers/kilo/models/meta-llama/llama-3.1-70b-instruct.toml b/providers/kilo/models/meta-llama/llama-3.1-70b-instruct.toml index 6fc67901197..881a90e0d15 100644 --- a/providers/kilo/models/meta-llama/llama-3.1-70b-instruct.toml +++ b/providers/kilo/models/meta-llama/llama-3.1-70b-instruct.toml @@ -8,4 +8,4 @@ output = 0.4 [limit] context = 131_072 -output = 8_192 +output = 16_384 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 653d45fd9d1..27422e128e0 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.65472 -output = 1.96416 -cache_read = 0.021824 +input = 0.57948 +output = 1.73844 +cache_read = 0.018438 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index b6ea0fea56d..2237628e2fc 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7735 -output = 2.431 -cache_read = 0.127075 +input = 0.7728 +output = 2.4288 +cache_read = 0.14352 [limit] context = 1_048_576 -output = 943_718 +output = 131_072 [modalities] input = ["text"] From 88bb3d960afafee3ef107fac71100acb51ef1201 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:21:48 +0800 Subject: [PATCH 177/392] fix(deepseek): align V4 Pro effort with GA documentation (#7564) * fix(deepseek): align V4 Pro effort with GA documentation * chore: retrigger review workflow The previous review run failed while installing opencode ("Failed to fetch version information"), so no code review ran. Co-Authored-By: Claude Opus 5 --------- Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- providers/deepseek/models/deepseek-v4-pro.toml | 13 +++++++++---- providers/deepseek/provider.toml | 6 +++--- 2 files changed, 12 insertions(+), 7 deletions(-) diff --git a/providers/deepseek/models/deepseek-v4-pro.toml b/providers/deepseek/models/deepseek-v4-pro.toml index 4d5aff0281b..6a81c8584d2 100644 --- a/providers/deepseek/models/deepseek-v4-pro.toml +++ b/providers/deepseek/models/deepseek-v4-pro.toml @@ -1,18 +1,23 @@ +# Toggle: thinking.type = enabled|disabled on /chat/completions. +# Effort: reasoning_effort = low|high|max; default high. +# Anthropic: output_config.effort = low|high|max; budget ignored. +# The 2026-08-13 V4-Pro GA announcement explicitly introduces all three levels; +# the pricing page maps API model deepseek-v4-pro to DeepSeek-V4-Pro-0813. +# https://api-docs.deepseek.com/news/news260813/ (accessed 2026-09-20) +# https://api-docs.deepseek.com/guides/thinking_mode/ (accessed 2026-09-20) +# https://api-docs.deepseek.com/quick_start/pricing/ (accessed 2026-09-20) # Reasoning tokens are billed at the output rate (no separate CoT price). # `completion_tokens_details.reasoning_tokens` is a subset of completion_tokens. # https://api-docs.deepseek.com/quick_start/pricing/ (accessed 2026-08-12) base_model = "deepseek/deepseek-v4-pro-0813" name = "DeepSeek V4 Pro" -# OpenAI: `thinking.type = enabled|disabled`, `reasoning_effort = high|max`. -# Anthropic: `thinking.type`, `output_config.effort = high|max`; budget ignored. -# https://api-docs.deepseek.com/api/create-chat-completion (accessed 2026-06-25) [[reasoning_options]] type = "toggle" [[reasoning_options]] type = "effort" -values = ["high", "max"] +values = ["low", "high", "max"] [interleaved] field = "reasoning_content" diff --git a/providers/deepseek/provider.toml b/providers/deepseek/provider.toml index 11731946bfd..76274ae9de4 100644 --- a/providers/deepseek/provider.toml +++ b/providers/deepseek/provider.toml @@ -2,11 +2,11 @@ name = "DeepSeek" env = ["DEEPSEEK_API_KEY"] npm = "@ai-sdk/openai-compatible" # OpenAI Chat is POST `/chat/completions`: `thinking.type = enabled|disabled` -# and `reasoning_effort = low|high|max`. Flash maps low→low; Pro maps low→high; -# xhigh maps to high (flash) or max (pro). +# and `reasoning_effort = low|high|max`. Both current models map minimal→low, +# medium/xhigh→high, and ultra→max; low/high/max remain distinct native levels. # Anthropic Messages is POST `/anthropic/v1/messages`: `thinking.type` and # `output_config.effort = low|high|max`; `thinking.budget_tokens` is ignored. -# https://api-docs.deepseek.com/guides/thinking_mode/ (accessed 2026-08-02) +# https://api-docs.deepseek.com/guides/thinking_mode/ (accessed 2026-09-20) # https://api-docs.deepseek.com/guides/anthropic_api (accessed 2026-06-25) doc = "https://api-docs.deepseek.com/quick_start/pricing" api = "https://api.deepseek.com" From 843685c20f25423b52f5a85d15f340c2d2099c9a Mon Sep 17 00:00:00 2001 From: Pedro Santos Date: Mon, 21 Sep 2026 16:22:26 +0100 Subject: [PATCH 178/392] fix(zai): add missing glm-4.6v-flash provider model (#7592) Resolves broken zhipuai symlink to zai/glm-4.6v-flash.toml. Fixes #1470 Co-authored-by: Pedro Santos <18473317+pc-gs@users.noreply.github.com> --- providers/zai/models/glm-4.6v-flash.toml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 providers/zai/models/glm-4.6v-flash.toml diff --git a/providers/zai/models/glm-4.6v-flash.toml b/providers/zai/models/glm-4.6v-flash.toml new file mode 100644 index 00000000000..d9793eab5e8 --- /dev/null +++ b/providers/zai/models/glm-4.6v-flash.toml @@ -0,0 +1,14 @@ +base_model = "zhipuai/glm-4.6v-flash" +# Free Flash tier on Z.AI; cost shape mirrors providers/zai/models/glm-4.5-flash.toml. +# Lab metadata (limits/modalities) lives in models/zhipuai/glm-4.6v-flash.toml. +# https://z.ai/blog/glm-4.6v +# https://huggingface.co/zai-org/GLM-4.6V-Flash + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 From d0b52220faa55db8fdf0c9d3ede02d79a78dabc5 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:22:58 +0800 Subject: [PATCH 179/392] feat(aihubmix): add step-3.7-flash (#7559) Co-authored-by: chenxue --- providers/aihubmix/models/step-3.7-flash.toml | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) create mode 100644 providers/aihubmix/models/step-3.7-flash.toml diff --git a/providers/aihubmix/models/step-3.7-flash.toml b/providers/aihubmix/models/step-3.7-flash.toml new file mode 100644 index 00000000000..d6dcca8c632 --- /dev/null +++ b/providers/aihubmix/models/step-3.7-flash.toml @@ -0,0 +1,18 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://platform.stepfun.com/docs/zh/guides/models/step-3.7-flash +# Effort: reasoning_effort = low|medium|high on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "stepfun/step-3.7-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 0.22 +output = 1.32 +cache_read = 0.044 From 137714d0fe5c9e0af9a18a5bb1f03359289e8cee Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:23:11 +0800 Subject: [PATCH 180/392] feat(aihubmix): add minimax-m3 (#7558) Co-authored-by: chenxue --- providers/aihubmix/models/minimax-m3.toml | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 providers/aihubmix/models/minimax-m3.toml diff --git a/providers/aihubmix/models/minimax-m3.toml b/providers/aihubmix/models/minimax-m3.toml new file mode 100644 index 00000000000..2103f7586f0 --- /dev/null +++ b/providers/aihubmix/models/minimax-m3.toml @@ -0,0 +1,17 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://platform.minimaxi.com/docs/api-reference/text-chat-openai.md +# Toggle: enable_thinking = true|false on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "minimax/MiniMax-M3" +structured_output = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.288 +output = 1.152 From 1da5b84d0bb6d17235ff9235d44222c528bad1d5 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:23:46 +0800 Subject: [PATCH 181/392] feat(aihubmix): add gemini-3.1-flash-image (#7556) Co-authored-by: chenxue --- .../aihubmix/models/gemini-3.1-flash-image.toml | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 providers/aihubmix/models/gemini-3.1-flash-image.toml diff --git a/providers/aihubmix/models/gemini-3.1-flash-image.toml b/providers/aihubmix/models/gemini-3.1-flash-image.toml new file mode 100644 index 00000000000..1c51818cb89 --- /dev/null +++ b/providers/aihubmix/models/gemini-3.1-flash-image.toml @@ -0,0 +1,17 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-image +# Effort: reasoning_effort = minimal|high on /v1/chat/completions. +# Gemini native: generationConfig.thinkingConfig.thinkingLevel uses the corresponding uppercase levels. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.1-flash-image" + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.5 +output = 3 From 1fc9f0f6b4cd03ce9bd8bb95e382da88fcdc8ad4 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:23:53 +0800 Subject: [PATCH 182/392] feat(aihubmix): add hy3 (#7555) Co-authored-by: chenxue --- providers/aihubmix/models/hy3.toml | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 providers/aihubmix/models/hy3.toml diff --git a/providers/aihubmix/models/hy3.toml b/providers/aihubmix/models/hy3.toml new file mode 100644 index 00000000000..a5f55cbba88 --- /dev/null +++ b/providers/aihubmix/models/hy3.toml @@ -0,0 +1,20 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://huggingface.co/tencent/Hy3 +# Effort: reasoning_effort = none|low|high on /v1/chat/completions. +# Off is effort=none; AIHubMix no_think is normalized to none, so no separate toggle. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "tencent/hy3" +structured_output = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high"] + +[cost] +input = 0.1562 +output = 0.6248 +cache_read = 0.03905 From 4c5584c0449f80108f32c15b74272aee55f1bd2e Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:24:04 +0800 Subject: [PATCH 183/392] feat(aihubmix): add muse-spark-1.1 (#7554) * feat(aihubmix): add muse-spark-1.1 * docs(aihubmix): cite Muse Spark 1.1 audio input declaration --------- Co-authored-by: chenxue --- providers/aihubmix/models/muse-spark-1.1.toml | 20 +++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 providers/aihubmix/models/muse-spark-1.1.toml diff --git a/providers/aihubmix/models/muse-spark-1.1.toml b/providers/aihubmix/models/muse-spark-1.1.toml new file mode 100644 index 00000000000..f0c00f709bf --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.1.toml @@ -0,0 +1,20 @@ +# Sources (accessed 2026-09-20): +# AIHubMix catalog row model_id=muse-spark-1.1 declares +# input_modalities="text,image,video,audio,pdf". Audio is a host-catalog override +# relative to the lab entry; this is a catalog declaration, not an inference test. +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://dev.meta.ai/docs/models/ +# Effort: reasoning_effort = minimal|low|medium|high|xhigh on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "meta/muse-spark-1.1" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[cost] +input = 1.375 +output = 4.675 + +[modalities] +input = ["text", "image", "video", "audio", "pdf"] From d5fb44d21e599898970deb7e3c6851df089082c5 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:24:12 +0800 Subject: [PATCH 184/392] feat(aihubmix): add gemini-3.5-flash-lite (#7553) * feat(aihubmix): add gemini-3.5-flash-lite * docs(aihubmix): cite exact catalog price for Gemini 3.5 Flash Lite --------- Co-authored-by: chenxue --- .../models/gemini-3.5-flash-lite.toml | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) create mode 100644 providers/aihubmix/models/gemini-3.5-flash-lite.toml diff --git a/providers/aihubmix/models/gemini-3.5-flash-lite.toml b/providers/aihubmix/models/gemini-3.5-flash-lite.toml new file mode 100644 index 00000000000..19c43bfae2f --- /dev/null +++ b/providers/aihubmix/models/gemini-3.5-flash-lite.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-20): +# AIHubMix catalog row model_id=gemini-3.5-flash-lite publishes +# pricing={"input":0.3,"output":2.499999,"cache_read":0.03}. +# Preserve the host's exact published price; Google's 2.50 is a different host. +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://ai.google.dev/gemini-api/docs/thinking +# Effort: reasoning_effort = minimal|low|medium|high on /v1/chat/completions. +# Gemini native: generationConfig.thinkingConfig.thinkingLevel uses the corresponding uppercase levels. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.5-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0.3 +output = 2.499999 +cache_read = 0.03 From 5d137a902e47a1f1de8df73832ee9a7d1b160f1b Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:24:19 +0800 Subject: [PATCH 185/392] feat(aihubmix): add gemini-3.1-flash-lite-image (#7552) Co-authored-by: chenxue --- .../models/gemini-3.1-flash-lite-image.toml | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) create mode 100644 providers/aihubmix/models/gemini-3.1-flash-lite-image.toml diff --git a/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml b/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml new file mode 100644 index 00000000000..14141e81cc6 --- /dev/null +++ b/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml @@ -0,0 +1,18 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite-image +# Effort: reasoning_effort = minimal|high on /v1/chat/completions. +# Gemini native: generationConfig.thinkingConfig.thinkingLevel uses the corresponding uppercase levels. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.1-flash-lite-image" + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.25 +output = 1.5 +cache_read = 0.025 From 90db5bf0fc3c2af0a4460570a3be94a48939ac29 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:24:29 +0800 Subject: [PATCH 186/392] feat(aihubmix): add muse-spark-1.2 (#7551) Co-authored-by: chenxue --- providers/aihubmix/models/muse-spark-1.2.toml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 providers/aihubmix/models/muse-spark-1.2.toml diff --git a/providers/aihubmix/models/muse-spark-1.2.toml b/providers/aihubmix/models/muse-spark-1.2.toml new file mode 100644 index 00000000000..01d1fa8c78a --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.2.toml @@ -0,0 +1,14 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://dev.meta.ai/docs/overview/ +# Effort: reasoning_effort = minimal|low|medium|high|xhigh on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "meta/muse-spark-1.2" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[cost] +input = 1.375 +output = 4.675 From 21e40fe94fc66aa22261400a90096b493bc1d9dd Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:24:37 +0800 Subject: [PATCH 187/392] feat(aihubmix): add qwen3.8-omni-flash (#7550) Co-authored-by: chenxue --- .../aihubmix/models/qwen3.8-omni-flash.toml | 23 +++++++++++++++++++ 1 file changed, 23 insertions(+) create mode 100644 providers/aihubmix/models/qwen3.8-omni-flash.toml diff --git a/providers/aihubmix/models/qwen3.8-omni-flash.toml b/providers/aihubmix/models/qwen3.8-omni-flash.toml new file mode 100644 index 00000000000..8dea0f811e5 --- /dev/null +++ b/providers/aihubmix/models/qwen3.8-omni-flash.toml @@ -0,0 +1,23 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://www.alibabacloud.com/help/en/model-studio/qwen-omni +# Toggle: enable_thinking = true|false on /v1/chat/completions. +# Effort: reasoning_effort = low|medium|xhigh on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "alibaba/qwen3.8-omni-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 0.1126 +output = 0.380025 +cache_read = 0.014075 +cache_write = 0.175937 From 0578dc285fd5a2d1d929b3a698e1c406e77c455d Mon Sep 17 00:00:00 2001 From: Pessimist228 Date: Mon, 21 Sep 2026 17:24:50 +0200 Subject: [PATCH 188/392] llmtech: rename model to nvidia/Qwen3.8-27B-NVFP4 (#7549) * llmtech: canonical model id is nvidia/Qwen3.8-27B-NVFP4 The served build is NVIDIA's own NVFP4 checkpoint, not the community one. The previous id stays accepted by the API as a synonym, so nothing breaks for anyone who has not switched yet. * llmtech: drop the old unsloth/ model path --- .../llmtech/models/{unsloth => nvidia}/Qwen3.8-27B-NVFP4.toml | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename providers/llmtech/models/{unsloth => nvidia}/Qwen3.8-27B-NVFP4.toml (100%) diff --git a/providers/llmtech/models/unsloth/Qwen3.8-27B-NVFP4.toml b/providers/llmtech/models/nvidia/Qwen3.8-27B-NVFP4.toml similarity index 100% rename from providers/llmtech/models/unsloth/Qwen3.8-27B-NVFP4.toml rename to providers/llmtech/models/nvidia/Qwen3.8-27B-NVFP4.toml From ba1ee716d838a37b985a48b6805e7237e88fe755 Mon Sep 17 00:00:00 2001 From: "C.C." Date: Mon, 21 Sep 2026 23:25:03 +0800 Subject: [PATCH 189/392] feat(vivgrid): add viv-fast and deepseek-v4.1-flash models (#7626) - viv-fast: new Vivgrid first-party SLM (1M context, 256K output, text+image input, reasoning effort low/high/max), lab entry under models/vivgrid/ plus provider pricing at $0.13/$0.40 with $0.05 cache read - deepseek-v4.1-flash: base_model override on deepseek/deepseek-v4.1-flash with Vivgrid pricing ($0.31/$1.23, $0.01 cache read), DeepSeek-native reasoning controls (thinking toggle + effort low/high/max) and reasoning_content interleaved, matching the existing deepseek-v4-flash entry on the same surface --- models/vivgrid/viv-fast.toml | 26 +++++++++++++++++++ .../vivgrid/models/deepseek-v4.1-flash.toml | 20 ++++++++++++++ providers/vivgrid/models/viv-fast.toml | 10 +++++++ 3 files changed, 56 insertions(+) create mode 100644 models/vivgrid/viv-fast.toml create mode 100644 providers/vivgrid/models/deepseek-v4.1-flash.toml create mode 100644 providers/vivgrid/models/viv-fast.toml diff --git a/models/vivgrid/viv-fast.toml b/models/vivgrid/viv-fast.toml new file mode 100644 index 00000000000..9bed45cd435 --- /dev/null +++ b/models/vivgrid/viv-fast.toml @@ -0,0 +1,26 @@ +# Sources: +# - https://docs.vivgrid.com/models/viv-fast +# - https://docs.vivgrid.com/api/model-api +name = "Viv Fast" +description = "Fast coding model" +release_date = "2026-09-21" +last_updated = "2026-09-21" +reasoning = true +knowledge = "2026-09" +temperature = false +open_weights = false +tool_call = true +structured_output = true +attachment = true + +[limit] +context = 1_000_000 +output = 256_000 + +[modalities] +input = ["text", "image"] +output = ["text"] + +[[links]] +label = "Model docs" +url = "https://docs.vivgrid.com/models/viv-fast" diff --git a/providers/vivgrid/models/deepseek-v4.1-flash.toml b/providers/vivgrid/models/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..276f6360406 --- /dev/null +++ b/providers/vivgrid/models/deepseek-v4.1-flash.toml @@ -0,0 +1,20 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/deepseek-v4.1-flash (accessed 2026-09-21) +base_model = "deepseek/deepseek-v4.1-flash" + +# Toggle: thinking.type = enabled|disabled +# Effort: reasoning_effort = low|high|max +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.31 +output = 1.23 +cache_read = 0.01 diff --git a/providers/vivgrid/models/viv-fast.toml b/providers/vivgrid/models/viv-fast.toml new file mode 100644 index 00000000000..354c69bf873 --- /dev/null +++ b/providers/vivgrid/models/viv-fast.toml @@ -0,0 +1,10 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/viv-fast (accessed 2026-09-21) +base_model = "vivgrid/viv-fast" +# Effort: reasoning_effort = low|high|max +reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] + +[cost] +input = 0.13 +output = 0.40 +cache_read = 0.05 From 5d395d498aa4352ae6ee6ad40da45cfd8049ba1e Mon Sep 17 00:00:00 2001 From: bukowskiiiiiiiiiiiii Date: Mon, 21 Sep 2026 23:25:52 +0800 Subject: [PATCH 190/392] feat(stepfun-step-plan): add step-5-preview (#7544) StepFun's new flagship base model, available on the Step Plan gateway (api.stepfun.com/step_plan/v1). Lab metadata: 1M-token context/input/ output, text/image/video input, reasoning + tool calling + structured output, closed weights. Provider override: reasoning_effort accepts low/medium/high; reasoning side channel is reasoning_content. Sources: - https://platform.stepfun.com/docs/zh/guides/models/step-5-preview - https://api.stepfun.com/step_plan/v1/models (accessed 2026-09-20) bun validate passes. Co-authored-by: BUKOWSKIREAL --- models/stepfun/step-5-preview.toml | 20 +++++++++++++++++++ .../models/step-5-preview.toml | 11 ++++++++++ providers/stepfun-step-plan/provider.toml | 2 +- 3 files changed, 32 insertions(+), 1 deletion(-) create mode 100644 models/stepfun/step-5-preview.toml create mode 100644 providers/stepfun-step-plan/models/step-5-preview.toml diff --git a/models/stepfun/step-5-preview.toml b/models/stepfun/step-5-preview.toml new file mode 100644 index 00000000000..e4718be0938 --- /dev/null +++ b/models/stepfun/step-5-preview.toml @@ -0,0 +1,20 @@ +name = "Step 5 Preview" +description = "StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window" +# release_date is StepFun's API catalog first-appearance timestamp (created=1789564127, +# 2026-09-16 13:08 UTC) — no dated launch announcement was found on the model docs page. +release_date = "2026-09-16" +last_updated = "2026-09-20" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[limit] +context = 1_000_000 +input = 1_000_000 +output = 1_000_000 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/stepfun-step-plan/models/step-5-preview.toml b/providers/stepfun-step-plan/models/step-5-preview.toml new file mode 100644 index 00000000000..684d9f67224 --- /dev/null +++ b/providers/stepfun-step-plan/models/step-5-preview.toml @@ -0,0 +1,11 @@ +# Chat `reasoning_effort` accepts low/medium/high (accessed 2026-09-20). +# https://platform.stepfun.com/docs/zh/guides/models/step-5-preview +# Step Plan is subscription/credit based with no public per-token price, so no [cost]. +base_model = "stepfun/step-5-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/stepfun-step-plan/provider.toml b/providers/stepfun-step-plan/provider.toml index ebcf390c442..1c0546449a3 100644 --- a/providers/stepfun-step-plan/provider.toml +++ b/providers/stepfun-step-plan/provider.toml @@ -4,7 +4,7 @@ npm = "@ai-sdk/openai-compatible" # Reasoning HTTP format (accessed 2026-06-25): # Step Plan exposes POST /step_plan/v1/chat/completions with top-level # `reasoning_effort` and POST /step_plan/v1/messages with -# `output_config.effort`. step-3.7-flash accepts low/medium/high; +# `output_config.effort`. step-3.7-flash and step-5-preview accept low/medium/high; # step-3.5-flash and step-3.5-flash-2603 accept low/high. No plan Responses # endpoint is listed. # Source: From 9b23514872a602c5edbd2ff05de796f636b2e59d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 15:26:20 +0000 Subject: [PATCH 191/392] chore(sync): update Kilo model catalog (#7637) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 3 +-- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 2 files changed, 6 insertions(+), 7 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index d4d63d57a55..01800e18158 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -11,5 +11,4 @@ output = 3.96 cache_read = 0.044 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 27422e128e0..a92bf5149a1 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.57948 -output = 1.73844 -cache_read = 0.018438 +input = 0.57816 +output = 1.73448 +cache_read = 0.019272 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] From c49c7b1f831ce90e0ca964206a11fca47034988b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 15:26:41 +0000 Subject: [PATCH 192/392] chore(sync): update OpenRouter model catalog (#7635) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 +++---- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 4 files changed, 13 insertions(+), 14 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index b044540752e..f9ab26fd006 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.05698 -output = 0.11396 -cache_read = 0.011396 +input = 0.05544 +output = 0.11088 +cache_read = 0.011088 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index c6c87510247..32fc69b9875 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,10 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57948 -output = 1.73844 -cache_read = 0.018438 +input = 0.57816 +output = 1.73448 +cache_read = 0.019272 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index ae6a4dbd88c..b7feb0289f9 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.948126 -output = 1.896252 -cache_read = 0.079011 +input = 0.942906 +output = 1.885812 +cache_read = 0.078576 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index d4c1d5ba628..9b9fc44533c 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57948 -output = 1.73844 -cache_read = 0.018438 +input = 0.57816 +output = 1.73448 +cache_read = 0.019272 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] From 1e4a5a889fb69680b00a315afa65e0518a3977f1 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:26:47 +0800 Subject: [PATCH 193/392] feat(aihubmix): add longcat-2.0 (#7557) Co-authored-by: chenxue --- providers/aihubmix/models/longcat-2.0.toml | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 providers/aihubmix/models/longcat-2.0.toml diff --git a/providers/aihubmix/models/longcat-2.0.toml b/providers/aihubmix/models/longcat-2.0.toml new file mode 100644 index 00000000000..4f0b9ae311d --- /dev/null +++ b/providers/aihubmix/models/longcat-2.0.toml @@ -0,0 +1,17 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://github.com/meituan-longcat/LongCat-2.0 +# Toggle: enable_thinking = true|false on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "meituan/longcat-2.0" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.7746 +output = 3.0984 +cache_read = 0.015492 From 12465a3b1719412cc202a20b1d02a5f9f61672af Mon Sep 17 00:00:00 2001 From: NoahChen Date: Mon, 21 Sep 2026 23:26:55 +0800 Subject: [PATCH 194/392] feat(siliconflow): sync model catalog with SiliconFlow international site (#7625) * feat(siliconflow): sync catalog with SiliconFlow international site (2026-09) Add 13 models: DeepSeek-V4-Flash-Vision-Exp, DeepSeek-V4-Pro-0813, DeepSeek-V4-Flash-0731, Qwen3.8-2.4T-A95B, GLM-5.3, GLM-5.3-Flash, Kimi-K3, Kimi-K2.7-Code, MiniMax-M3, gemma-4-12B-it, Nex-N2-Pro, meituan-longcat/LongCat-2.0, tencent/Hy3 (replaces Hy3-preview). Remove 5 models no longer listed on SiliconFlow international: ERNIE-4.5-300B-A47B, Qwen3-VL-235B-A22B-Instruct/Thinking, Qwen3-235B-A22B-Thinking-2507, Hy3-preview. Update pricing to current rates for DeepSeek-V3.2, DeepSeek-V4-Flash, DeepSeek-V4-Pro, Kimi-K2.6, GLM-5.1, GLM-5.2. Add lab metadata for google/gemma-4-12b-it and nex-agi/nex-n2-pro. Sources: https://www.siliconflow.com/models, https://www.siliconflow.com/pricing, https://docs.siliconflow.com/en/release-notes/overview.md * fix(siliconflow): address auto-review items on reasoning options - LongCat-2.0: drop toggle (SF does not enumerate enable_thinking for LongCat), keep budget_tokens with leading wire-path comment - MiniMax-M3, Qwen3.8-2.4T-A95B: add leading wire-path comments citing SF's thinking_budget surface; SF exposes no reasoning_effort param - Hy3: drop stale open_weights=false override inherited from Hy3-preview (lab entry tencent/hy3 is open weights) - nex-agi/nex-n2-pro lab: attachment=true, consistent with image input modality * fix(siliconflow): address round-2 auto-review items - Narrow inherited modalities to SF's actual surface: MiniMax-M3, Kimi-K3, Kimi-K2.7-Code, GLM-5.3-Flash now serve text+image only (per each model page's "Support image input: Yes"; no video/pdf on SF) - Hy3: drop redundant [modalities] (identical to lab), document that 262K/262K limits are SF-advertised deltas vs lab 256K/128K - GLM-5.1: cite exact price row (input 1.19 / cached 0.6 / output 3.74 from the GLM-5.1 model page) - cache_read=0.6 is SF's published price - V4-Pro / K2.7-Code / GLM-5.2: note costs are published USD list prices from siliconflow.com/pricing, not FX conversions * fix(siliconflow): address round-3 auto-review items - GLM-5.3 / GLM-5.3-Flash: effort low|high|max, matching the first-party zhipuai GLM-5.3 entry (5.2's high|max was generation-specific) - gemma-4-12B-it: drop [limit].context (identical to lab), keep only the real output delta --------- Co-authored-by: NoahChen <288864333+noahchen2002@users.noreply.github.com> --- models/google/gemma-4-12b-it.toml | 23 ++++++++++++++++ models/nex-agi/nex-n2-pro.toml | 23 ++++++++++++++++ .../models/MiniMaxAI/MiniMax-M3.toml | 26 +++++++++++++++++++ .../Qwen/Qwen3-235B-A22B-Thinking-2507.toml | 24 ----------------- .../Qwen/Qwen3-VL-235B-A22B-Instruct.toml | 23 ---------------- .../Qwen/Qwen3-VL-235B-A22B-Thinking.toml | 24 ----------------- .../models/Qwen/Qwen3.8-2.4T-A95B.toml | 16 ++++++++++++ .../models/baidu/ERNIE-4.5-300B-A47B.toml | 23 ---------------- .../models/deepseek-ai/DeepSeek-V3.2.toml | 1 + .../deepseek-ai/DeepSeek-V4-Flash-0731.toml | 12 +++++++++ .../DeepSeek-V4-Flash-Vision-Exp.toml | 12 +++++++++ .../models/deepseek-ai/DeepSeek-V4-Flash.toml | 2 +- .../deepseek-ai/DeepSeek-V4-Pro-0813.toml | 12 +++++++++ .../models/deepseek-ai/DeepSeek-V4-Pro.toml | 8 +++--- .../models/google/gemma-4-12B-it.toml | 16 ++++++++++++ .../models/meituan-longcat/LongCat-2.0.toml | 18 +++++++++++++ .../models/moonshotai/Kimi-K2.6.toml | 4 +-- .../models/moonshotai/Kimi-K2.7-Code.toml | 21 +++++++++++++++ .../models/moonshotai/Kimi-K3.toml | 22 ++++++++++++++++ .../models/nex-agi/Nex-N2-Pro.toml | 15 +++++++++++ .../models/tencent/Hy3-preview.toml | 18 ------------- providers/siliconflow/models/tencent/Hy3.toml | 15 +++++++++++ .../siliconflow/models/zai-org/GLM-5.1.toml | 11 +++++--- .../siliconflow/models/zai-org/GLM-5.2.toml | 7 +++-- .../models/zai-org/GLM-5.3-Flash.toml | 24 +++++++++++++++++ .../siliconflow/models/zai-org/GLM-5.3.toml | 16 ++++++++++++ 26 files changed, 292 insertions(+), 124 deletions(-) create mode 100644 models/google/gemma-4-12b-it.toml create mode 100644 models/nex-agi/nex-n2-pro.toml create mode 100644 providers/siliconflow/models/MiniMaxAI/MiniMax-M3.toml delete mode 100644 providers/siliconflow/models/Qwen/Qwen3-235B-A22B-Thinking-2507.toml delete mode 100644 providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Instruct.toml delete mode 100644 providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Thinking.toml create mode 100644 providers/siliconflow/models/Qwen/Qwen3.8-2.4T-A95B.toml delete mode 100644 providers/siliconflow/models/baidu/ERNIE-4.5-300B-A47B.toml create mode 100644 providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-0731.toml create mode 100644 providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp.toml create mode 100644 providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml create mode 100644 providers/siliconflow/models/google/gemma-4-12B-it.toml create mode 100644 providers/siliconflow/models/meituan-longcat/LongCat-2.0.toml create mode 100644 providers/siliconflow/models/moonshotai/Kimi-K2.7-Code.toml create mode 100644 providers/siliconflow/models/moonshotai/Kimi-K3.toml create mode 100644 providers/siliconflow/models/nex-agi/Nex-N2-Pro.toml delete mode 100644 providers/siliconflow/models/tencent/Hy3-preview.toml create mode 100644 providers/siliconflow/models/tencent/Hy3.toml create mode 100644 providers/siliconflow/models/zai-org/GLM-5.3-Flash.toml create mode 100644 providers/siliconflow/models/zai-org/GLM-5.3.toml diff --git a/models/google/gemma-4-12b-it.toml b/models/google/gemma-4-12b-it.toml new file mode 100644 index 00000000000..af2b5f0d680 --- /dev/null +++ b/models/google/gemma-4-12b-it.toml @@ -0,0 +1,23 @@ +name = "Gemma 4 12B IT" +description = "Compact Gemma 4 instruction model for open, self-hosted chat and reasoning" +family = "gemma" +release_date = "2026-06-09" +last_updated = "2026-06-09" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/google/gemma-4-12B-it" diff --git a/models/nex-agi/nex-n2-pro.toml b/models/nex-agi/nex-n2-pro.toml new file mode 100644 index 00000000000..3b134613e5f --- /dev/null +++ b/models/nex-agi/nex-n2-pro.toml @@ -0,0 +1,23 @@ +# https://huggingface.co/nex-agi/Nex-N2-Pro +name = "Nex-N2-Pro" +description = "Open agentic MoE model (397B total, 17B active) for coding, tool use, and research workflows" +release_date = "2026-06-02" +last_updated = "2026-06-02" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[limit] +context = 262_144 +output = 262_144 + +[modalities] +input = ["text", "image"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/nex-agi/Nex-N2-Pro" diff --git a/providers/siliconflow/models/MiniMaxAI/MiniMax-M3.toml b/providers/siliconflow/models/MiniMaxAI/MiniMax-M3.toml new file mode 100644 index 00000000000..748a1991b69 --- /dev/null +++ b/providers/siliconflow/models/MiniMaxAI/MiniMax-M3.toml @@ -0,0 +1,26 @@ +# Budget: thinking_budget (integer reasoning tokens, 128..32768) — SiliconFlow documents +# thinking_budget for all reasoning models; enable_thinking is not enumerated for MiniMax +# models on this host, so no toggle. +# Sources: https://www.siliconflow.com/models (2026-09-08), +# https://docs.siliconflow.com/en/api-reference/chat-completions/chat-completions +base_model = "minimax/MiniMax-M3" + + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.06 + +[limit] +output = 131_000 + +# SiliconFlow serves image input but not video for this model +# (model page "Support image input: Yes"): https://www.siliconflow.com/models/minimax-m3 +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/siliconflow/models/Qwen/Qwen3-235B-A22B-Thinking-2507.toml b/providers/siliconflow/models/Qwen/Qwen3-235B-A22B-Thinking-2507.toml deleted file mode 100644 index fb446e7d261..00000000000 --- a/providers/siliconflow/models/Qwen/Qwen3-235B-A22B-Thinking-2507.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Qwen/Qwen3-235B-A22B-Thinking-2507" -description = "Qwen reasoning model for deliberate problem solving, math, and coding" -family = "qwen" -release_date = "2025-07-28" -last_updated = "2025-11-25" -attachment = false -reasoning = true -reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.13 -output = 0.6 - -[limit] -context = 262_000 -output = 262_000 - -[modalities] -input = ["text"] -output = ["text"] \ No newline at end of file diff --git a/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Instruct.toml b/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Instruct.toml deleted file mode 100644 index 8c74b4bceca..00000000000 --- a/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Instruct.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "Qwen/Qwen3-VL-235B-A22B-Instruct" -description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" -family = "qwen" -release_date = "2025-10-04" -last_updated = "2025-11-25" -attachment = true -reasoning = false -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.3 -output = 1.5 - -[limit] -context = 262_000 -output = 262_000 - -[modalities] -input = ["text", "image"] -output = ["text"] \ No newline at end of file diff --git a/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Thinking.toml b/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Thinking.toml deleted file mode 100644 index 03bb1766276..00000000000 --- a/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Thinking.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Qwen/Qwen3-VL-235B-A22B-Thinking" -description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" -family = "qwen" -release_date = "2025-10-04" -last_updated = "2025-11-25" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.45 -output = 3.5 - -[limit] -context = 262_000 -output = 262_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/siliconflow/models/Qwen/Qwen3.8-2.4T-A95B.toml b/providers/siliconflow/models/Qwen/Qwen3.8-2.4T-A95B.toml new file mode 100644 index 00000000000..86601f6baa1 --- /dev/null +++ b/providers/siliconflow/models/Qwen/Qwen3.8-2.4T-A95B.toml @@ -0,0 +1,16 @@ +# Budget: thinking_budget (integer reasoning tokens, 128..32768) — SiliconFlow exposes no +# reasoning_effort parameter; thinking_budget is documented for all reasoning models. +# Sources: https://www.siliconflow.com/models (2026-09-08), +# https://docs.siliconflow.com/en/api-reference/chat-completions/chat-completions +base_model = "alibaba/qwen3.8-2.4t-a95b" + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[cost] +input = 2.0 +output = 6.0 +cache_read = 0.25 + +[limit] +context = 1_049_000 +output = 131_000 diff --git a/providers/siliconflow/models/baidu/ERNIE-4.5-300B-A47B.toml b/providers/siliconflow/models/baidu/ERNIE-4.5-300B-A47B.toml deleted file mode 100644 index 5309d4c0402..00000000000 --- a/providers/siliconflow/models/baidu/ERNIE-4.5-300B-A47B.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "baidu/ERNIE-4.5-300B-A47B" -description = "Tool-capable chat model for instruction following and agentic application workflows" -family = "ernie" -release_date = "2025-07-02" -last_updated = "2025-11-25" -attachment = false -reasoning = false -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.28 -output = 1.1 - -[limit] -context = 131_000 -output = 131_000 - -[modalities] -input = ["text"] -output = ["text"] \ No newline at end of file diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V3.2.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V3.2.toml index a05b736c528..84bd5c7b342 100644 --- a/providers/siliconflow/models/deepseek-ai/DeepSeek-V3.2.toml +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V3.2.toml @@ -14,6 +14,7 @@ open_weights = false [cost] input = 0.27 output = 0.42 +cache_read = 0.135 [limit] context = 164_000 diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-0731.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-0731.toml new file mode 100644 index 00000000000..be8c8bf0744 --- /dev/null +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-0731.toml @@ -0,0 +1,12 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "deepseek/deepseek-v4-flash-0731" + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.22 +output = 0.66 +cache_read = 0.014 diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp.toml new file mode 100644 index 00000000000..674c8102988 --- /dev/null +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp.toml @@ -0,0 +1,12 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "deepseek/deepseek-v4-flash-vision-exp" + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.44 +output = 1.32 +cache_read = 0.028 diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash.toml index 744b022d3f0..14ac45494a9 100644 --- a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash.toml +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash.toml @@ -6,6 +6,6 @@ reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] field = "reasoning_content" [cost] -input = 0.14 +input = 0.13 output = 0.28 cache_read = 0.028 diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml new file mode 100644 index 00000000000..dd2c092e9dc --- /dev/null +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml @@ -0,0 +1,12 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "deepseek/deepseek-v4-pro-0813" + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1.32 +output = 3.96 +cache_read = 0.044 diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro.toml index a573cf5c2af..9994892360a 100644 --- a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro.toml +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro.toml @@ -1,3 +1,5 @@ +# Cost values are the published USD list prices on SiliconFlow international +# (https://www.siliconflow.com/pricing, accessed 2026-09-08), not FX conversions by this PR. base_model = "deepseek/deepseek-v4-pro" reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] @@ -6,6 +8,6 @@ reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] field = "reasoning_content" [cost] -input = 1.74 -output = 3.48 -cache_read = 0.145 +input = 1.50162 +output = 3.135 +cache_read = 0.135 diff --git a/providers/siliconflow/models/google/gemma-4-12B-it.toml b/providers/siliconflow/models/google/gemma-4-12B-it.toml new file mode 100644 index 00000000000..05329524ad5 --- /dev/null +++ b/providers/siliconflow/models/google/gemma-4-12B-it.toml @@ -0,0 +1,16 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "google/gemma-4-12b-it" +attachment = false +reasoning = false +open_weights = false + +[cost] +input = 0.10 +output = 0.30 + +[limit] +output = 262_144 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/siliconflow/models/meituan-longcat/LongCat-2.0.toml b/providers/siliconflow/models/meituan-longcat/LongCat-2.0.toml new file mode 100644 index 00000000000..dfe567a7db0 --- /dev/null +++ b/providers/siliconflow/models/meituan-longcat/LongCat-2.0.toml @@ -0,0 +1,18 @@ +# Budget: thinking_budget (integer reasoning tokens, 128..32768) — SiliconFlow documents +# thinking_budget for all reasoning models and enumerates enable_thinking only for specific +# models (LongCat not among them), so no toggle on this host. +# Sources: https://www.siliconflow.com/models (2026-09-08), +# https://docs.siliconflow.com/en/api-reference/chat-completions/chat-completions +base_model = "meituan/longcat-2.0" +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.75 +output = 2.95 +cache_read = 0.015 + +[limit] +context = 1_049_000 diff --git a/providers/siliconflow/models/moonshotai/Kimi-K2.6.toml b/providers/siliconflow/models/moonshotai/Kimi-K2.6.toml index 73e5e28af8b..48e6aaf149e 100644 --- a/providers/siliconflow/models/moonshotai/Kimi-K2.6.toml +++ b/providers/siliconflow/models/moonshotai/Kimi-K2.6.toml @@ -16,8 +16,8 @@ field = "reasoning_content" [cost] input = 0.77 -output = 4.0 -cache_read = 0.2 +output = 3.4 +cache_read = 0.14 [limit] context = 262_000 diff --git a/providers/siliconflow/models/moonshotai/Kimi-K2.7-Code.toml b/providers/siliconflow/models/moonshotai/Kimi-K2.7-Code.toml new file mode 100644 index 00000000000..f70f197f92d --- /dev/null +++ b/providers/siliconflow/models/moonshotai/Kimi-K2.7-Code.toml @@ -0,0 +1,21 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "moonshotai/kimi-k2.7-code" + + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.85916 +output = 3.8 +cache_read = 0.17993 + +# Cost values are the published USD list prices on SiliconFlow international +# (https://www.siliconflow.com/pricing), not FX conversions by this PR. +# SiliconFlow serves image input but not video for this model +# (model page "Support image input: Yes"): https://www.siliconflow.com/models/kimi-k2-7-code +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/siliconflow/models/moonshotai/Kimi-K3.toml b/providers/siliconflow/models/moonshotai/Kimi-K3.toml new file mode 100644 index 00000000000..daa54bb61f7 --- /dev/null +++ b/providers/siliconflow/models/moonshotai/Kimi-K3.toml @@ -0,0 +1,22 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "moonshotai/kimi-k3" + + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 2.7 +output = 13.5 +cache_read = 0.27 + +[limit] +output = 262_000 + +# SiliconFlow serves image input but not video for this model +# (model page "Support image input: Yes"): https://www.siliconflow.com/models/kimi-k3 +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/siliconflow/models/nex-agi/Nex-N2-Pro.toml b/providers/siliconflow/models/nex-agi/Nex-N2-Pro.toml new file mode 100644 index 00000000000..f160fdf834d --- /dev/null +++ b/providers/siliconflow/models/nex-agi/Nex-N2-Pro.toml @@ -0,0 +1,15 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "nex-agi/nex-n2-pro" + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.5 +output = 2.5 +cache_read = 0.25 + +[limit] +output = 256_000 diff --git a/providers/siliconflow/models/tencent/Hy3-preview.toml b/providers/siliconflow/models/tencent/Hy3-preview.toml deleted file mode 100644 index fddc3766f5a..00000000000 --- a/providers/siliconflow/models/tencent/Hy3-preview.toml +++ /dev/null @@ -1,18 +0,0 @@ -base_model = "tencent/hy3-preview" -attachment = false -reasoning = true -reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] -open_weights = false - -[cost] -input = 0.066 -cache_read = 0.029 -output = 0.26 - -[limit] -context = 262_144 -output = 262_144 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/siliconflow/models/tencent/Hy3.toml b/providers/siliconflow/models/tencent/Hy3.toml new file mode 100644 index 00000000000..f304cbd9060 --- /dev/null +++ b/providers/siliconflow/models/tencent/Hy3.toml @@ -0,0 +1,15 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +# Limits below are this host's real deltas vs the lab entry (256K/128K): +# SiliconFlow advertises 262K total context and 262K max output +# (https://www.siliconflow.com/models/hy3). +base_model = "tencent/hy3" +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[cost] +input = 0.132 +cache_read = 0.033 +output = 0.528 + +[limit] +context = 262_144 +output = 262_144 diff --git a/providers/siliconflow/models/zai-org/GLM-5.1.toml b/providers/siliconflow/models/zai-org/GLM-5.1.toml index 5a341072c0d..248e165981b 100644 --- a/providers/siliconflow/models/zai-org/GLM-5.1.toml +++ b/providers/siliconflow/models/zai-org/GLM-5.1.toml @@ -1,3 +1,6 @@ +# Prices verified against SiliconFlow international, GLM-5.1 model page +# (https://www.siliconflow.com/models/glm-5-1, accessed 2026-09-08): +# input $1.19, cached input $0.6, output $3.74 per M tokens. name = "zai-org/GLM-5.1" description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" family = "glm" @@ -15,14 +18,14 @@ open_weights = true field = "reasoning_content" [cost] -input = 1.4 -output = 4.4 -cache_read = 0.26 +input = 1.19 +output = 3.74 +cache_read = 0.6 cache_write = 0 [limit] context = 205_000 -output = 205_000 +output = 131_000 [modalities] input = ["text"] diff --git a/providers/siliconflow/models/zai-org/GLM-5.2.toml b/providers/siliconflow/models/zai-org/GLM-5.2.toml index 1a6b9e26746..4f4b4d198e8 100644 --- a/providers/siliconflow/models/zai-org/GLM-5.2.toml +++ b/providers/siliconflow/models/zai-org/GLM-5.2.toml @@ -1,3 +1,6 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +# Cost values are the published USD list prices on SiliconFlow international +# (https://www.siliconflow.com/pricing, accessed 2026-09-08), not FX conversions by this PR. base_model = "zhipuai/glm-5.2" reasoning_options = [{ type = "effort", values = ["high", "max"] }] @@ -5,8 +8,8 @@ reasoning_options = [{ type = "effort", values = ["high", "max"] }] field = "reasoning_content" [cost] -input = 1.4 -output = 4.4 +input = 1.302 +output = 4.092 cache_read = 0.26 cache_write = 0 diff --git a/providers/siliconflow/models/zai-org/GLM-5.3-Flash.toml b/providers/siliconflow/models/zai-org/GLM-5.3-Flash.toml new file mode 100644 index 00000000000..afb2e9853b8 --- /dev/null +++ b/providers/siliconflow/models/zai-org/GLM-5.3-Flash.toml @@ -0,0 +1,24 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "zhipuai/glm-5.3-flash" + + +reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.03 +cache_write = 0 + +[limit] +context = 1_049_000 +output = 262_000 + +# SiliconFlow serves image input but not video/pdf for this model +# (model page "Support image input: Yes"): https://www.siliconflow.com/models/glm-5-3-flash +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/siliconflow/models/zai-org/GLM-5.3.toml b/providers/siliconflow/models/zai-org/GLM-5.3.toml new file mode 100644 index 00000000000..3de6268387c --- /dev/null +++ b/providers/siliconflow/models/zai-org/GLM-5.3.toml @@ -0,0 +1,16 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "zhipuai/glm-5.3" +reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1.4 +output = 4.4 +cache_read = 0.26 +cache_write = 0 + +[limit] +context = 1_049_000 +output = 262_000 From e64a2898d35f1f4b219fbd04b9d3017764e0298e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 15:27:31 +0000 Subject: [PATCH 195/392] chore(sync): update Eden AI model catalog (#7636) Co-authored-by: opencode-agent[bot] --- providers/edenai/models/deepseek/deepseek-v4-pro.toml | 2 +- .../infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml | 4 ++-- .../models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml | 4 ++-- providers/edenai/models/ionos/openai/gpt-oss-120b.toml | 4 ++-- providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml | 4 ++-- providers/edenai/models/scaleway/gpt-oss-120b.toml | 4 ++-- providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml | 4 ++-- 7 files changed, 13 insertions(+), 13 deletions(-) diff --git a/providers/edenai/models/deepseek/deepseek-v4-pro.toml b/providers/edenai/models/deepseek/deepseek-v4-pro.toml index 043e1bb1f41..05723a127e4 100644 --- a/providers/edenai/models/deepseek/deepseek-v4-pro.toml +++ b/providers/edenai/models/deepseek/deepseek-v4-pro.toml @@ -2,7 +2,7 @@ base_model = "deepseek/deepseek-v4-pro" [[reasoning_options]] type = "effort" -values = ["none", "high", "max"] +values = ["none", "low", "high", "max"] [cost] input = 0.66 diff --git a/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml b/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml index 9b2ca243154..beffb59e2ed 100644 --- a/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml +++ b/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml @@ -5,8 +5,8 @@ tool_call = false structured_output = false [cost] -input = 0.3438 -output = 0.4584 +input = 0.3447 +output = 0.4596 [limit] context = 100_000 diff --git a/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml b/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml index b5b5ca537eb..11dee7e0f74 100644 --- a/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml +++ b/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml @@ -4,5 +4,5 @@ tool_call = false structured_output = false [cost] -input = 0.7449 -output = 0.7449 +input = 0.74685 +output = 0.74685 diff --git a/providers/edenai/models/ionos/openai/gpt-oss-120b.toml b/providers/edenai/models/ionos/openai/gpt-oss-120b.toml index fb25f513d77..2bf7b7f0f21 100644 --- a/providers/edenai/models/ionos/openai/gpt-oss-120b.toml +++ b/providers/edenai/models/ionos/openai/gpt-oss-120b.toml @@ -8,5 +8,5 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.1719 -output = 0.7449 +input = 0.17235 +output = 0.74685 diff --git a/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml b/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml index ed1db7ea77d..d761571f13a 100644 --- a/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml +++ b/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml @@ -7,8 +7,8 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.4584 -output = 0.9168 +input = 0.4596 +output = 0.9192 [limit] context = 256_000 diff --git a/providers/edenai/models/scaleway/gpt-oss-120b.toml b/providers/edenai/models/scaleway/gpt-oss-120b.toml index df8da020b1b..dbff07167cc 100644 --- a/providers/edenai/models/scaleway/gpt-oss-120b.toml +++ b/providers/edenai/models/scaleway/gpt-oss-120b.toml @@ -7,8 +7,8 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.1719 -output = 0.6876 +input = 0.17235 +output = 0.6894 [limit] context = 128_000 diff --git a/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml b/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml index da84cd257a7..8d733529a6c 100644 --- a/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml +++ b/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml @@ -3,5 +3,5 @@ name = "Llama-3.3-70B-Instruct (Scaleway)" structured_output = false [cost] -input = 1.0314 -output = 1.0314 +input = 1.0341 +output = 1.0341 From 66635ce3d1efdad42d03046d1b58685956bc793d Mon Sep 17 00:00:00 2001 From: Kali Norby Date: Mon, 21 Sep 2026 08:30:37 -0700 Subject: [PATCH 196/392] Update neuralwatt catalog: add 8 models, deprecate retired GLM-5.2 family and V4 Pro (#7482) --- .../models/deepseek-v4-flash-flex.toml | 5 +-- .../models/deepseek-v4-flash-speed.toml | 26 ++++++++++++++ .../neuralwatt/models/deepseek-v4-flash.toml | 4 ++- .../neuralwatt/models/deepseek-v4-pro.toml | 11 +++--- .../models/deepseek-v4.1-flash-flex.toml | 26 ++++++++++++++ .../models/deepseek-v4.1-flash.toml | 25 +++++++++++++ providers/neuralwatt/models/glm-5.2-fast.toml | 3 ++ providers/neuralwatt/models/glm-5.2-flex.toml | 3 ++ .../models/glm-5.2-short-fast-flex.toml | 3 ++ .../neuralwatt/models/glm-5.2-short-fast.toml | 3 ++ .../neuralwatt/models/glm-5.2-short-flex.toml | 3 ++ .../neuralwatt/models/glm-5.2-short.toml | 3 ++ providers/neuralwatt/models/glm-5.2.toml | 3 ++ .../neuralwatt/models/glm-5.3-flash-flex.toml | 34 ++++++++++++++++++ .../neuralwatt/models/glm-5.3-flash.toml | 35 +++++++++++++++++++ providers/neuralwatt/models/glm-5.3-flex.toml | 30 ++++++++++++++++ providers/neuralwatt/models/glm-5.3.toml | 8 ++--- .../neuralwatt/models/qwen-3.8-27b-flex.toml | 33 +++++++++++++++++ providers/neuralwatt/models/qwen-3.8-27b.toml | 11 +++--- .../neuralwatt/models/qwen3.6-35b-flex.toml | 32 +++++++++++++++++ 20 files changed, 278 insertions(+), 23 deletions(-) create mode 100644 providers/neuralwatt/models/deepseek-v4-flash-speed.toml create mode 100644 providers/neuralwatt/models/deepseek-v4.1-flash-flex.toml create mode 100644 providers/neuralwatt/models/deepseek-v4.1-flash.toml create mode 100644 providers/neuralwatt/models/glm-5.3-flash-flex.toml create mode 100644 providers/neuralwatt/models/glm-5.3-flash.toml create mode 100644 providers/neuralwatt/models/glm-5.3-flex.toml create mode 100644 providers/neuralwatt/models/qwen-3.8-27b-flex.toml create mode 100644 providers/neuralwatt/models/qwen3.6-35b-flex.toml diff --git a/providers/neuralwatt/models/deepseek-v4-flash-flex.toml b/providers/neuralwatt/models/deepseek-v4-flash-flex.toml index 090d4d2f3f7..fae8619652d 100644 --- a/providers/neuralwatt/models/deepseek-v4-flash-flex.toml +++ b/providers/neuralwatt/models/deepseek-v4-flash-flex.toml @@ -1,5 +1,6 @@ # Flex-tier variant of deepseek-v4-flash: same model, context window -# (1,048,560), and output cap (65,536) as the standard tier; requested via the +# (1,048,560), and output cap (393,216, raised from 65,536 per GET /v1/models +# checked 2026-09-18) as the standard tier; requested via the # `-flex` model ID or service_tier="flex" and requires stream=true # (non-streaming requests fall through to the standard tier). # Cost is the flex rate: 0.65x of standard pricing (35% off) per the flex-tier @@ -26,4 +27,4 @@ cache_read = 0.0182 [limit] context = 1_048_560 -output = 65_536 +output = 393_216 diff --git a/providers/neuralwatt/models/deepseek-v4-flash-speed.toml b/providers/neuralwatt/models/deepseek-v4-flash-speed.toml new file mode 100644 index 00000000000..adf0fbd952c --- /dev/null +++ b/providers/neuralwatt/models/deepseek-v4-flash-speed.toml @@ -0,0 +1,26 @@ +# Speed variant serving the DeepSeek V4 Flash 0731 checkpoint: same pricing +# as deepseek-v4-flash ($0.14/$0.28/$0.028) with a 393,216-token output cap, +# per the public GET /v1/models metadata (checked 2026-09-18). Text-only +# input on this host. Reasoning matches the standard tier: reasoning_effort +# = none|high|max (default none, reasoning off by default; aliases +# low/minimal/medium -> high, xhigh -> max). Off is effort=none, so no +# separate toggle. thinking_token_budget is rejected with a 400 on this +# generation, so it is not declared. +# https://portal.neuralwatt.com/docs/api/chat-completions +# https://portal.neuralwatt.com/docs/api/models +base_model = "deepseek/deepseek-v4-flash-0731" +name = "DeepSeek V4 Flash (Speed)" +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high", "max"] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.028 + +[limit] +context = 1_048_560 +output = 393_216 diff --git a/providers/neuralwatt/models/deepseek-v4-flash.toml b/providers/neuralwatt/models/deepseek-v4-flash.toml index 5d30b1cbdca..4a32afaee6c 100644 --- a/providers/neuralwatt/models/deepseek-v4-flash.toml +++ b/providers/neuralwatt/models/deepseek-v4-flash.toml @@ -8,6 +8,8 @@ # is rejected with a 400 on this model, so it is not declared. # https://portal.neuralwatt.com/docs/api/models # https://portal.neuralwatt.com/pricing +# Output cap raised to 393,216 tokens per GET /v1/models (checked +# 2026-09-18; was 65,536). # https://portal.neuralwatt.com/docs/api/chat-completions base_model = "deepseek/deepseek-v4-flash" interleaved = true @@ -23,4 +25,4 @@ cache_read = 0.028 [limit] context = 1_048_560 -output = 65_536 +output = 393_216 diff --git a/providers/neuralwatt/models/deepseek-v4-pro.toml b/providers/neuralwatt/models/deepseek-v4-pro.toml index 7869e9bb888..7b3f48e037a 100644 --- a/providers/neuralwatt/models/deepseek-v4-pro.toml +++ b/providers/neuralwatt/models/deepseek-v4-pro.toml @@ -1,8 +1,6 @@ -# Preview rollout: deepseek-v4-pro is in private preview on Neuralwatt — -# visible to paid users after requesting access per model card, per the -# Preview Models guide. Pricing, output cap, and cache-read from the -# authenticated GET /v1/models metadata (checked 2026-08-27); V4-Pro is not -# yet in the public catalog or token-pricing table. +# Deprecated: Neuralwatt retired deepseek-v4-pro from GET /v1/models +# (checked 2026-09-18); it never left private preview. Retained for pricing +# history. # Reasoning per the chat-completions per-model table (docs verified against # the live API on 2026-08-26): reasoning_effort accepts none|low|high|max as # distinct levels with low as the default (reasoning on by default); aliases @@ -10,9 +8,8 @@ # accepted on this model (unlike deepseek-v4-flash, which rejects it); no # bounds or disable sentinel documented. # https://portal.neuralwatt.com/docs/api/chat-completions -# https://portal.neuralwatt.com/docs/guides/preview-models base_model = "deepseek/deepseek-v4-pro" -status = "beta" +status = "deprecated" interleaved = true # Effort: reasoning_effort = none|low|high|max (off is effort=none; aliases diff --git a/providers/neuralwatt/models/deepseek-v4.1-flash-flex.toml b/providers/neuralwatt/models/deepseek-v4.1-flash-flex.toml new file mode 100644 index 00000000000..a3dcc0edcd1 --- /dev/null +++ b/providers/neuralwatt/models/deepseek-v4.1-flash-flex.toml @@ -0,0 +1,26 @@ +# Flex-tier variant of deepseek-v4.1-flash: same model, context (1,048,560), +# vision, and effort ladder as the standard tier; requested via the `-flex` +# model ID or service_tier="flex" and requires stream=true (non-streaming +# requests fall through to the standard tier). Cost is the flex rate: 0.65x +# of standard pricing (35% off) per the flex-tier guide, applied to the +# standard rates advertised by GET /v1/models ($0.15/$0.6/$0.015, checked +# 2026-09-18). thinking_token_budget is rejected with a 400 on the V4-Flash +# generation and is not documented for V4.1-Flash, so it is not declared. +# https://portal.neuralwatt.com/docs/guides/flex-tier +# https://portal.neuralwatt.com/docs/api/chat-completions +base_model = "deepseek/deepseek-v4.1-flash" +name = "DeepSeek V4.1 Flash Flex" +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high", "xhigh", "max"] + +[cost] +input = 0.0975 +output = 0.39 +cache_read = 0.00975 + +[limit] +context = 1_048_560 +output = 393_216 diff --git a/providers/neuralwatt/models/deepseek-v4.1-flash.toml b/providers/neuralwatt/models/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..240f6508802 --- /dev/null +++ b/providers/neuralwatt/models/deepseek-v4.1-flash.toml @@ -0,0 +1,25 @@ +# Pricing, limits, vision, and effort ladder from the public GET /v1/models +# metadata (checked 2026-09-18): $0.15/$0.6/$0.015, 1,048,560-token context, +# 393,216-token output cap, 20-image cap. Effort: reasoning_effort = +# none|low|high|xhigh|max (default high, reasoning on by default; aliases +# medium -> high, minimal -> low). Off is effort=none, so no separate toggle. +# thinking_token_budget is rejected with a 400 on the V4-Flash generation +# (checked 2026-08-26) and is not documented for V4.1-Flash, so it is not +# declared. +# https://portal.neuralwatt.com/docs/api/chat-completions +# https://portal.neuralwatt.com/docs/api/models +base_model = "deepseek/deepseek-v4.1-flash" +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high", "xhigh", "max"] + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.015 + +[limit] +context = 1_048_560 +output = 393_216 diff --git a/providers/neuralwatt/models/glm-5.2-fast.toml b/providers/neuralwatt/models/glm-5.2-fast.toml index a61cf125702..241dd01adcd 100644 --- a/providers/neuralwatt/models/glm-5.2-fast.toml +++ b/providers/neuralwatt/models/glm-5.2-fast.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Fast" @@ -5,6 +7,7 @@ description = "Efficient GLM model for fast reasoning, coding, and agent workflo release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2-flex.toml b/providers/neuralwatt/models/glm-5.2-flex.toml index e42b7fb1cc5..b8fce8e6ba2 100644 --- a/providers/neuralwatt/models/glm-5.2-flex.toml +++ b/providers/neuralwatt/models/glm-5.2-flex.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Flex" @@ -5,6 +7,7 @@ description = "Flagship GLM model for hybrid reasoning, coding, and agentic engi release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2-short-fast-flex.toml b/providers/neuralwatt/models/glm-5.2-short-fast-flex.toml index eea25d5f58e..f463c21d078 100644 --- a/providers/neuralwatt/models/glm-5.2-short-fast-flex.toml +++ b/providers/neuralwatt/models/glm-5.2-short-fast-flex.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Short Fast Flex" @@ -5,6 +7,7 @@ description = "Efficient GLM model for fast reasoning, coding, and agent workflo release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2-short-fast.toml b/providers/neuralwatt/models/glm-5.2-short-fast.toml index fa48e039e59..20da2db40c4 100644 --- a/providers/neuralwatt/models/glm-5.2-short-fast.toml +++ b/providers/neuralwatt/models/glm-5.2-short-fast.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Short Fast" @@ -5,6 +7,7 @@ description = "Efficient GLM model for fast reasoning, coding, and agent workflo release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2-short-flex.toml b/providers/neuralwatt/models/glm-5.2-short-flex.toml index 604df60b11d..0b1cc1ed110 100644 --- a/providers/neuralwatt/models/glm-5.2-short-flex.toml +++ b/providers/neuralwatt/models/glm-5.2-short-flex.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Short Flex" @@ -5,6 +7,7 @@ description = "Flagship GLM model for hybrid reasoning, coding, and agentic engi release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2-short.toml b/providers/neuralwatt/models/glm-5.2-short.toml index 26127431e9c..a324bae38c5 100644 --- a/providers/neuralwatt/models/glm-5.2-short.toml +++ b/providers/neuralwatt/models/glm-5.2-short.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Short" @@ -5,6 +7,7 @@ description = "Flagship GLM model for hybrid reasoning, coding, and agentic engi release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2.toml b/providers/neuralwatt/models/glm-5.2.toml index 0b7b1191535..b83cbca3c04 100644 --- a/providers/neuralwatt/models/glm-5.2.toml +++ b/providers/neuralwatt/models/glm-5.2.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2" @@ -5,6 +7,7 @@ description = "Flagship GLM model for hybrid reasoning, coding, and agentic engi release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.3-flash-flex.toml b/providers/neuralwatt/models/glm-5.3-flash-flex.toml new file mode 100644 index 00000000000..502beae0882 --- /dev/null +++ b/providers/neuralwatt/models/glm-5.3-flash-flex.toml @@ -0,0 +1,34 @@ +# Flex-tier variant of glm-5.3-flash: same model, context (1,048,560), +# vision, and effort ladder as the standard tier; requested via the `-flex` +# model ID or service_tier="flex" and requires stream=true (non-streaming +# requests fall through to the standard tier). Cost is the flex rate: 0.65x +# of standard pricing (35% off) per the flex-tier guide, applied to the +# standard rates advertised by GET /v1/models ($0.15/$0.5/$0.03, checked +# 2026-09-18). JSON mode is off on this host, unlike the lab default. +# https://portal.neuralwatt.com/docs/guides/flex-tier +# https://portal.neuralwatt.com/docs/api/chat-completions +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3 Flash Flex" +structured_output = false + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.0975 +output = 0.325 +cache_read = 0.0195 + +[limit] +context = 1_048_560 +output = 1_048_560 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/neuralwatt/models/glm-5.3-flash.toml b/providers/neuralwatt/models/glm-5.3-flash.toml new file mode 100644 index 00000000000..8f87fce7fcc --- /dev/null +++ b/providers/neuralwatt/models/glm-5.3-flash.toml @@ -0,0 +1,35 @@ +# Exited preview 2026-09-11. Pricing, vision, and effort ladder from the +# public GET /v1/models metadata (checked 2026-09-18): $0.15/$0.5/$0.03, +# 1,048,560-token context, 20-image cap. Reasoning cannot be disabled +# (reasoning.mandatory = true), so there is no none level and no toggle; +# low, high and max are distinct efforts, with minimal/medium/xhigh accepted +# as aliases onto them. Budget: thinking_token_budget (integer reasoning +# tokens). Neuralwatt serves text+image input only (lab metadata also lists +# video/pdf). JSON mode is off on this host, unlike the lab default. +# https://portal.neuralwatt.com/docs/api/chat-completions +# https://portal.neuralwatt.com/models/glm-5.3-flash +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3 Flash" +structured_output = false + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.03 + +[limit] +context = 1_048_560 +output = 1_048_560 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/neuralwatt/models/glm-5.3-flex.toml b/providers/neuralwatt/models/glm-5.3-flex.toml new file mode 100644 index 00000000000..fa4a8aaef2c --- /dev/null +++ b/providers/neuralwatt/models/glm-5.3-flex.toml @@ -0,0 +1,30 @@ +# Flex-tier variant of glm-5.3: same model, context (1,048,560), and effort +# ladder as the standard tier; requested via the `-flex` model ID or +# service_tier="flex" and requires stream=true (non-streaming requests fall +# through to the standard tier). Cost is the flex rate: 0.65x of standard +# pricing (35% off) per the flex-tier guide, applied to the standard rates +# advertised by GET /v1/models ($1.45/$4.5/$0.145, checked 2026-09-18). +# Budget: thinking_token_budget (integer reasoning tokens) +# https://portal.neuralwatt.com/docs/guides/flex-tier +# https://portal.neuralwatt.com/docs/api/chat-completions +base_model = "zhipuai/glm-5.3" +name = "GLM 5.3 Flex" +structured_output = false + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.9425 +output = 2.925 +cache_read = 0.09425 + +[limit] +context = 1_048_560 +output = 1_048_560 diff --git a/providers/neuralwatt/models/glm-5.3.toml b/providers/neuralwatt/models/glm-5.3.toml index 700f54f026e..5e77d0a249d 100644 --- a/providers/neuralwatt/models/glm-5.3.toml +++ b/providers/neuralwatt/models/glm-5.3.toml @@ -1,17 +1,13 @@ -# Preview rollout: glm-5.3 is a gated preview on Neuralwatt, visible to -# accounts granted access per the Preview Models guide. Its list price is -# Neuralwatt's GLM 5.2 parity default and is marked for review at launch, -# per the model description in GET /v1/models (checked 2026-08-30). +# Exited preview 2026-09-10; the preview-parity list price ($1.45/$4.5/$0.145) +# was confirmed as the launch price per GET /v1/models (checked 2026-09-18). # Reasoning cannot be disabled (reasoning.mandatory = true), so there is no # none level and no toggle. The API supports low, high and max as distinct # efforts; minimal, medium and xhigh are accepted as aliases onto them. # Budget: thinking_token_budget (integer reasoning tokens) # https://portal.neuralwatt.com/docs/api/chat-completions -# https://portal.neuralwatt.com/docs/guides/preview-models base_model = "zhipuai/glm-5.3" name = "GLM 5.3" structured_output = false -status = "beta" interleaved = true diff --git a/providers/neuralwatt/models/qwen-3.8-27b-flex.toml b/providers/neuralwatt/models/qwen-3.8-27b-flex.toml new file mode 100644 index 00000000000..7021462ac7f --- /dev/null +++ b/providers/neuralwatt/models/qwen-3.8-27b-flex.toml @@ -0,0 +1,33 @@ +# Flex-tier variant of qwen-3.8-27b: same model, context (262,128), vision, +# and effort ladder as the standard tier; requested via the `-flex` model ID +# or service_tier="flex" and requires stream=true (non-streaming requests +# fall through to the standard tier). Cost is the flex rate: 0.65x of +# standard pricing (35% off) per the flex-tier guide, applied to the standard +# rates advertised by GET /v1/models ($0.45/$3.2/$0.25, checked 2026-09-18). +# Budget: thinking_token_budget (integer reasoning tokens) +# Neuralwatt serves text+image input only (lab metadata also lists video). +# https://portal.neuralwatt.com/docs/guides/flex-tier +# https://portal.neuralwatt.com/docs/api/chat-completions +base_model = "alibaba/qwen3.8-27b" +name = "Qwen3.8 27B Flex" +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.2925 +output = 2.08 +cache_read = 0.1625 + +[limit] +context = 262_128 +output = 131_072 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/neuralwatt/models/qwen-3.8-27b.toml b/providers/neuralwatt/models/qwen-3.8-27b.toml index baa1ccb1e23..e2749dc1438 100644 --- a/providers/neuralwatt/models/qwen-3.8-27b.toml +++ b/providers/neuralwatt/models/qwen-3.8-27b.toml @@ -1,6 +1,5 @@ -# Preview rollout: qwen-3.8-27b is in preview on Neuralwatt — early-access -# program, granted via the portal enroll page; absent from the public -# /v1/models catalog until access is granted. +# Left preview and joined the public GET /v1/models catalog; the output cap +# was raised from 65,536 to 131,072 tokens (checked 2026-09-18). # Effort: reasoning_effort = none|low|medium|xhigh (default xhigh, reasoning # on by default); aliases max/xhigh/high -> xhigh and minimal -> low, per the # chat-completions per-model table (docs verified against the live API on @@ -9,12 +8,10 @@ # model. # Neuralwatt serves text+image input only (lab metadata also lists video). # Pricing, context, output cap, and effort ladder from the model page and the -# authenticated GET /v1/models metadata (checked 2026-08-27): +# GET /v1/models metadata (checked 2026-09-18): # https://portal.neuralwatt.com/models/qwen-3.8-27b # https://portal.neuralwatt.com/docs/api/chat-completions -# https://portal.neuralwatt.com/docs/guides/preview-models base_model = "alibaba/qwen3.8-27b" -status = "beta" interleaved = true [[reasoning_options]] @@ -31,7 +28,7 @@ cache_read = 0.25 [limit] context = 262_128 -output = 65_536 +output = 131_072 [modalities] input = ["text", "image"] diff --git a/providers/neuralwatt/models/qwen3.6-35b-flex.toml b/providers/neuralwatt/models/qwen3.6-35b-flex.toml new file mode 100644 index 00000000000..3f8bd238eca --- /dev/null +++ b/providers/neuralwatt/models/qwen3.6-35b-flex.toml @@ -0,0 +1,32 @@ +# Flex-tier variant of qwen3.6-35b: same model, context (131,056), vision, +# and effort ladder as the standard tier; requested via the `-flex` model ID +# or service_tier="flex" and requires stream=true (non-streaming requests +# fall through to the standard tier). Cost is the flex rate: 0.65x of +# standard pricing (35% off) per the flex-tier guide, applied to the standard +# rates advertised by GET /v1/models ($0.29/$1.15/$0.029, checked 2026-09-18). +# Budget: thinking_token_budget (integer reasoning tokens) +# https://portal.neuralwatt.com/docs/guides/flex-tier +# https://portal.neuralwatt.com/docs/api/chat-completions +base_model = "alibaba/qwen3.6-35b-a3b" +name = "Qwen3.6 35B Flex" + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1885 +output = 0.7475 +cache_read = 0.01885 + +[limit] +context = 131_056 +output = 131_056 + +[modalities] +input = ["text", "image"] From 6c97e53166f543afe27be8cf75d99fa64b171a0a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 10:30:43 -0500 Subject: [PATCH 197/392] chore(sync): update Hugging Face model catalog (#7562) * chore(sync): update Hugging Face model catalog * fix(huggingface): add Hy4 reasoning options --------- Co-authored-by: opencode-agent[bot] Co-authored-by: rekram1-node --- .../huggingface/models/tencent/Hy4-preview.toml | 13 +++++++++++++ 1 file changed, 13 insertions(+) create mode 100644 providers/huggingface/models/tencent/Hy4-preview.toml diff --git a/providers/huggingface/models/tencent/Hy4-preview.toml b/providers/huggingface/models/tencent/Hy4-preview.toml new file mode 100644 index 00000000000..5a2ca915a5b --- /dev/null +++ b/providers/huggingface/models/tencent/Hy4-preview.toml @@ -0,0 +1,13 @@ +# Effort: reasoning_effort = none|high; high is the default. +# https://huggingface.co/tencent/Hy4-preview#quickstart +base_model = "tencent/hy4-preview" +description = "Tencent Hy reasoning model for coding, instruction following, and agent tasks" +structured_output = true +reasoning_options = [{ type = "effort", values = ["none", "high"] }] + +[cost] +input = 0.834 +output = 2.501 + +[limit] +context = 1_000_000 From 7c89bbad28ce8171e9f4853fed74edd8c8bf47a6 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 10:36:03 -0500 Subject: [PATCH 198/392] fix: [missing-model] xai: grok-imagine-image-quality (#7391) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- models/xai/grok-imagine-image-quality.toml | 26 +++++++++++++++++++ .../models/grok-imagine-image-quality.toml | 8 ++++++ 2 files changed, 34 insertions(+) create mode 100644 models/xai/grok-imagine-image-quality.toml create mode 100644 providers/xai/models/grok-imagine-image-quality.toml diff --git a/models/xai/grok-imagine-image-quality.toml b/models/xai/grok-imagine-image-quality.toml new file mode 100644 index 00000000000..46eb1539995 --- /dev/null +++ b/models/xai/grok-imagine-image-quality.toml @@ -0,0 +1,26 @@ +# Sources: +# - https://docs.x.ai/docs/models +# - https://docs.x.ai/developers/models/grok-imagine-image-quality +# - https://docs.x.ai/developers/pricing +# - https://docs.x.ai/docs/guides/image-generation +# Pricing: $0.05/image (1K), $0.07/image (2K); image input $0.01/image (not token-based; no [cost] authored) +# Release: dated alias grok-imagine-image-quality-20260403; also aliased as grok-imagine-image-quality-latest, grok-imagine-image-pro + +name = "Grok Imagine Image Quality" +description = "Higher-fidelity Grok Imagine image model for prompt-driven generation, editing, and visual design workflows" +family = "grok" +release_date = "2026-04-03" +last_updated = "2026-04-03" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 16_000 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["image"] diff --git a/providers/xai/models/grok-imagine-image-quality.toml b/providers/xai/models/grok-imagine-image-quality.toml new file mode 100644 index 00000000000..caf513b6e28 --- /dev/null +++ b/providers/xai/models/grok-imagine-image-quality.toml @@ -0,0 +1,8 @@ +# Sources: +# - https://docs.x.ai/docs/models +# - https://docs.x.ai/developers/models/grok-imagine-image-quality +# - https://docs.x.ai/developers/pricing +# Pricing: $0.05/image (1K), $0.07/image (2K); image input $0.01/image (not token-based; no [cost] authored) +# Aliases: grok-imagine-image-quality-20260403, grok-imagine-image-quality-latest, grok-imagine-image-pro + +base_model = "xai/grok-imagine-image-quality" From 3afe1fe9568450183252db8300b85c1de0887368 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:36:11 +0800 Subject: [PATCH 199/392] feat(aihubmix): add gpt-6-astra (#7422) AIHubMix serves this route but the catalog carried no aihubmix entry for it. Cost comes from the provider's public model listing and the reasoning controls from its published model-data index; limit, modalities, attachment and tool_call inherit from models/openai/gpt-6-astra.toml rather than being repeated here. Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- providers/aihubmix/models/gpt-6-astra.toml | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) create mode 100644 providers/aihubmix/models/gpt-6-astra.toml diff --git a/providers/aihubmix/models/gpt-6-astra.toml b/providers/aihubmix/models/gpt-6-astra.toml new file mode 100644 index 00000000000..b005bff58b8 --- /dev/null +++ b/providers/aihubmix/models/gpt-6-astra.toml @@ -0,0 +1,22 @@ +# Effort: low|medium|high|xhigh|max +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "openai/gpt-6-astra" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 20 +output = 75 +cache_read = 2 +cache_write = 25 From 7815e59aed1c50ab5eb312b47502894b7a56aea3 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:36:21 +0800 Subject: [PATCH 200/392] feat(aihubmix): add gemini-3.8-flash (#7423) * feat(aihubmix): add gemini-3.8-flash AIHubMix serves this route but the catalog carried no aihubmix entry for it. Cost comes from the provider's public model listing and the reasoning controls from its published model-data index; limit, modalities, attachment and tool_call inherit from models/google/gemini-3.8-flash.toml rather than being repeated here. Co-Authored-By: Claude Opus 5 * aihubmix/gemini-3.8-flash: drop budget_tokens from reasoning_options The bare budget_tokens entry came from a generic multi-protocol field dump, not from model-level evidence: first-party providers/google/models/gemini-3.8-flash.toml and the AIHubMix peer gemini-3.7-flash are both effort-only, and Google's documented reasoning-token budgets apply to the 2.5 family, not 3.8. Effort low|medium|high is unchanged and is the model-level control. The leading comment loses the budget field paths along with it. Co-Authored-By: Claude Opus 5 --------- Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- providers/aihubmix/models/gemini-3.8-flash.toml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 providers/aihubmix/models/gemini-3.8-flash.toml diff --git a/providers/aihubmix/models/gemini-3.8-flash.toml b/providers/aihubmix/models/gemini-3.8-flash.toml new file mode 100644 index 00000000000..9657ff10785 --- /dev/null +++ b/providers/aihubmix/models/gemini-3.8-flash.toml @@ -0,0 +1,14 @@ +# Effort: low|medium|high +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.8-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 From cd17c40386ecc955e977b534a3d1fa399b22668b Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:36:28 +0800 Subject: [PATCH 201/392] feat(aihubmix): add hy4-preview (#7424) AIHubMix serves this route but the catalog carried no aihubmix entry for it. Cost comes from the provider's public model listing and the reasoning controls from its published model-data index; limit, modalities, attachment and tool_call inherit from models/tencent/hy4-preview.toml rather than being repeated here. Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- providers/aihubmix/models/hy4-preview.toml | 28 ++++++++++++++++++++++ 1 file changed, 28 insertions(+) create mode 100644 providers/aihubmix/models/hy4-preview.toml diff --git a/providers/aihubmix/models/hy4-preview.toml b/providers/aihubmix/models/hy4-preview.toml new file mode 100644 index 00000000000..dc0f2ad6ac7 --- /dev/null +++ b/providers/aihubmix/models/hy4-preview.toml @@ -0,0 +1,28 @@ +# Off is effort=none; graded levels — no toggle. The same off elsewhere: +# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11); +# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path. +# Effort: none|high +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# Budget: +# integer $.reasoning.max_tokens on /v1/chat/completions; $.thinking.budget_tokens >= 1024 on /v1/messages; +# integer $.generationConfig.thinkingConfig.thinkingBudget on the Gemini path (-1 dynamic, 0 off where supported); +# the Responses path carries effort but has no reasoning-token budget field. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "tencent/hy4-preview" +structured_output = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.845 +output = 2.535 +cache_read = 0.04225 From 88245969a40fc5610915ffd9fcbb3f162d26d3e0 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:36:37 +0800 Subject: [PATCH 202/392] feat(aihubmix): add muse-spark-1.3 (#7425) * feat(aihubmix): add muse-spark-1.3 AIHubMix serves this route but the catalog carried no aihubmix entry for it. Cost comes from the provider's public model listing and the reasoning controls from its published model-data index; limit, modalities, attachment and tool_call inherit from models/meta/muse-spark-1.3.toml rather than being repeated here. Co-Authored-By: Claude Opus 5 * aihubmix/muse-spark-1.3: drop budget_tokens from reasoning_options The bare budget_tokens entry came from a generic multi-protocol field dump rather than a model-level control: first-party Meta and the established relays of meta/muse-spark-1.3 are effort-only, and the same-host AIHubMix effort-only peers omit it as well. The effort set stays at minimal|low|medium|high|xhigh. `max` is deliberately not listed: Meta's 1.3 launch post says "Previously available reasoning modes are available today with max reasoning coming shortly after we finish additional safety testing", so the "Muse Spark 1.3 (max)" benchmark label is not an API availability statement yet. Co-Authored-By: Claude Opus 5 --------- Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- providers/aihubmix/models/muse-spark-1.3.toml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 providers/aihubmix/models/muse-spark-1.3.toml diff --git a/providers/aihubmix/models/muse-spark-1.3.toml b/providers/aihubmix/models/muse-spark-1.3.toml new file mode 100644 index 00000000000..21edfd6c678 --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.3.toml @@ -0,0 +1,14 @@ +# Effort: minimal|low|medium|high|xhigh +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "meta/muse-spark-1.3" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[cost] +input = 1.375 +output = 4.675 +cache_read = 0.165 From 96655bf2b6494f39d2203a91cc97e532a2ae3e27 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:36:44 +0800 Subject: [PATCH 203/392] feat(aihubmix): add deepseek-v4-flash-0731-fast (#7429) AIHubMix serves this route but the catalog carried no aihubmix entry for it. Cost comes from the provider's public model listing and the reasoning controls from its published model-data index; limit, modalities, attachment and tool_call inherit from models/deepseek/deepseek-v4-flash-0731.toml rather than being repeated here. Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- .../models/deepseek-v4-flash-0731-fast.toml | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) create mode 100644 providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml diff --git a/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml b/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml new file mode 100644 index 00000000000..db3b1b7c1e9 --- /dev/null +++ b/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml @@ -0,0 +1,24 @@ +# Toggle: +# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11); +# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path. +# Effort: low|high|max +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "deepseek/deepseek-v4-flash-0731" +name = "DeepSeek V4 Flash 0731 Fast" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.28 +output = 1.4 +cache_read = 0.07 From 63cca5fee192c2f25b7876af96cb9a0093250d94 Mon Sep 17 00:00:00 2001 From: Kastan Day Date: Mon, 21 Sep 2026 08:37:35 -0700 Subject: [PATCH 204/392] fix(cloudflare): correct catalogue reasoning efforts & `base_model`s (#7454) * fix(cloudflare): refresh catalogue reasoning and canonical links * fix(cloudflare): sync GLM 4.7 toggle and expand reasoning tables * fix(cloudflare): cap oversized context and output limits * fix(cloudflare): inherit SEA-LION description and clarify reasoning sources * docs(cloudflare): cite GLM thinking control at its catalogue entry * docs(cloudflare): cite dated DeepSeek effort controls * fix(cloudflare): omit fixed Kimi K2.7 effort control --- .../aisingapore/gemma-sea-lion-v4-27b-it.toml | 18 +----------- .../deepseek-ai/deepseek-v4-flash-0731.toml | 15 +++++----- .../@cf/deepseek-ai/deepseek-v4-pro-0813.toml | 10 +++++-- .../models/@cf/google/gemma-4-26b-a4b-it.toml | 12 +++++--- .../@cf/ibm-granite/granite-4.0-h-micro.toml | 13 +-------- .../mistral-small-3.1-24b-instruct.toml | 10 +------ .../models/@cf/moonshotai/kimi-k2.6.toml | 12 +++++--- .../models/@cf/moonshotai/kimi-k2.7-code.toml | 4 ++- .../@cf/nvidia/nemotron-3-120b-a12b.toml | 8 ++++- .../models/@cf/qwen/qwen3.8-27b.toml | 8 ++++- .../models/@cf/zai-org/glm-4.7-flash.toml | 29 ++++++------------- .../models/@cf/zai-org/glm-5.2.toml | 7 ++--- .../models/@cf/zai-org/glm-5.3-flash.toml | 23 ++++++--------- .../models/@cf/zai-org/glm-5.3.toml | 13 +++++++-- 14 files changed, 82 insertions(+), 100 deletions(-) diff --git a/providers/cloudflare-workers-ai/models/@cf/aisingapore/gemma-sea-lion-v4-27b-it.toml b/providers/cloudflare-workers-ai/models/@cf/aisingapore/gemma-sea-lion-v4-27b-it.toml index 8976b92b55a..3fe191a4d99 100644 --- a/providers/cloudflare-workers-ai/models/@cf/aisingapore/gemma-sea-lion-v4-27b-it.toml +++ b/providers/cloudflare-workers-ai/models/@cf/aisingapore/gemma-sea-lion-v4-27b-it.toml @@ -1,23 +1,7 @@ +base_model = "aisingapore/gemma-sea-lion-v4-27b-it" name = "Gemma Sea Lion V4 27B It" -description = "Open Gemma instruction model for efficient chat and self-hosted deployments" -family = "gemma" -release_date = "2025-09-23" -last_updated = "2025-09-23" -attachment = false -reasoning = false -temperature = true -tool_call = false structured_output = false -open_weights = true [cost] input = 0.351 output = 0.555 - -[limit] -context = 128_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml index b72f1489e71..2385a1bc60b 100644 --- a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml +++ b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml @@ -1,13 +1,14 @@ -# Toggle: thinking.type = enabled|disabled. -# Effort: reasoning_effort = high|max. +# Effort: reasoning_effort = none|low|high|max; none disables thinking. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# Context/output capped to Cloudflare Workers AI context window (1,048,576). +# Search reports 1,310,720, exceeding AI Gateway max_completion_tokens. +# This dated release adds a distinct low effort; the older preview does not. +# Creator: https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731#chat-template base_model = "deepseek/deepseek-v4-flash-0731" -[[reasoning_options]] -type = "toggle" - [[reasoning_options]] type = "effort" -values = ["high", "max"] +values = ["none", "low", "high", "max"] [cost] input = 0.44 @@ -15,5 +16,5 @@ output = 1.32 cache_read = 0.014 [limit] -context = 1_310_720 +context = 1_048_576 output = 1_048_576 diff --git a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-pro-0813.toml b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-pro-0813.toml index b28c06ffbe1..32c7c6708b0 100644 --- a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-pro-0813.toml +++ b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-pro-0813.toml @@ -1,8 +1,12 @@ -# Toggle: thinking.type = enabled|disabled. -# Effort: reasoning_effort = high|max. +# Effort: reasoning_effort = none|low|high|max; none disables thinking. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# This dated release adds a distinct low effort; the older preview does not. +# Creator: https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813#chat-template base_model = "deepseek/deepseek-v4-pro-0813" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["high", "max"] }] +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high", "max"] [cost] input = 1.32 diff --git a/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml b/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml index 817334b2cee..2c6a8e4386f 100644 --- a/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml +++ b/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml @@ -1,11 +1,15 @@ -# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and -# `chat_template_kwargs.enable_thinking = true|false`; no budget is documented. -# https://developers.cloudflare.com/workers-ai/models/gemma-4-26b-a4b-it/sync-input.json (accessed 2026-06-25) +# Effort: reasoning_effort = none|high; none disables thinking. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# Per-model Search config normalizes low/medium to high; these are aliases, +# not distinct levels despite the generic sync-input.json effort enum. base_model = "google/gemma-4-26b-a4b-it" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] interleaved = true +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + [cost] input = 0.1 output = 0.3 diff --git a/providers/cloudflare-workers-ai/models/@cf/ibm-granite/granite-4.0-h-micro.toml b/providers/cloudflare-workers-ai/models/@cf/ibm-granite/granite-4.0-h-micro.toml index a43a661e427..1aa972501cf 100644 --- a/providers/cloudflare-workers-ai/models/@cf/ibm-granite/granite-4.0-h-micro.toml +++ b/providers/cloudflare-workers-ai/models/@cf/ibm-granite/granite-4.0-h-micro.toml @@ -1,14 +1,7 @@ +base_model = "ibm/granite-4-h-micro" name = "Granite 4.0 H Micro" description = "Efficient model for low-latency assistance, extraction, and routine automation" -family = "granite" -release_date = "2025-10-07" -last_updated = "2025-10-07" -attachment = false -reasoning = false -temperature = true -tool_call = true structured_output = false -open_weights = true [cost] input = 0.017 @@ -17,7 +10,3 @@ output = 0.112 [limit] context = 131_000 output = 131_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/cloudflare-workers-ai/models/@cf/mistralai/mistral-small-3.1-24b-instruct.toml b/providers/cloudflare-workers-ai/models/@cf/mistralai/mistral-small-3.1-24b-instruct.toml index e456a0e8625..7ed149c081e 100644 --- a/providers/cloudflare-workers-ai/models/@cf/mistralai/mistral-small-3.1-24b-instruct.toml +++ b/providers/cloudflare-workers-ai/models/@cf/mistralai/mistral-small-3.1-24b-instruct.toml @@ -1,23 +1,15 @@ +base_model = "mistral/mistral-small-3-1-24b-instruct-2503" name = "Mistral Small 3.1 24B Instruct" description = "Efficient Mistral model for fast chat, extraction, and production assistants" -family = "mistral-small" -release_date = "2025-03-18" -last_updated = "2025-03-18" attachment = false -reasoning = false -temperature = true -tool_call = true structured_output = false -open_weights = true [cost] input = 0.351 output = 0.555 [limit] -context = 128_000 output = 128_000 [modalities] input = ["text"] -output = ["text"] diff --git a/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml b/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml index 3c4c22a8305..6148d7218ca 100644 --- a/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml +++ b/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml @@ -1,9 +1,13 @@ -# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and -# `chat_template_kwargs.thinking = true|false`; no budget is documented. -# https://developers.cloudflare.com/workers-ai/models/kimi-k2.6/sync-input.json (accessed 2026-06-25) +# Effort: reasoning_effort = none|high; none selects instant mode. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# Per-model Search config normalizes low/medium to high; these are aliases, +# not distinct levels despite the generic sync-input.json effort enum. base_model = "moonshotai/kimi-k2.6" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] [interleaved] field = "reasoning_content" diff --git a/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.7-code.toml b/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.7-code.toml index a06c8bc9a29..43a8509e3a6 100644 --- a/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.7-code.toml +++ b/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.7-code.toml @@ -1,6 +1,8 @@ +# Thinking is mandatory; high is the only effective effort, so aliases add no caller control. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). base_model = "moonshotai/kimi-k2.7-code" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] temperature = true +reasoning_options = [] [cost] input = 0.95 diff --git a/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml b/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml index 554decfc2e3..a75a99e22b1 100644 --- a/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml @@ -5,9 +5,15 @@ base_model = "nvidia/nemotron-3-super-120b-a12b" name = "Nemotron 3 Super 120B" structured_output = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] interleaved = true +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.5 output = 1.5 diff --git a/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3.8-27b.toml b/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3.8-27b.toml index 307aa38fb79..f491dd473a3 100644 --- a/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3.8-27b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3.8-27b.toml @@ -5,7 +5,13 @@ base_model = "alibaba/qwen3.8-27b" description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "xhigh"] }] + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] [cost] input = 0.45 diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml index 197fa7c94b5..37f0d8a203e 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml @@ -1,32 +1,21 @@ -# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and -# `chat_template_kwargs.enable_thinking = true|false`; no budget is documented. -# https://developers.cloudflare.com/workers-ai/models/glm-4.7-flash/sync-input.json (accessed 2026-06-25) - -name = "GLM-4.7-Flash" +# Toggle: chat_template_kwargs.enable_thinking = true|false (default: true). +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# Workers AI's per-model config exposes only the binary thinking toggle; +# the generic sync-input.json low/medium/high enum is not an effort selector here. +# Creator template: https://huggingface.co/zai-org/GLM-4.7-Flash/blob/main/chat_template.jinja +base_model = "zhipuai/glm-4.7-flash" description = "Efficient GLM model for fast reasoning, coding, and agent workflows" -family = "glm-flash" -release_date = "2026-01-19" -last_updated = "2026-01-19" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true structured_output = true -knowledge = "2025-04" -open_weights = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.0605 output = 0.4 [limit] context = 131_072 -output = 131_072 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.2.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.2.toml index 5642ab6e4ea..48337ad55f9 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.2.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.2.toml @@ -1,14 +1,13 @@ +# Effort: reasoning_effort = none|high|max; none disables thinking. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). # Context and output limits verified against Cloudflare Workers AI's OpenAI-compatible endpoint. # https://developers.cloudflare.com/workers-ai/models/glm-5.2/ base_model = "zhipuai/glm-5.2" name = "Glm 5.2" -[[reasoning_options]] -type = "toggle" - [[reasoning_options]] type = "effort" -values = ["low", "medium", "high"] +values = ["none", "high", "max"] [cost] input = 1.4 diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml index 000ebf8a49a..83a85d6c195 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml @@ -1,18 +1,14 @@ +# Effort: reasoning_effort = low|high|max; thinking is mandatory. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). # Context/output capped to Cloudflare Workers AI context window (1,048,576). -# Catalog/search previously reported 1,310,720, which exceeds AI Gateway max_completion_tokens. -# https://developers.cloudflare.com/workers-ai/models/glm-5.3-flash/ +# Search reports 1,310,720, exceeding AI Gateway max_completion_tokens. +base_model = "zhipuai/glm-5.3-flash" name = "Glm 5.3 Flash" description = "GLM vision model for visual reasoning, documents, and multimodal agents" -family = "glm-flash" -release_date = "2026-08-26" -last_updated = "2026-08-26" -attachment = true -reasoning = true -temperature = true -tool_call = true -structured_output = true -open_weights = true -reasoning_options = [] + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] [cost] input = 0.15 @@ -20,9 +16,8 @@ output = 0.5 cache_read = 0.03 [limit] -context = 1_310_720 +context = 1_048_576 output = 1_048_576 [modalities] input = ["text", "image"] -output = ["text"] diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml index a5f20bbaf80..a66154848f3 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml @@ -1,7 +1,14 @@ +# Effort: reasoning_effort = low|high|max; thinking is mandatory. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# Context/output capped to Cloudflare Workers AI context window (1,048,576). +# Search reports 1,310,720, exceeding AI Gateway max_completion_tokens. base_model = "zhipuai/glm-5.3" name = "Glm 5.3" description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] [cost] input = 1.4 @@ -9,5 +16,5 @@ output = 4.4 cache_read = 0.26 [limit] -context = 1_310_720 -output = 1_310_720 +context = 1_048_576 +output = 1_048_576 From e52dfc2abcc74fcb39c9a2bd7736f998517dc518 Mon Sep 17 00:00:00 2001 From: Daniel Chen Date: Mon, 21 Sep 2026 08:45:33 -0700 Subject: [PATCH 205/392] feat(opencode):added grok-4.7 to go and zen --- models/xai/grok-4.7.toml | 20 ++++++++++++++++++++ providers/opencode-go/models/grok-4.7.toml | 19 +++++++++++++++++++ 2 files changed, 39 insertions(+) create mode 100644 models/xai/grok-4.7.toml create mode 100644 providers/opencode-go/models/grok-4.7.toml diff --git a/models/xai/grok-4.7.toml b/models/xai/grok-4.7.toml new file mode 100644 index 00000000000..1142f86cd13 --- /dev/null +++ b/models/xai/grok-4.7.toml @@ -0,0 +1,20 @@ +name = "Grok 4.7" +description = "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects" +family = "grok" +knowledge = "2026-02-01" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[limit] +context = 500_000 +output = 500_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/opencode-go/models/grok-4.7.toml b/providers/opencode-go/models/grok-4.7.toml new file mode 100644 index 00000000000..edde035f699 --- /dev/null +++ b/providers/opencode-go/models/grok-4.7.toml @@ -0,0 +1,19 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 + +[provider] +npm = "@ai-sdk/openai" From 739a4e206b88ea0699c94bb57ea3b138bc23c743 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:29:16 +0000 Subject: [PATCH 206/392] chore(sync): update Cloudflare Workers AI model catalog (#7640) Co-authored-by: opencode-agent[bot] --- .../models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml | 2 +- .../cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml | 2 +- providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml index 2385a1bc60b..48cdf638dfe 100644 --- a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml +++ b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml @@ -16,5 +16,5 @@ output = 1.32 cache_read = 0.014 [limit] -context = 1_048_576 +context = 1_310_720 output = 1_048_576 diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml index 83a85d6c195..13c5aacae4c 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml @@ -16,7 +16,7 @@ output = 0.5 cache_read = 0.03 [limit] -context = 1_048_576 +context = 1_310_720 output = 1_048_576 [modalities] diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml index a66154848f3..456f1fb0549 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml @@ -16,5 +16,5 @@ output = 4.4 cache_read = 0.26 [limit] -context = 1_048_576 +context = 1_310_720 output = 1_048_576 From 68310f550497b27d52e2f53c8f828fb63e3d42c4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:29:29 +0000 Subject: [PATCH 207/392] chore(sync): update xAI model catalog (#7644) Co-authored-by: opencode-agent[bot] --- providers/xai/models/grok-imagine-image-quality.toml | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/providers/xai/models/grok-imagine-image-quality.toml b/providers/xai/models/grok-imagine-image-quality.toml index caf513b6e28..1be5068f356 100644 --- a/providers/xai/models/grok-imagine-image-quality.toml +++ b/providers/xai/models/grok-imagine-image-quality.toml @@ -4,5 +4,8 @@ # - https://docs.x.ai/developers/pricing # Pricing: $0.05/image (1K), $0.07/image (2K); image input $0.01/image (not token-based; no [cost] authored) # Aliases: grok-imagine-image-quality-20260403, grok-imagine-image-quality-latest, grok-imagine-image-pro - base_model = "xai/grok-imagine-image-quality" + +[modalities] +input = ["text", "image", "pdf"] +output = ["image", "pdf"] From 39e41897358590fdd14291475e0d3f4725f536d6 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:29:33 +0000 Subject: [PATCH 208/392] chore(sync): update OpenRouter model catalog (#7645) Co-authored-by: opencode-agent[bot] --- .../models/anthropic/claude-opus-4.toml | 31 ------------------- .../models/anthropic/claude-sonnet-4.toml | 2 +- .../models/deepseek/deepseek-v4-pro-0813.toml | 6 ++-- .../models/deepseek/deepseek-v4-pro.toml | 6 ++-- providers/openrouter/models/tencent/hy3.toml | 6 ++-- .../openrouter/models/x-ai/grok-4.7.toml | 23 ++++++++++++++ .../models/~deepseek/deepseek-pro-latest.toml | 6 ++-- .../openrouter/models/~x-ai/grok-latest.toml | 12 +++---- 8 files changed, 42 insertions(+), 50 deletions(-) delete mode 100644 providers/openrouter/models/anthropic/claude-opus-4.toml create mode 100644 providers/openrouter/models/x-ai/grok-4.7.toml diff --git a/providers/openrouter/models/anthropic/claude-opus-4.toml b/providers/openrouter/models/anthropic/claude-opus-4.toml deleted file mode 100644 index 094f7f59203..00000000000 --- a/providers/openrouter/models/anthropic/claude-opus-4.toml +++ /dev/null @@ -1,31 +0,0 @@ -# Toggle: reasoning.enabled = true|false -# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens -name = "Claude Opus 4" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -family = "claude-opus" -release_date = "2025-05-22" -last_updated = "2025-05-22" -attachment = true -reasoning = true -temperature = true -tool_call = true -structured_output = false -knowledge = "2025-01-31" -open_weights = false - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 15 -output = 75 -cache_read = 1.5 -cache_write = 18.75 - -[limit] -context = 200_000 -output = 32_000 - -[modalities] -input = ["image", "text", "pdf"] -output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-sonnet-4.toml b/providers/openrouter/models/anthropic/claude-sonnet-4.toml index 7baa5f4f635..b7b54a299a9 100644 --- a/providers/openrouter/models/anthropic/claude-sonnet-4.toml +++ b/providers/openrouter/models/anthropic/claude-sonnet-4.toml @@ -30,7 +30,7 @@ cache_read = 0.6 cache_write = 7.5 [limit] -context = 1_000_000 +context = 200_000 output = 64_000 [modalities] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 32fc69b9875..13dfedbfc61 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57816 -output = 1.73448 -cache_read = 0.019272 +input = 0.57288 +output = 1.71864 +cache_read = 0.019096 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index b7feb0289f9..a705f18ca62 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.942906 -output = 1.885812 -cache_read = 0.078576 +input = 0.936294 +output = 1.872588 +cache_read = 0.078025 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/tencent/hy3.toml b/providers/openrouter/models/tencent/hy3.toml index 80ecfdd3aa8..f61fa3175e9 100644 --- a/providers/openrouter/models/tencent/hy3.toml +++ b/providers/openrouter/models/tencent/hy3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.132 -output = 0.528 -cache_read = 0.033 +input = 0.0825 +output = 0.33 +cache_read = 0.020625 [limit] context = 262_144 diff --git a/providers/openrouter/models/x-ai/grok-4.7.toml b/providers/openrouter/models/x-ai/grok-4.7.toml new file mode 100644 index 00000000000..49d2235d6f9 --- /dev/null +++ b/providers/openrouter/models/x-ai/grok-4.7.toml @@ -0,0 +1,23 @@ +base_model = "xai/grok-4.7" +description = "Grok model for agentic tool use, reasoning, coding, and live assistance" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 1.6 +output = 4.8 +cache_read = 0.4 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 3.2 +output = 9.6 +cache_read = 0.8 + +[limit] +output = 450_000 + +[modalities] +input = ["text", "image", "pdf"] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 9b9fc44533c..3a21dc3b34a 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57816 -output = 1.73448 -cache_read = 0.019272 +input = 0.57288 +output = 1.71864 +cache_read = 0.019096 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~x-ai/grok-latest.toml b/providers/openrouter/models/~x-ai/grok-latest.toml index ef7efee492e..2ce36baf40e 100644 --- a/providers/openrouter/models/~x-ai/grok-latest.toml +++ b/providers/openrouter/models/~x-ai/grok-latest.toml @@ -15,15 +15,15 @@ type = "effort" values = ["low", "medium", "high", "xhigh"] [cost] -input = 2 -output = 6 -cache_read = 0.5 +input = 1.6 +output = 4.8 +cache_read = 0.4 [[cost.tiers]] tier = { type = "context", size = 200_000 } -input = 4 -output = 12 -cache_read = 1 +input = 3.2 +output = 9.6 +cache_read = 0.8 [limit] context = 500_000 From b6a3717a04c18450eacb27fd10a5aa5c1d2bd01c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:29:37 +0000 Subject: [PATCH 209/392] chore(sync): update Kilo model catalog (#7647) Co-authored-by: opencode-agent[bot] --- .../kilo/models/anthropic/claude-opus-4.toml | 29 ------------------- providers/kilo/models/tencent/hy3.toml | 6 ++-- providers/kilo/models/x-ai/grok-4.7.toml | 17 +++++++++++ .../models/~deepseek/deepseek-pro-latest.toml | 6 ++-- providers/kilo/models/~x-ai/grok-latest.toml | 6 ++-- 5 files changed, 26 insertions(+), 38 deletions(-) delete mode 100644 providers/kilo/models/anthropic/claude-opus-4.toml create mode 100644 providers/kilo/models/x-ai/grok-4.7.toml diff --git a/providers/kilo/models/anthropic/claude-opus-4.toml b/providers/kilo/models/anthropic/claude-opus-4.toml deleted file mode 100644 index 6e622f068ab..00000000000 --- a/providers/kilo/models/anthropic/claude-opus-4.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Anthropic: Claude Opus 4 ($$$$)" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -family = "claude-opus" -release_date = "2025-05-22" -last_updated = "2025-05-22" -attachment = true -reasoning = true -temperature = true -tool_call = true -structured_output = false -open_weights = false - -[[reasoning_options]] -type = "effort" -values = ["none", "high"] - -[cost] -input = 15 -output = 75 -cache_read = 1.5 -cache_write = 18.75 - -[limit] -context = 200_000 -output = 32_000 - -[modalities] -input = ["image", "text", "pdf"] -output = ["text"] diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index 2bdfa3cbd7b..e96e3110ab1 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.132 -output = 0.528 -cache_read = 0.033 +input = 0.0825 +output = 0.33 +cache_read = 0.020625 [limit] context = 262_144 diff --git a/providers/kilo/models/x-ai/grok-4.7.toml b/providers/kilo/models/x-ai/grok-4.7.toml new file mode 100644 index 00000000000..bdb4be1fce7 --- /dev/null +++ b/providers/kilo/models/x-ai/grok-4.7.toml @@ -0,0 +1,17 @@ +base_model = "xai/grok-4.7" +description = "Grok 4.7 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM." + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 1.6 +output = 4.8 +cache_read = 0.4 + +[limit] +output = 450_000 + +[modalities] +input = ["text", "image", "pdf"] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index a92bf5149a1..d1321db247a 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.57816 -output = 1.73448 -cache_read = 0.019272 +input = 0.57288 +output = 1.71864 +cache_read = 0.019096 [limit] context = 1_024_000 diff --git a/providers/kilo/models/~x-ai/grok-latest.toml b/providers/kilo/models/~x-ai/grok-latest.toml index 5f182ec62c7..85f00745faa 100644 --- a/providers/kilo/models/~x-ai/grok-latest.toml +++ b/providers/kilo/models/~x-ai/grok-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "medium", "high", "xhigh"] [cost] -input = 2 -output = 6 -cache_read = 0.5 +input = 1.6 +output = 4.8 +cache_read = 0.4 [limit] context = 500_000 From b23f4e5f2da6b78079466db4fd331823097f46d7 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:29:44 +0000 Subject: [PATCH 210/392] chore(sync): update EmpirioLabs AI model catalog (#7646) Co-authored-by: opencode-agent[bot] --- providers/empiriolabs/models/step-5-preview.toml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) create mode 100644 providers/empiriolabs/models/step-5-preview.toml diff --git a/providers/empiriolabs/models/step-5-preview.toml b/providers/empiriolabs/models/step-5-preview.toml new file mode 100644 index 00000000000..c4ca4d28659 --- /dev/null +++ b/providers/empiriolabs/models/step-5-preview.toml @@ -0,0 +1,16 @@ +base_model = "stepfun/step-5-preview" +base_model_omit = ["limit.input"] +temperature = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1 +output = 2.7 +cache_read = 0.05 + +[limit] +context = 1_024_000 +output = 131_072 From 9e4e5ce97e4a014184f6bb76db7c2a8ed35a22aa Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:29:53 +0000 Subject: [PATCH 211/392] chore(sync): update NanoGPT model catalog (#7642) Co-authored-by: opencode-agent[bot] --- providers/nano-gpt/models/gemma-4-12b-it.toml | 11 +--------- .../models/stepfun/step-5-preview.toml | 20 +------------------ 2 files changed, 2 insertions(+), 29 deletions(-) diff --git a/providers/nano-gpt/models/gemma-4-12b-it.toml b/providers/nano-gpt/models/gemma-4-12b-it.toml index 6959fb985f1..3d3dd5e5581 100644 --- a/providers/nano-gpt/models/gemma-4-12b-it.toml +++ b/providers/nano-gpt/models/gemma-4-12b-it.toml @@ -1,13 +1,6 @@ +base_model = "google/gemma-4-12b-it" name = "Gemma 4 12B Instruct" -description = "Google's Gemma 4 12B Instruct is an open-weight multimodal model for text, image, audio, and video understanding, with tool calling and structured output support." -family = "gemma" -release_date = "2026-08-01" -last_updated = "2026-08-01" -attachment = true reasoning = false -tool_call = true -structured_output = true -open_weights = true [cost] input = 0.05 @@ -17,8 +10,6 @@ cache_read = 0.025 [limit] context = 131_072 input = 131_072 -output = 32_768 [modalities] input = ["text", "image", "video", "audio"] -output = ["text"] diff --git a/providers/nano-gpt/models/stepfun/step-5-preview.toml b/providers/nano-gpt/models/stepfun/step-5-preview.toml index e6b122e57eb..20ebdd2b522 100644 --- a/providers/nano-gpt/models/stepfun/step-5-preview.toml +++ b/providers/nano-gpt/models/stepfun/step-5-preview.toml @@ -1,13 +1,4 @@ -name = "Step 5 Preview" -description = "Step 5 Preview is StepFun's 600B sparse MoE frontier model for production-scale agents, activating 27B parameters per token. It is built for software engineering, long-horizon tool use, research, professional knowledge work, and finance, with native text, image, and video understanding and a 1M-token context window. ⚠️ Note: This model routes through StepFun, so privacy and logging guarantees may be limited." -family = "step" -release_date = "2026-09-20" -last_updated = "2026-09-20" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = false +base_model = "stepfun/step-5-preview" [[reasoning_options]] type = "effort" @@ -17,12 +8,3 @@ values = ["low", "medium", "high"] input = 1 output = 2.7 cache_read = 0.05 - -[limit] -context = 1_000_000 -input = 1_000_000 -output = 1_000_000 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] From ce2cb0547484fb62bee8813352123d5a6a5f550b Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 11:40:51 -0500 Subject: [PATCH 212/392] fix: [missing-model] xai: grok-4.7 (#7648) * fix: [missing-model] xai: grok-4.7 * Add 'pdf' to input modalities in grok-4.7.toml * Delete modalities input from grok-4.7.toml Removed modalities section from grok-4.7.toml --------- Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: Aiden Cline <63023139+rekram1-node@users.noreply.github.com> --- models/xai/grok-4.7.toml | 5 +++-- providers/xai/models/grok-4.7.toml | 18 ++++++++++++++++++ 2 files changed, 21 insertions(+), 2 deletions(-) create mode 100644 providers/xai/models/grok-4.7.toml diff --git a/models/xai/grok-4.7.toml b/models/xai/grok-4.7.toml index 1142f86cd13..7872d4c7219 100644 --- a/models/xai/grok-4.7.toml +++ b/models/xai/grok-4.7.toml @@ -1,7 +1,8 @@ +# Sources: https://docs.x.ai/developers/models/grok-4.7, https://docs.x.ai/developers/models name = "Grok 4.7" description = "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects" family = "grok" -knowledge = "2026-02-01" +knowledge = "2026-05" release_date = "2026-09-21" last_updated = "2026-09-21" attachment = true @@ -16,5 +17,5 @@ context = 500_000 output = 500_000 [modalities] -input = ["text", "image"] +input = ["text", "image", "pdf"] output = ["text"] diff --git a/providers/xai/models/grok-4.7.toml b/providers/xai/models/grok-4.7.toml new file mode 100644 index 00000000000..c205cfd8209 --- /dev/null +++ b/providers/xai/models/grok-4.7.toml @@ -0,0 +1,18 @@ +# Sources: https://docs.x.ai/developers/models/grok-4.7, https://docs.x.ai/developers/pricing, and https://docs.x.ai/developers/model-capabilities/text/reasoning +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 + From 1c042a3aaf0a037ba4fdf7332af31481db0c2974 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 11:50:28 -0500 Subject: [PATCH 213/392] chore(sync): update Vercel AI Gateway model catalog (#7641) * chore(sync): update Vercel AI Gateway model catalog * fix(sync): preserve Vercel tiered pricing --------- Co-authored-by: opencode-agent[bot] Co-authored-by: rekram1-node --- packages/core/src/sync/providers/vercel.ts | 69 +++++++++++++++++-- packages/core/test/sync.test.ts | 28 +++++++- .../models/alibaba/qwen-3.6-max-preview.toml | 7 ++ .../models/alibaba/qwen3-coder-plus.toml | 18 +++++ .../vercel/models/alibaba/qwen3-coder.toml | 12 ++++ .../models/alibaba/qwen3-max-preview.toml | 12 ++++ .../models/alibaba/qwen3-max-thinking.toml | 18 ++++- .../vercel/models/alibaba/qwen3-max.toml | 12 ++++ .../vercel/models/alibaba/qwen3.5-plus.toml | 7 ++ .../vercel/models/alibaba/qwen3.6-plus.toml | 7 ++ .../vercel/models/alibaba/qwen3.7-flash.toml | 14 ++++ .../vercel/models/alibaba/qwen3.7-plus.toml | 7 ++ .../models/anthropic/claude-sonnet-4.toml | 7 ++ .../vercel/models/bytedance/seed-1.6.toml | 6 ++ .../vercel/models/bytedance/seed-1.8.toml | 14 +++- .../models/google/gemini-3.1-pro-preview.toml | 6 ++ .../vercel/models/openai/gpt-5.4-pro.toml | 5 ++ providers/vercel/models/openai/gpt-5.4.toml | 12 +++- .../vercel/models/openai/gpt-5.5-pro.toml | 5 ++ providers/vercel/models/openai/gpt-5.5.toml | 6 ++ .../models/openai/gpt-5.6-luna-fast.toml | 7 ++ .../vercel/models/openai/gpt-5.6-luna.toml | 7 ++ .../models/openai/gpt-5.6-sol-fast.toml | 7 ++ .../vercel/models/openai/gpt-5.6-sol.toml | 7 ++ .../models/openai/gpt-5.6-terra-fast.toml | 7 ++ .../vercel/models/openai/gpt-5.6-terra.toml | 7 ++ .../models/openai/gpt-6-astra-fast.toml | 2 +- .../vercel/models/sakana/fugu-ultra-v2.toml | 11 ++- .../vercel/models/sakana/fugu-ultra.toml | 6 ++ .../spacexai/grok-4.20-multi-agent-beta.toml | 6 ++ .../spacexai/grok-4.20-multi-agent.toml | 6 ++ .../grok-4.20-non-reasoning-beta.toml | 6 ++ .../spacexai/grok-4.20-non-reasoning.toml | 6 ++ .../spacexai/grok-4.20-reasoning-beta.toml | 6 ++ .../models/spacexai/grok-4.20-reasoning.toml | 6 ++ .../vercel/models/spacexai/grok-4.3.toml | 7 ++ .../vercel/models/spacexai/grok-4.5.toml | 7 ++ .../vercel/models/spacexai/grok-4.6.toml | 7 ++ .../vercel/models/spacexai/grok-4.7.toml | 13 ++++ .../models/spacexai/grok-build-0.1.toml | 6 ++ .../vercel/models/stepfun/step-5-preview.toml | 12 +--- 41 files changed, 396 insertions(+), 25 deletions(-) create mode 100644 providers/vercel/models/spacexai/grok-4.7.toml diff --git a/packages/core/src/sync/providers/vercel.ts b/packages/core/src/sync/providers/vercel.ts index fdf8be631dd..e7800f77e32 100644 --- a/packages/core/src/sync/providers/vercel.ts +++ b/packages/core/src/sync/providers/vercel.ts @@ -237,17 +237,75 @@ function price(value: string | undefined) { : undefined; } +function tieredPrice(value: string | undefined, tiers: z.infer[] | undefined) { + const base = price(value); + const normalized = (tiers ?? []) + .map((tier, index, values) => ({ + start: tier.min ?? (index === 0 ? 0 : values[index - 1]?.max ?? 0), + cost: price(tier.cost), + })) + .filter((tier): tier is { start: number; cost: number } => tier.cost !== undefined) + .sort((a, b) => a.start - b.start); + + return { + base: normalized[0]?.cost ?? base, + thresholds: normalized.map((tier) => tier.start).filter((start) => start > 0), + at(threshold: number) { + return normalized.findLast((tier) => tier.start <= threshold)?.cost ?? base; + }, + }; +} + function buildCost(pricing: VercelModel["pricing"], existing?: ExistingModel["cost"]) { - const input = price(pricing?.input_tiers?.[0]?.cost ?? pricing?.input); - const output = price(pricing?.output_tiers?.[0]?.cost ?? pricing?.output); + const hasPricingTiers = [ + pricing?.input_tiers, + pricing?.output_tiers, + pricing?.input_cache_read_tiers, + pricing?.input_cache_write_tiers, + ].some((tiers) => (tiers?.length ?? 0) > 0); + const inputPrice = tieredPrice(pricing?.input, pricing?.input_tiers); + const outputPrice = tieredPrice(pricing?.output, pricing?.output_tiers); + const cacheReadPrice = tieredPrice(pricing?.input_cache_read, pricing?.input_cache_read_tiers); + const cacheWritePrice = tieredPrice(pricing?.input_cache_write, pricing?.input_cache_write_tiers); + const input = inputPrice.base; + const output = outputPrice.base; if (input === undefined || output === undefined) return undefined; + + const thresholds = new Set([ + ...inputPrice.thresholds, + ...outputPrice.thresholds, + ...cacheReadPrice.thresholds, + ...cacheWritePrice.thresholds, + ]); + const tiers: NonNullable["tiers"]> = []; + let previous = { + input, + output, + cache_read: cacheReadPrice.base, + cache_write: cacheWritePrice.base, + }; + for (const size of [...thresholds].sort((a, b) => a - b)) { + const tierInput = inputPrice.at(size); + const tierOutput = outputPrice.at(size); + if (tierInput === undefined || tierOutput === undefined) continue; + const current = { + input: tierInput, + output: tierOutput, + cache_read: cacheReadPrice.at(size), + cache_write: cacheWritePrice.at(size), + }; + if (JSON.stringify(current) === JSON.stringify(previous)) continue; + tiers.push({ tier: { type: "context", size }, ...current }); + previous = current; + } + return { input, output, reasoning: existing?.reasoning, - cache_read: price(pricing?.input_cache_read_tiers?.[0]?.cost ?? pricing?.input_cache_read), - cache_write: price(pricing?.input_cache_write_tiers?.[0]?.cost ?? pricing?.input_cache_write), - tiers: existing?.tiers, + cache_read: cacheReadPrice.base, + cache_write: cacheWritePrice.base, + tiers: hasPricingTiers ? (tiers.length > 0 ? tiers : undefined) : existing?.tiers, }; } @@ -291,6 +349,7 @@ function sameVercelModel(current: ExistingModel, desired: SyncedModel) { [current.cost?.output, desiredModel.cost?.output, true], [current.cost?.cache_read, desiredModel.cost?.cache_read, true], [current.cost?.cache_write, desiredModel.cost?.cache_write, true], + [current.cost?.tiers, desiredModel.cost?.tiers], [current.limit?.context, desiredModel.limit?.context], [current.limit?.input, desiredModel.limit?.input], [current.limit?.output, desiredModel.limit?.output], diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index d0609631969..0281aac4808 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -4423,7 +4423,7 @@ test("retains Merge Gateway models missing from an API-key-scoped response", () expect(mergeGateway.deleteMissing).toBe(false); }); -test("parses Vercel pricing tiers with an implicit zero minimum", () => { +test("translates Vercel pricing tiers with an implicit zero minimum", () => { const [model] = vercel.parseModels({ data: [{ id: "openai/gpt-5.6-luna", @@ -4440,14 +4440,36 @@ test("parses Vercel pricing tiers with an implicit zero minimum", () => { { cost: "0.0000001", max: 272_000 }, { cost: "0.0000002", min: 272_000 }, ], + input_tiers: [ + { cost: "0.000001", max: 272_000 }, + { cost: "0.000002", min: 272_000 }, + ], + output_tiers: [ + { cost: "0.000006", max: 272_000 }, + { cost: "0.000009", min: 272_000 }, + ], }, }], }); expect(model).toBeDefined(); - expect(buildVercelModel(model!, undefined)).toMatchObject({ - cost: { input: 1, output: 6, cache_read: 0.1 }, + const synced = buildVercelModel(model!, undefined); + expect(synced).toMatchObject({ + cost: { + input: 1, + output: 6, + cache_read: 0.1, + tiers: [{ + tier: { type: "context", size: 272_000 }, + input: 2, + output: 9, + cache_read: 0.2, + }], + }, }); + expect(vercel.sameModel?.({ + cost: { input: 1, output: 6, cache_read: 0.1 }, + }, synced)).toBe(false); }); test("Vercel factored models inherit temperature from base metadata", () => { diff --git a/providers/vercel/models/alibaba/qwen-3.6-max-preview.toml b/providers/vercel/models/alibaba/qwen-3.6-max-preview.toml index 2978e3e117c..df7e9852286 100644 --- a/providers/vercel/models/alibaba/qwen-3.6-max-preview.toml +++ b/providers/vercel/models/alibaba/qwen-3.6-max-preview.toml @@ -23,6 +23,13 @@ output = 7.8 cache_read = 0.13 cache_write = 1.625 +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 2 +output = 12 +cache_read = 0.2 +cache_write = 2.5 + [limit] context = 240_000 output = 64_000 diff --git a/providers/vercel/models/alibaba/qwen3-coder-plus.toml b/providers/vercel/models/alibaba/qwen3-coder-plus.toml index 06f877de2eb..656712e8607 100644 --- a/providers/vercel/models/alibaba/qwen3-coder-plus.toml +++ b/providers/vercel/models/alibaba/qwen3-coder-plus.toml @@ -5,5 +5,23 @@ input = 1 output = 5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 32_001 } +input = 1.8 +output = 9 +cache_read = 0.36 + +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 3 +output = 15 +cache_read = 0.6 + +[[cost.tiers]] +tier = { type = "context", size = 256_001 } +input = 6 +output = 60 +cache_read = 1.2 + [limit] context = 1_000_000 diff --git a/providers/vercel/models/alibaba/qwen3-coder.toml b/providers/vercel/models/alibaba/qwen3-coder.toml index e61dd0d67dd..f5fe46e649f 100644 --- a/providers/vercel/models/alibaba/qwen3-coder.toml +++ b/providers/vercel/models/alibaba/qwen3-coder.toml @@ -16,6 +16,18 @@ input = 1.5 output = 7.5 cache_read = 0.3 +[[cost.tiers]] +tier = { type = "context", size = 32_001 } +input = 2.7 +output = 13.5 +cache_read = 0.54 + +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 4.5 +output = 22.5 +cache_read = 0.9 + [limit] context = 262_144 output = 65_536 diff --git a/providers/vercel/models/alibaba/qwen3-max-preview.toml b/providers/vercel/models/alibaba/qwen3-max-preview.toml index 94b902c8984..69a704070b5 100644 --- a/providers/vercel/models/alibaba/qwen3-max-preview.toml +++ b/providers/vercel/models/alibaba/qwen3-max-preview.toml @@ -15,6 +15,18 @@ input = 1.2 output = 6 cache_read = 0.24 +[[cost.tiers]] +tier = { type = "context", size = 32_001 } +input = 2.4 +output = 12 +cache_read = 0.48 + +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 3 +output = 15 +cache_read = 0.6 + [limit] context = 262_144 output = 32_768 diff --git a/providers/vercel/models/alibaba/qwen3-max-thinking.toml b/providers/vercel/models/alibaba/qwen3-max-thinking.toml index db97c26cd56..b00cd1597a9 100644 --- a/providers/vercel/models/alibaba/qwen3-max-thinking.toml +++ b/providers/vercel/models/alibaba/qwen3-max-thinking.toml @@ -5,17 +5,33 @@ release_date = "2026-01-23" last_updated = "2025-01" attachment = false reasoning = true -reasoning_options = [{ type = "budget_tokens", min = 1, max = 81_920 }] temperature = true tool_call = true knowledge = "2025-01" open_weights = true +[[reasoning_options]] +type = "budget_tokens" +min = 1 +max = 81_920 + [cost] input = 1.2 output = 6 cache_read = 0.24 +[[cost.tiers]] +tier = { type = "context", size = 32_001 } +input = 2.4 +output = 12 +cache_read = 0.48 + +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 3 +output = 15 +cache_read = 0.6 + [limit] context = 256_000 output = 65_536 diff --git a/providers/vercel/models/alibaba/qwen3-max.toml b/providers/vercel/models/alibaba/qwen3-max.toml index 74f80d21720..556b71930db 100644 --- a/providers/vercel/models/alibaba/qwen3-max.toml +++ b/providers/vercel/models/alibaba/qwen3-max.toml @@ -5,5 +5,17 @@ input = 1.2 output = 6 cache_read = 0.24 +[[cost.tiers]] +tier = { type = "context", size = 32_001 } +input = 2.4 +output = 12 +cache_read = 0.48 + +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 3 +output = 15 +cache_read = 0.6 + [limit] output = 32_768 diff --git a/providers/vercel/models/alibaba/qwen3.5-plus.toml b/providers/vercel/models/alibaba/qwen3.5-plus.toml index fab6054e8ca..3f4c69a844a 100644 --- a/providers/vercel/models/alibaba/qwen3.5-plus.toml +++ b/providers/vercel/models/alibaba/qwen3.5-plus.toml @@ -16,6 +16,13 @@ output = 2.5 cache_read = 0.04 cache_write = 0.5 +[[cost.tiers]] +tier = { type = "context", size = 256_001 } +input = 0.5 +output = 3 +cache_read = 0.05 +cache_write = 0.625 + [limit] output = 64_000 diff --git a/providers/vercel/models/alibaba/qwen3.6-plus.toml b/providers/vercel/models/alibaba/qwen3.6-plus.toml index 1b29656202b..e87a6735aae 100644 --- a/providers/vercel/models/alibaba/qwen3.6-plus.toml +++ b/providers/vercel/models/alibaba/qwen3.6-plus.toml @@ -15,6 +15,13 @@ output = 3 cache_read = 0.05 cache_write = 0.625 +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 2 +output = 6 +cache_read = 0.2 +cache_write = 2.5 + [limit] output = 64_000 diff --git a/providers/vercel/models/alibaba/qwen3.7-flash.toml b/providers/vercel/models/alibaba/qwen3.7-flash.toml index 774d5fe725d..8aaf3abadf1 100644 --- a/providers/vercel/models/alibaba/qwen3.7-flash.toml +++ b/providers/vercel/models/alibaba/qwen3.7-flash.toml @@ -10,6 +10,20 @@ output = 0.13 cache_read = 0.006 cache_write = 0.038 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.1 +output = 0.4 +cache_read = 0.02 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.2 +output = 0.8 +cache_read = 0.04 +cache_write = 0.25 + [limit] context = 991_000 output = 64_000 diff --git a/providers/vercel/models/alibaba/qwen3.7-plus.toml b/providers/vercel/models/alibaba/qwen3.7-plus.toml index 819ab65c9e3..4511b24804c 100644 --- a/providers/vercel/models/alibaba/qwen3.7-plus.toml +++ b/providers/vercel/models/alibaba/qwen3.7-plus.toml @@ -24,5 +24,12 @@ output = 1.6 cache_read = 0.08 cache_write = 0.5 +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 1.2 +output = 4.8 +cache_read = 0.24 +cache_write = 1.5 + [modalities] input = ["text", "image", "pdf"] diff --git a/providers/vercel/models/anthropic/claude-sonnet-4.toml b/providers/vercel/models/anthropic/claude-sonnet-4.toml index df6a061b4b1..61e522dfd97 100644 --- a/providers/vercel/models/anthropic/claude-sonnet-4.toml +++ b/providers/vercel/models/anthropic/claude-sonnet-4.toml @@ -7,6 +7,13 @@ output = 15 cache_read = 0.3 cache_write = 3.75 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 6 +output = 22.5 +cache_read = 0.6 +cache_write = 7.5 + [limit] context = 1_000_000 output = 8_192 diff --git a/providers/vercel/models/bytedance/seed-1.6.toml b/providers/vercel/models/bytedance/seed-1.6.toml index c0346917f50..bc03000e407 100644 --- a/providers/vercel/models/bytedance/seed-1.6.toml +++ b/providers/vercel/models/bytedance/seed-1.6.toml @@ -18,6 +18,12 @@ input = 0.25 output = 2 cache_read = 0.05 +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 0.5 +output = 4 +cache_read = 0.05 + [limit] context = 256_000 output = 32_000 diff --git a/providers/vercel/models/bytedance/seed-1.8.toml b/providers/vercel/models/bytedance/seed-1.8.toml index 1fa077191ce..7e02aaccd52 100644 --- a/providers/vercel/models/bytedance/seed-1.8.toml +++ b/providers/vercel/models/bytedance/seed-1.8.toml @@ -5,17 +5,29 @@ release_date = "2025-09-01" last_updated = "2025-10" attachment = false reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] temperature = true tool_call = true knowledge = "2024-10" open_weights = false +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + [cost] input = 0.25 output = 2 cache_read = 0.05 +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 0.5 +output = 4 +cache_read = 0.05 + [limit] context = 256_000 output = 64_000 diff --git a/providers/vercel/models/google/gemini-3.1-pro-preview.toml b/providers/vercel/models/google/gemini-3.1-pro-preview.toml index 31fde20cebe..742f9779a46 100644 --- a/providers/vercel/models/google/gemini-3.1-pro-preview.toml +++ b/providers/vercel/models/google/gemini-3.1-pro-preview.toml @@ -9,6 +9,12 @@ input = 2 output = 12 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 4 +output = 18 +cache_read = 0.4 + [limit] context = 1_000_000 output = 64_000 diff --git a/providers/vercel/models/openai/gpt-5.4-pro.toml b/providers/vercel/models/openai/gpt-5.4-pro.toml index 802e2b01e7d..8f49caeaef4 100644 --- a/providers/vercel/models/openai/gpt-5.4-pro.toml +++ b/providers/vercel/models/openai/gpt-5.4-pro.toml @@ -10,5 +10,10 @@ values = ["medium", "high", "xhigh"] input = 30 output = 180 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 60 +output = 270 + [modalities] input = ["text", "image", "pdf"] diff --git a/providers/vercel/models/openai/gpt-5.4.toml b/providers/vercel/models/openai/gpt-5.4.toml index 117744507a5..e80519bdd16 100644 --- a/providers/vercel/models/openai/gpt-5.4.toml +++ b/providers/vercel/models/openai/gpt-5.4.toml @@ -1,9 +1,17 @@ base_model = "openai/gpt-5.4" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] name = "GPT 5.4" -temperature = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 2.5 output = 15 cache_read = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 5 +output = 22.5 +cache_read = 0.5 diff --git a/providers/vercel/models/openai/gpt-5.5-pro.toml b/providers/vercel/models/openai/gpt-5.5-pro.toml index effd8f55a5a..b124e744fda 100644 --- a/providers/vercel/models/openai/gpt-5.5-pro.toml +++ b/providers/vercel/models/openai/gpt-5.5-pro.toml @@ -10,6 +10,11 @@ values = ["medium", "high", "xhigh"] input = 30 output = 180 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 60 +output = 270 + [limit] context = 1_000_000 input = 872_000 diff --git a/providers/vercel/models/openai/gpt-5.5.toml b/providers/vercel/models/openai/gpt-5.5.toml index 4d37b9c3632..6fa3472701a 100644 --- a/providers/vercel/models/openai/gpt-5.5.toml +++ b/providers/vercel/models/openai/gpt-5.5.toml @@ -11,6 +11,12 @@ input = 5 output = 30 cache_read = 0.5 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 10 +output = 45 +cache_read = 1 + [limit] context = 1_000_000 input = 872_000 diff --git a/providers/vercel/models/openai/gpt-5.6-luna-fast.toml b/providers/vercel/models/openai/gpt-5.6-luna-fast.toml index 0677db18d76..74dd96c0e31 100644 --- a/providers/vercel/models/openai/gpt-5.6-luna-fast.toml +++ b/providers/vercel/models/openai/gpt-5.6-luna-fast.toml @@ -10,3 +10,10 @@ input = 0.4 output = 2.4 cache_read = 0.04 cache_write = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.8 +output = 3.6 +cache_read = 0.08 +cache_write = 1 diff --git a/providers/vercel/models/openai/gpt-5.6-luna.toml b/providers/vercel/models/openai/gpt-5.6-luna.toml index f33d0f6f9b5..e48b3985f6c 100644 --- a/providers/vercel/models/openai/gpt-5.6-luna.toml +++ b/providers/vercel/models/openai/gpt-5.6-luna.toml @@ -12,3 +12,10 @@ input = 0.2 output = 1.2 cache_read = 0.02 cache_write = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.4 +output = 1.8 +cache_read = 0.04 +cache_write = 0.5 diff --git a/providers/vercel/models/openai/gpt-5.6-sol-fast.toml b/providers/vercel/models/openai/gpt-5.6-sol-fast.toml index 007699dc75e..d989b000e02 100644 --- a/providers/vercel/models/openai/gpt-5.6-sol-fast.toml +++ b/providers/vercel/models/openai/gpt-5.6-sol-fast.toml @@ -10,3 +10,10 @@ input = 8 output = 40 cache_read = 0.8 cache_write = 10 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 16 +output = 60 +cache_read = 1.6 +cache_write = 20 diff --git a/providers/vercel/models/openai/gpt-5.6-sol.toml b/providers/vercel/models/openai/gpt-5.6-sol.toml index b07d7577a9a..ed5b73a9699 100644 --- a/providers/vercel/models/openai/gpt-5.6-sol.toml +++ b/providers/vercel/models/openai/gpt-5.6-sol.toml @@ -12,3 +12,10 @@ input = 4 output = 20 cache_read = 0.4 cache_write = 5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 8 +output = 30 +cache_read = 0.8 +cache_write = 10 diff --git a/providers/vercel/models/openai/gpt-5.6-terra-fast.toml b/providers/vercel/models/openai/gpt-5.6-terra-fast.toml index c5e03cad536..e2caf6406e9 100644 --- a/providers/vercel/models/openai/gpt-5.6-terra-fast.toml +++ b/providers/vercel/models/openai/gpt-5.6-terra-fast.toml @@ -10,3 +10,10 @@ input = 4 output = 24 cache_read = 0.4 cache_write = 5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 8 +output = 36 +cache_read = 0.8 +cache_write = 10 diff --git a/providers/vercel/models/openai/gpt-5.6-terra.toml b/providers/vercel/models/openai/gpt-5.6-terra.toml index 7c036e01d1c..cc47098412f 100644 --- a/providers/vercel/models/openai/gpt-5.6-terra.toml +++ b/providers/vercel/models/openai/gpt-5.6-terra.toml @@ -12,3 +12,10 @@ input = 2 output = 12 cache_read = 0.2 cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 18 +cache_read = 0.4 +cache_write = 5 diff --git a/providers/vercel/models/openai/gpt-6-astra-fast.toml b/providers/vercel/models/openai/gpt-6-astra-fast.toml index ad67cd2908a..bbb8660be6c 100644 --- a/providers/vercel/models/openai/gpt-6-astra-fast.toml +++ b/providers/vercel/models/openai/gpt-6-astra-fast.toml @@ -16,4 +16,4 @@ tier = { type = "context", size = 272_001 } input = 40 output = 150 cache_read = 4 -cache_write = 25 +cache_write = 50 diff --git a/providers/vercel/models/sakana/fugu-ultra-v2.toml b/providers/vercel/models/sakana/fugu-ultra-v2.toml index f77055d8e6e..3ada0b1935d 100644 --- a/providers/vercel/models/sakana/fugu-ultra-v2.toml +++ b/providers/vercel/models/sakana/fugu-ultra-v2.toml @@ -9,13 +9,22 @@ attachment = true reasoning = true tool_call = true open_weights = false -reasoning_options = [{ type = "effort", values = ["high", "xhigh", "max"] }] + +[[reasoning_options]] +type = "effort" +values = ["high", "xhigh", "max"] [cost] input = 5 output = 30 cache_read = 0.5 +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 10 +output = 45 +cache_read = 1 + [limit] context = 1_000_000 output = 1_000_000 diff --git a/providers/vercel/models/sakana/fugu-ultra.toml b/providers/vercel/models/sakana/fugu-ultra.toml index 099d441fbe6..f36e9177ff8 100644 --- a/providers/vercel/models/sakana/fugu-ultra.toml +++ b/providers/vercel/models/sakana/fugu-ultra.toml @@ -8,5 +8,11 @@ input = 5 output = 30 cache_read = 0.5 +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 10 +output = 45 +cache_read = 1 + [limit] output = 1_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-multi-agent-beta.toml b/providers/vercel/models/spacexai/grok-4.20-multi-agent-beta.toml index ed37aa69ee5..d45eab3701e 100644 --- a/providers/vercel/models/spacexai/grok-4.20-multi-agent-beta.toml +++ b/providers/vercel/models/spacexai/grok-4.20-multi-agent-beta.toml @@ -14,6 +14,12 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-multi-agent.toml b/providers/vercel/models/spacexai/grok-4.20-multi-agent.toml index 37a06048cc0..a90774a3c3f 100644 --- a/providers/vercel/models/spacexai/grok-4.20-multi-agent.toml +++ b/providers/vercel/models/spacexai/grok-4.20-multi-agent.toml @@ -14,6 +14,12 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-non-reasoning-beta.toml b/providers/vercel/models/spacexai/grok-4.20-non-reasoning-beta.toml index acd189124d0..ac3eafaaecd 100644 --- a/providers/vercel/models/spacexai/grok-4.20-non-reasoning-beta.toml +++ b/providers/vercel/models/spacexai/grok-4.20-non-reasoning-beta.toml @@ -13,6 +13,12 @@ input = 1.25 output = 2.5 cache_read = 0.4 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-non-reasoning.toml b/providers/vercel/models/spacexai/grok-4.20-non-reasoning.toml index 468d77b85b2..0e9a2baf86f 100644 --- a/providers/vercel/models/spacexai/grok-4.20-non-reasoning.toml +++ b/providers/vercel/models/spacexai/grok-4.20-non-reasoning.toml @@ -13,6 +13,12 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-reasoning-beta.toml b/providers/vercel/models/spacexai/grok-4.20-reasoning-beta.toml index 19522d5f419..691eded5322 100644 --- a/providers/vercel/models/spacexai/grok-4.20-reasoning-beta.toml +++ b/providers/vercel/models/spacexai/grok-4.20-reasoning-beta.toml @@ -14,6 +14,12 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-reasoning.toml b/providers/vercel/models/spacexai/grok-4.20-reasoning.toml index 8fb2843f07d..e6c7e7ae17b 100644 --- a/providers/vercel/models/spacexai/grok-4.20-reasoning.toml +++ b/providers/vercel/models/spacexai/grok-4.20-reasoning.toml @@ -14,6 +14,12 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.3.toml b/providers/vercel/models/spacexai/grok-4.3.toml index d379c871d7f..1ce53c49412 100644 --- a/providers/vercel/models/spacexai/grok-4.3.toml +++ b/providers/vercel/models/spacexai/grok-4.3.toml @@ -1,5 +1,6 @@ base_model = "xai/grok-4.3" description = "Grok model for agentic tool use, reasoning, coding, and live assistance" + [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] @@ -9,5 +10,11 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] output = 1_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.5.toml b/providers/vercel/models/spacexai/grok-4.5.toml index 2eb4ebf0d6e..ec9457541d2 100644 --- a/providers/vercel/models/spacexai/grok-4.5.toml +++ b/providers/vercel/models/spacexai/grok-4.5.toml @@ -1,5 +1,6 @@ base_model = "xai/grok-4.5" description = "Grok model for agentic tool use, reasoning, coding, and live assistance" + [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] @@ -9,5 +10,11 @@ input = 2 output = 6 cache_read = 0.3 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 4 +output = 12 +cache_read = 0.6 + [modalities] input = ["text", "image", "pdf"] diff --git a/providers/vercel/models/spacexai/grok-4.6.toml b/providers/vercel/models/spacexai/grok-4.6.toml index 713ac05e4b3..58cede40b86 100644 --- a/providers/vercel/models/spacexai/grok-4.6.toml +++ b/providers/vercel/models/spacexai/grok-4.6.toml @@ -1,5 +1,6 @@ base_model = "xai/grok-4.6" description = "Grok model for agentic tool use, reasoning, coding, and live assistance" + [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] @@ -8,3 +9,9 @@ values = ["low", "medium", "high"] input = 2 output = 6 cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 4 +output = 12 +cache_read = 1 diff --git a/providers/vercel/models/spacexai/grok-4.7.toml b/providers/vercel/models/spacexai/grok-4.7.toml new file mode 100644 index 00000000000..defc4659f57 --- /dev/null +++ b/providers/vercel/models/spacexai/grok-4.7.toml @@ -0,0 +1,13 @@ +base_model = "xai/grok-4.7" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[cost] +input = 1.2 +output = 3.6 +cache_read = 0.3 + +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.4 +output = 7.2 +cache_read = 0.6 diff --git a/providers/vercel/models/spacexai/grok-build-0.1.toml b/providers/vercel/models/spacexai/grok-build-0.1.toml index dfcdb207ca7..ec232622716 100644 --- a/providers/vercel/models/spacexai/grok-build-0.1.toml +++ b/providers/vercel/models/spacexai/grok-build-0.1.toml @@ -7,5 +7,11 @@ input = 1 output = 2 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2 +output = 4 +cache_read = 0.4 + [modalities] input = ["text", "image"] diff --git a/providers/vercel/models/stepfun/step-5-preview.toml b/providers/vercel/models/stepfun/step-5-preview.toml index 8db9a1d1622..ce090c9a924 100644 --- a/providers/vercel/models/stepfun/step-5-preview.toml +++ b/providers/vercel/models/stepfun/step-5-preview.toml @@ -1,22 +1,12 @@ -name = "Step 5 Preview" +base_model = "stepfun/step-5-preview" description = "StepFun flash model for efficient multimodal reasoning, coding, and tool use" -family = "step" -release_date = "2026-09-20" -last_updated = "2026-09-20" -attachment = true reasoning = false tool_call = false -open_weights = false [cost] input = 1 output = 2.7 cache_read = 0.05 -[limit] -context = 1_000_000 -output = 1_000_000 - [modalities] input = ["text", "image"] -output = ["text"] From a2217f3a6b4a30e33e74ba10169f83fda5af76fb Mon Sep 17 00:00:00 2001 From: Johnny Chadda Date: Mon, 21 Sep 2026 18:53:49 +0200 Subject: [PATCH 214/392] Use pooled model IDs for Opper and refresh catalog (#6542) * Use pooled model IDs for Opper and refresh catalog * Add verified Muse max effort and document pool overrides --- .../models/anthropic/claude-haiku-4-5.toml | 13 --------- .../models/anthropic/claude-opus-4-7.toml | 15 ---------- .../opper/models/anthropic/claude-opus-5.toml | 15 ---------- .../models/anthropic/claude-sonnet-4-5.toml | 13 --------- .../models/anthropic/claude-sonnet-4-6.toml | 15 ---------- .../models/anthropic/claude-sonnet-5.toml | 15 ---------- providers/opper/models/claude-fable-5-1.toml | 15 ++++++++++ .../{anthropic => }/claude-fable-5.toml | 6 ++-- providers/opper/models/claude-haiku-4-5.toml | 18 ++++++++++++ .../{anthropic => }/claude-opus-4-5.toml | 6 ++-- .../{anthropic => }/claude-opus-4-6.toml | 6 ++-- providers/opper/models/claude-opus-4-7.toml | 18 ++++++++++++ .../{anthropic => }/claude-opus-4-8.toml | 6 ++-- providers/opper/models/claude-opus-5.toml | 15 ++++++++++ providers/opper/models/claude-sonnet-4-5.toml | 23 +++++++++++++++ providers/opper/models/claude-sonnet-4-6.toml | 15 ++++++++++ providers/opper/models/claude-sonnet-5.toml | 18 ++++++++++++ providers/opper/models/deepseek-v4-flash.toml | 12 ++++++++ providers/opper/models/deepseek-v4-pro.toml | 15 ++++++++++ providers/opper/models/devstral-2512.toml | 11 ++++++++ .../opper/models/gemini-3-flash-preview.toml | 13 +++++++++ .../opper/models/gemini-3.1-pro-preview.toml | 20 +++++++++++++ .../opper/models/gemini-3.5-flash-lite.toml | 13 +++++++++ providers/opper/models/gemini-3.5-flash.toml | 13 +++++++++ providers/opper/models/gemini-3.7-flash.toml | 13 +++++++++ providers/opper/models/gemini-3.8-flash.toml | 16 +++++++++++ .../models/gemini/gemini-3-flash-preview.toml | 10 ------- .../models/gemini/gemini-3.1-pro-preview.toml | 16 ----------- .../models/gemini/gemini-3.5-flash-lite.toml | 10 ------- .../opper/models/gemini/gemini-3.5-flash.toml | 10 ------- providers/opper/models/gemma-4-31b-it.toml | 16 +++++++++++ providers/opper/models/glm-5.2.toml | 12 ++++++++ providers/opper/models/glm-5.3-flash.toml | 21 ++++++++++++++ providers/opper/models/glm-5.3.toml | 14 ++++++++++ .../opper/models/gpt-5.3-chat-latest.toml | 9 ++++++ providers/opper/models/gpt-5.3-codex.toml | 18 ++++++++++++ providers/opper/models/gpt-5.4-mini.toml | 14 ++++++++++ providers/opper/models/gpt-5.4-nano.toml | 14 ++++++++++ providers/opper/models/gpt-5.4-pro.toml | 18 ++++++++++++ providers/opper/models/gpt-5.4.toml | 20 +++++++++++++ providers/opper/models/gpt-5.5-pro.toml | 18 ++++++++++++ providers/opper/models/gpt-5.5.toml | 20 +++++++++++++ providers/opper/models/gpt-5.6-luna.toml | 25 +++++++++++++++++ providers/opper/models/gpt-5.6-sol.toml | 25 +++++++++++++++++ providers/opper/models/gpt-5.6-terra.toml | 25 +++++++++++++++++ providers/opper/models/gpt-6-astra.toml | 25 +++++++++++++++++ providers/opper/models/gpt-oss-120b.toml | 16 +++++++++++ providers/opper/models/gpt-oss-20b.toml | 16 +++++++++++ providers/opper/models/grok-4.3.toml | 18 ++++++++++++ providers/opper/models/grok-4.5.toml | 15 ++++++++++ providers/opper/models/grok-4.6.toml | 20 +++++++++++++ providers/opper/models/grok-build-0.1.toml | 15 ++++++++++ providers/opper/models/kimi-k3.toml | 16 +++++++++++ providers/opper/models/minimax-m2.7.toml | 16 +++++++++++ providers/opper/models/minimax-m3.toml | 28 +++++++++++++++++++ providers/opper/models/minimax/m3.toml | 24 ---------------- .../opper/models/mistral-large-2512.toml | 11 ++++++++ .../opper/models/mistral-small-2603.toml | 16 +++++++++++ .../opper/models/mistral/devstral-2512.toml | 5 ---- .../models/mistral/mistral-large-2512.toml | 5 ---- .../models/mistral/mistral-small-2603.toml | 9 ------ providers/opper/models/moonshot/kimi-k3.toml | 13 --------- .../models/{meta => }/muse-spark-1.2.toml | 5 ++++ providers/opper/models/muse-spark-1.3.toml | 15 ++++++++++ .../models/openai/gpt-5.3-chat-latest.toml | 6 ---- .../opper/models/openai/gpt-5.3-codex.toml | 8 ------ .../opper/models/openai/gpt-5.4-mini.toml | 8 ------ .../opper/models/openai/gpt-5.4-nano.toml | 8 ------ .../opper/models/openai/gpt-5.4-pro.toml | 12 -------- providers/opper/models/openai/gpt-5.4.toml | 14 ---------- .../opper/models/openai/gpt-5.5-pro.toml | 12 -------- providers/opper/models/openai/gpt-5.5.toml | 14 ---------- .../opper/models/openai/gpt-5.6-luna.toml | 16 ----------- .../opper/models/openai/gpt-5.6-sol.toml | 16 ----------- .../opper/models/openai/gpt-5.6-terra.toml | 16 ----------- .../opper/models/perplexity/sonar-pro.toml | 5 ---- .../perplexity/sonar-reasoning-pro.toml | 7 ----- providers/opper/models/perplexity/sonar.toml | 5 ---- providers/opper/models/qwen3-coder-next.toml | 6 ++++ providers/opper/models/qwen3.6-35b-a3b.toml | 15 ++++++++++ providers/opper/models/qwen3.8-2.4t-a95b.toml | 13 +++++++++ providers/opper/models/qwen3.8-27b.toml | 17 +++++++++++ providers/opper/models/qwen3.8-max.toml | 20 +++++++++++++ providers/opper/models/sonar-pro.toml | 14 ++++++++++ .../opper/models/sonar-reasoning-pro.toml | 17 +++++++++++ providers/opper/models/sonar.toml | 7 +++++ .../models/vertexai/gemini-3.7-flash-eu.toml | 11 -------- .../models/vertexai/gemini-3.7-flash.toml | 10 ------- providers/opper/models/xai/grok-4.3.toml | 19 ------------- providers/opper/models/xai/grok-4.5.toml | 19 ------------- providers/opper/models/xai/grok-4.6.toml | 19 ------------- .../opper/models/xai/grok-build-0.1.toml | 17 ----------- providers/opper/provider.toml | 23 +++++++++------ 93 files changed, 885 insertions(+), 450 deletions(-) delete mode 100644 providers/opper/models/anthropic/claude-haiku-4-5.toml delete mode 100644 providers/opper/models/anthropic/claude-opus-4-7.toml delete mode 100644 providers/opper/models/anthropic/claude-opus-5.toml delete mode 100644 providers/opper/models/anthropic/claude-sonnet-4-5.toml delete mode 100644 providers/opper/models/anthropic/claude-sonnet-4-6.toml delete mode 100644 providers/opper/models/anthropic/claude-sonnet-5.toml create mode 100644 providers/opper/models/claude-fable-5-1.toml rename providers/opper/models/{anthropic => }/claude-fable-5.toml (50%) create mode 100644 providers/opper/models/claude-haiku-4-5.toml rename providers/opper/models/{anthropic => }/claude-opus-4-5.toml (50%) rename providers/opper/models/{anthropic => }/claude-opus-4-6.toml (50%) create mode 100644 providers/opper/models/claude-opus-4-7.toml rename providers/opper/models/{anthropic => }/claude-opus-4-8.toml (50%) create mode 100644 providers/opper/models/claude-opus-5.toml create mode 100644 providers/opper/models/claude-sonnet-4-5.toml create mode 100644 providers/opper/models/claude-sonnet-4-6.toml create mode 100644 providers/opper/models/claude-sonnet-5.toml create mode 100644 providers/opper/models/deepseek-v4-flash.toml create mode 100644 providers/opper/models/deepseek-v4-pro.toml create mode 100644 providers/opper/models/devstral-2512.toml create mode 100644 providers/opper/models/gemini-3-flash-preview.toml create mode 100644 providers/opper/models/gemini-3.1-pro-preview.toml create mode 100644 providers/opper/models/gemini-3.5-flash-lite.toml create mode 100644 providers/opper/models/gemini-3.5-flash.toml create mode 100644 providers/opper/models/gemini-3.7-flash.toml create mode 100644 providers/opper/models/gemini-3.8-flash.toml delete mode 100644 providers/opper/models/gemini/gemini-3-flash-preview.toml delete mode 100644 providers/opper/models/gemini/gemini-3.1-pro-preview.toml delete mode 100644 providers/opper/models/gemini/gemini-3.5-flash-lite.toml delete mode 100644 providers/opper/models/gemini/gemini-3.5-flash.toml create mode 100644 providers/opper/models/gemma-4-31b-it.toml create mode 100644 providers/opper/models/glm-5.2.toml create mode 100644 providers/opper/models/glm-5.3-flash.toml create mode 100644 providers/opper/models/glm-5.3.toml create mode 100644 providers/opper/models/gpt-5.3-chat-latest.toml create mode 100644 providers/opper/models/gpt-5.3-codex.toml create mode 100644 providers/opper/models/gpt-5.4-mini.toml create mode 100644 providers/opper/models/gpt-5.4-nano.toml create mode 100644 providers/opper/models/gpt-5.4-pro.toml create mode 100644 providers/opper/models/gpt-5.4.toml create mode 100644 providers/opper/models/gpt-5.5-pro.toml create mode 100644 providers/opper/models/gpt-5.5.toml create mode 100644 providers/opper/models/gpt-5.6-luna.toml create mode 100644 providers/opper/models/gpt-5.6-sol.toml create mode 100644 providers/opper/models/gpt-5.6-terra.toml create mode 100644 providers/opper/models/gpt-6-astra.toml create mode 100644 providers/opper/models/gpt-oss-120b.toml create mode 100644 providers/opper/models/gpt-oss-20b.toml create mode 100644 providers/opper/models/grok-4.3.toml create mode 100644 providers/opper/models/grok-4.5.toml create mode 100644 providers/opper/models/grok-4.6.toml create mode 100644 providers/opper/models/grok-build-0.1.toml create mode 100644 providers/opper/models/kimi-k3.toml create mode 100644 providers/opper/models/minimax-m2.7.toml create mode 100644 providers/opper/models/minimax-m3.toml delete mode 100644 providers/opper/models/minimax/m3.toml create mode 100644 providers/opper/models/mistral-large-2512.toml create mode 100644 providers/opper/models/mistral-small-2603.toml delete mode 100644 providers/opper/models/mistral/devstral-2512.toml delete mode 100644 providers/opper/models/mistral/mistral-large-2512.toml delete mode 100644 providers/opper/models/mistral/mistral-small-2603.toml delete mode 100644 providers/opper/models/moonshot/kimi-k3.toml rename providers/opper/models/{meta => }/muse-spark-1.2.toml (51%) create mode 100644 providers/opper/models/muse-spark-1.3.toml delete mode 100644 providers/opper/models/openai/gpt-5.3-chat-latest.toml delete mode 100644 providers/opper/models/openai/gpt-5.3-codex.toml delete mode 100644 providers/opper/models/openai/gpt-5.4-mini.toml delete mode 100644 providers/opper/models/openai/gpt-5.4-nano.toml delete mode 100644 providers/opper/models/openai/gpt-5.4-pro.toml delete mode 100644 providers/opper/models/openai/gpt-5.4.toml delete mode 100644 providers/opper/models/openai/gpt-5.5-pro.toml delete mode 100644 providers/opper/models/openai/gpt-5.5.toml delete mode 100644 providers/opper/models/openai/gpt-5.6-luna.toml delete mode 100644 providers/opper/models/openai/gpt-5.6-sol.toml delete mode 100644 providers/opper/models/openai/gpt-5.6-terra.toml delete mode 100644 providers/opper/models/perplexity/sonar-pro.toml delete mode 100644 providers/opper/models/perplexity/sonar-reasoning-pro.toml delete mode 100644 providers/opper/models/perplexity/sonar.toml create mode 100644 providers/opper/models/qwen3-coder-next.toml create mode 100644 providers/opper/models/qwen3.6-35b-a3b.toml create mode 100644 providers/opper/models/qwen3.8-2.4t-a95b.toml create mode 100644 providers/opper/models/qwen3.8-27b.toml create mode 100644 providers/opper/models/qwen3.8-max.toml create mode 100644 providers/opper/models/sonar-pro.toml create mode 100644 providers/opper/models/sonar-reasoning-pro.toml create mode 100644 providers/opper/models/sonar.toml delete mode 100644 providers/opper/models/vertexai/gemini-3.7-flash-eu.toml delete mode 100644 providers/opper/models/vertexai/gemini-3.7-flash.toml delete mode 100644 providers/opper/models/xai/grok-4.3.toml delete mode 100644 providers/opper/models/xai/grok-4.5.toml delete mode 100644 providers/opper/models/xai/grok-4.6.toml delete mode 100644 providers/opper/models/xai/grok-build-0.1.toml diff --git a/providers/opper/models/anthropic/claude-haiku-4-5.toml b/providers/opper/models/anthropic/claude-haiku-4-5.toml deleted file mode 100644 index 63e9f09896d..00000000000 --- a/providers/opper/models/anthropic/claude-haiku-4-5.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "anthropic/claude-haiku-4-5" -structured_output = true - -reasoning_options = [] - -[cost] -input = 1 -output = 5 -cache_read = 0.1 -cache_write = 1.25 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-opus-4-7.toml b/providers/opper/models/anthropic/claude-opus-4-7.toml deleted file mode 100644 index d4f86815775..00000000000 --- a/providers/opper/models/anthropic/claude-opus-4-7.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "anthropic/claude-opus-4-7" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - -[cost] -input = 5 -output = 25 -cache_read = 0.5 -cache_write = 6.25 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-opus-5.toml b/providers/opper/models/anthropic/claude-opus-5.toml deleted file mode 100644 index 634c2f987e6..00000000000 --- a/providers/opper/models/anthropic/claude-opus-5.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "anthropic/claude-opus-5" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - -[cost] -input = 5 -output = 25 -cache_read = 0.5 -cache_write = 6.25 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-sonnet-4-5.toml b/providers/opper/models/anthropic/claude-sonnet-4-5.toml deleted file mode 100644 index fdc69c21166..00000000000 --- a/providers/opper/models/anthropic/claude-sonnet-4-5.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "anthropic/claude-sonnet-4-5" -structured_output = true - -reasoning_options = [] - -[cost] -input = 3 -output = 15 -cache_read = 0.3 -cache_write = 3.75 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-sonnet-4-6.toml b/providers/opper/models/anthropic/claude-sonnet-4-6.toml deleted file mode 100644 index aeed8195b8f..00000000000 --- a/providers/opper/models/anthropic/claude-sonnet-4-6.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "anthropic/claude-sonnet-4-6" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "max"] - -[cost] -input = 3 -output = 15 -cache_read = 0.3 -cache_write = 3.75 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-sonnet-5.toml b/providers/opper/models/anthropic/claude-sonnet-5.toml deleted file mode 100644 index c49fbcbdf8a..00000000000 --- a/providers/opper/models/anthropic/claude-sonnet-5.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "anthropic/claude-sonnet-5" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - -[cost] -input = 2 -output = 10 -cache_read = 0.2 -cache_write = 2.5 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/claude-fable-5-1.toml b/providers/opper/models/claude-fable-5-1.toml new file mode 100644 index 00000000000..a17d2fb8c29 --- /dev/null +++ b/providers/opper/models/claude-fable-5-1.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-fable-5-1" +structured_output = true + +reasoning_options = [] + +[cost] +input = 10 +output = 50 +cache_read = 0.25 +cache_write = 12.5 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-fable-5.toml b/providers/opper/models/claude-fable-5.toml similarity index 50% rename from providers/opper/models/anthropic/claude-fable-5.toml rename to providers/opper/models/claude-fable-5.toml index 78ec64e8c7f..2bd15839fc0 100644 --- a/providers/opper/models/anthropic/claude-fable-5.toml +++ b/providers/opper/models/claude-fable-5.toml @@ -1,9 +1,9 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. base_model = "anthropic/claude-fable-5" structured_output = true -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] +reasoning_options = [] [cost] input = 10 diff --git a/providers/opper/models/claude-haiku-4-5.toml b/providers/opper/models/claude-haiku-4-5.toml new file mode 100644 index 00000000000..79bd6828653 --- /dev/null +++ b/providers/opper/models/claude-haiku-4-5.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-haiku-4-5" +structured_output = true + +reasoning_options = [] + +[cost] +input = 1.1 +output = 5.5 +cache_read = 0.11 +cache_write = 1.375 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-opus-4-5.toml b/providers/opper/models/claude-opus-4-5.toml similarity index 50% rename from providers/opper/models/anthropic/claude-opus-4-5.toml rename to providers/opper/models/claude-opus-4-5.toml index 45ac31901cf..fe7f973c895 100644 --- a/providers/opper/models/anthropic/claude-opus-4-5.toml +++ b/providers/opper/models/claude-opus-4-5.toml @@ -1,9 +1,9 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. base_model = "anthropic/claude-opus-4-5" structured_output = true -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] +reasoning_options = [] [cost] input = 5 diff --git a/providers/opper/models/anthropic/claude-opus-4-6.toml b/providers/opper/models/claude-opus-4-6.toml similarity index 50% rename from providers/opper/models/anthropic/claude-opus-4-6.toml rename to providers/opper/models/claude-opus-4-6.toml index dd56adf14b9..548c9efc22f 100644 --- a/providers/opper/models/anthropic/claude-opus-4-6.toml +++ b/providers/opper/models/claude-opus-4-6.toml @@ -1,9 +1,9 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. base_model = "anthropic/claude-opus-4-6" structured_output = true -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "max"] +reasoning_options = [] [cost] input = 5 diff --git a/providers/opper/models/claude-opus-4-7.toml b/providers/opper/models/claude-opus-4-7.toml new file mode 100644 index 00000000000..8b222c7b3c4 --- /dev/null +++ b/providers/opper/models/claude-opus-4-7.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-opus-4-7" +structured_output = true + +reasoning_options = [] + +[cost] +input = 5.5 +output = 27.5 +cache_read = 0.55 +cache_write = 6.875 + +[limit] +output = 64000 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-opus-4-8.toml b/providers/opper/models/claude-opus-4-8.toml similarity index 50% rename from providers/opper/models/anthropic/claude-opus-4-8.toml rename to providers/opper/models/claude-opus-4-8.toml index 83e1e7059c1..0d7904267ce 100644 --- a/providers/opper/models/anthropic/claude-opus-4-8.toml +++ b/providers/opper/models/claude-opus-4-8.toml @@ -1,9 +1,9 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. base_model = "anthropic/claude-opus-4-8" structured_output = true -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] +reasoning_options = [] [cost] input = 5 diff --git a/providers/opper/models/claude-opus-5.toml b/providers/opper/models/claude-opus-5.toml new file mode 100644 index 00000000000..052e3ac7cd9 --- /dev/null +++ b/providers/opper/models/claude-opus-5.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-opus-5" +structured_output = true + +reasoning_options = [] + +[cost] +input = 5.5 +output = 27.5 +cache_read = 0.55 +cache_write = 6.875 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/claude-sonnet-4-5.toml b/providers/opper/models/claude-sonnet-4-5.toml new file mode 100644 index 00000000000..953685007ad --- /dev/null +++ b/providers/opper/models/claude-sonnet-4-5.toml @@ -0,0 +1,23 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "anthropic/claude-sonnet-4-5" +structured_output = true + +reasoning_options = [] + +[cost] +input = 3.3 +output = 16.5 +cache_read = 0.33 +cache_write = 4.125 + +[[cost.tiers]] +tier = { type = "context", size = 200000 } +input = 6.6 +output = 24.75 +cache_read = 0.66 +cache_write = 8.25 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/claude-sonnet-4-6.toml b/providers/opper/models/claude-sonnet-4-6.toml new file mode 100644 index 00000000000..8efa496556f --- /dev/null +++ b/providers/opper/models/claude-sonnet-4-6.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-sonnet-4-6" +structured_output = true + +reasoning_options = [] + +[cost] +input = 3.3 +output = 16.5 +cache_read = 0.33 +cache_write = 4.125 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/claude-sonnet-5.toml b/providers/opper/models/claude-sonnet-5.toml new file mode 100644 index 00000000000..3b91951e85f --- /dev/null +++ b/providers/opper/models/claude-sonnet-5.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-sonnet-5" +structured_output = true + +reasoning_options = [] + +[cost] +input = 2.2 +output = 11 +cache_read = 0.22 +cache_write = 2.75 + +[limit] +output = 64000 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/deepseek-v4-flash.toml b/providers/opper/models/deepseek-v4-flash.toml new file mode 100644 index 00000000000..b332afbbd8d --- /dev/null +++ b/providers/opper/models/deepseek-v4-flash.toml @@ -0,0 +1,12 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "deepseek/deepseek-v4-flash" + +reasoning_options = [] + +[cost] +input = 0.25 +output = 0.66 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/deepseek-v4-pro.toml b/providers/opper/models/deepseek-v4-pro.toml new file mode 100644 index 00000000000..fb6c5faa0c1 --- /dev/null +++ b/providers/opper/models/deepseek-v4-pro.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "deepseek/deepseek-v4-pro" + +reasoning_options = [] + +[cost] +input = 1.78812 +output = 3.57624 + +[limit] +output = 65536 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/devstral-2512.toml b/providers/opper/models/devstral-2512.toml new file mode 100644 index 00000000000..d3595c76c26 --- /dev/null +++ b/providers/opper/models/devstral-2512.toml @@ -0,0 +1,11 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +base_model = "mistral/devstral-2512" +structured_output = true + +[cost] +input = 0.4 +output = 2 + +[limit] +context = 256000 +output = 8192 diff --git a/providers/opper/models/gemini-3-flash-preview.toml b/providers/opper/models/gemini-3-flash-preview.toml new file mode 100644 index 00000000000..0e6941f334a --- /dev/null +++ b/providers/opper/models/gemini-3-flash-preview.toml @@ -0,0 +1,13 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemini-3-flash-preview" + +reasoning_options = [] + +[cost] +input = 0.5 +output = 3 +cache_read = 0.05 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini-3.1-pro-preview.toml b/providers/opper/models/gemini-3.1-pro-preview.toml new file mode 100644 index 00000000000..e25eda06eb1 --- /dev/null +++ b/providers/opper/models/gemini-3.1-pro-preview.toml @@ -0,0 +1,20 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "google/gemini-3.1-pro-preview" + +reasoning_options = [] + +[cost] +input = 2 +output = 12 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 200000 } +input = 4 +output = 18 +cache_read = 0.4 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini-3.5-flash-lite.toml b/providers/opper/models/gemini-3.5-flash-lite.toml new file mode 100644 index 00000000000..0f10955993d --- /dev/null +++ b/providers/opper/models/gemini-3.5-flash-lite.toml @@ -0,0 +1,13 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemini-3.5-flash-lite" + +reasoning_options = [] + +[cost] +input = 0.3 +output = 2.5 +cache_read = 0.03 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini-3.5-flash.toml b/providers/opper/models/gemini-3.5-flash.toml new file mode 100644 index 00000000000..b597560116a --- /dev/null +++ b/providers/opper/models/gemini-3.5-flash.toml @@ -0,0 +1,13 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemini-3.5-flash" + +reasoning_options = [] + +[cost] +input = 1.5 +output = 9 +cache_read = 0.15 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini-3.7-flash.toml b/providers/opper/models/gemini-3.7-flash.toml new file mode 100644 index 00000000000..514a0c8cd9f --- /dev/null +++ b/providers/opper/models/gemini-3.7-flash.toml @@ -0,0 +1,13 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemini-3.7-flash" + +reasoning_options = [] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini-3.8-flash.toml b/providers/opper/models/gemini-3.8-flash.toml new file mode 100644 index 00000000000..b7dfba7907c --- /dev/null +++ b/providers/opper/models/gemini-3.8-flash.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# All five members have params but omit params.temperature; Opper drops sampling controls +# for such cards before dispatch, so temperature is accepted but not respected. +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemini-3.8-flash" +temperature = false + +reasoning_options = [] + +[cost] +input = 0.825 +output = 4.125 +cache_read = 0.0825 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini/gemini-3-flash-preview.toml b/providers/opper/models/gemini/gemini-3-flash-preview.toml deleted file mode 100644 index b5b4efd521b..00000000000 --- a/providers/opper/models/gemini/gemini-3-flash-preview.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3-flash-preview" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 0.5 -output = 3 -cache_read = 0.05 diff --git a/providers/opper/models/gemini/gemini-3.1-pro-preview.toml b/providers/opper/models/gemini/gemini-3.1-pro-preview.toml deleted file mode 100644 index 8cb3844011e..00000000000 --- a/providers/opper/models/gemini/gemini-3.1-pro-preview.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemini-3.1-pro-preview" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 2 -output = 12 -cache_read = 0.2 - -[[cost.tiers]] -tier = { size = 200_000 } -input = 4.00 -output = 18.00 -cache_read = 0.40 diff --git a/providers/opper/models/gemini/gemini-3.5-flash-lite.toml b/providers/opper/models/gemini/gemini-3.5-flash-lite.toml deleted file mode 100644 index e5452932870..00000000000 --- a/providers/opper/models/gemini/gemini-3.5-flash-lite.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3.5-flash-lite" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 0.3 -output = 2.5 -cache_read = 0.03 diff --git a/providers/opper/models/gemini/gemini-3.5-flash.toml b/providers/opper/models/gemini/gemini-3.5-flash.toml deleted file mode 100644 index aa9842db8a7..00000000000 --- a/providers/opper/models/gemini/gemini-3.5-flash.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3.5-flash" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 1.5 -output = 9 -cache_read = 0.15 diff --git a/providers/opper/models/gemma-4-31b-it.toml b/providers/opper/models/gemma-4-31b-it.toml new file mode 100644 index 00000000000..8ee52039c78 --- /dev/null +++ b/providers/opper/models/gemma-4-31b-it.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemma-4-31b-it" + +reasoning_options = [] + +[cost] +input = 0.46488 +output = 2.44062 + +[limit] +context = 256000 +output = 8192 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/glm-5.2.toml b/providers/opper/models/glm-5.2.toml new file mode 100644 index 00000000000..bc1044a1917 --- /dev/null +++ b/providers/opper/models/glm-5.2.toml @@ -0,0 +1,12 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "zhipuai/glm-5.2" + +reasoning_options = [] + +[cost] +input = 1.62708 +output = 5.811 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/glm-5.3-flash.toml b/providers/opper/models/glm-5.3-flash.toml new file mode 100644 index 00000000000..c7db468e1b7 --- /dev/null +++ b/providers/opper/models/glm-5.3-flash.toml @@ -0,0 +1,21 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "zhipuai/glm-5.3-flash" +attachment = false +structured_output = false + +reasoning_options = [] + +[cost] +input = 0.2 +output = 0.5 +cache_read = 0.07 + +[limit] +output = 128000 + +[modalities] +input = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/glm-5.3.toml b/providers/opper/models/glm-5.3.toml new file mode 100644 index 00000000000..b6c8534060c --- /dev/null +++ b/providers/opper/models/glm-5.3.toml @@ -0,0 +1,14 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: members do not share a supported effort level. +base_model = "zhipuai/glm-5.3" +structured_output = false + +reasoning_options = [] + +[cost] +input = 1.75 +output = 4.6488 +cache_read = 0.44 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.3-chat-latest.toml b/providers/opper/models/gpt-5.3-chat-latest.toml new file mode 100644 index 00000000000..7568b754041 --- /dev/null +++ b/providers/opper/models/gpt-5.3-chat-latest.toml @@ -0,0 +1,9 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Live pool call on 2026-09-08 reports this upstream model has been deprecated. +base_model = "openai/gpt-5.3-chat-latest" +status = "deprecated" + +[cost] +input = 1.75 +output = 14 +cache_read = 0.175 diff --git a/providers/opper/models/gpt-5.3-codex.toml b/providers/opper/models/gpt-5.3-codex.toml new file mode 100644 index 00000000000..ea0de06a8c5 --- /dev/null +++ b/providers/opper/models/gpt-5.3-codex.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; OpenAI native levels on all pool members. +# Catalog context equals the lab input cap; retain the lab total context window. +base_model = "openai/gpt-5.3-codex" +temperature = false + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 1.75 +output = 14 +cache_read = 0.175 + +[modalities] +input = ["text", "pdf"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.4-mini.toml b/providers/opper/models/gpt-5.4-mini.toml new file mode 100644 index 00000000000..5daca15ba01 --- /dev/null +++ b/providers/opper/models/gpt-5.4-mini.toml @@ -0,0 +1,14 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Native OpenAI effort levels verified through this single-member pool on 2026-09-08. +# Effort: top-level reasoning_effort; levels shared by the pool members. +base_model = "openai/gpt-5.4-mini" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 0.75 +output = 4.5 +cache_read = 0.075 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.4-nano.toml b/providers/opper/models/gpt-5.4-nano.toml new file mode 100644 index 00000000000..dbbf8bf940a --- /dev/null +++ b/providers/opper/models/gpt-5.4-nano.toml @@ -0,0 +1,14 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; OpenAI native levels on all pool members. +base_model = "openai/gpt-5.4-nano" +temperature = false + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 0.2 +output = 1.25 +cache_read = 0.02 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.4-pro.toml b/providers/opper/models/gpt-5.4-pro.toml new file mode 100644 index 00000000000..03bb664a101 --- /dev/null +++ b/providers/opper/models/gpt-5.4-pro.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.4-pro" + +reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] + +[cost] +input = 30 +output = 180 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 60 +output = 270 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.4.toml b/providers/opper/models/gpt-5.4.toml new file mode 100644 index 00000000000..b83170741b4 --- /dev/null +++ b/providers/opper/models/gpt-5.4.toml @@ -0,0 +1,20 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.4" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 2.5 +output = 15 +cache_read = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 5 +output = 22.5 +cache_read = 0.5 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.5-pro.toml b/providers/opper/models/gpt-5.5-pro.toml new file mode 100644 index 00000000000..254bf5b66c9 --- /dev/null +++ b/providers/opper/models/gpt-5.5-pro.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Native OpenAI effort levels verified through this single-member pool on 2026-09-08. +# All pool members publish flat rates: no pricing.thresholds or input_surcharge_threshold_tokens. +# Opper billing uses this schedule; native context surcharges do not apply to these pools. +# Effort: top-level reasoning_effort; levels shared by the pool members. +base_model = "openai/gpt-5.5-pro" + +reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] + +[cost] +input = 30 +output = 180 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.5.toml b/providers/opper/models/gpt-5.5.toml new file mode 100644 index 00000000000..b05f12abb7f --- /dev/null +++ b/providers/opper/models/gpt-5.5.toml @@ -0,0 +1,20 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.5" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 5 +output = 30 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 10 +output = 45 +cache_read = 1 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.6-luna.toml b/providers/opper/models/gpt-5.6-luna.toml new file mode 100644 index 00000000000..254ff2a8555 --- /dev/null +++ b/providers/opper/models/gpt-5.6-luna.toml @@ -0,0 +1,25 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.6-luna" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.2 +output = 1.2 +cache_read = 0.02 +cache_write = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 0.4 +output = 1.8 +cache_read = 0.04 +cache_write = 0.5 + +[limit] +context = 1000000 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.6-sol.toml b/providers/opper/models/gpt-5.6-sol.toml new file mode 100644 index 00000000000..e7b4f94f5f8 --- /dev/null +++ b/providers/opper/models/gpt-5.6-sol.toml @@ -0,0 +1,25 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.6-sol" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 5 +output = 30 +cache_read = 0.5 +cache_write = 6.25 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 10 +output = 45 +cache_read = 1 +cache_write = 12.5 + +[limit] +context = 1000000 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.6-terra.toml b/providers/opper/models/gpt-5.6-terra.toml new file mode 100644 index 00000000000..209fe14cd29 --- /dev/null +++ b/providers/opper/models/gpt-5.6-terra.toml @@ -0,0 +1,25 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.6-terra" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 2 +output = 12 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 4 +output = 18 +cache_read = 0.4 +cache_write = 5 + +[limit] +context = 1000000 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-6-astra.toml b/providers/opper/models/gpt-6-astra.toml new file mode 100644 index 00000000000..62f6292c6cb --- /dev/null +++ b/providers/opper/models/gpt-6-astra.toml @@ -0,0 +1,25 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-6-astra" + +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 20 +output = 75 +cache_read = 2 +cache_write = 25 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-oss-120b.toml b/providers/opper/models/gpt-oss-120b.toml new file mode 100644 index 00000000000..1a576848f91 --- /dev/null +++ b/providers/opper/models/gpt-oss-120b.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "openai/gpt-oss-120b" + +reasoning_options = [] + +[cost] +input = 1.1622 +output = 4.88124 + +[limit] +context = 128000 +output = 8192 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-oss-20b.toml b/providers/opper/models/gpt-oss-20b.toml new file mode 100644 index 00000000000..ce3715afc7b --- /dev/null +++ b/providers/opper/models/gpt-oss-20b.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "openai/gpt-oss-20b" + +reasoning_options = [] + +[cost] +input = 0.11622 +output = 0.488124 + +[limit] +context = 128000 +output = 8192 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/grok-4.3.toml b/providers/opper/models/grok-4.3.toml new file mode 100644 index 00000000000..e40b4396000 --- /dev/null +++ b/providers/opper/models/grok-4.3.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# All pool members publish flat rates: no pricing.thresholds or input_surcharge_threshold_tokens. +# Opper billing uses this schedule; native context surcharges do not apply to these pools. +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "xai/grok-4.3" + +reasoning_options = [] + +[cost] +input = 1.25 +output = 2.5 +cache_read = 0.2 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/grok-4.5.toml b/providers/opper/models/grok-4.5.toml new file mode 100644 index 00000000000..8cb4220459f --- /dev/null +++ b/providers/opper/models/grok-4.5.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# All pool members publish flat rates: no pricing.thresholds or input_surcharge_threshold_tokens. +# Opper billing uses this schedule; native context surcharges do not apply to these pools. +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "xai/grok-4.5" + +reasoning_options = [] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/grok-4.6.toml b/providers/opper/models/grok-4.6.toml new file mode 100644 index 00000000000..36c860ca4c1 --- /dev/null +++ b/providers/opper/models/grok-4.6.toml @@ -0,0 +1,20 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "xai/grok-4.6" + +reasoning_options = [] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200000 } +input = 4 +output = 12 +cache_read = 1 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/grok-build-0.1.toml b/providers/opper/models/grok-build-0.1.toml new file mode 100644 index 00000000000..13eef952286 --- /dev/null +++ b/providers/opper/models/grok-build-0.1.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# All pool members publish flat rates: no pricing.thresholds or input_surcharge_threshold_tokens. +# Opper billing uses this schedule; native context surcharges do not apply to these pools. +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "xai/grok-build-0.1" + +reasoning_options = [] + +[cost] +input = 1 +output = 2 +cache_read = 0.2 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/kimi-k3.toml b/providers/opper/models/kimi-k3.toml new file mode 100644 index 00000000000..b4519ce43f6 --- /dev/null +++ b/providers/opper/models/kimi-k3.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: members do not share a supported effort level. +base_model = "moonshotai/kimi-k3" +attachment = false + +reasoning_options = [] + +[cost] +input = 3 +output = 15 + +[modalities] +input = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/minimax-m2.7.toml b/providers/opper/models/minimax-m2.7.toml new file mode 100644 index 00000000000..ed41ef98618 --- /dev/null +++ b/providers/opper/models/minimax-m2.7.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "minimax/MiniMax-M2.7" +structured_output = true + +reasoning_options = [] + +[cost] +input = 0.69732 +output = 2.78928 + +[limit] +context = 196608 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/minimax-m3.toml b/providers/opper/models/minimax-m3.toml new file mode 100644 index 00000000000..fae074d323c --- /dev/null +++ b/providers/opper/models/minimax-m3.toml @@ -0,0 +1,28 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "minimax/MiniMax-M3" +structured_output = true + +reasoning_options = [] + +[cost] +input = 0.6 +output = 2.4 +cache_read = 0.12 + +[[cost.tiers]] +tier = { type = "context", size = 524288 } +input = 1.2 +output = 4.8 +cache_read = 0.24 + +[limit] +context = 1000000 +output = 131072 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/minimax/m3.toml b/providers/opper/models/minimax/m3.toml deleted file mode 100644 index d53af696874..00000000000 --- a/providers/opper/models/minimax/m3.toml +++ /dev/null @@ -1,24 +0,0 @@ -# Opper billed schedule, verified against GET /v3/compat/models and /v3/models (2026-08-20): -# base 0.6/2.4 (cache 0.12) up to 524_288, then 2x. This differs from MiniMax first-party -# list (0.30/1.20 base, upper band 0.60/2.40); the delta is flagged to Opper catalog -# maintainers and this entry will be updated if the billed schedule changes. -base_model = "minimax/MiniMax-M3" - -reasoning_options = [] - -[cost] -input = 0.6 -output = 2.4 -cache_read = 0.12 - -[[cost.tiers]] -tier = { size = 524_288 } -input = 1.2 -output = 4.8 -cache_read = 0.24 - -[limit] -context = 1_048_576 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/mistral-large-2512.toml b/providers/opper/models/mistral-large-2512.toml new file mode 100644 index 00000000000..1f67818b4a1 --- /dev/null +++ b/providers/opper/models/mistral-large-2512.toml @@ -0,0 +1,11 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +base_model = "mistral/mistral-large-2512" +structured_output = true + +[cost] +input = 0.5 +output = 1.5 + +[limit] +context = 256000 +output = 8192 diff --git a/providers/opper/models/mistral-small-2603.toml b/providers/opper/models/mistral-small-2603.toml new file mode 100644 index 00000000000..cff65202966 --- /dev/null +++ b/providers/opper/models/mistral-small-2603.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "mistral/mistral-small-2603" +structured_output = false + +reasoning_options = [] + +[cost] +input = 0.5811 +output = 2.44062 + +[limit] +output = 8192 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/mistral/devstral-2512.toml b/providers/opper/models/mistral/devstral-2512.toml deleted file mode 100644 index 62661ddc309..00000000000 --- a/providers/opper/models/mistral/devstral-2512.toml +++ /dev/null @@ -1,5 +0,0 @@ -base_model = "mistral/devstral-2512" - -[cost] -input = 0.4 -output = 2 diff --git a/providers/opper/models/mistral/mistral-large-2512.toml b/providers/opper/models/mistral/mistral-large-2512.toml deleted file mode 100644 index 49df75a6342..00000000000 --- a/providers/opper/models/mistral/mistral-large-2512.toml +++ /dev/null @@ -1,5 +0,0 @@ -base_model = "mistral/mistral-large-2512" - -[cost] -input = 0.5 -output = 1.5 diff --git a/providers/opper/models/mistral/mistral-small-2603.toml b/providers/opper/models/mistral/mistral-small-2603.toml deleted file mode 100644 index c82dfc68357..00000000000 --- a/providers/opper/models/mistral/mistral-small-2603.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "mistral/mistral-small-2603" - -[[reasoning_options]] -type = "effort" -values = ["none", "high"] - -[cost] -input = 0.15 -output = 0.6 diff --git a/providers/opper/models/moonshot/kimi-k3.toml b/providers/opper/models/moonshot/kimi-k3.toml deleted file mode 100644 index 0862dd21e4c..00000000000 --- a/providers/opper/models/moonshot/kimi-k3.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "moonshotai/kimi-k3" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 3 -output = 15 -cache_read = 0.3 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/meta/muse-spark-1.2.toml b/providers/opper/models/muse-spark-1.2.toml similarity index 51% rename from providers/opper/models/meta/muse-spark-1.2.toml rename to providers/opper/models/muse-spark-1.2.toml index aea877da54a..200e6da0293 100644 --- a/providers/opper/models/meta/muse-spark-1.2.toml +++ b/providers/opper/models/muse-spark-1.2.toml @@ -1,3 +1,5 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. base_model = "meta/muse-spark-1.2" reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high", "xhigh"] }] @@ -6,3 +8,6 @@ reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "h input = 1.25 output = 4.25 cache_read = 0.15 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/muse-spark-1.3.toml b/providers/opper/models/muse-spark-1.3.toml new file mode 100644 index 00000000000..bba0703ecde --- /dev/null +++ b/providers/opper/models/muse-spark-1.3.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Standard-tier pool forwards native effort unchanged; max verified live on 2026-09-08. +# The catalog reasoning.supported list omits max, but the serving path accepts it. +# Effort: top-level reasoning_effort; levels shared by the pool members. +base_model = "meta/muse-spark-1.3" + +reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 1.25 +output = 4.25 +cache_read = 0.15 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/openai/gpt-5.3-chat-latest.toml b/providers/opper/models/openai/gpt-5.3-chat-latest.toml deleted file mode 100644 index 197284ef5f0..00000000000 --- a/providers/opper/models/openai/gpt-5.3-chat-latest.toml +++ /dev/null @@ -1,6 +0,0 @@ -base_model = "openai/gpt-5.3-chat-latest" - -[cost] -input = 1.75 -output = 14 -cache_read = 0.175 diff --git a/providers/opper/models/openai/gpt-5.3-codex.toml b/providers/opper/models/openai/gpt-5.3-codex.toml deleted file mode 100644 index e85f5642a44..00000000000 --- a/providers/opper/models/openai/gpt-5.3-codex.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-5.3-codex" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] - -[cost] -input = 1.75 -output = 14 -cache_read = 0.175 diff --git a/providers/opper/models/openai/gpt-5.4-mini.toml b/providers/opper/models/openai/gpt-5.4-mini.toml deleted file mode 100644 index a21229d609e..00000000000 --- a/providers/opper/models/openai/gpt-5.4-mini.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-5.4-mini" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] - -[cost] -input = 0.75 -output = 4.5 -cache_read = 0.075 diff --git a/providers/opper/models/openai/gpt-5.4-nano.toml b/providers/opper/models/openai/gpt-5.4-nano.toml deleted file mode 100644 index 0db45e05b9e..00000000000 --- a/providers/opper/models/openai/gpt-5.4-nano.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-5.4-nano" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] - -[cost] -input = 0.2 -output = 1.25 -cache_read = 0.02 diff --git a/providers/opper/models/openai/gpt-5.4-pro.toml b/providers/opper/models/openai/gpt-5.4-pro.toml deleted file mode 100644 index 6da8f6aa095..00000000000 --- a/providers/opper/models/openai/gpt-5.4-pro.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/gpt-5.4-pro" - -reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] - -[cost] -input = 30 -output = 180 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 60.00 -output = 270.00 diff --git a/providers/opper/models/openai/gpt-5.4.toml b/providers/opper/models/openai/gpt-5.4.toml deleted file mode 100644 index 7398a595b17..00000000000 --- a/providers/opper/models/openai/gpt-5.4.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "openai/gpt-5.4" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] - -[cost] -input = 2.5 -output = 15 -cache_read = 0.25 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 5.00 -output = 22.50 -cache_read = 0.50 diff --git a/providers/opper/models/openai/gpt-5.5-pro.toml b/providers/opper/models/openai/gpt-5.5-pro.toml deleted file mode 100644 index 77d5371cb8e..00000000000 --- a/providers/opper/models/openai/gpt-5.5-pro.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/gpt-5.5-pro" - -reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] - -[cost] -input = 30 -output = 180 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 60.00 -output = 270.00 diff --git a/providers/opper/models/openai/gpt-5.5.toml b/providers/opper/models/openai/gpt-5.5.toml deleted file mode 100644 index a58539b3e09..00000000000 --- a/providers/opper/models/openai/gpt-5.5.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "openai/gpt-5.5" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] - -[cost] -input = 5 -output = 30 -cache_read = 0.5 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 10.00 -output = 45.00 -cache_read = 1.00 diff --git a/providers/opper/models/openai/gpt-5.6-luna.toml b/providers/opper/models/openai/gpt-5.6-luna.toml deleted file mode 100644 index 07d577afa36..00000000000 --- a/providers/opper/models/openai/gpt-5.6-luna.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "openai/gpt-5.6-luna" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] - -[cost] -input = 0.2 -output = 1.2 -cache_read = 0.02 -cache_write = 0.25 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 0.40 -output = 1.80 -cache_read = 0.04 -cache_write = 0.50 diff --git a/providers/opper/models/openai/gpt-5.6-sol.toml b/providers/opper/models/openai/gpt-5.6-sol.toml deleted file mode 100644 index 31470933d23..00000000000 --- a/providers/opper/models/openai/gpt-5.6-sol.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "openai/gpt-5.6-sol" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] - -[cost] -input = 5 -output = 30 -cache_read = 0.5 -cache_write = 6.25 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 10.00 -output = 45.00 -cache_read = 1.00 -cache_write = 12.50 diff --git a/providers/opper/models/openai/gpt-5.6-terra.toml b/providers/opper/models/openai/gpt-5.6-terra.toml deleted file mode 100644 index 9af64197094..00000000000 --- a/providers/opper/models/openai/gpt-5.6-terra.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "openai/gpt-5.6-terra" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] - -[cost] -input = 2 -output = 12 -cache_read = 0.2 -cache_write = 2.5 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 4.00 -output = 18.00 -cache_read = 0.40 -cache_write = 5.00 diff --git a/providers/opper/models/perplexity/sonar-pro.toml b/providers/opper/models/perplexity/sonar-pro.toml deleted file mode 100644 index f03972e3868..00000000000 --- a/providers/opper/models/perplexity/sonar-pro.toml +++ /dev/null @@ -1,5 +0,0 @@ -base_model = "perplexity/sonar-pro" - -[cost] -input = 3 -output = 15 diff --git a/providers/opper/models/perplexity/sonar-reasoning-pro.toml b/providers/opper/models/perplexity/sonar-reasoning-pro.toml deleted file mode 100644 index 6189d9d9ff8..00000000000 --- a/providers/opper/models/perplexity/sonar-reasoning-pro.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "perplexity/sonar-reasoning-pro" - -reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] - -[cost] -input = 2 -output = 8 diff --git a/providers/opper/models/perplexity/sonar.toml b/providers/opper/models/perplexity/sonar.toml deleted file mode 100644 index a6c921f515e..00000000000 --- a/providers/opper/models/perplexity/sonar.toml +++ /dev/null @@ -1,5 +0,0 @@ -base_model = "perplexity/sonar" - -[cost] -input = 1 -output = 1 diff --git a/providers/opper/models/qwen3-coder-next.toml b/providers/opper/models/qwen3-coder-next.toml new file mode 100644 index 00000000000..2816f8c2b7e --- /dev/null +++ b/providers/opper/models/qwen3-coder-next.toml @@ -0,0 +1,6 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +base_model = "alibaba/qwen3-coder-next" + +[cost] +input = 0.5811 +output = 2.3244 diff --git a/providers/opper/models/qwen3.6-35b-a3b.toml b/providers/opper/models/qwen3.6-35b-a3b.toml new file mode 100644 index 00000000000..f7e98234caf --- /dev/null +++ b/providers/opper/models/qwen3.6-35b-a3b.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "alibaba/qwen3.6-35b-a3b" + +reasoning_options = [] + +[cost] +input = 0.248 +output = 1.485 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/qwen3.8-2.4t-a95b.toml b/providers/opper/models/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..db70485f41e --- /dev/null +++ b/providers/opper/models/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,13 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "alibaba/qwen3.8-2.4t-a95b" + +reasoning_options = [] + +[cost] +input = 2.5 +output = 6 +cache_read = 0.63 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/qwen3.8-27b.toml b/providers/opper/models/qwen3.8-27b.toml new file mode 100644 index 00000000000..8cd1f5f6882 --- /dev/null +++ b/providers/opper/models/qwen3.8-27b.toml @@ -0,0 +1,17 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +base_model = "alibaba/qwen3.8-27b" +attachment = false +structured_output = false + +reasoning_options = [{ type = "effort", values = ["low", "medium", "xhigh"] }] + +[cost] +input = 0.5811 +output = 3 + +[modalities] +input = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/qwen3.8-max.toml b/providers/opper/models/qwen3.8-max.toml new file mode 100644 index 00000000000..65905a41b73 --- /dev/null +++ b/providers/opper/models/qwen3.8-max.toml @@ -0,0 +1,20 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide graded effort or toggle: members share only reasoning_effort=none. +base_model = "alibaba/qwen3.8-max" +structured_output = true + +reasoning_options = [] + +[cost] +input = 2 +output = 6 +cache_read = 0.25 + +[limit] +context = 983616 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/sonar-pro.toml b/providers/opper/models/sonar-pro.toml new file mode 100644 index 00000000000..de476cca476 --- /dev/null +++ b/providers/opper/models/sonar-pro.toml @@ -0,0 +1,14 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +base_model = "perplexity/sonar-pro" +attachment = false +structured_output = true + +[cost] +input = 3 +output = 15 + +[limit] +output = 8000 + +[modalities] +input = ["text"] diff --git a/providers/opper/models/sonar-reasoning-pro.toml b/providers/opper/models/sonar-reasoning-pro.toml new file mode 100644 index 00000000000..98d5cd92398 --- /dev/null +++ b/providers/opper/models/sonar-reasoning-pro.toml @@ -0,0 +1,17 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "perplexity/sonar-reasoning-pro" +attachment = false +structured_output = true + +reasoning_options = [] + +[cost] +input = 2 +output = 8 + +[modalities] +input = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/sonar.toml b/providers/opper/models/sonar.toml new file mode 100644 index 00000000000..7242f639443 --- /dev/null +++ b/providers/opper/models/sonar.toml @@ -0,0 +1,7 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +base_model = "perplexity/sonar" +structured_output = true + +[cost] +input = 1 +output = 1 diff --git a/providers/opper/models/vertexai/gemini-3.7-flash-eu.toml b/providers/opper/models/vertexai/gemini-3.7-flash-eu.toml deleted file mode 100644 index 03e8b29cab7..00000000000 --- a/providers/opper/models/vertexai/gemini-3.7-flash-eu.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "google/gemini-3.7-flash" -name = "Gemini 3.7 Flash (EU)" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 0.75 -output = 3.75 -cache_read = 0.075 diff --git a/providers/opper/models/vertexai/gemini-3.7-flash.toml b/providers/opper/models/vertexai/gemini-3.7-flash.toml deleted file mode 100644 index e9bacad26c9..00000000000 --- a/providers/opper/models/vertexai/gemini-3.7-flash.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3.7-flash" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 0.75 -output = 3.75 -cache_read = 0.075 diff --git a/providers/opper/models/xai/grok-4.3.toml b/providers/opper/models/xai/grok-4.3.toml deleted file mode 100644 index 329b82bb4ac..00000000000 --- a/providers/opper/models/xai/grok-4.3.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "xai/grok-4.3" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high"] - -[cost] -input = 1.25 -output = 2.5 -cache_read = 0.2 - -[[cost.tiers]] -tier = { size = 200_000 } -input = 2.5 -output = 5 -cache_read = 0.4 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/xai/grok-4.5.toml b/providers/opper/models/xai/grok-4.5.toml deleted file mode 100644 index fd10a1caace..00000000000 --- a/providers/opper/models/xai/grok-4.5.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "xai/grok-4.5" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 2 -output = 6 -cache_read = 0.3 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 4 -output = 12 -cache_read = 0.6 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/xai/grok-4.6.toml b/providers/opper/models/xai/grok-4.6.toml deleted file mode 100644 index 1f8386fd483..00000000000 --- a/providers/opper/models/xai/grok-4.6.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "xai/grok-4.6" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh"] - -[cost] -input = 2 -output = 6 -cache_read = 0.5 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 4 -output = 12 -cache_read = 1 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/xai/grok-build-0.1.toml b/providers/opper/models/xai/grok-build-0.1.toml deleted file mode 100644 index b2c4aeeafad..00000000000 --- a/providers/opper/models/xai/grok-build-0.1.toml +++ /dev/null @@ -1,17 +0,0 @@ -base_model = "xai/grok-build-0.1" - -reasoning_options = [] - -[cost] -input = 1 -output = 2 -cache_read = 0.2 - -[[cost.tiers]] -tier = { size = 200_000 } -input = 2.00 -output = 4.00 -cache_read = 0.40 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/provider.toml b/providers/opper/provider.toml index 7d14eff2de3..7ccd9102a07 100644 --- a/providers/opper/provider.toml +++ b/providers/opper/provider.toml @@ -1,13 +1,20 @@ +# Sources (2026-09-08): https://api.opper.ai/v3/models?limit=0&type=llm +# and authenticated https://api.opper.ai/v3/compat/models?type=model,pool. +# Model IDs are bare Opper pool names; Opper selects the underlying provider. +# Prices are USD/MTok ceilings across pool members, not a fixed rate for every +# request. Cache-read discounts are listed only when all priced members offer +# them. Context tiers include the highest marginal rates and whole-request +# surcharges, so estimates can exceed the actual bill. No currency conversion +# is needed: the catalog already converts non-USD provider prices to USD. +# Limits do not exceed either the lab model or the pool's smallest known limit; +# input modalities are restricted to those shared by the lab and pool members. +# POST /v3/compat/chat/completions accepts top-level reasoning_effort. Native +# thinking toggles and reasoning-token budgets are not exposed on this surface. +# Effort options describe common pool controls; [] means no pool-wide control, +# including pools where some adapters ignore effort. Reasoning output, when +# present, uses message.reasoning_content or streaming delta.reasoning_content. name = "Opper" env = ["OPPER_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://api.opper.ai/v3/compat" -# Reasoning HTTP format (measured against POST /v3/compat/chat/completions, 2026-08-19): -# top-level reasoning_effort takes an effort string that is passed through to the -# upstream model API unchanged, so each model accepts its native set (Anthropic: -# low|medium|high|max; OpenAI GPT-5.x: none|low|medium|high|xhigh and max where the -# model supports it; Gemini: minimal|low|medium|high). Budget-token values are not -# accepted on this surface, and no separate reasoning on/off toggle is exposed. Models with no upstream effort control accept the -# parameter without effect, so those entries use reasoning_options = []. -# Pricing is mirrored from GET /v3/compat/models (USD per token, no gateway markup). doc = "https://opper.ai/models" From 4c98cacfe9bdeec580eba60827bf696c97c77f0c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 17:23:22 +0000 Subject: [PATCH 215/392] chore(sync): update OpenRouter model catalog (#7654) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 ++++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/ibm-granite/granite-4.2-8b.toml | 6 +++--- providers/openrouter/models/qwen/qwen3.8-27b.toml | 6 +++--- providers/openrouter/models/x-ai/grok-4.7.toml | 3 --- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 6 files changed, 17 insertions(+), 19 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 13dfedbfc61..745a5e03d0f 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57288 -output = 1.71864 -cache_read = 0.019096 +input = 0.57156 +output = 1.71468 +cache_read = 0.018186 [limit] context = 1_048_576 +output = 393_216 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index a705f18ca62..9c2770326f2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.936294 -output = 1.872588 -cache_read = 0.078025 +input = 0.931074 +output = 1.862148 +cache_read = 0.07759 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/ibm-granite/granite-4.2-8b.toml b/providers/openrouter/models/ibm-granite/granite-4.2-8b.toml index 1b683dd536e..62943cfaadb 100644 --- a/providers/openrouter/models/ibm-granite/granite-4.2-8b.toml +++ b/providers/openrouter/models/ibm-granite/granite-4.2-8b.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.1 -output = 0.15 -cache_read = 0.05 +input = 0.06 +output = 0.25 +cache_read = 0.015 [limit] context = 131_072 diff --git a/providers/openrouter/models/qwen/qwen3.8-27b.toml b/providers/openrouter/models/qwen/qwen3.8-27b.toml index 64a9a2f6128..d1ddcffa2ed 100644 --- a/providers/openrouter/models/qwen/qwen3.8-27b.toml +++ b/providers/openrouter/models/qwen/qwen3.8-27b.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "medium", "xhigh"] [cost] -input = 0.2 -output = 2.5 -cache_read = 0.05 +input = 0.42 +output = 3 +cache_read = 0.085 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/x-ai/grok-4.7.toml b/providers/openrouter/models/x-ai/grok-4.7.toml index 49d2235d6f9..3ffa2ff7ae3 100644 --- a/providers/openrouter/models/x-ai/grok-4.7.toml +++ b/providers/openrouter/models/x-ai/grok-4.7.toml @@ -18,6 +18,3 @@ cache_read = 0.8 [limit] output = 450_000 - -[modalities] -input = ["text", "image", "pdf"] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 3a21dc3b34a..24cf56881f7 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57288 -output = 1.71864 -cache_read = 0.019096 +input = 0.57156 +output = 1.71468 +cache_read = 0.018186 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] From 85d8b02bfd7ea081f462e62c7a875545d12be4d9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 17:23:25 +0000 Subject: [PATCH 216/392] chore(sync): update Venice model catalog (#7653) Co-authored-by: opencode-agent[bot] --- providers/venice/models/grok-4-7.toml | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) create mode 100644 providers/venice/models/grok-4-7.toml diff --git a/providers/venice/models/grok-4-7.toml b/providers/venice/models/grok-4-7.toml new file mode 100644 index 00000000000..e405114dc5c --- /dev/null +++ b/providers/venice/models/grok-4-7.toml @@ -0,0 +1,24 @@ +base_model = "xai/grok-4.7" +description = "Grok model for agentic tool use, reasoning, coding, and live assistance" +release_date = "2026-09-16" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2.27 +output = 6.8 +cache_read = 0.57 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4.53 +output = 13.6 +cache_read = 1.13 + +[limit] +output = 200_000 + +[modalities] +input = ["text", "image"] From ddc239ff7d95cc769723f2ee5be6a4139e3120de Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 17:23:38 +0000 Subject: [PATCH 217/392] chore(sync): update Vercel AI Gateway model catalog (#7655) Co-authored-by: opencode-agent[bot] --- providers/vercel/models/spacexai/grok-4.7.toml | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/providers/vercel/models/spacexai/grok-4.7.toml b/providers/vercel/models/spacexai/grok-4.7.toml index defc4659f57..95c1c2ce193 100644 --- a/providers/vercel/models/spacexai/grok-4.7.toml +++ b/providers/vercel/models/spacexai/grok-4.7.toml @@ -1,5 +1,8 @@ base_model = "xai/grok-4.7" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 1.2 @@ -11,3 +14,6 @@ tier = { type = "context", size = 200_001 } input = 2.4 output = 7.2 cache_read = 0.6 + +[modalities] +input = ["text", "image"] From c8bc0ea7c90ae825c251476236d1403e4d64292b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 17:24:06 +0000 Subject: [PATCH 218/392] chore(sync): update Kilo model catalog (#7656) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 3 ++- providers/kilo/models/qwen/qwen3.8-27b.toml | 1 + providers/kilo/models/x-ai/grok-4.7.toml | 3 --- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 4 files changed, 8 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index 01800e18158..d4d63d57a55 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -11,4 +11,5 @@ output = 3.96 cache_read = 0.044 [limit] -context = 1_024_000 +context = 1_048_576 +output = 393_216 diff --git a/providers/kilo/models/qwen/qwen3.8-27b.toml b/providers/kilo/models/qwen/qwen3.8-27b.toml index 9b18fb98b33..820f69fb194 100644 --- a/providers/kilo/models/qwen/qwen3.8-27b.toml +++ b/providers/kilo/models/qwen/qwen3.8-27b.toml @@ -12,4 +12,5 @@ cache_read = 0.085 cache_write = 0.53125 [limit] +context = 1_000_000 output = 131_072 diff --git a/providers/kilo/models/x-ai/grok-4.7.toml b/providers/kilo/models/x-ai/grok-4.7.toml index bdb4be1fce7..7e904bb941d 100644 --- a/providers/kilo/models/x-ai/grok-4.7.toml +++ b/providers/kilo/models/x-ai/grok-4.7.toml @@ -12,6 +12,3 @@ cache_read = 0.4 [limit] output = 450_000 - -[modalities] -input = ["text", "image", "pdf"] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index d1321db247a..c10a096a2ae 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.57288 -output = 1.71864 -cache_read = 0.019096 +input = 0.57156 +output = 1.71468 +cache_read = 0.018186 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 393_216 [modalities] input = ["text"] From c4eefc3346ada0c358f9c5f7878cdf0cda22d9cc Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:31:30 +0000 Subject: [PATCH 219/392] chore(sync): update OpenRouter model catalog (#7663) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash-0731.toml | 2 +- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/moonshotai/kimi-k3.toml | 6 +++--- .../openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml | 5 +++-- providers/openrouter/models/z-ai/glm-5.3-flash.toml | 7 ++++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 4 ++-- providers/openrouter/models/~z-ai/glm-flash-latest.toml | 4 ++-- 9 files changed, 24 insertions(+), 22 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml index be2f907d844..48200f680e3 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml @@ -11,7 +11,7 @@ values = ["low", "high", "max"] [cost] input = 0.04 -output = 0.16 +output = 0.32 cache_read = 0.016 [limit] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 745a5e03d0f..c11c30e3519 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57156 -output = 1.71468 -cache_read = 0.018186 +input = 0.56628 +output = 1.69884 +cache_read = 0.018018 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 9c2770326f2..399e63186f7 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.931074 -output = 1.862148 -cache_read = 0.07759 +input = 0.924462 +output = 1.848924 +cache_read = 0.077039 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index 22b1407542a..c11eb6c3dae 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.7 -output = 8.5 -cache_read = 0.17 +input = 3 +output = 15 +cache_read = 0.3 [limit] output = 943_718 diff --git a/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml b/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml index 297dfb56987..bf732029012 100644 --- a/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml +++ b/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml @@ -7,8 +7,9 @@ structured_output = true type = "toggle" [cost] -input = 0.06 -output = 0.24 +input = 0.05 +output = 0.2 +cache_read = 0.03 [limit] output = 235_929 diff --git a/providers/openrouter/models/z-ai/glm-5.3-flash.toml b/providers/openrouter/models/z-ai/glm-5.3-flash.toml index f82829f96ea..6fd3f066cfd 100644 --- a/providers/openrouter/models/z-ai/glm-5.3-flash.toml +++ b/providers/openrouter/models/z-ai/glm-5.3-flash.toml @@ -6,12 +6,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.09 -output = 0.3 -cache_read = 0.018 +input = 0.075 +output = 0.25 +cache_read = 0.02 [limit] context = 1_310_720 +output = 102_400 [modalities] input = ["text", "image", "video"] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 24cf56881f7..427614e0228 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57156 -output = 1.71468 -cache_read = 0.018186 +input = 0.56628 +output = 1.69884 +cache_read = 0.018018 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index c98a00af617..867340e2671 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -21,8 +21,8 @@ values = ["low", "high", "max"] [cost] input = 0.04 -output = 0.16 -cache_read = 0.016 +output = 0.2 +cache_read = 0.01 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~z-ai/glm-flash-latest.toml b/providers/openrouter/models/~z-ai/glm-flash-latest.toml index 8f90bbaaf93..ac623f94e52 100644 --- a/providers/openrouter/models/~z-ai/glm-flash-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-flash-latest.toml @@ -17,11 +17,11 @@ values = ["low", "high", "max"] [cost] input = 0.075 output = 0.25 -cache_read = 0.015 +cache_read = 0.02 [limit] context = 1_310_720 -output = 943_718 +output = 102_400 [modalities] input = ["text", "image", "video"] From d626e17c832129c7b6f8f1f5005234f167b9867c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:31:37 +0000 Subject: [PATCH 220/392] chore(sync): update DevPass (LLM Gateway) model catalog (#7661) Co-authored-by: opencode-agent[bot] --- .../llmgateway/models/glm-5.3-flash.toml | 6 ++-- providers/llmgateway/models/grok-4-7.toml | 28 +++++++++++++++++++ 2 files changed, 31 insertions(+), 3 deletions(-) create mode 100644 providers/llmgateway/models/grok-4-7.toml diff --git a/providers/llmgateway/models/glm-5.3-flash.toml b/providers/llmgateway/models/glm-5.3-flash.toml index d46987e8acc..5aecfa64069 100644 --- a/providers/llmgateway/models/glm-5.3-flash.toml +++ b/providers/llmgateway/models/glm-5.3-flash.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] -input = 0.088 -output = 0.25 -cache_read = 0.025 +input = 0.07 +output = 0.19 +cache_read = 0.015 [limit] context = 1_048_576 diff --git a/providers/llmgateway/models/grok-4-7.toml b/providers/llmgateway/models/grok-4-7.toml new file mode 100644 index 00000000000..4e89d64b78f --- /dev/null +++ b/providers/llmgateway/models/grok-4-7.toml @@ -0,0 +1,28 @@ +name = "Grok 4.7" +description = "Grok model for agentic tool use, reasoning, coding, and live assistance" +family = "grok" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[limit] +context = 500_000 +output = 500_000 + +[modalities] +input = ["text", "image"] +output = ["text"] From d7ce2f138822ee40b5b2b51c8ab1ba74b2d56c20 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:31:45 +0000 Subject: [PATCH 221/392] chore(sync): update Kilo model catalog (#7659) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/moonshotai/kimi-k3.toml | 6 +++--- providers/kilo/models/z-ai/glm-5.3-flash.toml | 1 + providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- .../kilo/models/~deepseek/deepseek-v4-flash-latest.toml | 4 ++-- providers/kilo/models/~z-ai/glm-flash-latest.toml | 4 ++-- 5 files changed, 11 insertions(+), 10 deletions(-) diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index fd81e087ca6..1ebab0a52b7 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.7 -output = 8.5 -cache_read = 0.17 +input = 3 +output = 15 +cache_read = 0.3 [limit] output = 943_718 diff --git a/providers/kilo/models/z-ai/glm-5.3-flash.toml b/providers/kilo/models/z-ai/glm-5.3-flash.toml index 0104ea56ade..1bc5d8b7a69 100644 --- a/providers/kilo/models/z-ai/glm-5.3-flash.toml +++ b/providers/kilo/models/z-ai/glm-5.3-flash.toml @@ -12,6 +12,7 @@ cache_read = 0.03 [limit] context = 1_048_576 +output = 102_400 [modalities] input = ["text", "image", "video"] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index c10a096a2ae..191b436787a 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.57156 -output = 1.71468 -cache_read = 0.018186 +input = 0.56628 +output = 1.69884 +cache_read = 0.018018 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 7ead984259a..49d602e67e6 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -16,8 +16,8 @@ values = ["none", "low", "high", "max"] [cost] input = 0.04 -output = 0.16 -cache_read = 0.016 +output = 0.2 +cache_read = 0.01 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~z-ai/glm-flash-latest.toml b/providers/kilo/models/~z-ai/glm-flash-latest.toml index aee70deebb0..5035574d3d5 100644 --- a/providers/kilo/models/~z-ai/glm-flash-latest.toml +++ b/providers/kilo/models/~z-ai/glm-flash-latest.toml @@ -17,11 +17,11 @@ values = ["low", "high", "max"] [cost] input = 0.075 output = 0.25 -cache_read = 0.015 +cache_read = 0.02 [limit] context = 1_048_576 -output = 943_718 +output = 102_400 [modalities] input = ["text", "image", "video"] From e30e2fc6e88d1e5ad06a4780a9516f1de179cdc8 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:31:47 +0000 Subject: [PATCH 222/392] chore(sync): update Merge Gateway model catalog (#7662) Co-authored-by: opencode-agent[bot] --- .../merge-gateway/models/qwen/qwen3.5-27b.toml | 7 ++++--- providers/merge-gateway/models/xai/grok-4.7.toml | 13 +++++++++++++ 2 files changed, 17 insertions(+), 3 deletions(-) create mode 100644 providers/merge-gateway/models/xai/grok-4.7.toml diff --git a/providers/merge-gateway/models/qwen/qwen3.5-27b.toml b/providers/merge-gateway/models/qwen/qwen3.5-27b.toml index f0885b3822b..d7d8d42decc 100644 --- a/providers/merge-gateway/models/qwen/qwen3.5-27b.toml +++ b/providers/merge-gateway/models/qwen/qwen3.5-27b.toml @@ -1,5 +1,6 @@ # Source: https://api-gateway.merge.dev/v1/models?model=qwen%2Fqwen3.5-27b (accessed 2026-07-21) base_model = "alibaba/qwen3.5-27b" +attachment = false [[reasoning_options]] type = "toggle" @@ -10,8 +11,8 @@ output = 0.688 cache_read = 0.0172 [limit] -context = 256_000 -output = 64_000 +context = 131_072 +output = 32_768 [modalities] -input = ["text", "image"] +input = ["text"] diff --git a/providers/merge-gateway/models/xai/grok-4.7.toml b/providers/merge-gateway/models/xai/grok-4.7.toml new file mode 100644 index 00000000000..19465981ea7 --- /dev/null +++ b/providers/merge-gateway/models/xai/grok-4.7.toml @@ -0,0 +1,13 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[modalities] +input = ["text", "image"] From c28d1e58ad585c41613d7762b8a8ab438b6bf737 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:31:49 +0000 Subject: [PATCH 223/392] chore(sync): update Charm Hyper model catalog (#7660) Co-authored-by: opencode-agent[bot] --- providers/hyper/models/deepseek-v4.1-flash.toml | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/providers/hyper/models/deepseek-v4.1-flash.toml b/providers/hyper/models/deepseek-v4.1-flash.toml index c703c4fcf0c..7fd6fdfb3b1 100644 --- a/providers/hyper/models/deepseek-v4.1-flash.toml +++ b/providers/hyper/models/deepseek-v4.1-flash.toml @@ -2,7 +2,7 @@ base_model = "deepseek/deepseek-v4.1-flash" [[reasoning_options]] type = "effort" -values = ["low", "high", "max"] +values = ["low", "high", "xhigh"] [cost] input = 0.3 @@ -10,4 +10,5 @@ output = 1.2 cache_read = 0.03 [limit] +context = 1_048_576 output = 32_768 From d7d8109e69c5bf10cf7fc6172e04d64234ea180b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 18:31:52 +0000 Subject: [PATCH 224/392] chore(sync): update LLM Gateway model catalog (#7664) Co-authored-by: opencode-agent[bot] --- .../models/gonka24/glm-5.3-flash.toml | 20 +++++++++++++ .../models/xai/grok-4-7.toml | 28 +++++++++++++++++++ 2 files changed, 48 insertions(+) create mode 100644 providers/llmgateway-providers/models/gonka24/glm-5.3-flash.toml create mode 100644 providers/llmgateway-providers/models/xai/grok-4-7.toml diff --git a/providers/llmgateway-providers/models/gonka24/glm-5.3-flash.toml b/providers/llmgateway-providers/models/gonka24/glm-5.3-flash.toml new file mode 100644 index 00000000000..921d7dc252d --- /dev/null +++ b/providers/llmgateway-providers/models/gonka24/glm-5.3-flash.toml @@ -0,0 +1,20 @@ +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3 Flash (Gonka24)" +attachment = false +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.07 +output = 0.19 +cache_read = 0.015 + +[limit] +context = 200_000 +output = 16_384 + +[modalities] +input = ["text"] diff --git a/providers/llmgateway-providers/models/xai/grok-4-7.toml b/providers/llmgateway-providers/models/xai/grok-4-7.toml new file mode 100644 index 00000000000..03bd9afa268 --- /dev/null +++ b/providers/llmgateway-providers/models/xai/grok-4-7.toml @@ -0,0 +1,28 @@ +name = "Grok 4.7 (xAI)" +description = "Grok model for agentic tool use, reasoning, coding, and live assistance" +family = "grok" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[limit] +context = 500_000 +output = 500_000 + +[modalities] +input = ["text", "image"] +output = ["text"] From 422b34ab53646b470469b1bc9900a252f288b079 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:35:19 -0500 Subject: [PATCH 225/392] fix(sync): ingest aiand pricing tiers (#7650) Co-authored-by: rekram1-node --- packages/core/src/sync/providers/aiand.ts | 44 ++++++++++++----- packages/core/test/aiand.test.ts | 58 ++++++++++++++++++++++- 2 files changed, 87 insertions(+), 15 deletions(-) diff --git a/packages/core/src/sync/providers/aiand.ts b/packages/core/src/sync/providers/aiand.ts index 9545bd91e0f..860bbc50944 100644 --- a/packages/core/src/sync/providers/aiand.ts +++ b/packages/core/src/sync/providers/aiand.ts @@ -31,6 +31,28 @@ const FeedReasoningOption = z.union([ CatalogReasoningOption, ]); +const PRICE = z.number().nonnegative(); +const FeedCostFields = { + input: PRICE, + output: PRICE, + reasoning: PRICE.optional(), + cache_read: PRICE.optional(), + cache_write: PRICE.optional(), + input_audio: PRICE.optional(), + output_audio: PRICE.optional(), +}; +const FeedCostTier = z + .object({ + ...FeedCostFields, + tier: z + .object({ + type: z.literal("context").default("context"), + size: z.number().int().nonnegative(), + }) + .passthrough(), + }) + .passthrough(); + export const AiandModel = z .object({ id: z.string().min(1), @@ -45,13 +67,7 @@ export const AiandModel = z temperature: z.boolean(), tool_call: z.boolean(), structured_output: z.boolean().optional(), - cost: z - .object({ - input: z.number().nonnegative(), - output: z.number().nonnegative(), - cache_read: z.number().nonnegative().optional(), - }) - .passthrough(), + cost: z.object({ ...FeedCostFields, tiers: z.array(FeedCostTier).optional() }).passthrough(), limit: z .object({ context: z.number().int().positive(), @@ -182,12 +198,14 @@ export function buildAiandModel( cost: { input: model.cost.input, output: model.cost.output, - reasoning: authored?.cost?.reasoning, - cache_read: model.cost.cache_read, - cache_write: authored?.cost?.cache_write, - input_audio: authored?.cost?.input_audio, - output_audio: authored?.cost?.output_audio, - tiers: authored?.cost?.tiers, + reasoning: model.cost.reasoning ?? authored?.cost?.reasoning, + cache_read: model.cost.cache_read ?? authored?.cost?.cache_read, + cache_write: model.cost.cache_write ?? authored?.cost?.cache_write, + input_audio: model.cost.input_audio ?? authored?.cost?.input_audio, + output_audio: model.cost.output_audio ?? authored?.cost?.output_audio, + // Unlike auxiliary prices, tiers are a complete source assertion: an + // omitted list means flat pricing and clears stale authored tiers. + tiers: model.cost.tiers, }, // Absence means active on the feed, and the feed owns deprecation: a // curated alpha/beta survives omission, a curated deprecated does not, diff --git a/packages/core/test/aiand.test.ts b/packages/core/test/aiand.test.ts index d4e0c74c1fb..9e531806203 100644 --- a/packages/core/test/aiand.test.ts +++ b/packages/core/test/aiand.test.ts @@ -173,13 +173,67 @@ test("skippedNotice stays silent on a clean sync", () => { test("authored-only cost fields ride along; feed prices are authoritative", () => { const built = buildAiandModel( aiandModel(), - { cost: { input: 9, output: 9, cache_write: 0.5 }, limit: { context: 1, input: 128_000, output: 1 } }, + { + cost: { input: 9, output: 9, reasoning: 0.75, cache_write: 0.5, input_audio: 1.25 }, + limit: { context: 1, input: 128_000, output: 1 }, + }, null, ); - expect(built.cost).toMatchObject({ input: 0.15, output: 0.25, cache_write: 0.5 }); + expect(built.cost).toMatchObject({ + input: 0.15, + output: 0.25, + reasoning: 0.75, + cache_read: 0.08, + cache_write: 0.5, + input_audio: 1.25, + }); expect(built.limit).toEqual({ context: 1_048_576, input: 128_000, output: 384_000 }); }); +test("feed cost tiers and auxiliary prices replace authored values", () => { + const tiers = [{ + tier: { type: "context" as const, size: 200_000 }, + input: 0.3, + output: 0.5, + cache_read: 0.16, + }]; + const model = AiandModel.parse({ + ...aiandModel(), + cost: { + ...aiandModel().cost, + reasoning: 0.4, + cache_write: 0.12, + output_audio: 2, + tiers, + }, + }); + const built = buildAiandModel(model, { + cost: { + input: 9, + output: 9, + reasoning: 9, + cache_write: 9, + output_audio: 9, + tiers: [{ tier: { type: "context", size: 100_000 }, input: 9, output: 9 }], + }, + }, null); + + expect(built.cost).toMatchObject({ reasoning: 0.4, cache_write: 0.12, output_audio: 2 }); + expect(built.cost?.tiers).toEqual(tiers); +}); + +test("omitted feed cost tiers clear authored tiers", () => { + const built = buildAiandModel(aiandModel(), { + cost: { + input: 9, + output: 9, + tiers: [{ tier: { type: "context", size: 200_000 }, input: 9, output: 9 }], + }, + }, null); + + expect(built.cost?.tiers).toBeUndefined(); +}); + test("a non-reasoning feed model omits reasoning_options entirely", () => { const built = buildAiandModel( aiandModel({ reasoning: false, reasoning_options: undefined }), From 68fbad48e4bdba5168048d80c8c3962ae4583f91 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:36:00 -0500 Subject: [PATCH 226/392] fix(sync): use Requesty pricing bands (#7652) Co-authored-by: rekram1-node --- packages/core/src/sync/providers/requesty.ts | 9 ++- packages/core/test/requesty.test.ts | 78 ++++++++++++++++++++ 2 files changed, 83 insertions(+), 4 deletions(-) diff --git a/packages/core/src/sync/providers/requesty.ts b/packages/core/src/sync/providers/requesty.ts index 00f6dbcc990..d8b0bdf4de5 100644 --- a/packages/core/src/sync/providers/requesty.ts +++ b/packages/core/src/sync/providers/requesty.ts @@ -184,8 +184,9 @@ function reasoningOptions( } function buildCost(model: RequestyModel): SyncedFullModel["cost"] { - const input = model.input_price; - const output = model.output_price; + const base = model.pricing?.[0] ?? model; + const input = base.input_price; + const output = base.output_price; if (input == null || output == null) return undefined; const tiers = (model.pricing ?? []).slice(1).map((band) => ({ @@ -198,8 +199,8 @@ function buildCost(model: RequestyModel): SyncedFullModel["cost"] { return { input: pricePerMillion(input), output: pricePerMillion(output), - cache_read: chargedPricePerMillion(model.cached_price), - cache_write: chargedPricePerMillion(model.caching_price), + cache_read: chargedPricePerMillion(base.cached_price), + cache_write: chargedPricePerMillion(base.caching_price), tiers: tiers.length > 0 ? tiers : undefined, }; } diff --git a/packages/core/test/requesty.test.ts b/packages/core/test/requesty.test.ts index 597b6de65c8..52eaec2ccde 100644 --- a/packages/core/test/requesty.test.ts +++ b/packages/core/test/requesty.test.ts @@ -49,3 +49,81 @@ test.each(["claude-fable-5.1", "claude-fable-5.1@eu"])( }); }, ); + +test("uses the first Requesty pricing band as the base cost", () => { + const model = buildRequestyModel(RequestyModel.parse({ + id: "requesty-priced-model", + created: Date.parse("2026-09-01") / 1_000, + description: "Requesty description", + context_window: 1_000_000, + max_output_tokens: 128_000, + input_price: 0.000001, + output_price: 0.000002, + cached_price: 0.000003, + caching_price: 0.000004, + pricing: [ + { + prompt_tokens_threshold: 0, + input_price: 0.00001, + output_price: 0.00005, + cached_price: 0.00000025, + caching_price: 0.0000125, + }, + { + prompt_tokens_threshold: 200_000, + input_price: 0.00002, + output_price: 0.000075, + cached_price: 0.0000005, + caching_price: 0.000025, + }, + ], + })); + + expect(model.cost).toEqual({ + input: 10, + output: 50, + cache_read: 0.25, + cache_write: 12.5, + tiers: [{ + tier: { type: "context", size: 200_000 }, + input: 20, + output: 75, + cache_read: 0.5, + cache_write: 25, + }], + }); +}); + +test("uses Requesty pricing bands when top-level prices are null", () => { + const model = buildRequestyModel(RequestyModel.parse({ + id: "requesty-priced-model", + created: Date.parse("2026-09-01") / 1_000, + description: "Requesty description", + context_window: 1_000_000, + max_output_tokens: 128_000, + input_price: null, + output_price: null, + pricing: [ + { + prompt_tokens_threshold: 0, + input_price: 0.00001, + output_price: 0.00005, + }, + { + prompt_tokens_threshold: 200_000, + input_price: null, + output_price: null, + }, + ], + })); + + expect(model.cost).toEqual({ + input: 10, + output: 50, + tiers: [{ + tier: { type: "context", size: 200_000 }, + input: 10, + output: 50, + }], + }); +}); From b449ad12403ef4903ed837d6c9fb8ea3b10b25f8 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:38:26 -0500 Subject: [PATCH 227/392] fix(sync): clear stale CrossModel pricing tiers (#7649) Co-authored-by: rekram1-node --- .../core/src/sync/providers/crossmodel.ts | 7 ++-- packages/core/test/sync.test.ts | 32 +++++++++++++++++++ 2 files changed, 36 insertions(+), 3 deletions(-) diff --git a/packages/core/src/sync/providers/crossmodel.ts b/packages/core/src/sync/providers/crossmodel.ts index e405a811a8f..0a1f3d067fe 100644 --- a/packages/core/src/sync/providers/crossmodel.ts +++ b/packages/core/src/sync/providers/crossmodel.ts @@ -193,8 +193,9 @@ export function buildCrossModel( // CrossModel serves threshold-tiered pricing. The lowest-threshold tier is the // headline [cost]; every higher tier maps to a [[cost.tiers]] context band // (threshold -> tier size), so tier pricing stays fresh on each sync instead of - // being frozen at hand-authored values. Fall back to the existing tiers only - // when the API reports none. + // being frozen at hand-authored values. When the API reports usable pricing, + // its tier list is authoritative; fall back to the existing cost only when the + // source pricing is absent or unusable. const tiers = [...(model.pricing?.tiers ?? [])].sort( (a, b) => (a.threshold ?? 0) - (b.threshold ?? 0), ); @@ -210,7 +211,7 @@ export function buildCrossModel( .filter((entry): entry is NonNullable => entry !== undefined); const cost = base !== undefined - ? { ...base, tiers: contextTiers.length > 0 ? contextTiers : existing?.cost?.tiers } + ? { ...base, tiers: contextTiers.length > 0 ? contextTiers : undefined } : existing?.cost; // Every served model reports a context window; without one (and no existing diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 0281aac4808..3da3fe65f8f 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -449,6 +449,38 @@ test("syncs CrossModel's structured-output capability", () => { }); }); +test("clears stale CrossModel context tiers only when source pricing is usable", () => { + const existing: ExistingModel = { + base_model: "alibaba/qwen3.8-max", + cost: { + input: 9, + output: 27, + tiers: [ + { + tier: { type: "context", size: 200_000 }, + input: 18, + output: 54, + }, + ], + }, + }; + + const authoritative = buildCrossModel(crossModelModel(), existing); + const absent = buildCrossModel(crossModelModel({ pricing: undefined }), existing); + const unusable = buildCrossModel( + crossModelModel({ + pricing: { + tiers: [{ threshold: 0, input_micro_per_1m: 1_880_000 }], + }, + }), + existing, + ); + + expect(authoritative?.cost).toEqual({ input: 1.88, output: 5.63 }); + expect(absent?.cost).toEqual(existing.cost); + expect(unusable?.cost).toEqual(existing.cost); +}); + test("parses CrossModel's nullable reasoning controls", () => { const parsed = CrossModelResponse.parse({ data: [ From a7332d5569a1398c8177ec7e4db3b7c5d3dc0677 Mon Sep 17 00:00:00 2001 From: Arpit Mishra <63418443+ArpitMishra17@users.noreply.github.com> Date: Tue, 22 Sep 2026 00:08:49 +0530 Subject: [PATCH 228/392] feat(stepfun-ai): add step-5-preview (#7657) StepFun's flagship base model, verified live on the Global PAYG gateway (api.stepfun.ai/v1). Reuses existing lab metadata (models/stepfun/step-5-preview.toml): 1M-token context/input, text/image/video input, reasoning + tool calling + structured output, closed weights. Provider override: reasoning_effort accepts low/medium/high; reasoning side channel is reasoning_content; per-token cost input 1.00, output 2.70, cache_read 0.05 USD/MTok. Standalone file (not symlinked to providers/stepfun) because China (api.stepfun.com) is unverified with a Global key (401). Global Step Plan (step_plan/v1) verified absent for this model, so not added there. Sources: - https://platform.stepfun.ai/docs/en/guides/models/step-5-preview - https://platform.stepfun.ai/docs/en/guides/pricing/details - GET https://api.stepfun.ai/v1/models (accessed 2026-09-21) bun validate passes. --- .../stepfun-ai/models/step-5-preview.toml | 21 +++++++++++++++++++ providers/stepfun-ai/provider.toml | 4 ++-- 2 files changed, 23 insertions(+), 2 deletions(-) create mode 100644 providers/stepfun-ai/models/step-5-preview.toml diff --git a/providers/stepfun-ai/models/step-5-preview.toml b/providers/stepfun-ai/models/step-5-preview.toml new file mode 100644 index 00000000000..c7a645ba40a --- /dev/null +++ b/providers/stepfun-ai/models/step-5-preview.toml @@ -0,0 +1,21 @@ +# Chat `reasoning_effort`, Messages `output_config.effort`, and Responses +# `reasoning.effort` accept low/medium/high (accessed 2026-09-21). +# https://platform.stepfun.ai/docs/en/guides/models/step-5-preview +# Verified live: GET https://api.stepfun.ai/v1/models lists step-5-preview +# with reasoning_effort_support_list ["low", "medium", "high"] and +# max_input_tokens 1024000; POST /v1/chat/completions succeeds. +# Pricing (USD/MTok): input 1.00, output 2.70, cache_read 0.05 (95% cache +# discount), matching existing mirrors and vendor reporting. +base_model = "stepfun/step-5-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1 +output = 2.7 +cache_read = 0.05 diff --git a/providers/stepfun-ai/provider.toml b/providers/stepfun-ai/provider.toml index 0c84a7aaa5d..84b890902aa 100644 --- a/providers/stepfun-ai/provider.toml +++ b/providers/stepfun-ai/provider.toml @@ -4,8 +4,8 @@ npm = "@ai-sdk/openai-compatible" # Reasoning HTTP format (accessed 2026-06-29): # POST /v1/chat/completions uses top-level `reasoning_effort`; POST /v1/messages # uses `output_config.effort`; POST /v1/responses uses `reasoning.effort`. -# Values are low/medium/high for step-3.7-flash; step-3.5-flash-2603 accepts -# low/high. Responses supports only step-3.7-flash. Chat returns reasoning at +# Values are low/medium/high for step-3.7-flash and step-5-preview; step-3.5-flash-2603 accepts +# low/high. Responses supports step-3.7-flash and step-5-preview. Chat returns reasoning at # `choices[].message.reasoning` or streamed `choices[].delta.reasoning`; # `reasoning_format` is general (default), the latter using `reasoning_content`. # Responses streams `response.reasoning_text.delta`. From 36a09cd2cdb684821d660f5b9062ca941b6a36f7 Mon Sep 17 00:00:00 2001 From: thatdevguy <46411187+chrissalomon@users.noreply.github.com> Date: Mon, 21 Sep 2026 14:39:05 -0400 Subject: [PATCH 229/392] Add Tempr as a provider, with its Anthropic models (#7638) Tempr's AI Gateway (api.temprhq.io/v1) is a BYOK, OpenAI-compatible relay: requests go to the caller's own key on the named provider, with no markup on tokens. These 14 Anthropic models base_model the lab entries with Anthropic's own pricing; reasoning_options are what Tempr's GET /v1/models reports, with a leading comment giving the fields a caller sets. --- providers/tempr/logo.svg | 4 ++++ .../models/anthropic/claude-fable-5-1.toml | 16 +++++++++++++++ .../models/anthropic/claude-fable-5.toml | 16 +++++++++++++++ .../anthropic/claude-haiku-4-5-20251001.toml | 18 +++++++++++++++++ .../models/anthropic/claude-haiku-4-5.toml | 18 +++++++++++++++++ .../anthropic/claude-opus-4-5-20251101.toml | 20 +++++++++++++++++++ .../models/anthropic/claude-opus-4-5.toml | 20 +++++++++++++++++++ .../models/anthropic/claude-opus-4-6.toml | 20 +++++++++++++++++++ .../models/anthropic/claude-opus-4-7.toml | 18 +++++++++++++++++ .../models/anthropic/claude-opus-4-8.toml | 18 +++++++++++++++++ .../tempr/models/anthropic/claude-opus-5.toml | 16 +++++++++++++++ .../anthropic/claude-sonnet-4-5-20250929.toml | 18 +++++++++++++++++ .../models/anthropic/claude-sonnet-4-5.toml | 18 +++++++++++++++++ .../models/anthropic/claude-sonnet-4-6.toml | 20 +++++++++++++++++++ .../models/anthropic/claude-sonnet-5.toml | 18 +++++++++++++++++ providers/tempr/provider.toml | 5 +++++ 16 files changed, 263 insertions(+) create mode 100644 providers/tempr/logo.svg create mode 100644 providers/tempr/models/anthropic/claude-fable-5-1.toml create mode 100644 providers/tempr/models/anthropic/claude-fable-5.toml create mode 100644 providers/tempr/models/anthropic/claude-haiku-4-5-20251001.toml create mode 100644 providers/tempr/models/anthropic/claude-haiku-4-5.toml create mode 100644 providers/tempr/models/anthropic/claude-opus-4-5-20251101.toml create mode 100644 providers/tempr/models/anthropic/claude-opus-4-5.toml create mode 100644 providers/tempr/models/anthropic/claude-opus-4-6.toml create mode 100644 providers/tempr/models/anthropic/claude-opus-4-7.toml create mode 100644 providers/tempr/models/anthropic/claude-opus-4-8.toml create mode 100644 providers/tempr/models/anthropic/claude-opus-5.toml create mode 100644 providers/tempr/models/anthropic/claude-sonnet-4-5-20250929.toml create mode 100644 providers/tempr/models/anthropic/claude-sonnet-4-5.toml create mode 100644 providers/tempr/models/anthropic/claude-sonnet-4-6.toml create mode 100644 providers/tempr/models/anthropic/claude-sonnet-5.toml create mode 100644 providers/tempr/provider.toml diff --git a/providers/tempr/logo.svg b/providers/tempr/logo.svg new file mode 100644 index 00000000000..a9fb18e6892 --- /dev/null +++ b/providers/tempr/logo.svg @@ -0,0 +1,4 @@ + + + + diff --git a/providers/tempr/models/anthropic/claude-fable-5-1.toml b/providers/tempr/models/anthropic/claude-fable-5-1.toml new file mode 100644 index 00000000000..78d27e34265 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-fable-5-1.toml @@ -0,0 +1,16 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = low|medium|high|xhigh|max +base_model = "anthropic/claude-fable-5-1" +structured_output = true + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 10 +output = 50 +cache_read = 0.25 +cache_write = 12.5 diff --git a/providers/tempr/models/anthropic/claude-fable-5.toml b/providers/tempr/models/anthropic/claude-fable-5.toml new file mode 100644 index 00000000000..03847400952 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-fable-5.toml @@ -0,0 +1,16 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = low|medium|high|xhigh|max +base_model = "anthropic/claude-fable-5" +structured_output = true + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 diff --git a/providers/tempr/models/anthropic/claude-haiku-4-5-20251001.toml b/providers/tempr/models/anthropic/claude-haiku-4-5-20251001.toml new file mode 100644 index 00000000000..c2502f29a34 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-haiku-4-5-20251001.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high (sized to a thinking budget) +base_model = "anthropic/claude-haiku-4-5-20251001" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 1 +output = 5 +cache_read = 0.1 +cache_write = 1.25 diff --git a/providers/tempr/models/anthropic/claude-haiku-4-5.toml b/providers/tempr/models/anthropic/claude-haiku-4-5.toml new file mode 100644 index 00000000000..102da98f92e --- /dev/null +++ b/providers/tempr/models/anthropic/claude-haiku-4-5.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high (sized to a thinking budget) +base_model = "anthropic/claude-haiku-4-5" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 1 +output = 5 +cache_read = 0.1 +cache_write = 1.25 diff --git a/providers/tempr/models/anthropic/claude-opus-4-5-20251101.toml b/providers/tempr/models/anthropic/claude-opus-4-5-20251101.toml new file mode 100644 index 00000000000..1ff845fa823 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-4-5-20251101.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high (alias: top-level reasoning_effort) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high +base_model = "anthropic/claude-opus-4-5-20251101" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high"] }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-opus-4-5.toml b/providers/tempr/models/anthropic/claude-opus-4-5.toml new file mode 100644 index 00000000000..35b1bae9829 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-4-5.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high (alias: top-level reasoning_effort) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high +base_model = "anthropic/claude-opus-4-5" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high"] }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-opus-4-6.toml b/providers/tempr/models/anthropic/claude-opus-4-6.toml new file mode 100644 index 00000000000..bd4664a3a5f --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-4-6.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high|max (alias: top-level reasoning_effort) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high|max; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high|max +base_model = "anthropic/claude-opus-4-6" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high", "max"] }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-opus-4-7.toml b/providers/tempr/models/anthropic/claude-opus-4-7.toml new file mode 100644 index 00000000000..5796ecdbe51 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-4-7.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = none|low|medium|high|xhigh|max +base_model = "anthropic/claude-opus-4-7" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-opus-4-8.toml b/providers/tempr/models/anthropic/claude-opus-4-8.toml new file mode 100644 index 00000000000..23f9c6ab18e --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-4-8.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = none|low|medium|high|xhigh|max +base_model = "anthropic/claude-opus-4-8" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-opus-5.toml b/providers/tempr/models/anthropic/claude-opus-5.toml new file mode 100644 index 00000000000..ca74b5d1101 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-5.toml @@ -0,0 +1,16 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = low|medium|high|xhigh|max +base_model = "anthropic/claude-opus-5" +structured_output = true + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-sonnet-4-5-20250929.toml b/providers/tempr/models/anthropic/claude-sonnet-4-5-20250929.toml new file mode 100644 index 00000000000..8cda1b84187 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-sonnet-4-5-20250929.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high (sized to a thinking budget) +base_model = "anthropic/claude-sonnet-4-5-20250929" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 diff --git a/providers/tempr/models/anthropic/claude-sonnet-4-5.toml b/providers/tempr/models/anthropic/claude-sonnet-4-5.toml new file mode 100644 index 00000000000..aba63782362 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-sonnet-4-5.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high (sized to a thinking budget) +base_model = "anthropic/claude-sonnet-4-5" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 diff --git a/providers/tempr/models/anthropic/claude-sonnet-4-6.toml b/providers/tempr/models/anthropic/claude-sonnet-4-6.toml new file mode 100644 index 00000000000..396e2f09c38 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-sonnet-4-6.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high|max (alias: top-level reasoning_effort) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high|max; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high|max +base_model = "anthropic/claude-sonnet-4-6" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high", "max"] }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 diff --git a/providers/tempr/models/anthropic/claude-sonnet-5.toml b/providers/tempr/models/anthropic/claude-sonnet-5.toml new file mode 100644 index 00000000000..a7908c1fc7c --- /dev/null +++ b/providers/tempr/models/anthropic/claude-sonnet-5.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = none|low|medium|high|xhigh|max +base_model = "anthropic/claude-sonnet-5" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 diff --git a/providers/tempr/provider.toml b/providers/tempr/provider.toml new file mode 100644 index 00000000000..70ceee60f53 --- /dev/null +++ b/providers/tempr/provider.toml @@ -0,0 +1,5 @@ +name = "Tempr" +npm = "@ai-sdk/openai-compatible" +env = ["TEMPR_API_KEY"] +api = "https://api.temprhq.io/v1" +doc = "https://temprhq.io/docs/gateway-reference.html" From fa5f6eadf9b90935f601e06ab46a806fd04c969c Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:40:54 -0500 Subject: [PATCH 230/392] fix: [missing-model] github-copilot: grok-4.7 (#7665) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- providers/github-copilot/models/grok-4.7.toml | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) create mode 100644 providers/github-copilot/models/grok-4.7.toml diff --git a/providers/github-copilot/models/grok-4.7.toml b/providers/github-copilot/models/grok-4.7.toml new file mode 100644 index 00000000000..25d6bf546e2 --- /dev/null +++ b/providers/github-copilot/models/grok-4.7.toml @@ -0,0 +1,24 @@ +# Sources: +# - https://github.blog/changelog/2026-09-21-grok-4-7-is-now-available-in-github-copilot/ +# - https://docs.github.com/en/copilot/reference/copilot-billing/models-and-pricing +# - https://docs.github.com/en/copilot/reference/ai-models/supported-models +# - https://docs.x.ai/developers/models/grok-4.7 +# - https://docs.x.ai/developers/pricing +base_model = "xai/grok-4.7" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 + +[limit] +context = 500_000 +input = 372_000 +output = 128_000 From 9312e146098c165f7ac3e418be02d2a13cf0fff8 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 19:24:31 +0000 Subject: [PATCH 231/392] chore(sync): update Kilo model catalog (#7667) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 1 - .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 2 files changed, 5 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index d4d63d57a55..007b654034e 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -12,4 +12,3 @@ cache_read = 0.044 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 191b436787a..6caf71dbf01 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.56628 -output = 1.69884 -cache_read = 0.018018 +input = 0.558624 +output = 1.675872 +cache_read = 0.018621 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] From c8cf3b64cae40e537cfae939318c95eebb6785c5 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 19:25:24 +0000 Subject: [PATCH 232/392] chore(sync): update OpenRouter model catalog (#7668) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 +++---- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 3 files changed, 10 insertions(+), 11 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index c11c30e3519..7c5828b16e2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,10 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.56628 -output = 1.69884 -cache_read = 0.018018 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 399e63186f7..5af84d1accb 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.924462 -output = 1.848924 -cache_read = 0.077039 +input = 0.922722 +output = 1.845444 +cache_read = 0.076894 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 427614e0228..1db14b9b08b 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.56628 -output = 1.69884 -cache_read = 0.018018 +input = 0.558624 +output = 1.675872 +cache_read = 0.018621 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] From cdd70092cb739402465d19122438bd656473752e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 19:26:14 +0000 Subject: [PATCH 233/392] chore(sync): update Merge Gateway model catalog (#7666) Co-authored-by: opencode-agent[bot] --- providers/merge-gateway/models/qwen/qwen3.5-122b-a10b.toml | 7 ++++--- providers/merge-gateway/models/qwen/qwen3.5-397b-a17b.toml | 7 ++++--- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/providers/merge-gateway/models/qwen/qwen3.5-122b-a10b.toml b/providers/merge-gateway/models/qwen/qwen3.5-122b-a10b.toml index b86a0e589cd..5ef98e7a3a0 100644 --- a/providers/merge-gateway/models/qwen/qwen3.5-122b-a10b.toml +++ b/providers/merge-gateway/models/qwen/qwen3.5-122b-a10b.toml @@ -1,6 +1,7 @@ # Source: https://api-gateway.merge.dev/v1/models?model=qwen%2Fqwen3.5-122b-a10b (accessed 2026-07-21) base_model = "alibaba/qwen3.5-122b-a10b" name = "Qwen3.5 122B A10B" +attachment = false [[reasoning_options]] type = "toggle" @@ -11,8 +12,8 @@ output = 0.917 cache_read = 0.023 [limit] -context = 256_000 -output = 64_000 +context = 131_072 +output = 32_768 [modalities] -input = ["text", "image"] +input = ["text"] diff --git a/providers/merge-gateway/models/qwen/qwen3.5-397b-a17b.toml b/providers/merge-gateway/models/qwen/qwen3.5-397b-a17b.toml index 0264c6bc733..84592d1b8dd 100644 --- a/providers/merge-gateway/models/qwen/qwen3.5-397b-a17b.toml +++ b/providers/merge-gateway/models/qwen/qwen3.5-397b-a17b.toml @@ -2,6 +2,7 @@ # Selected route reasoning.controls = ["thinking"], disable_supported = true. base_model = "alibaba/qwen3.5-397b-a17b" name = "Qwen3.5 397B A17B" +attachment = false [[reasoning_options]] type = "toggle" @@ -12,8 +13,8 @@ output = 1.032 cache_read = 0.0344 [limit] -context = 256_000 -output = 64_000 +context = 131_072 +output = 32_768 [modalities] -input = ["text", "image"] +input = ["text"] From 3feb2f1df5f738390e575795c5c0e1b8134f2bf4 Mon Sep 17 00:00:00 2001 From: Jack Date: Tue, 22 Sep 2026 03:45:13 +0800 Subject: [PATCH 234/392] feat: add MiMo V2.6 models --- models/xiaomi/mimo-v2.6-flash.toml | 19 +++++++++++++++++++ models/xiaomi/mimo-v2.6-pro.toml | 19 +++++++++++++++++++ .../opencode-go/models/mimo-v2.6-flash.toml | 14 ++++++++++++++ .../opencode-go/models/mimo-v2.6-pro.toml | 14 ++++++++++++++ .../opencode/models/mimo-v2.6-flash-free.toml | 15 +++++++++++++++ 5 files changed, 81 insertions(+) create mode 100644 models/xiaomi/mimo-v2.6-flash.toml create mode 100644 models/xiaomi/mimo-v2.6-pro.toml create mode 100644 providers/opencode-go/models/mimo-v2.6-flash.toml create mode 100644 providers/opencode-go/models/mimo-v2.6-pro.toml create mode 100644 providers/opencode/models/mimo-v2.6-flash-free.toml diff --git a/models/xiaomi/mimo-v2.6-flash.toml b/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..3c23f154d83 --- /dev/null +++ b/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,19 @@ +name = "MiMo-V2.6-Flash" +description = "MiMo Flash model for multimodal coding agents and long-context automation" +family = "mimo" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +knowledge = "2024-12" +open_weights = false + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "audio", "video"] +output = ["text"] diff --git a/models/xiaomi/mimo-v2.6-pro.toml b/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..43fd862667d --- /dev/null +++ b/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,19 @@ +name = "MiMo-V2.6-Pro" +description = "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution" +family = "mimo" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = false +reasoning = true +temperature = true +tool_call = true +knowledge = "2024-12" +open_weights = false + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/opencode-go/models/mimo-v2.6-flash.toml b/providers/opencode-go/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..018c58c0b68 --- /dev/null +++ b/providers/opencode-go/models/mimo-v2.6-flash.toml @@ -0,0 +1,14 @@ +base_model = "xiaomi/mimo-v2.6-flash" +reasoning_options = [] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[limit] +context = 1_000_000 +output = 128_000 diff --git a/providers/opencode-go/models/mimo-v2.6-pro.toml b/providers/opencode-go/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..40f32a90e8b --- /dev/null +++ b/providers/opencode-go/models/mimo-v2.6-pro.toml @@ -0,0 +1,14 @@ +base_model = "xiaomi/mimo-v2.6-pro" +attachment = true +reasoning_options = [] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.003625 + +[limit] +output = 128_000 diff --git a/providers/opencode/models/mimo-v2.6-flash-free.toml b/providers/opencode/models/mimo-v2.6-flash-free.toml new file mode 100644 index 00000000000..753e0075aa2 --- /dev/null +++ b/providers/opencode/models/mimo-v2.6-flash-free.toml @@ -0,0 +1,15 @@ +base_model = "xiaomi/mimo-v2.6-flash" +name = "MiMo-V2.6-Flash Free" +reasoning_options = [] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 + +[limit] +context = 200_000 +output = 32_000 From aa6265846839066b9eb7019b406924ad517ee242 Mon Sep 17 00:00:00 2001 From: Jack Date: Tue, 22 Sep 2026 03:48:04 +0800 Subject: [PATCH 235/392] Update knowledge date to 2026-09-22 --- models/xiaomi/mimo-v2.6-pro.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/models/xiaomi/mimo-v2.6-pro.toml b/models/xiaomi/mimo-v2.6-pro.toml index 43fd862667d..85d587e9f63 100644 --- a/models/xiaomi/mimo-v2.6-pro.toml +++ b/models/xiaomi/mimo-v2.6-pro.toml @@ -7,7 +7,7 @@ attachment = false reasoning = true temperature = true tool_call = true -knowledge = "2024-12" +knowledge = "2026-09-22" open_weights = false [limit] From 546269d6e0b90b8b825528f1b1a66c76af4eae8c Mon Sep 17 00:00:00 2001 From: Jack Date: Tue, 22 Sep 2026 03:49:20 +0800 Subject: [PATCH 236/392] Update knowledge date and input modalities --- models/xiaomi/mimo-v2.6-flash.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/models/xiaomi/mimo-v2.6-flash.toml b/models/xiaomi/mimo-v2.6-flash.toml index 3c23f154d83..f065c84b627 100644 --- a/models/xiaomi/mimo-v2.6-flash.toml +++ b/models/xiaomi/mimo-v2.6-flash.toml @@ -7,7 +7,7 @@ attachment = true reasoning = true temperature = true tool_call = true -knowledge = "2024-12" +knowledge = "2026-09-22" open_weights = false [limit] @@ -15,5 +15,5 @@ context = 1_048_576 output = 131_072 [modalities] -input = ["text", "image", "audio", "video"] +input = ["text", "image", "audio", "video", "pdf"] output = ["text"] From 98646c1bf4091865e478e2bcd4f809ba57ca4340 Mon Sep 17 00:00:00 2001 From: Jack Date: Tue, 22 Sep 2026 03:50:14 +0800 Subject: [PATCH 237/392] Enable attachment and expand input modalities Updated attachment status and input modalities. --- models/xiaomi/mimo-v2.6-pro.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/models/xiaomi/mimo-v2.6-pro.toml b/models/xiaomi/mimo-v2.6-pro.toml index 85d587e9f63..fea580df360 100644 --- a/models/xiaomi/mimo-v2.6-pro.toml +++ b/models/xiaomi/mimo-v2.6-pro.toml @@ -3,7 +3,7 @@ description = "Stronger MiMo Pro tier for multimodal reasoning and coding-agent family = "mimo" release_date = "2026-09-22" last_updated = "2026-09-22" -attachment = false +attachment = true reasoning = true temperature = true tool_call = true @@ -15,5 +15,5 @@ context = 1_048_576 output = 131_072 [modalities] -input = ["text"] +input = ["text", "image", "audio", "video", "pdf"] output = ["text"] From 093bb9e02a61b81a3cd5e95f0bfcc71d3add7b76 Mon Sep 17 00:00:00 2001 From: Jack Date: Tue, 22 Sep 2026 04:02:39 +0800 Subject: [PATCH 238/392] fix: remove redundant MiMo override --- providers/opencode-go/models/mimo-v2.6-pro.toml | 1 - 1 file changed, 1 deletion(-) diff --git a/providers/opencode-go/models/mimo-v2.6-pro.toml b/providers/opencode-go/models/mimo-v2.6-pro.toml index 40f32a90e8b..82f9688f50f 100644 --- a/providers/opencode-go/models/mimo-v2.6-pro.toml +++ b/providers/opencode-go/models/mimo-v2.6-pro.toml @@ -1,5 +1,4 @@ base_model = "xiaomi/mimo-v2.6-pro" -attachment = true reasoning_options = [] [interleaved] From 544b829052f24ce5dbef80d498a6e9823aa6426d Mon Sep 17 00:00:00 2001 From: Jack Date: Tue, 22 Sep 2026 04:04:52 +0800 Subject: [PATCH 239/392] fix: inherit MiMo Go limits --- providers/opencode-go/models/mimo-v2.6-flash.toml | 4 ---- providers/opencode-go/models/mimo-v2.6-pro.toml | 3 --- 2 files changed, 7 deletions(-) diff --git a/providers/opencode-go/models/mimo-v2.6-flash.toml b/providers/opencode-go/models/mimo-v2.6-flash.toml index 018c58c0b68..06b0dcf1faa 100644 --- a/providers/opencode-go/models/mimo-v2.6-flash.toml +++ b/providers/opencode-go/models/mimo-v2.6-flash.toml @@ -8,7 +8,3 @@ field = "reasoning_content" input = 0.14 output = 0.28 cache_read = 0.0028 - -[limit] -context = 1_000_000 -output = 128_000 diff --git a/providers/opencode-go/models/mimo-v2.6-pro.toml b/providers/opencode-go/models/mimo-v2.6-pro.toml index 82f9688f50f..7513e538ec8 100644 --- a/providers/opencode-go/models/mimo-v2.6-pro.toml +++ b/providers/opencode-go/models/mimo-v2.6-pro.toml @@ -8,6 +8,3 @@ field = "reasoning_content" input = 0.435 output = 0.87 cache_read = 0.003625 - -[limit] -output = 128_000 From 702c1c4c6d93f358871143b8e5c94bb49a474160 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 20:26:21 +0000 Subject: [PATCH 240/392] chore(sync): update OpenRouter model catalog (#7673) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-pro.toml | 6 ++-- .../models/xiaomi/mimo-v2.6-flash.toml | 16 ++++++++++ .../xiaomi/mimo-v2.6-pro-ultraspeed.toml | 29 +++++++++++++++++++ .../models/xiaomi/mimo-v2.6-pro.toml | 16 ++++++++++ .../models/~deepseek/deepseek-pro-latest.toml | 6 ++-- 5 files changed, 67 insertions(+), 6 deletions(-) create mode 100644 providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml create mode 100644 providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml create mode 100644 providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 5af84d1accb..81e8b644948 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.922722 -output = 1.845444 -cache_read = 0.076894 +input = 0.91263 +output = 1.82526 +cache_read = 0.076053 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml b/providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..3e52e783fa7 --- /dev/null +++ b/providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,16 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +base_model = "xiaomi/mimo-v2.6-flash" +description = "MiMo flash model for fast multimodal assistance and agent workflows" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..40cfae78edb --- /dev/null +++ b/providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,29 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +name = "MiMo-V2.6-Pro-UltraSpeed" +description = "MiMo pro model for strong multimodal reasoning and agent execution" +family = "mimo" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 4.35 +output = 8.7 +cache_read = 0.036 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "video", "audio"] +output = ["text"] diff --git a/providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml b/providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..2f40e81f221 --- /dev/null +++ b/providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,16 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +base_model = "xiaomi/mimo-v2.6-pro" +description = "MiMo pro model for strong multimodal reasoning and agent execution" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 1db14b9b08b..1c92257a1ff 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.558624 -output = 1.675872 -cache_read = 0.018621 +input = 0.5544 +output = 1.6632 +cache_read = 0.01848 [limit] context = 1_048_576 From 256a8e05011870931c8ed2bf092b0691bb01928d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 20:26:34 +0000 Subject: [PATCH 241/392] chore(sync): update Kilo model catalog (#7672) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/xiaomi/mimo-v2.6-flash.toml | 15 +++++++++++++++ providers/kilo/models/xiaomi/mimo-v2.6-pro.toml | 15 +++++++++++++++ .../models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 3 files changed, 33 insertions(+), 3 deletions(-) create mode 100644 providers/kilo/models/xiaomi/mimo-v2.6-flash.toml create mode 100644 providers/kilo/models/xiaomi/mimo-v2.6-pro.toml diff --git a/providers/kilo/models/xiaomi/mimo-v2.6-flash.toml b/providers/kilo/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..b4165fb24ad --- /dev/null +++ b/providers/kilo/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,15 @@ +base_model = "xiaomi/mimo-v2.6-flash" +description = "MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for..." +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/kilo/models/xiaomi/mimo-v2.6-pro.toml b/providers/kilo/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..e3e0a47689b --- /dev/null +++ b/providers/kilo/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,15 @@ +base_model = "xiaomi/mimo-v2.6-pro" +description = "MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding..." +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 6caf71dbf01..b5bfc21fba2 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.558624 -output = 1.675872 -cache_read = 0.018621 +input = 0.5544 +output = 1.6632 +cache_read = 0.01848 [limit] context = 1_024_000 From ba6fdc8e65d2b62122b9f888e0c79eb8380b52a1 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 20:26:43 +0000 Subject: [PATCH 242/392] chore(sync): update Deep Infra model catalog (#7671) Co-authored-by: opencode-agent[bot] --- providers/deepinfra/models/tencent/Hy3.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/deepinfra/models/tencent/Hy3.toml b/providers/deepinfra/models/tencent/Hy3.toml index 141b04fc694..743a6959ec1 100644 --- a/providers/deepinfra/models/tencent/Hy3.toml +++ b/providers/deepinfra/models/tencent/Hy3.toml @@ -3,9 +3,9 @@ structured_output = true reasoning_options = [] [cost] -input = 0.14 -output = 0.58 -cache_read = 0.035 +input = 0.13 +output = 0.53 +cache_read = 0.033 [limit] context = 262_144 From 38cb3a87413159c919b6616ebeebb9280c7ee0d6 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:25:20 +0000 Subject: [PATCH 243/392] chore(sync): update Kilo model catalog (#7676) Co-authored-by: opencode-agent[bot] --- .../xiaomi/mimo-v2.6-pro-ultraspeed.toml | 28 +++++++++++++++++++ .../models/~deepseek/deepseek-pro-latest.toml | 6 ++-- 2 files changed, 31 insertions(+), 3 deletions(-) create mode 100644 providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml diff --git a/providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..08e541fac0d --- /dev/null +++ b/providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,28 @@ +name = "Xiaomi: MiMo-V2.6-Pro-UltraSpeed" +description = "MiMo-V2.6-Pro-UltraSpeed is the fast speed edition of Xiaomi's flagship foundation model, MiMo-V2.6-Pro. Built from the same 1T MiMo-V2.6-Pro checkpoint, it matches the original model in quality while delivering roughly 10x..." +family = "mimo" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 4.35 +output = 8.7 +cache_read = 0.036 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "video", "audio"] +output = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index b5bfc21fba2..3563cca9cda 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.5544 -output = 1.6632 -cache_read = 0.01848 +input = 0.54648 +output = 1.63944 +cache_read = 0.018216 [limit] context = 1_024_000 From 276327558e857b8f3340572d2cd94a089031e1c5 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:25:24 +0000 Subject: [PATCH 244/392] chore(sync): update NanoGPT model catalog (#7677) Co-authored-by: opencode-agent[bot] --- providers/nano-gpt/models/x-ai/grok-4.7.toml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) create mode 100644 providers/nano-gpt/models/x-ai/grok-4.7.toml diff --git a/providers/nano-gpt/models/x-ai/grok-4.7.toml b/providers/nano-gpt/models/x-ai/grok-4.7.toml new file mode 100644 index 00000000000..81b4067b343 --- /dev/null +++ b/providers/nano-gpt/models/x-ai/grok-4.7.toml @@ -0,0 +1,16 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 1.6 +output = 4.8 +cache_read = 0.4 + +[limit] +input = 500_000 + +[modalities] +input = ["text", "image"] From 5d374d40b34fd48c28957a94cd6b397419a3cfd8 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:25:27 +0000 Subject: [PATCH 245/392] chore(sync): update OpenRouter model catalog (#7679) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 81e8b644948..4a55e5fac13 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.91263 -output = 1.82526 -cache_read = 0.076053 +input = 0.900798 +output = 1.801596 +cache_read = 0.075067 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 1c92257a1ff..37d1b27ede8 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5544 -output = 1.6632 -cache_read = 0.01848 +input = 0.54648 +output = 1.63944 +cache_read = 0.018216 [limit] context = 1_048_576 From deef71c3c18f440764ba19d58a2dc462579cd90e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 22:25:50 +0000 Subject: [PATCH 246/392] chore(sync): update Kilo model catalog (#7683) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/z-ai/glm-5.3-flash.toml | 2 +- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- .../kilo/models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- 4 files changed, 8 insertions(+), 8 deletions(-) diff --git a/providers/kilo/models/z-ai/glm-5.3-flash.toml b/providers/kilo/models/z-ai/glm-5.3-flash.toml index 1bc5d8b7a69..c464db465b8 100644 --- a/providers/kilo/models/z-ai/glm-5.3-flash.toml +++ b/providers/kilo/models/z-ai/glm-5.3-flash.toml @@ -12,7 +12,7 @@ cache_read = 0.03 [limit] context = 1_048_576 -output = 102_400 +output = 943_718 [modalities] input = ["text", "image", "video"] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 3563cca9cda..85d8d21cda2 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.54648 -output = 1.63944 -cache_read = 0.018216 +input = 0.53856 +output = 1.61568 +cache_read = 0.017952 [limit] context = 1_024_000 diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 49d602e67e6..8aac86d2dac 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -16,7 +16,7 @@ values = ["none", "low", "high", "max"] [cost] input = 0.04 -output = 0.2 +output = 0.4 cache_read = 0.01 [limit] diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 631c1758bcd..afff7105d92 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.7 -output = 8.5 -cache_read = 0.17 +input = 1.5 +output = 7.5 +cache_read = 0.15 [limit] context = 1_048_576 From 59955f4459ca03e3b7c824021ff4d502be6cba17 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 22:25:53 +0000 Subject: [PATCH 247/392] chore(sync): update OpenRouter model catalog (#7682) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-0731.toml | 2 +- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.3-flash.toml | 8 ++++---- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- 6 files changed, 15 insertions(+), 15 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml index 48200f680e3..3de935d9215 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml @@ -11,7 +11,7 @@ values = ["low", "high", "max"] [cost] input = 0.04 -output = 0.32 +output = 0.64 cache_read = 0.016 [limit] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 4a55e5fac13..43ba15830d0 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.900798 -output = 1.801596 -cache_read = 0.075067 +input = 0.892272 +output = 1.784544 +cache_read = 0.074356 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-5.3-flash.toml b/providers/openrouter/models/z-ai/glm-5.3-flash.toml index 6fd3f066cfd..188a84f8fb5 100644 --- a/providers/openrouter/models/z-ai/glm-5.3-flash.toml +++ b/providers/openrouter/models/z-ai/glm-5.3-flash.toml @@ -6,13 +6,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.075 -output = 0.25 -cache_read = 0.02 +input = 0.15 +output = 0.5 +cache_read = 0.05 [limit] context = 1_310_720 -output = 102_400 +output = 943_718 [modalities] input = ["text", "image", "video"] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 37d1b27ede8..3783e77eceb 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.54648 -output = 1.63944 -cache_read = 0.018216 +input = 0.53856 +output = 1.61568 +cache_read = 0.017952 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 867340e2671..6d98cbc2838 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -21,7 +21,7 @@ values = ["low", "high", "max"] [cost] input = 0.04 -output = 0.2 +output = 0.4 cache_read = 0.01 [limit] diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 7031b748f98..0bb8650abd5 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.7 -output = 8.5 -cache_read = 0.17 +input = 1.5 +output = 7.5 +cache_read = 0.15 [limit] context = 1_048_576 From 64964841a65dbd1f262ede752a702ec018ef3084 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 22:26:20 +0000 Subject: [PATCH 248/392] chore(sync): update EmpirioLabs AI model catalog (#7681) Co-authored-by: opencode-agent[bot] --- .../empiriolabs/models/mimo-v2-6-flash.toml | 17 +++++++++++++++++ providers/empiriolabs/models/mimo-v2-6-pro.toml | 17 +++++++++++++++++ 2 files changed, 34 insertions(+) create mode 100644 providers/empiriolabs/models/mimo-v2-6-flash.toml create mode 100644 providers/empiriolabs/models/mimo-v2-6-pro.toml diff --git a/providers/empiriolabs/models/mimo-v2-6-flash.toml b/providers/empiriolabs/models/mimo-v2-6-flash.toml new file mode 100644 index 00000000000..ab796f4d7ef --- /dev/null +++ b/providers/empiriolabs/models/mimo-v2-6-flash.toml @@ -0,0 +1,17 @@ +base_model = "xiaomi/mimo-v2.6-flash" +name = "MiMo V2.6 Flash" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.7 +output = 1.4 +cache_read = 0.7 + +[limit] +context = 1_000_000 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/empiriolabs/models/mimo-v2-6-pro.toml b/providers/empiriolabs/models/mimo-v2-6-pro.toml new file mode 100644 index 00000000000..04dcc7ceb0e --- /dev/null +++ b/providers/empiriolabs/models/mimo-v2-6-pro.toml @@ -0,0 +1,17 @@ +base_model = "xiaomi/mimo-v2.6-pro" +name = "MiMo V2.6 Pro" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 2.175 +output = 4.35 +cache_read = 2.175 + +[limit] +context = 1_000_000 + +[modalities] +input = ["text", "image", "video", "audio"] From d51cf7c4cc7bb39fe5e82247625d161f9f120376 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:25:04 +0000 Subject: [PATCH 249/392] chore(sync): update Kilo model catalog (#7685) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/meta-llama/llama-4-maverick.toml | 2 +- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/meta-llama/llama-4-maverick.toml b/providers/kilo/models/meta-llama/llama-4-maverick.toml index 537ab8cd5fd..fab1dd0fc03 100644 --- a/providers/kilo/models/meta-llama/llama-4-maverick.toml +++ b/providers/kilo/models/meta-llama/llama-4-maverick.toml @@ -15,7 +15,7 @@ input = 0.1875 output = 0.6525 [limit] -context = 1_048_576 +context = 128_000 output = 16_384 [modalities] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 85d8d21cda2..022394b5a93 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.53856 -output = 1.61568 -cache_read = 0.017952 +input = 0.53196 +output = 1.59588 +cache_read = 0.016926 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 393_216 [modalities] input = ["text"] From 32107e3c9042bc6ee3c1d9ed21a2b92e738621cf Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:25:07 +0000 Subject: [PATCH 250/392] chore(sync): update OpenRouter model catalog (#7684) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/meta-llama/llama-4-maverick.toml | 4 ++-- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 43ba15830d0..3cb613ddd30 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.892272 -output = 1.784544 -cache_read = 0.074356 +input = 0.883746 +output = 1.767492 +cache_read = 0.073646 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/meta-llama/llama-4-maverick.toml b/providers/openrouter/models/meta-llama/llama-4-maverick.toml index de6e5240ff9..b9b42749c83 100644 --- a/providers/openrouter/models/meta-llama/llama-4-maverick.toml +++ b/providers/openrouter/models/meta-llama/llama-4-maverick.toml @@ -12,8 +12,8 @@ knowledge = "2024-08-31" open_weights = true [cost] -input = 0.2 -output = 0.8 +input = 0.1875 +output = 0.6525 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 3783e77eceb..85762922f96 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.53856 -output = 1.61568 -cache_read = 0.017952 +input = 0.53196 +output = 1.59588 +cache_read = 0.016926 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] From f71cbc84cad92ffdd2bc7b09445c5bdf24c7dc5b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 00:59:36 +0000 Subject: [PATCH 251/392] chore(sync): update OpenRouter model catalog (#7686) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/tencent/hy3.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.3.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- providers/openrouter/models/~z-ai/glm-latest.toml | 8 ++++---- 7 files changed, 23 insertions(+), 23 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index f9ab26fd006..6f817c5a0a0 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.05544 -output = 0.11088 -cache_read = 0.011088 +input = 0.088606 +output = 0.177212 +cache_read = 0.017721 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 3cb613ddd30..e7683af4617 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.883746 -output = 1.767492 -cache_read = 0.073646 +input = 0.95526 +output = 1.91052 +cache_read = 0.079605 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/tencent/hy3.toml b/providers/openrouter/models/tencent/hy3.toml index f61fa3175e9..80ecfdd3aa8 100644 --- a/providers/openrouter/models/tencent/hy3.toml +++ b/providers/openrouter/models/tencent/hy3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.0825 -output = 0.33 -cache_read = 0.020625 +input = 0.132 +output = 0.528 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index 6ca3ad0c4f6..b68509e1e67 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.91 -output = 2.86 -cache_read = 0.169 +input = 0.784 +output = 2.464 +cache_read = 0.1456 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 85762922f96..1db14b9b08b 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.53196 -output = 1.59588 -cache_read = 0.016926 +input = 0.558624 +output = 1.675872 +cache_read = 0.018621 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 6d98cbc2838..85c55eed894 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.04 -output = 0.4 -cache_read = 0.01 +input = 0.03 +output = 1 +cache_read = 0.012 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 709b8881ffa..12ade84d329 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7728 -output = 2.4288 -cache_read = 0.14352 +input = 0.7839 +output = 2.6532 +cache_read = 0.15678 [limit] context = 1_310_720 -output = 131_072 +output = 235_929 [modalities] input = ["text"] From 1a23b290742855a2960a4422da5c8116b9711618 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 00:59:39 +0000 Subject: [PATCH 252/392] chore(sync): update DigitalOcean model catalog (#7688) Co-authored-by: opencode-agent[bot] --- providers/digitalocean/models/deepseek-3.2.toml | 6 +++--- providers/digitalocean/models/deepseek-4-flash.toml | 6 +++--- providers/digitalocean/models/deepseek-v4-flash-0731.toml | 6 +++--- providers/digitalocean/models/deepseek-v4-pro.toml | 6 +++--- providers/digitalocean/models/glm-5.2.toml | 6 +++--- providers/digitalocean/models/glm-5.3.toml | 6 +++--- providers/digitalocean/models/kimi-k3.toml | 6 +++--- providers/digitalocean/models/mimo-v2.5-pro.toml | 6 +++--- providers/digitalocean/models/openai-gpt-oss-120b.toml | 4 ++-- 9 files changed, 26 insertions(+), 26 deletions(-) diff --git a/providers/digitalocean/models/deepseek-3.2.toml b/providers/digitalocean/models/deepseek-3.2.toml index 86a0601e001..d8d92fa42fa 100644 --- a/providers/digitalocean/models/deepseek-3.2.toml +++ b/providers/digitalocean/models/deepseek-3.2.toml @@ -18,9 +18,9 @@ type = "effort" values = ["none", "low", "medium", "high"] [cost] -input = 0.25 -output = 0.8 -cache_read = 0.075 +input = 0.5 +output = 1.6 +cache_read = 0.15 [limit] context = 163_840 diff --git a/providers/digitalocean/models/deepseek-4-flash.toml b/providers/digitalocean/models/deepseek-4-flash.toml index 2e033308f91..143346a359b 100644 --- a/providers/digitalocean/models/deepseek-4-flash.toml +++ b/providers/digitalocean/models/deepseek-4-flash.toml @@ -10,9 +10,9 @@ tool_call = true open_weights = false [cost] -input = 0.0679 -output = 0.168 -cache_read = 0.0168 +input = 0.14 +output = 0.28 +cache_read = 0.028 [limit] context = 1_048_576 diff --git a/providers/digitalocean/models/deepseek-v4-flash-0731.toml b/providers/digitalocean/models/deepseek-v4-flash-0731.toml index 74ce3b0ae91..2e9e64c5632 100644 --- a/providers/digitalocean/models/deepseek-v4-flash-0731.toml +++ b/providers/digitalocean/models/deepseek-v4-flash-0731.toml @@ -5,9 +5,9 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.08 -output = 0.252 -cache_read = 0.0252 +input = 0.14 +output = 0.28 +cache_read = 0.028 [limit] context = 1_048_576 diff --git a/providers/digitalocean/models/deepseek-v4-pro.toml b/providers/digitalocean/models/deepseek-v4-pro.toml index 8a4edb98729..b4ebe0207c1 100644 --- a/providers/digitalocean/models/deepseek-v4-pro.toml +++ b/providers/digitalocean/models/deepseek-v4-pro.toml @@ -19,9 +19,9 @@ type = "effort" values = ["low", "medium", "high", "xhigh"] [cost] -input = 0.87 -output = 1.74 -cache_read = 0.174 +input = 1.74 +output = 3.48 +cache_read = 0.348 [limit] context = 1_048_576 diff --git a/providers/digitalocean/models/glm-5.2.toml b/providers/digitalocean/models/glm-5.2.toml index 3f01602c179..eb3b3d1b78d 100644 --- a/providers/digitalocean/models/glm-5.2.toml +++ b/providers/digitalocean/models/glm-5.2.toml @@ -8,9 +8,9 @@ type = "effort" values = ["medium", "high", "xhigh"] [cost] -input = 0.7 -output = 2.2 -cache_read = 0.105 +input = 1.4 +output = 4.4 +cache_read = 0.21 [limit] context = 262_144 diff --git a/providers/digitalocean/models/glm-5.3.toml b/providers/digitalocean/models/glm-5.3.toml index 9ba7811a008..6172a61a5ab 100644 --- a/providers/digitalocean/models/glm-5.3.toml +++ b/providers/digitalocean/models/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.95 -output = 3.4 -cache_read = 0.2 +input = 1.4 +output = 4.4 +cache_read = 0.26 [limit] context = 1_048_576 diff --git a/providers/digitalocean/models/kimi-k3.toml b/providers/digitalocean/models/kimi-k3.toml index 7533a7423e2..62d0f7b7404 100644 --- a/providers/digitalocean/models/kimi-k3.toml +++ b/providers/digitalocean/models/kimi-k3.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.55 -output = 12.95 -cache_read = 0.285 +input = 3 +output = 15 +cache_read = 0.3 [modalities] input = ["text", "image"] diff --git a/providers/digitalocean/models/mimo-v2.5-pro.toml b/providers/digitalocean/models/mimo-v2.5-pro.toml index 20f6f6db3ea..365235fb39f 100644 --- a/providers/digitalocean/models/mimo-v2.5-pro.toml +++ b/providers/digitalocean/models/mimo-v2.5-pro.toml @@ -9,9 +9,9 @@ type = "effort" values = ["none", "high"] [cost] -input = 0.4 -output = 1.5 -cache_read = 0.08 +input = 0.8 +output = 3 +cache_read = 0.16 [limit] context = 262_144 diff --git a/providers/digitalocean/models/openai-gpt-oss-120b.toml b/providers/digitalocean/models/openai-gpt-oss-120b.toml index fc45d3aac8b..4af4fef37ad 100644 --- a/providers/digitalocean/models/openai-gpt-oss-120b.toml +++ b/providers/digitalocean/models/openai-gpt-oss-120b.toml @@ -19,8 +19,8 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.055 -output = 0.385 +input = 0.1 +output = 0.7 cache_read = 0.02 [limit] From a16b62ad1be404f4f10308b71807787b8afa2732 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 00:59:46 +0000 Subject: [PATCH 253/392] chore(sync): update Kilo model catalog (#7687) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/tencent/hy3.toml | 6 +++--- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- .../models/~deepseek/deepseek-v4-flash-latest.toml | 6 +++--- providers/kilo/models/~z-ai/glm-latest.toml | 10 +++++----- 4 files changed, 16 insertions(+), 16 deletions(-) diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index e96e3110ab1..c3469c1e9dc 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.0825 -output = 0.33 -cache_read = 0.020625 +input = 0.13 +output = 0.53 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 022394b5a93..6caf71dbf01 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.53196 -output = 1.59588 -cache_read = 0.016926 +input = 0.558624 +output = 1.675872 +cache_read = 0.018621 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 8aac86d2dac..ca12f499908 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.04 -output = 0.4 -cache_read = 0.01 +input = 0.03 +output = 1 +cache_read = 0.012 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 2237628e2fc..6a2c43df8a6 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7728 -output = 2.4288 -cache_read = 0.14352 +input = 0.7839 +output = 2.6532 +cache_read = 0.15678 [limit] -context = 1_048_576 -output = 131_072 +context = 262_144 +output = 235_929 [modalities] input = ["text"] From 59590e31fb40ac41a9ec4fb81af471c0e4883bb9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 01:35:29 +0000 Subject: [PATCH 254/392] chore(sync): update Kilo model catalog (#7690) Co-authored-by: opencode-agent[bot] --- .../models/~deepseek/deepseek-v4-flash-latest.toml | 4 ++-- providers/kilo/models/~z-ai/glm-latest.toml | 10 +++++----- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index ca12f499908..0412502ff5d 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -16,8 +16,8 @@ values = ["none", "low", "high", "max"] [cost] input = 0.03 -output = 1 -cache_read = 0.012 +output = 0.8 +cache_read = 0.008 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 6a2c43df8a6..51780ddf2fb 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7839 -output = 2.6532 -cache_read = 0.15678 +input = 0.7 +output = 2.2 +cache_read = 0.13 [limit] -context = 262_144 -output = 235_929 +context = 1_048_576 +output = 131_072 [modalities] input = ["text"] From d577dc96de1773f5849f093eb1a4c05088ef7276 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 01:35:33 +0000 Subject: [PATCH 255/392] chore(sync): update OpenRouter model catalog (#7689) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- .../models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- providers/openrouter/models/~z-ai/glm-latest.toml | 8 ++++---- 4 files changed, 11 insertions(+), 11 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..b8d53b0809c 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 1.32 +output = 3.96 +cache_read = 0.044 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 16f058ed841..854b27b6a72 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.15 -output = 0.6 -cache_read = 0.003 +input = 0.3 +output = 1.2 +cache_read = 0.006 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 85c55eed894..81420955903 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -21,7 +21,7 @@ values = ["low", "high", "max"] [cost] input = 0.03 -output = 1 +output = 0.8 cache_read = 0.012 [limit] diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 12ade84d329..afa82f5ee86 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7839 -output = 2.6532 -cache_read = 0.15678 +input = 0.7826 +output = 2.4596 +cache_read = 0.14534 [limit] context = 1_310_720 -output = 235_929 +output = 131_072 [modalities] input = ["text"] From d0a0af81e26a783730ba502b6fbaf7b9ec52714b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 01:35:49 +0000 Subject: [PATCH 256/392] chore(sync): update NanoGPT model catalog (#7691) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index b4cce19afb3..a62995b0beb 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.22 -output = 0.66 -cache_read = 0.007 +input = 0.44 +output = 1.32 +cache_read = 0.014 [limit] context = 1_048_576 From 63b9be07808ebb2da0f0c158456d892d56663e6e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:24:58 -0500 Subject: [PATCH 257/392] chore(sync): update CoreWeave model catalog (#7643) Co-authored-by: opencode-agent[bot] --- .../models/deepseek-ai/DeepSeek-V4.1-Flash.toml | 15 +++++++++++++++ 1 file changed, 15 insertions(+) create mode 100644 providers/wandb/models/deepseek-ai/DeepSeek-V4.1-Flash.toml diff --git a/providers/wandb/models/deepseek-ai/DeepSeek-V4.1-Flash.toml b/providers/wandb/models/deepseek-ai/DeepSeek-V4.1-Flash.toml new file mode 100644 index 00000000000..0c25c1c3eec --- /dev/null +++ b/providers/wandb/models/deepseek-ai/DeepSeek-V4.1-Flash.toml @@ -0,0 +1,15 @@ +base_model = "deepseek/deepseek-v4.1-flash" +description = "DeepSeek V4.1 Flash is a multimodal MoE model for coding, reasoning, and agentic workloads with long contexts." +family = "deepseek" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.2 +output = 0.65 +cache_read = 0.03 + +[limit] +context = 1_048_576 +output = 1_048_576 From 7fb291189f66de31dd1edb34dbc3d25597995d38 Mon Sep 17 00:00:00 2001 From: Divy Date: Mon, 21 Sep 2026 19:25:26 -0700 Subject: [PATCH 258/392] coralbricks: retire Kimi K3, add DeepSeek V4.1 Flash, publish cache-write rates (#7599) * remove(coralbricks): retire Kimi K3 Issue: CoralBricks retired kimi-k3 on 2026-09-20 with a hard 404. Fix: delete the stale provider model entry. * add(coralbricks): DeepSeek V4.1 Flash (deepseek-v4.1-flash-fast-fp4) Public on inference.coralbricks.ai since 2026-09-21: 1M context, $0.22 in / $0.66 out per 1M tokens, cached input free, cache writes $0.07 (0.3x input), text only, served as an MXFP4 checkpoint. * coralbricks: DeepSeek V4.1 Flash at $0.30 / $1.20, declare reasoning options CoralBricks set the published rate for deepseek-v4.1-flash-fast-fp4 to $0.30 input / $1.20 output per 1M tokens, with cache writes at $0.09. The entry inherits reasoning = true from the lab baseline but declared no reasoning_options, which failed validate. Declare the thinking toggle and low / high / max effort, each verified against the CoralBricks endpoint. Co-Authored-By: Claude Opus 5 (1M context) * coralbricks: DeepSeek V4.1 Flash reasoning via reasoning_effort, with sources On the CoralBricks endpoint reasoning is off unless reasoning_effort is set, and thinking.type has no effect, so the control is effort none | low | high | max rather than a toggle. Adds a header with the pricing sources and the verified wire path. Co-Authored-By: Claude Opus 5 (1M context) * coralbricks: DeepSeek V4.1 Flash interleaved reasoning, no attachments Reasoning comes back in reasoning_content, as on the other CoralBricks reasoners, so interleaved = true. The endpoint serves this model text-only and rejects image parts with a 400, so attachment = false rather than the base model's true. Co-Authored-By: Claude Opus 5 (1M context) * coralbricks: publish cache-write rates for every model https://www.coralbricks.ai/pricing bills the novel tokens of a request at the cache-write rate, 1.5x input: GLM 5.3 $1.68, GLM 5.3 Flash $0.23, GPT-OSS 120B $0.18 per 1M. Add cache_write to those three entries. DeepSeek V4.1 Flash already carries its $0.09. Its header now states the effort field only, without claims about the default state. Co-Authored-By: Claude Opus 5 (1M context) * coralbricks: DeepSeek V4.1 Flash reasoning is opt-in; add medium effort Reasoning on this endpoint is opt-in by design: with no reasoning_effort, or none, the model answers without reasoning; low, medium, high and max turn it on. State that in the header and declare medium, which the endpoint accepts. Co-Authored-By: Claude Opus 5 (1M context) * coralbricks: DeepSeek V4.1 Flash effort ladder back to low / high / max The endpoint accepts medium, but acceptance is not a distinct depth; keep the lab ladder plus the host's none (off). Co-Authored-By: Claude Opus 5 (1M context) --------- Co-authored-by: Claude Opus 5 (1M context) --- .../models/deepseek-v4.1-flash-fast-fp4.toml | 25 +++++++++++++++++++ .../coralbricks/models/glm-5.3-flash-fp4.toml | 1 + providers/coralbricks/models/glm-5.3-fp4.toml | 1 + .../coralbricks/models/gpt-oss-120b.toml | 1 + providers/coralbricks/models/kimi-k3.toml | 10 -------- 5 files changed, 28 insertions(+), 10 deletions(-) create mode 100644 providers/coralbricks/models/deepseek-v4.1-flash-fast-fp4.toml delete mode 100644 providers/coralbricks/models/kimi-k3.toml diff --git a/providers/coralbricks/models/deepseek-v4.1-flash-fast-fp4.toml b/providers/coralbricks/models/deepseek-v4.1-flash-fast-fp4.toml new file mode 100644 index 00000000000..9c63cb4a2db --- /dev/null +++ b/providers/coralbricks/models/deepseek-v4.1-flash-fast-fp4.toml @@ -0,0 +1,25 @@ +# CoralBricks pricing per https://www.coralbricks.ai/pricing and https://www.coralbricks.ai/api/public/models (accessed 2026-09-21): +# input $0.30, output $1.20, cache write $0.09 per 1M tokens; cached input is free. +# Effort: reasoning_effort = none|low|high|max. Reasoning is opt-in: with no reasoning_effort, or none, the model +# answers without reasoning. thinking.type is ignored by this endpoint. +# Verified live against POST https://inference.coralbricks.ai/v1/chat/completions (2026-09-21): low, high and max +# return HTTP 200 with reasoning_content; none and no field return 0 reasoning tokens. +base_model = "deepseek/deepseek-v4.1-flash" +name = "DeepSeek V4.1 Flash FP4" +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] +interleaved = true +attachment = false + +[limit] +context = 1_048_576 +output = 131_072 + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0 +cache_write = 0.09 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/coralbricks/models/glm-5.3-flash-fp4.toml b/providers/coralbricks/models/glm-5.3-flash-fp4.toml index a7836490ddd..2c52bfdcb00 100644 --- a/providers/coralbricks/models/glm-5.3-flash-fp4.toml +++ b/providers/coralbricks/models/glm-5.3-flash-fp4.toml @@ -12,6 +12,7 @@ output = 131_072 input = 0.15 output = 0.5 cache_read = 0 +cache_write = 0.23 [modalities] input = ["text", "image", "video"] diff --git a/providers/coralbricks/models/glm-5.3-fp4.toml b/providers/coralbricks/models/glm-5.3-fp4.toml index 2488afc97cb..90e669e9d54 100644 --- a/providers/coralbricks/models/glm-5.3-fp4.toml +++ b/providers/coralbricks/models/glm-5.3-fp4.toml @@ -12,3 +12,4 @@ output = 131_072 input = 1.12 output = 4.4 cache_read = 0 +cache_write = 1.68 diff --git a/providers/coralbricks/models/gpt-oss-120b.toml b/providers/coralbricks/models/gpt-oss-120b.toml index 8fd3f30e11e..80b28acc1b5 100644 --- a/providers/coralbricks/models/gpt-oss-120b.toml +++ b/providers/coralbricks/models/gpt-oss-120b.toml @@ -8,3 +8,4 @@ interleaved = true input = 0.12 output = 0.6 cache_read = 0 +cache_write = 0.18 diff --git a/providers/coralbricks/models/kimi-k3.toml b/providers/coralbricks/models/kimi-k3.toml deleted file mode 100644 index 77c2aabdc7b..00000000000 --- a/providers/coralbricks/models/kimi-k3.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "moonshotai/kimi-k3" - -reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] -interleaved = true - -[cost] -input = 3.0 -output = 15.0 -cache_read = 0 From c6db867cb5b2b2d601b29a32d5705ab9e78d9685 Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Tue, 22 Sep 2026 10:26:28 +0800 Subject: [PATCH 259/392] feat(aihubmix): add gemini-3.6-flash (#7430) AIHubMix serves this route but the catalog carried no aihubmix entry for it. Cost comes from the provider's public model listing and the reasoning controls from its published model-data index; limit, modalities, attachment and tool_call inherit from models/google/gemini-3.6-flash.toml rather than being repeated here. Co-authored-by: chenxue Co-authored-by: Claude Opus 5 --- providers/aihubmix/models/gemini-3.6-flash.toml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 providers/aihubmix/models/gemini-3.6-flash.toml diff --git a/providers/aihubmix/models/gemini-3.6-flash.toml b/providers/aihubmix/models/gemini-3.6-flash.toml new file mode 100644 index 00000000000..d9c9722442d --- /dev/null +++ b/providers/aihubmix/models/gemini-3.6-flash.toml @@ -0,0 +1,14 @@ +# Effort: minimal|low|medium|high +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.6-flash" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 1.5 +output = 7.5 +cache_read = 0.15 From 226a53f287a2ca8e18803398f321bc14b9b95866 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:29:23 -0500 Subject: [PATCH 260/392] chore(sync): update Vercel AI Gateway model catalog (#7678) * chore(sync): update Vercel AI Gateway model catalog * fix(vercel): normalize MiMo V2.6 models * fix(vercel): add MiMo V2.6 reasoning efforts --------- Co-authored-by: opencode-agent[bot] Co-authored-by: rekram1-node --- models/xiaomi/mimo-v2.6-pro-ultraspeed.toml | 18 ++++++++++++++++++ .../models/deepseek/deepseek-v4.1-flash.toml | 2 +- .../vercel/models/xiaomi/mimo-v2.6-flash.toml | 13 +++++++++++++ .../xiaomi/mimo-v2.6-pro-ultraspeed.toml | 13 +++++++++++++ .../vercel/models/xiaomi/mimo-v2.6-pro.toml | 13 +++++++++++++ 5 files changed, 58 insertions(+), 1 deletion(-) create mode 100644 models/xiaomi/mimo-v2.6-pro-ultraspeed.toml create mode 100644 providers/vercel/models/xiaomi/mimo-v2.6-flash.toml create mode 100644 providers/vercel/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml create mode 100644 providers/vercel/models/xiaomi/mimo-v2.6-pro.toml diff --git a/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..a4ef4e05975 --- /dev/null +++ b/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,18 @@ +name = "MiMo-V2.6-Pro-UltraSpeed" +description = "MiMo pro model for strong multimodal reasoning and agent execution" +family = "mimo" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +open_weights = false + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "video", "audio"] +output = ["text"] diff --git a/providers/vercel/models/deepseek/deepseek-v4.1-flash.toml b/providers/vercel/models/deepseek/deepseek-v4.1-flash.toml index c185e36af68..5d9ec77c911 100644 --- a/providers/vercel/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/vercel/models/deepseek/deepseek-v4.1-flash.toml @@ -12,7 +12,7 @@ values = ["high", "xhigh"] [cost] input = 0.3 output = 1.2 -cache_read = 0.03 +cache_read = 0.007 [limit] context = 1_048_576 diff --git a/providers/vercel/models/xiaomi/mimo-v2.6-flash.toml b/providers/vercel/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..bb5496a5085 --- /dev/null +++ b/providers/vercel/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,13 @@ +# Effort: reasoning.effort = none|minimal|low|medium|high|xhigh|max +# https://mimo.mi.com/docs/en-US/api/chat/responses +base_model = "xiaomi/mimo-v2.6-flash" +name = "MiMo V2.6 Flash" +reasoning_options = [{ type = "effort", values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[modalities] +input = ["text", "image"] diff --git a/providers/vercel/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/providers/vercel/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..dc1eb9fa8fa --- /dev/null +++ b/providers/vercel/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,13 @@ +# Effort: reasoning.effort = none|minimal|low|medium|high|xhigh|max +# https://mimo.mi.com/docs/en-US/api/chat/responses +base_model = "xiaomi/mimo-v2.6-pro-ultraspeed" +name = "MiMo V2.6 Pro UltraSpeed" +reasoning_options = [{ type = "effort", values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4.35 +output = 8.7 +cache_read = 0.036 + +[modalities] +input = ["text", "image"] diff --git a/providers/vercel/models/xiaomi/mimo-v2.6-pro.toml b/providers/vercel/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..11f32cb1976 --- /dev/null +++ b/providers/vercel/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,13 @@ +# Effort: reasoning.effort = none|minimal|low|medium|high|xhigh|max +# https://mimo.mi.com/docs/en-US/api/chat/responses +base_model = "xiaomi/mimo-v2.6-pro" +name = "MiMo V2.6 Pro" +reasoning_options = [{ type = "effort", values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 + +[modalities] +input = ["text", "image"] From 15d02bdb540bc047ef8943d41bf59214477a14e8 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:30:37 -0500 Subject: [PATCH 261/392] fix(sync): preserve Workers AI pricing overrides (#7651) Co-authored-by: rekram1-node --- .../sync/providers/cloudflare-workers-ai.ts | 8 +-- .../core/test/cloudflare-workers-ai.test.ts | 56 +++++++++++++++++++ 2 files changed, 57 insertions(+), 7 deletions(-) create mode 100644 packages/core/test/cloudflare-workers-ai.test.ts diff --git a/packages/core/src/sync/providers/cloudflare-workers-ai.ts b/packages/core/src/sync/providers/cloudflare-workers-ai.ts index 252e9ae9cba..e68dd2b8e36 100644 --- a/packages/core/src/sync/providers/cloudflare-workers-ai.ts +++ b/packages/core/src/sync/providers/cloudflare-workers-ai.ts @@ -41,13 +41,7 @@ const CloudflareModel = z.object({ max_output_length: z.number().nullable().optional(), input_modalities: z.array(z.string()).optional(), output_modalities: z.array(z.string()).optional(), - pricing: z.object({ - prompt: z.string(), - completion: z.string(), - internal_reasoning: z.string().optional(), - input_cache_read: z.string().optional(), - input_cache_write: z.string().optional(), - }), + pricing: OpenRouterModel.shape.pricing, supported_features: z.array(z.string()).optional(), supported_sampling_parameters: z.array(z.string()).optional(), }).passthrough(); diff --git a/packages/core/test/cloudflare-workers-ai.test.ts b/packages/core/test/cloudflare-workers-ai.test.ts new file mode 100644 index 00000000000..e15416cb7ec --- /dev/null +++ b/packages/core/test/cloudflare-workers-ai.test.ts @@ -0,0 +1,56 @@ +import { expect, test } from "bun:test"; + +import { cloudflareWorkersAi } from "../src/sync/providers/cloudflare-workers-ai.js"; + +test("preserves OpenRouter pricing overrides as context tiers", () => { + const [source] = cloudflareWorkersAi.parseModels({ + result: { + data: [{ + id: "workers-ai/@cf/test/context-priced", + name: "Context Priced", + created: 1_782_777_600, + hugging_face_id: null, + knowledge_cutoff: null, + context_length: 256_000, + architecture: { + input_modalities: ["text"], + output_modalities: ["text"], + }, + pricing: { + prompt: "0.000001", + completion: "0.000002", + overrides: [{ + min_prompt_tokens: 128_000, + prompt: "0.000003", + completion: "0.000004", + }], + }, + top_provider: { + context_length: 256_000, + max_completion_tokens: 8_192, + }, + supported_parameters: [], + }], + }, + }); + + expect(source?.pricing.overrides).toEqual([{ + min_prompt_tokens: 128_000, + prompt: "0.000003", + completion: "0.000004", + }]); + + const translated = cloudflareWorkersAi.translateModel(source!, { + existing: () => undefined, + authored: () => undefined, + }); + expect(translated.model.cost).toEqual({ + input: 1, + output: 2, + tiers: [{ + tier: { type: "context", size: 128_000 }, + input: 3, + output: 4, + }], + }); +}); From 8a0fe46cc5e854bea2668da39a9a5faf267b3482 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:34:42 -0500 Subject: [PATCH 262/392] fix(alibaba-cn): correct Qwen context pricing (#7692) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Use Alibaba’s published Beijing USD tiers and align qwen3-max with the current 2026-01-23 alias metadata. Co-authored-by: rekram1-node Co-authored-by: tbourrillon <7868998+tbourrillon@users.noreply.github.com> --- providers/alibaba-cn/models/qwen3-max.toml | 40 +++++++++++++++---- .../alibaba-cn/models/qwen3.5-flash.toml | 29 ++++++++++---- providers/alibaba-cn/models/qwen3.5-plus.toml | 30 ++++++++++---- 3 files changed, 78 insertions(+), 21 deletions(-) diff --git a/providers/alibaba-cn/models/qwen3-max.toml b/providers/alibaba-cn/models/qwen3-max.toml index ffb4eb4582b..6f9229234b8 100644 --- a/providers/alibaba-cn/models/qwen3-max.toml +++ b/providers/alibaba-cn/models/qwen3-max.toml @@ -1,23 +1,49 @@ +# Sources (accessed 2026-09-22): +# https://help.aliyun.com/en/model-studio/model-pricing (China (Beijing) USD list prices) +# https://help.aliyun.com/en/model-studio/model-qwen3-max (alias, capabilities, and limits) +# https://help.aliyun.com/en/model-studio/deep-thinking +# qwen3-max currently aliases qwen3-max-2026-01-23. +# Toggle: enable_thinking true|false +# Budget: thinking_budget (integer reasoning tokens, max 81,920) name = "Qwen3 Max" description = "Flagship Qwen model for complex reasoning, coding, and agentic workflows" family = "qwen" -release_date = "2025-09-23" -last_updated = "2026-09-11" +release_date = "2026-01-23" +last_updated = "2026-09-22" attachment = false -reasoning = false +reasoning = true +structured_output = true temperature = true knowledge = "2025-04" tool_call = true open_weights = false +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" +max = 81_920 + +[interleaved] +field = "reasoning_content" + [cost] -input = 1.291 -output = 7.749 +input = 0.359 +output = 1.434 +reasoning = 1.434 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.574 +output = 2.294 +reasoning = 2.294 [[cost.tiers]] tier = { type = "context", size = 128_000 } -input = 2.153 -output = 12.915 +input = 1.004 +output = 4.014 +reasoning = 4.014 [limit] context = 262_144 diff --git a/providers/alibaba-cn/models/qwen3.5-flash.toml b/providers/alibaba-cn/models/qwen3.5-flash.toml index 51651ed2e75..4e8481bd07f 100644 --- a/providers/alibaba-cn/models/qwen3.5-flash.toml +++ b/providers/alibaba-cn/models/qwen3.5-flash.toml @@ -1,8 +1,14 @@ +# Sources (accessed 2026-09-22): +# https://help.aliyun.com/en/model-studio/model-pricing (China (Beijing) USD list prices) +# https://help.aliyun.com/en/model-studio/qwen3-5-flash (capabilities and limits) +# https://help.aliyun.com/en/model-studio/deep-thinking +# Toggle: enable_thinking true|false +# Budget: thinking_budget (integer reasoning tokens, max 81,920) name = "Qwen3.5 Flash" description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" family = "qwen" release_date = "2026-02-23" -last_updated = "2026-09-11" +last_updated = "2026-09-22" attachment = true reasoning = true structured_output = true @@ -18,16 +24,25 @@ type = "toggle" type = "budget_tokens" max = 81_920 +[interleaved] +field = "reasoning_content" + [cost] -input = 0.172 -output = 1.033 -reasoning = 1.033 +input = 0.029 +output = 0.287 +reasoning = 0.287 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.115 +output = 1.147 +reasoning = 1.147 [[cost.tiers]] tier = { type = "context", size = 256_000 } -input = 0.689 -output = 4.133 -reasoning = 4.133 +input = 0.172 +output = 1.72 +reasoning = 1.72 [limit] context = 1_000_000 diff --git a/providers/alibaba-cn/models/qwen3.5-plus.toml b/providers/alibaba-cn/models/qwen3.5-plus.toml index 56105ed505b..429ae670c66 100644 --- a/providers/alibaba-cn/models/qwen3.5-plus.toml +++ b/providers/alibaba-cn/models/qwen3.5-plus.toml @@ -1,10 +1,17 @@ +# Sources (accessed 2026-09-22): +# https://help.aliyun.com/en/model-studio/model-pricing (China (Beijing) USD list prices) +# https://help.aliyun.com/en/model-studio/qwen3-5-plus (capabilities and limits) +# https://help.aliyun.com/en/model-studio/deep-thinking +# Toggle: enable_thinking true|false +# Budget: thinking_budget (integer reasoning tokens, max 81,920) name = "Qwen3.5 Plus" description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" family = "qwen" release_date = "2026-02-16" -last_updated = "2026-09-11" -attachment = false +last_updated = "2026-09-22" +attachment = true reasoning = true +structured_output = true temperature = true knowledge = "2025-04" tool_call = true @@ -17,16 +24,25 @@ type = "toggle" type = "budget_tokens" max = 81_920 +[interleaved] +field = "reasoning_content" + [cost] +input = 0.115 +output = 0.688 +reasoning = 0.688 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } input = 0.287 -output = 1.722 -reasoning = 1.722 +output = 1.72 +reasoning = 1.72 [[cost.tiers]] tier = { type = "context", size = 256_000 } -input = 1.148 -output = 6.888 -reasoning = 6.888 +input = 0.573 +output = 3.44 +reasoning = 3.44 [limit] context = 1_000_000 From ec4b93ddb6b1f6b9607c099259db651fd62c4a6f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 02:35:03 +0000 Subject: [PATCH 263/392] chore(sync): update OpenRouter model catalog (#7693) Co-authored-by: opencode-agent[bot] --- .../xiaomi/mimo-v2.6-pro-ultraspeed.toml | 18 +----------------- providers/openrouter/models/z-ai/glm-5.3.toml | 6 +++--- .../~deepseek/deepseek-v4-flash-latest.toml | 2 +- .../openrouter/models/~z-ai/glm-latest.toml | 6 +++--- 4 files changed, 8 insertions(+), 24 deletions(-) diff --git a/providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml index 40cfae78edb..e7a245618f3 100644 --- a/providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml +++ b/providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -1,16 +1,8 @@ # Toggle: reasoning.enabled = true|false # https://openrouter.ai/docs/guides/best-practices/reasoning-tokens -name = "MiMo-V2.6-Pro-UltraSpeed" +base_model = "xiaomi/mimo-v2.6-pro-ultraspeed" description = "MiMo pro model for strong multimodal reasoning and agent execution" -family = "mimo" -release_date = "2026-09-21" -last_updated = "2026-09-21" -attachment = true -reasoning = true -temperature = true -tool_call = true structured_output = true -open_weights = false [[reasoning_options]] type = "toggle" @@ -19,11 +11,3 @@ type = "toggle" input = 4.35 output = 8.7 cache_read = 0.036 - -[limit] -context = 1_048_576 -output = 131_072 - -[modalities] -input = ["text", "image", "video", "audio"] -output = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index b68509e1e67..358dbfe6dac 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.784 -output = 2.464 -cache_read = 0.1456 +input = 0.7 +output = 2.2 +cache_read = 0.13 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 81420955903..6c267e8f600 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -22,7 +22,7 @@ values = ["low", "high", "max"] [cost] input = 0.03 output = 0.8 -cache_read = 0.012 +cache_read = 0.008 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index afa82f5ee86..6bcd917eec4 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7826 -output = 2.4596 -cache_read = 0.14534 +input = 0.7 +output = 2.2 +cache_read = 0.13 [limit] context = 1_310_720 From f909dbf12263e28a5e28517967bf28f7d00b3912 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 02:35:08 +0000 Subject: [PATCH 264/392] chore(sync): update EmpirioLabs AI model catalog (#7694) Co-authored-by: opencode-agent[bot] --- .../models/mimo-v2-6-pro-ultraspeed.toml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 providers/empiriolabs/models/mimo-v2-6-pro-ultraspeed.toml diff --git a/providers/empiriolabs/models/mimo-v2-6-pro-ultraspeed.toml b/providers/empiriolabs/models/mimo-v2-6-pro-ultraspeed.toml new file mode 100644 index 00000000000..e69d20993d0 --- /dev/null +++ b/providers/empiriolabs/models/mimo-v2-6-pro-ultraspeed.toml @@ -0,0 +1,14 @@ +base_model = "xiaomi/mimo-v2.6-pro-ultraspeed" +name = "MiMo V2.6 Pro UltraSpeed" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 21.75 +output = 43.5 +cache_read = 21.75 + +[limit] +context = 1_000_000 From 9ef10005b290d2dbbd8edcf5eb0ba198e72cef50 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 02:35:48 +0000 Subject: [PATCH 265/392] chore(sync): update Kilo model catalog (#7695) Co-authored-by: opencode-agent[bot] --- .../xiaomi/mimo-v2.6-pro-ultraspeed.toml | 18 +----------------- 1 file changed, 1 insertion(+), 17 deletions(-) diff --git a/providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml index 08e541fac0d..ad9bc52de83 100644 --- a/providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml +++ b/providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -1,14 +1,6 @@ -name = "Xiaomi: MiMo-V2.6-Pro-UltraSpeed" +base_model = "xiaomi/mimo-v2.6-pro-ultraspeed" description = "MiMo-V2.6-Pro-UltraSpeed is the fast speed edition of Xiaomi's flagship foundation model, MiMo-V2.6-Pro. Built from the same 1T MiMo-V2.6-Pro checkpoint, it matches the original model in quality while delivering roughly 10x..." -family = "mimo" -release_date = "2026-09-21" -last_updated = "2026-09-21" -attachment = true -reasoning = true -temperature = true -tool_call = true structured_output = true -open_weights = false [[reasoning_options]] type = "effort" @@ -18,11 +10,3 @@ values = ["none", "high"] input = 4.35 output = 8.7 cache_read = 0.036 - -[limit] -context = 1_048_576 -output = 131_072 - -[modalities] -input = ["text", "image", "video", "audio"] -output = ["text"] From fd3dedfe12242774c50647a1fdfd928269e32292 Mon Sep 17 00:00:00 2001 From: Aiden Cline <63023139+rekram1-node@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:37:18 -0500 Subject: [PATCH 266/392] chore: use Muse Spark in workflows (#7696) Co-authored-by: opencode-agent --- .github/workflows/ci-fixer.yml | 2 +- .github/workflows/issue-fixer.yml | 2 +- .github/workflows/opencode.yml | 3 ++- .github/workflows/pr-reviewer.yml | 2 +- 4 files changed, 5 insertions(+), 4 deletions(-) diff --git a/.github/workflows/ci-fixer.yml b/.github/workflows/ci-fixer.yml index 9d1205a126c..846da3efff8 100644 --- a/.github/workflows/ci-fixer.yml +++ b/.github/workflows/ci-fixer.yml @@ -160,7 +160,7 @@ jobs: Failed log excerpt: EOF cat "$LOG_FILE" - } | opencode run --agent ci-fixer -m opencode/grok-4.5 | tee "$RESPONSE_FILE" + } | opencode run --agent ci-fixer -m opencode/muse-spark-1.3#xhigh | tee "$RESPONSE_FILE" - name: Check changed paths if: steps.budget.outputs.run == 'true' && steps.budget-cache.outputs.cache-hit != 'true' diff --git a/.github/workflows/issue-fixer.yml b/.github/workflows/issue-fixer.yml index 17a2ce9a544..e4b26e201d8 100644 --- a/.github/workflows/issue-fixer.yml +++ b/.github/workflows/issue-fixer.yml @@ -60,7 +60,7 @@ jobs: + "If it is a feature request, a request to track a new kind of information, a question, or any miscellaneous non-catalog-data request, do not edit files. Respond briefly that it needs maintainer review and no automated fix was opened." ' "$ISSUE_FILE" > "$PROMPT_FILE" - opencode run --agent issue-fixer -m opencode/grok-4.5 --format json < "$PROMPT_FILE" | tee "$EVENTS_FILE" + opencode run --agent issue-fixer -m opencode/muse-spark-1.3#xhigh --format json < "$PROMPT_FILE" | tee "$EVENTS_FILE" if ! jq -ers 'map(select(.type == "text") | .part.text) | last | select(length > 0)' "$EVENTS_FILE" > "$RESPONSE_FILE"; then echo "Issue fixer did not produce a final response." >&2 diff --git a/.github/workflows/opencode.yml b/.github/workflows/opencode.yml index 24c819769a9..2e06c6beccb 100644 --- a/.github/workflows/opencode.yml +++ b/.github/workflows/opencode.yml @@ -27,4 +27,5 @@ jobs: env: OPENCODE_API_KEY: ${{ secrets.OPENCODE_API_KEY }} with: - model: opencode/grok-4.5 + model: opencode/muse-spark-1.3 + variant: xhigh diff --git a/.github/workflows/pr-reviewer.yml b/.github/workflows/pr-reviewer.yml index 48a5c853311..45336b5fdd6 100644 --- a/.github/workflows/pr-reviewer.yml +++ b/.github/workflows/pr-reviewer.yml @@ -78,7 +78,7 @@ jobs: export PR_REVIEW_READY_FILE rm -f "$PR_REVIEW_READY_FILE" - opencode run --agent pr-reviewer -m opencode/grok-4.5 --format json <<'EOF' | tee "$EVENTS_FILE" + opencode run --agent pr-reviewer -m opencode/muse-spark-1.3#xhigh --format json <<'EOF' | tee "$EVENTS_FILE" Review this pull request using the trusted reviewer instructions. Start with `.pr-review/pull-request.json`, `.pr-review/diff.patch`, `AGENTS.md`, and the contributing guidance in `README.md`. Read `sync.md`, the reasoning-options audit guide, schema code, and nearby base-revision files when relevant to the changed files. Use only the read, glob, grep, and mark-pr-ready tools. Return only the final review comment in the agent's required output format. Never include progress narration or passed-check summaries. EOF From a38ec312a02a31f9b111bea47856ad8a994d4b25 Mon Sep 17 00:00:00 2001 From: thatdevguy <46411187+chrissalomon@users.noreply.github.com> Date: Mon, 21 Sep 2026 22:37:44 -0400 Subject: [PATCH 267/392] Add Tempr's Google Gemini and Gemma models (#7674) * Add Tempr's Google Gemini models 16 models Google currently serves that Tempr's Gateway can reach: 14 chat models and the 2 embedding models, base_model-ing the lab entries with Google's own pricing. gemini-embedding-2 is text-only on Tempr's /v1/embeddings, so its entry says so. Co-Authored-By: Claude Opus 5 * tempr: drop the Gemini 2.5 models Google no longer serves to new users gemini-2.5-flash, gemini-2.5-flash-lite and gemini-2.5-pro are still in Google's model list, but a call answers 404 "This model ... is no longer available to new users" (checked 2026-09-21), so a new Tempr customer can't reach them. Their successors (gemini-3.6-flash, gemini-3.5-flash-lite, gemini-3.1-pro-preview) are already listed. Co-Authored-By: Claude Opus 5 * tempr: add the Gemma 4 models (gemma-4-31b-it, gemma-4-26b-a4b-it) The two Gemma 4 models the Gemini API serves, on the same Google key as the Gemini entries: base_model points at models/google/gemma-4-*, no [cost] (same as providers/google), and the thinking toggle Tempr's GET /v1/models reports. Tempr switches it with thinkingLevel "minimal"/"high", the only two levels Gemma 4 takes (live 2026-09-21). Co-Authored-By: Claude Opus 5 --------- Co-authored-by: Claude Opus 5 --- .../models/google/gemini-3-flash-preview.toml | 15 ++++++++++++++ .../models/google/gemini-3.1-flash-lite.toml | 15 ++++++++++++++ .../gemini-3.1-pro-preview-customtools.toml | 20 +++++++++++++++++++ .../models/google/gemini-3.1-pro-preview.toml | 20 +++++++++++++++++++ .../models/google/gemini-3.5-flash-lite.toml | 14 +++++++++++++ .../tempr/models/google/gemini-3.5-flash.toml | 15 ++++++++++++++ .../tempr/models/google/gemini-3.6-flash.toml | 15 ++++++++++++++ .../tempr/models/google/gemini-3.7-flash.toml | 15 ++++++++++++++ .../tempr/models/google/gemini-3.8-flash.toml | 15 ++++++++++++++ .../models/google/gemini-embedding-001.toml | 5 +++++ .../models/google/gemini-embedding-2.toml | 15 ++++++++++++++ .../models/google/gemini-flash-latest.toml | 15 ++++++++++++++ .../google/gemini-flash-lite-latest.toml | 14 +++++++++++++ .../models/google/gemma-4-26b-a4b-it.toml | 9 +++++++++ .../tempr/models/google/gemma-4-31b-it.toml | 9 +++++++++ 15 files changed, 211 insertions(+) create mode 100644 providers/tempr/models/google/gemini-3-flash-preview.toml create mode 100644 providers/tempr/models/google/gemini-3.1-flash-lite.toml create mode 100644 providers/tempr/models/google/gemini-3.1-pro-preview-customtools.toml create mode 100644 providers/tempr/models/google/gemini-3.1-pro-preview.toml create mode 100644 providers/tempr/models/google/gemini-3.5-flash-lite.toml create mode 100644 providers/tempr/models/google/gemini-3.5-flash.toml create mode 100644 providers/tempr/models/google/gemini-3.6-flash.toml create mode 100644 providers/tempr/models/google/gemini-3.7-flash.toml create mode 100644 providers/tempr/models/google/gemini-3.8-flash.toml create mode 100644 providers/tempr/models/google/gemini-embedding-001.toml create mode 100644 providers/tempr/models/google/gemini-embedding-2.toml create mode 100644 providers/tempr/models/google/gemini-flash-latest.toml create mode 100644 providers/tempr/models/google/gemini-flash-lite-latest.toml create mode 100644 providers/tempr/models/google/gemma-4-26b-a4b-it.toml create mode 100644 providers/tempr/models/google/gemma-4-31b-it.toml diff --git a/providers/tempr/models/google/gemini-3-flash-preview.toml b/providers/tempr/models/google/gemini-3-flash-preview.toml new file mode 100644 index 00000000000..af36c2e66df --- /dev/null +++ b/providers/tempr/models/google/gemini-3-flash-preview.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-3-flash-preview" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 0.5 +output = 3 +cache_read = 0.05 +input_audio = 1 diff --git a/providers/tempr/models/google/gemini-3.1-flash-lite.toml b/providers/tempr/models/google/gemini-3.1-flash-lite.toml new file mode 100644 index 00000000000..56979e347f7 --- /dev/null +++ b/providers/tempr/models/google/gemini-3.1-flash-lite.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-3.1-flash-lite" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 0.25 +output = 1.5 +cache_read = 0.025 +input_audio = 0.5 diff --git a/providers/tempr/models/google/gemini-3.1-pro-preview-customtools.toml b/providers/tempr/models/google/gemini-3.1-pro-preview-customtools.toml new file mode 100644 index 00000000000..cc288f0d5ca --- /dev/null +++ b/providers/tempr/models/google/gemini-3.1-pro-preview-customtools.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high +# /v1/responses: reasoning.effort = low|medium|high +base_model = "google/gemini-3.1-pro-preview-customtools" + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high"] }, +] + +[cost] +input = 2 +output = 12 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 200000 } +input = 4 +output = 18 +cache_read = 0.4 diff --git a/providers/tempr/models/google/gemini-3.1-pro-preview.toml b/providers/tempr/models/google/gemini-3.1-pro-preview.toml new file mode 100644 index 00000000000..e08f204c81c --- /dev/null +++ b/providers/tempr/models/google/gemini-3.1-pro-preview.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high +# /v1/responses: reasoning.effort = low|medium|high +base_model = "google/gemini-3.1-pro-preview" + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high"] }, +] + +[cost] +input = 2 +output = 12 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 200000 } +input = 4 +output = 18 +cache_read = 0.4 diff --git a/providers/tempr/models/google/gemini-3.5-flash-lite.toml b/providers/tempr/models/google/gemini-3.5-flash-lite.toml new file mode 100644 index 00000000000..6863ddd6378 --- /dev/null +++ b/providers/tempr/models/google/gemini-3.5-flash-lite.toml @@ -0,0 +1,14 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-3.5-flash-lite" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 0.3 +output = 2.5 +cache_read = 0.03 diff --git a/providers/tempr/models/google/gemini-3.5-flash.toml b/providers/tempr/models/google/gemini-3.5-flash.toml new file mode 100644 index 00000000000..bb1b51b4d5e --- /dev/null +++ b/providers/tempr/models/google/gemini-3.5-flash.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-3.5-flash" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 1.5 +output = 9 +cache_read = 0.15 +input_audio = 1.5 diff --git a/providers/tempr/models/google/gemini-3.6-flash.toml b/providers/tempr/models/google/gemini-3.6-flash.toml new file mode 100644 index 00000000000..aa1d7ce7069 --- /dev/null +++ b/providers/tempr/models/google/gemini-3.6-flash.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-3.6-flash" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 +input_audio = 0.75 diff --git a/providers/tempr/models/google/gemini-3.7-flash.toml b/providers/tempr/models/google/gemini-3.7-flash.toml new file mode 100644 index 00000000000..135e4ffe07a --- /dev/null +++ b/providers/tempr/models/google/gemini-3.7-flash.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high +# /v1/responses: reasoning.effort = low|medium|high +base_model = "google/gemini-3.7-flash" + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high"] }, +] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 +input_audio = 0.75 diff --git a/providers/tempr/models/google/gemini-3.8-flash.toml b/providers/tempr/models/google/gemini-3.8-flash.toml new file mode 100644 index 00000000000..44ee2192f92 --- /dev/null +++ b/providers/tempr/models/google/gemini-3.8-flash.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high +# /v1/responses: reasoning.effort = low|medium|high +base_model = "google/gemini-3.8-flash" + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high"] }, +] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 +input_audio = 0.75 diff --git a/providers/tempr/models/google/gemini-embedding-001.toml b/providers/tempr/models/google/gemini-embedding-001.toml new file mode 100644 index 00000000000..eb9ce08109f --- /dev/null +++ b/providers/tempr/models/google/gemini-embedding-001.toml @@ -0,0 +1,5 @@ +base_model = "google/gemini-embedding-001" + +[cost] +input = 0.15 +output = 0 diff --git a/providers/tempr/models/google/gemini-embedding-2.toml b/providers/tempr/models/google/gemini-embedding-2.toml new file mode 100644 index 00000000000..a9b158bc61f --- /dev/null +++ b/providers/tempr/models/google/gemini-embedding-2.toml @@ -0,0 +1,15 @@ +# Tempr serves this model on /v1/embeddings, which takes OpenAI's embeddings shape, and +# Gemini gets text input only there: image, audio, video and PDF input isn't reachable. +base_model = "google/gemini-embedding-2" +attachment = false + +[cost] +input = 0.2 +output = 0 + +[limit] +output = 1 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/tempr/models/google/gemini-flash-latest.toml b/providers/tempr/models/google/gemini-flash-latest.toml new file mode 100644 index 00000000000..2aa8e6ce3bd --- /dev/null +++ b/providers/tempr/models/google/gemini-flash-latest.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high +# /v1/responses: reasoning.effort = low|medium|high +base_model = "google/gemini-flash-latest" + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high"] }, +] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 +input_audio = 0.75 diff --git a/providers/tempr/models/google/gemini-flash-lite-latest.toml b/providers/tempr/models/google/gemini-flash-lite-latest.toml new file mode 100644 index 00000000000..0c9e9f3e9c8 --- /dev/null +++ b/providers/tempr/models/google/gemini-flash-lite-latest.toml @@ -0,0 +1,14 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-flash-lite-latest" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 0.3 +output = 2.5 +cache_read = 0.03 diff --git a/providers/tempr/models/google/gemma-4-26b-a4b-it.toml b/providers/tempr/models/google/gemma-4-26b-a4b-it.toml new file mode 100644 index 00000000000..89c5a30b22b --- /dev/null +++ b/providers/tempr/models/google/gemma-4-26b-a4b-it.toml @@ -0,0 +1,9 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# /v1/messages: thinking.type = enabled|disabled +# /v1/responses: reasoning.effort = "none" turns it off, any other level turns it on +base_model = "google/gemma-4-26b-a4b-it" + +reasoning_options = [ + { type = "toggle" }, +] diff --git a/providers/tempr/models/google/gemma-4-31b-it.toml b/providers/tempr/models/google/gemma-4-31b-it.toml new file mode 100644 index 00000000000..94549c23965 --- /dev/null +++ b/providers/tempr/models/google/gemma-4-31b-it.toml @@ -0,0 +1,9 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# /v1/messages: thinking.type = enabled|disabled +# /v1/responses: reasoning.effort = "none" turns it off, any other level turns it on +base_model = "google/gemma-4-31b-it" + +reasoning_options = [ + { type = "toggle" }, +] From d5bed5aa97cd45fd2221e1f86342f9f33e3775ca Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 03:31:21 +0000 Subject: [PATCH 268/392] chore(sync): update Kilo model catalog (#7702) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~z-ai/glm-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 51780ddf2fb..a1f67c33d90 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7 -output = 2.2 -cache_read = 0.13 +input = 0.7826 +output = 2.4596 +cache_read = 0.14534 [limit] context = 1_048_576 From 76aceeb8f429801d5cb278f81ab1e9ec56cc00a2 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 03:31:27 +0000 Subject: [PATCH 269/392] chore(sync): update OpenRouter model catalog (#7703) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/z-ai/glm-5.3.toml | 6 +++--- providers/openrouter/models/~z-ai/glm-latest.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index 358dbfe6dac..8b9af97ef26 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7 -output = 2.2 -cache_read = 0.13 +input = 0.84 +output = 2.64 +cache_read = 0.156 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 6bcd917eec4..afa82f5ee86 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7 -output = 2.2 -cache_read = 0.13 +input = 0.7826 +output = 2.4596 +cache_read = 0.14534 [limit] context = 1_310_720 From 409f66dc1588624d781ef566b44e34a10f83bc97 Mon Sep 17 00:00:00 2001 From: Aiden Cline <63023139+rekram1-node@users.noreply.github.com> Date: Mon, 21 Sep 2026 22:38:19 -0500 Subject: [PATCH 270/392] fix(workflows): pass OpenCode variant separately (#7705) Co-authored-by: opencode-agent --- .github/workflows/ci-fixer.yml | 2 +- .github/workflows/issue-fixer.yml | 2 +- .github/workflows/pr-reviewer.yml | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/ci-fixer.yml b/.github/workflows/ci-fixer.yml index 846da3efff8..5a94d2a3db2 100644 --- a/.github/workflows/ci-fixer.yml +++ b/.github/workflows/ci-fixer.yml @@ -160,7 +160,7 @@ jobs: Failed log excerpt: EOF cat "$LOG_FILE" - } | opencode run --agent ci-fixer -m opencode/muse-spark-1.3#xhigh | tee "$RESPONSE_FILE" + } | opencode run --agent ci-fixer -m opencode/muse-spark-1.3 --variant xhigh | tee "$RESPONSE_FILE" - name: Check changed paths if: steps.budget.outputs.run == 'true' && steps.budget-cache.outputs.cache-hit != 'true' diff --git a/.github/workflows/issue-fixer.yml b/.github/workflows/issue-fixer.yml index e4b26e201d8..4e1f144f164 100644 --- a/.github/workflows/issue-fixer.yml +++ b/.github/workflows/issue-fixer.yml @@ -60,7 +60,7 @@ jobs: + "If it is a feature request, a request to track a new kind of information, a question, or any miscellaneous non-catalog-data request, do not edit files. Respond briefly that it needs maintainer review and no automated fix was opened." ' "$ISSUE_FILE" > "$PROMPT_FILE" - opencode run --agent issue-fixer -m opencode/muse-spark-1.3#xhigh --format json < "$PROMPT_FILE" | tee "$EVENTS_FILE" + opencode run --agent issue-fixer -m opencode/muse-spark-1.3 --variant xhigh --format json < "$PROMPT_FILE" | tee "$EVENTS_FILE" if ! jq -ers 'map(select(.type == "text") | .part.text) | last | select(length > 0)' "$EVENTS_FILE" > "$RESPONSE_FILE"; then echo "Issue fixer did not produce a final response." >&2 diff --git a/.github/workflows/pr-reviewer.yml b/.github/workflows/pr-reviewer.yml index 45336b5fdd6..fe824e9d48d 100644 --- a/.github/workflows/pr-reviewer.yml +++ b/.github/workflows/pr-reviewer.yml @@ -78,7 +78,7 @@ jobs: export PR_REVIEW_READY_FILE rm -f "$PR_REVIEW_READY_FILE" - opencode run --agent pr-reviewer -m opencode/muse-spark-1.3#xhigh --format json <<'EOF' | tee "$EVENTS_FILE" + opencode run --agent pr-reviewer -m opencode/muse-spark-1.3 --variant xhigh --format json <<'EOF' | tee "$EVENTS_FILE" Review this pull request using the trusted reviewer instructions. Start with `.pr-review/pull-request.json`, `.pr-review/diff.patch`, `AGENTS.md`, and the contributing guidance in `README.md`. Read `sync.md`, the reasoning-options audit guide, schema code, and nearby base-revision files when relevant to the changed files. Use only the read, glob, grep, and mark-pr-ready tools. Return only the final review comment in the agent's required output format. Never include progress narration or passed-check summaries. EOF From f02d68065194a8fcb2e39b035ef7d4e5460d91bf Mon Sep 17 00:00:00 2001 From: Kastan Day Date: Mon, 21 Sep 2026 20:50:47 -0700 Subject: [PATCH 271/392] fix(sync): safely refresh Workers AI reasoning from search (#7453) Co-authored-by: opencode-agent --- .../sync/providers/cloudflare-workers-ai.ts | 296 +++++++++++++----- 1 file changed, 223 insertions(+), 73 deletions(-) diff --git a/packages/core/src/sync/providers/cloudflare-workers-ai.ts b/packages/core/src/sync/providers/cloudflare-workers-ai.ts index e68dd2b8e36..0ed0e391513 100644 --- a/packages/core/src/sync/providers/cloudflare-workers-ai.ts +++ b/packages/core/src/sync/providers/cloudflare-workers-ai.ts @@ -1,16 +1,25 @@ import { z } from "zod"; -import { readdirSync } from "node:fs"; +import { readFileSync, readdirSync } from "node:fs"; import path from "node:path"; -import type { ExistingModel, SyncedModel, SyncProvider } from "../index.js"; -import { - buildOpenRouterModel, - OpenRouterModel, - OpenRouterResponse, -} from "./openrouter.js"; +import type { ExistingModel, SyncedFullModel, SyncedModel, SyncProvider } from "../index.js"; +import { MissingReasoningOptionsError } from "../missing-reasoning-options.js"; +import { buildOpenRouterModel, OpenRouterModel } from "./openrouter.js"; const API_BASE = "https://api.cloudflare.com/client/v4/accounts"; -const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models"); +const TOGGLE_HEADER = "# Toggle: chat_template_kwargs.enable_thinking = true|false\n"; +// Models with a verified enable_thinking control. DeepSeek/Kimi use other wire paths; +// their authored comments are preserved below, never inferred from generic schemas. +// These are exact model IDs, not a default for future models or whole publishers. +const ENABLE_THINKING_MODELS = new Set([ + "@cf/google/gemma-4-26b-a4b-it", + "@cf/nvidia/nemotron-3-120b-a12b", + "@cf/qwen/qwen3.8-27b", + "@cf/zai-org/glm-4.7-flash", + "@cf/zai-org/glm-5.2", +]); +const ROOT_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", ".."); +const MODELS_DIR = path.join(ROOT_DIR, "models"); const metadataFilesByPublisher = new Map(); const METADATA_PUBLISHERS: Record = { "deepseek-ai": "deepseek", @@ -24,31 +33,50 @@ const METADATA_PUBLISHERS: Record = { "zai-org": "zhipuai", }; -const CloudflareOpenRouterResponse = z.object({ - result: z.union([OpenRouterResponse, z.array(OpenRouterModel)]).optional(), - result_info: z.object({ - page: z.number().optional(), - total_pages: z.number().optional(), - }).passthrough().optional(), -}).passthrough(); +const WorkersAiReasoning = OpenRouterModel.shape.reasoning.unwrap().partial({ mandatory: true }); +const WorkersAiModel = OpenRouterModel.extend({ reasoning: WorkersAiReasoning.optional() }); +type WorkersAiModel = z.infer; const CloudflareModel = z.object({ - id: z.string(), - name: z.string(), - created: z.number(), + id: z.string().trim().min(1), + name: z.string().trim().min(1), + created: z.number().int().nonnegative(), hugging_face_id: z.string().nullable().optional(), - context_length: z.number(), - max_output_length: z.number().nullable().optional(), - input_modalities: z.array(z.string()).optional(), - output_modalities: z.array(z.string()).optional(), - pricing: OpenRouterModel.shape.pricing, + context_length: z.number().int().positive(), + max_output_length: z.number().int().positive().nullable().optional(), + input_modalities: z.array(z.string().min(1)).min(1).optional(), + output_modalities: z.array(z.string().min(1)).min(1).optional(), + pricing: OpenRouterModel.shape.pricing.extend({ + prompt: z.string().refine(validPrice), + completion: z.string().refine(validPrice), + }), supported_features: z.array(z.string()).optional(), supported_sampling_parameters: z.array(z.string()).optional(), + // Validate reasoning separately so malformed controls do not discard the model. + reasoning: z.unknown().optional(), }).passthrough(); const CloudflareResponse = z.object({ - data: z.array(CloudflareModel), -}).passthrough(); + data: z.array(z.unknown()).optional(), + result: z.union([ + z.array(z.unknown()), + z.object({ data: z.array(z.unknown()) }), + ]).optional(), + success: z.literal(true).optional(), + result_info: z.object({ + total_pages: z.number().int().positive().optional(), + }).optional(), +}).refine((response) => response.data !== undefined || response.result !== undefined, { + message: "Cloudflare Workers AI response did not include model data", +}); + +function modelRows(response: z.infer) { + return response.data ?? (Array.isArray(response.result) ? response.result : response.result!.data); +} + +function validPrice(value: string) { + return value.trim() !== "" && Number.isFinite(Number(value)) && Number(value) >= 0; +} type CloudflareModel = z.infer; @@ -56,6 +84,8 @@ export const cloudflareWorkersAi = { id: "cloudflare-workers-ai", name: "Cloudflare Workers AI", modelsDir: "providers/cloudflare-workers-ai/models", + deleteMissing: false, + authoritativeHeaders: true, async fetchModels() { const accountID = process.env.CLOUDFLARE_WORKERS_AI_SYNC_ACCOUNT_ID; const token = process.env.CLOUDFLARE_WORKERS_AI_SYNC_API_TOKEN; @@ -66,50 +96,104 @@ export const cloudflareWorkersAi = { } const first = await fetchPage(accountID, token, 1); - const models = parseCloudflareModels(first); - const pageInfo = CloudflareOpenRouterResponse.safeParse(first).success - ? CloudflareOpenRouterResponse.parse(first).result_info - : undefined; - - for (let page = 2; page <= (pageInfo?.total_pages ?? 1); page++) { - models.push(...parseCloudflareModels(await fetchPage(accountID, token, page))); + if (first === undefined) throw new Error("Cloudflare Workers AI search returned no usable models"); + const rows = modelRows(first); + for (let page = 2; page <= (first.result_info?.total_pages ?? 1); page++) { + const response = await fetchPage(accountID, token, page); + // Keep successful pages; deleteMissing: false retains models on failed pages. + if (response !== undefined) rows.push(...modelRows(response)); } - return { data: models }; + return { data: rows }; }, parseModels(raw) { - return parseCloudflareModels(raw); + const models = parseCloudflareModels(raw); + if (models.length === 0) throw new Error("Cloudflare Workers AI search returned no usable models"); + return models; }, translateModel(model, context) { - const normalized = normalizeModel(model); - const id = normalized.id.replace(/^workers-ai\//, ""); + const id = model.id; + const existing = context.existing(id); + if (existing === undefined && hasReasoning(model) && workersAiReasoningOptions(model) === undefined) { + throw new MissingReasoningOptionsError(id, "Workers AI Search does not specify concrete reasoning controls; manual authoring is needed"); + } + const translated = buildWorkersAiModel(model, existing); + // Only read paths already present in the sync runner's catalogue map. + const header = existing === undefined ? "" : modelHeader(path.resolve(ROOT_DIR, this.modelsDir, `${id}.toml`)); + if (header === undefined) return undefined; + // Remove only the exact generated line; preserve every other leading comment. + const preservedHeader = header.replace(TOGGLE_HEADER, ""); + // Respect curated model-specific wire instructions, including Qwen's split lines. + const hasToggleWire = /\b(?:thinking\.type|chat_template_kwargs\.thinking|enable_thinking)\b/.test(preservedHeader); + const toggleHeader = ENABLE_THINKING_MODELS.has(id) ? TOGGLE_HEADER : ""; + if (translated.reasoning_options?.some((option) => option.type === "toggle") && !hasToggleWire && !toggleHeader) { + if (existing === undefined) { + throw new MissingReasoningOptionsError(id, "Workers AI toggle wire path needs manual verification"); + } + console.warn(`Keeping catalogue reasoning for ${id}: Workers AI toggle wire path is unknown`); + // Keep the reasoning boundary small: other usable properties still update. + return { + id, + model: buildWorkersAiModel({ ...model, reasoning: undefined }, existing), + header, + }; + } return { id, - model: buildWorkersAiModel(normalized, context.existing(id)), + model: translated, + header: (translated.reasoning_options?.some((option) => option.type === "toggle") && !hasToggleWire ? toggleHeader : "") + + preservedHeader, }; }, -} satisfies SyncProvider; +} satisfies SyncProvider; + +function hasReasoning(model: WorkersAiModel) { + return model.supported_parameters.includes("reasoning") || model.supported_parameters.includes("include_reasoning") + || model.reasoning?.mandatory !== undefined || model.reasoning?.supported_efforts !== undefined + || model.reasoning?.supports_max_tokens === true; +} + +function modelHeader(file: string) { + try { + const lines = readFileSync(file, "utf8").split("\n"); + const firstKey = lines.findIndex((line) => line.trim() !== "" && !line.trim().startsWith("#")); + return lines.slice(0, firstKey === -1 ? lines.length : firstKey).join("\n") + "\n"; + } catch (error) { + console.warn(`Skipping Workers AI model with an unreadable catalogue header: ${file}`, error); + return undefined; + } +} export function buildWorkersAiModel( - model: z.infer, + model: WorkersAiModel, existing: ExistingModel | undefined, ): SyncedModel { + const reasoningOptions = workersAiReasoningOptions(model); + const reasoning = reasoningOptions !== undefined + ? true + : existing?.reasoning ?? hasReasoning(model); const source = { ...model, + reasoning: undefined, + supported_parameters: [ + ...model.supported_parameters.filter((parameter) => !["reasoning", "include_reasoning"].includes(parameter)), + ...(reasoning ? ["reasoning"] : []), + ], name: existing?.name ?? model.name, top_provider: { ...model.top_provider, max_completion_tokens: existing?.limit?.output ?? model.top_provider.max_completion_tokens, }, }; - const synced = { - ...buildOpenRouterModel( - source, - existing, - existing?.base_model ?? resolveCloudflareBaseModel(model), - ), - reasoning_options: existing?.reasoning_options, - }; + // The shared builder uses these options when source.reasoning is omitted. + const existingWithReasoningOptions = reasoningOptions === undefined + ? existing + : { ...existing, reasoning_options: reasoningOptions }; + const synced = buildOpenRouterModel( + source, + existingWithReasoningOptions, + existing?.base_model ?? resolveCloudflareBaseModel(model), + ); if ("base_model" in synced) return synced; return { ...synced, @@ -123,7 +207,50 @@ export function buildWorkersAiModel( }; } -export function resolveCloudflareBaseModel(model: z.infer) { +function workersAiReasoningOptions({ id, reasoning }: WorkersAiModel): SyncedFullModel["reasoning_options"] { + // A null gateway allowlist does not identify concrete model controls. + // Preserve the catalogue instead of expanding it to every effort in the schema. + if (reasoning === undefined || reasoning.supported_efforts === null) return undefined; + + const options: NonNullable = []; + const efforts = reasoning.supported_efforts; + if (efforts?.length === 0) return undefined; + if (efforts === undefined && reasoning.supports_max_tokens !== true) { + // GLM-4.7-Flash's verified ConfigAPI/Search shape is { mandatory: false, + // default_enabled: true }; its only control is enable_thinking. The generic + // input schema's low/medium/high enum is not a model capability. + // Creator: https://huggingface.co/zai-org/GLM-4.7-Flash/blob/main/chat_template.jinja + // Require the explicit off-capability flag; absence of efforts alone says nothing. + return id === "@cf/zai-org/glm-4.7-flash" && reasoning.mandatory === false + ? [{ type: "toggle" }] + : undefined; + } + // Without either an explicit mandatory flag or a named off setting, a partial + // effort list cannot tell us whether replacing the catalogue would lose a toggle. + if (reasoning.mandatory === undefined && !efforts?.includes("none")) return undefined; + + if (reasoning.mandatory === false && !efforts?.includes("none")) { + options.push({ type: "toggle" }); + } + + const values = reasoning.mandatory ? efforts?.filter((value) => value !== "none") : efforts; + // One mandatory effective effort offers no caller choice. Unlike missing or + // empty metadata, a concrete singleton establishes this explicitly. + const fixedEffort = reasoning.mandatory === true && values?.length === 1; + if (values?.length && !fixedEffort) { + options.push({ type: "effort", values: [...values] }); + } + + if (reasoning.supports_max_tokens === true) { + // Explicit reasoning.max_tokens support, never inferred from output limits. + options.push({ type: "budget_tokens" }); + } + + // Empty control metadata never clears authored controls; a known fixed effort can. + return options.length > 0 || fixedEffort ? options : undefined; +} + +export function resolveCloudflareBaseModel(model: WorkersAiModel) { const [, publisher] = model.id.replace(/^workers-ai\//, "").split("/"); if (publisher === undefined) return undefined; @@ -157,39 +284,61 @@ async function fetchPage(accountID: string, token: string, page: number) { url.searchParams.set("per_page", "1000"); url.searchParams.set("page", String(page)); - const response = await fetch(url, { - headers: { Authorization: `Bearer ${token}` }, - }); - if (!response.ok) { - throw new Error( - `Cloudflare Workers AI models request failed: ${response.status} ${response.statusText}${await responseDetails(response)}`, - ); + for (let attempt = 1; attempt <= 4; attempt++) { + try { + const response = await fetch(url, { + headers: { Authorization: `Bearer ${token}` }, + signal: AbortSignal.timeout(30_000), + }); + if (!response.ok) { + throw new Error(`${response.status} ${response.statusText}${await responseDetails(response)}`); + } + return CloudflareResponse.parse(await response.json()); + } catch (error) { + console.warn( + `Workers AI search page ${page}, attempt ${attempt}/4 failed: ${error instanceof Error ? error.message : String(error)}`, + ); + if (attempt < 4) await Bun.sleep(30_000); + } } - return response.json(); } -function parseCloudflareModels(raw: unknown): CloudflareModel[] { - const cloudflare = CloudflareResponse.safeParse(raw); - if (cloudflare.success) return cloudflare.data.data; - - const direct = OpenRouterResponse.safeParse(raw); - if (direct.success) return direct.data.data.map((model) => CloudflareModel.parse(model)); - - const wrapped = CloudflareOpenRouterResponse.parse(raw); - if (wrapped.result === undefined) { - throw new Error("Cloudflare Workers AI response did not include model data"); - } - const models = Array.isArray(wrapped.result) ? wrapped.result : wrapped.result.data; - return models.map((model) => CloudflareModel.parse(model)); +function parseCloudflareModels(raw: unknown): WorkersAiModel[] { + const response = CloudflareResponse.parse(raw); + return modelRows(response).flatMap((row) => { + const parsed = CloudflareModel.safeParse(row); + if (!parsed.success) { + console.warn(`Skipping invalid Workers AI model: ${parsed.error.message}`); + return []; + } + try { + return [normalizeModel(parsed.data)]; + } catch (error) { + console.warn(`Skipping invalid Workers AI model ${parsed.data.id}: ${error instanceof Error ? error.message : String(error)}`); + return []; + } + }); } function normalizeModel(model: CloudflareModel) { - if ("architecture" in model && "top_provider" in model && "supported_parameters" in model) { - return OpenRouterModel.parse(model); + const parsed = WorkersAiReasoning.safeParse(model.reasoning); + const reasoning = parsed.success ? parsed.data : undefined; + if (!parsed.success && model.reasoning != null) { + console.warn(`Ignoring invalid Workers AI reasoning for ${model.id}: ${parsed.error.message}`); + } + const id = model.id.replace(/^workers-ai\//, ""); + const normalizedID = id.startsWith("@cf/") ? id : `@cf/${id}`; + if ("architecture" in model || "top_provider" in model || "supported_parameters" in model) { + const normalized = WorkersAiModel.parse({ ...model, id: normalizedID, reasoning }); + z.number().int().positive().nullable().parse(normalized.top_provider.max_completion_tokens); + if (normalized.architecture.input_modalities.length === 0 || normalized.architecture.output_modalities.length === 0) { + throw new Error("Model modalities must not be empty"); + } + return normalized; } - return OpenRouterModel.parse({ - id: model.id.startsWith("@cf/") ? model.id : `@cf/${model.id.replace(/^@cf\//, "")}`, + return WorkersAiModel.parse({ + id: normalizedID, name: model.name, created: model.created, hugging_face_id: model.hugging_face_id ?? null, @@ -208,6 +357,7 @@ function normalizeModel(model: CloudflareModel) { ...model.supported_sampling_parameters ?? [], ...model.supported_features ?? [], ], + reasoning, }); } From a483df9c693998f16f3042fcbdd66a9e43bbe840 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:09:23 -0500 Subject: [PATCH 272/392] chore(sync): update Pioneer model catalog (#7704) Co-authored-by: opencode-agent[bot] --- .../models/fastino/gliguard-LLMGuardrails-300M.toml | 11 +++++++++-- providers/pioneer/models/fastino/gliner2-base-v1.toml | 11 +++++++++-- .../pioneer/models/fastino/gliner2-large-v1.toml | 11 +++++++++-- .../models/fastino/gliner2-multi-large-v1.toml | 11 +++++++++-- .../pioneer/models/fastino/gliner2-multi-v1.toml | 11 +++++++++-- .../fastino/gliner2-privacy-filter-PII-multi.toml | 11 +++++++++-- 6 files changed, 54 insertions(+), 12 deletions(-) diff --git a/providers/pioneer/models/fastino/gliguard-LLMGuardrails-300M.toml b/providers/pioneer/models/fastino/gliguard-LLMGuardrails-300M.toml index de754b40e20..f455ac6d116 100644 --- a/providers/pioneer/models/fastino/gliguard-LLMGuardrails-300M.toml +++ b/providers/pioneer/models/fastino/gliguard-LLMGuardrails-300M.toml @@ -3,11 +3,18 @@ description = "Tool-capable chat model for instruction following and agentic app release_date = "2026-04-30" last_updated = "2026-04-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2-base-v1.toml b/providers/pioneer/models/fastino/gliner2-base-v1.toml index ef23d31a6ad..060cd035d56 100644 --- a/providers/pioneer/models/fastino/gliner2-base-v1.toml +++ b/providers/pioneer/models/fastino/gliner2-base-v1.toml @@ -3,11 +3,18 @@ description = "Tool-capable chat model for instruction following and agentic app release_date = "2025-06-30" last_updated = "2025-06-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2-large-v1.toml b/providers/pioneer/models/fastino/gliner2-large-v1.toml index 086e826101d..93f13f5d9c0 100644 --- a/providers/pioneer/models/fastino/gliner2-large-v1.toml +++ b/providers/pioneer/models/fastino/gliner2-large-v1.toml @@ -3,11 +3,18 @@ description = "Flagship model for demanding analysis, coding, and production age release_date = "2025-06-30" last_updated = "2025-06-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2-multi-large-v1.toml b/providers/pioneer/models/fastino/gliner2-multi-large-v1.toml index 74367df5b6c..5e3d3050671 100644 --- a/providers/pioneer/models/fastino/gliner2-multi-large-v1.toml +++ b/providers/pioneer/models/fastino/gliner2-multi-large-v1.toml @@ -3,11 +3,18 @@ description = "Flagship model for demanding analysis, coding, and production age release_date = "2025-11-30" last_updated = "2025-11-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2-multi-v1.toml b/providers/pioneer/models/fastino/gliner2-multi-v1.toml index 70b2d9553eb..76f6cdfd84a 100644 --- a/providers/pioneer/models/fastino/gliner2-multi-v1.toml +++ b/providers/pioneer/models/fastino/gliner2-multi-v1.toml @@ -3,11 +3,18 @@ description = "Tool-capable chat model for instruction following and agentic app release_date = "2025-11-30" last_updated = "2025-11-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2-privacy-filter-PII-multi.toml b/providers/pioneer/models/fastino/gliner2-privacy-filter-PII-multi.toml index 707d407fa98..af0a2ddc10a 100644 --- a/providers/pioneer/models/fastino/gliner2-privacy-filter-PII-multi.toml +++ b/providers/pioneer/models/fastino/gliner2-privacy-filter-PII-multi.toml @@ -3,11 +3,18 @@ description = "Tool-capable chat model for instruction following and agentic app release_date = "2026-04-30" last_updated = "2026-04-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] From f3fe9e21d073c676dc712961bb3c8470d0c42739 Mon Sep 17 00:00:00 2001 From: suxcode <1729303158@qq.com> Date: Tue, 22 Sep 2026 12:09:36 +0800 Subject: [PATCH 273/392] =?UTF-8?q?feat(xiaomi)=20:=20=E5=A2=9E=E5=8A=A0?= =?UTF-8?q?=20MiMo=20V2.6=20=E5=8E=9F=E7=94=9F=E4=B8=8E=E8=AE=A2=E9=98=85?= =?UTF-8?q?=E6=A8=A1=E5=9E=8B=20(#7700)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit edit by gpt-5 --- .../models/mimo-v2.6-flash.toml | 12 +++++++ .../models/mimo-v2.6-pro.toml | 12 +++++++ .../models/mimo-v2.6-flash.toml | 12 +++++++ .../models/mimo-v2.6-pro.toml | 12 +++++++ .../models/mimo-v2.6-flash.toml | 12 +++++++ .../models/mimo-v2.6-pro.toml | 12 +++++++ providers/xiaomi/models/mimo-v2.6-flash.toml | 32 +++++++++++++++++++ .../models/mimo-v2.6-pro-ultraspeed.toml | 31 ++++++++++++++++++ providers/xiaomi/models/mimo-v2.6-pro.toml | 32 +++++++++++++++++++ 9 files changed, 167 insertions(+) create mode 100644 providers/xiaomi-token-plan-ams/models/mimo-v2.6-flash.toml create mode 100644 providers/xiaomi-token-plan-ams/models/mimo-v2.6-pro.toml create mode 100644 providers/xiaomi-token-plan-cn/models/mimo-v2.6-flash.toml create mode 100644 providers/xiaomi-token-plan-cn/models/mimo-v2.6-pro.toml create mode 100644 providers/xiaomi-token-plan-sgp/models/mimo-v2.6-flash.toml create mode 100644 providers/xiaomi-token-plan-sgp/models/mimo-v2.6-pro.toml create mode 100644 providers/xiaomi/models/mimo-v2.6-flash.toml create mode 100644 providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml create mode 100644 providers/xiaomi/models/mimo-v2.6-pro.toml diff --git a/providers/xiaomi-token-plan-ams/models/mimo-v2.6-flash.toml b/providers/xiaomi-token-plan-ams/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..4129697bc51 --- /dev/null +++ b/providers/xiaomi-token-plan-ams/models/mimo-v2.6-flash.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/en-US/tokenplan/Token%20Plan/quick-access +base_model = "xiaomi/mimo-v2.6-flash" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi-token-plan-ams/models/mimo-v2.6-pro.toml b/providers/xiaomi-token-plan-ams/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..62ef14092f6 --- /dev/null +++ b/providers/xiaomi-token-plan-ams/models/mimo-v2.6-pro.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/en-US/tokenplan/Token%20Plan/quick-access +base_model = "xiaomi/mimo-v2.6-pro" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi-token-plan-cn/models/mimo-v2.6-flash.toml b/providers/xiaomi-token-plan-cn/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..091c178cc51 --- /dev/null +++ b/providers/xiaomi-token-plan-cn/models/mimo-v2.6-flash.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/zh-CN/api/chat/openai-api +base_model = "xiaomi/mimo-v2.6-flash" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi-token-plan-cn/models/mimo-v2.6-pro.toml b/providers/xiaomi-token-plan-cn/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..27c7d777207 --- /dev/null +++ b/providers/xiaomi-token-plan-cn/models/mimo-v2.6-pro.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/zh-CN/api/chat/openai-api +base_model = "xiaomi/mimo-v2.6-pro" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-flash.toml b/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..4129697bc51 --- /dev/null +++ b/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-flash.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/en-US/tokenplan/Token%20Plan/quick-access +base_model = "xiaomi/mimo-v2.6-flash" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-pro.toml b/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..62ef14092f6 --- /dev/null +++ b/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-pro.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/en-US/tokenplan/Token%20Plan/quick-access +base_model = "xiaomi/mimo-v2.6-pro" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi/models/mimo-v2.6-flash.toml b/providers/xiaomi/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..338e97b8d17 --- /dev/null +++ b/providers/xiaomi/models/mimo-v2.6-flash.toml @@ -0,0 +1,32 @@ +# Toggle: thinking.type = enabled|disabled +# Sources: https://mimo.mi.com/models/zh-CN/mimo-v2.6-flash and https://mimo.mi.com/docs/zh-CN/api/chat/openai-api +name = "MiMo-V2.6-Flash" +description = "MiMo Flash model for multimodal coding agents and long-context automation" +family = "mimo" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +knowledge = "2026-09-22" +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "audio", "video", "pdf"] +output = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml b/providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..f1d91260627 --- /dev/null +++ b/providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,31 @@ +# Toggle: thinking.type = enabled|disabled +# Sources: https://mimo.mi.com/models/zh-CN/mimo-v2.6-pro-ultraspeed and https://mimo.mi.com/docs/zh-CN/api/chat/openai-api +name = "MiMo-V2.6-Pro-UltraSpeed" +description = "MiMo pro model for strong multimodal reasoning and agent execution" +family = "mimo" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 4.35 +output = 8.7 +cache_read = 0.036 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "video", "audio"] +output = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/xiaomi/models/mimo-v2.6-pro.toml b/providers/xiaomi/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..8debc33d06f --- /dev/null +++ b/providers/xiaomi/models/mimo-v2.6-pro.toml @@ -0,0 +1,32 @@ +# Toggle: thinking.type = enabled|disabled +# Sources: https://mimo.mi.com/models/zh-CN/mimo-v2.6-pro and https://mimo.mi.com/docs/zh-CN/api/chat/openai-api +name = "MiMo-V2.6-Pro" +description = "MiMo Pro model for multimodal coding agents and long-context automation" +family = "mimo" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +knowledge = "2026-09-22" +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "audio", "video", "pdf"] +output = ["text"] + +[interleaved] +field = "reasoning_content" From ee0566f601c9036ae67782a8f2e16f4b67a12123 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 04:30:22 +0000 Subject: [PATCH 274/392] chore(sync): update OpenRouter model catalog (#7706) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-pro-0813.toml | 6 ++-- .../models/deepseek/deepseek-v4.1-flash.toml | 6 ++-- .../models/nex-agi/nex-n2.5-mini.toml | 28 +++++++++++++++++++ .../models/nex-agi/nex-n2.5-pro.toml | 28 +++++++++++++++++++ .../models/~deepseek/deepseek-pro-latest.toml | 6 ++-- .../models/~z-ai/glm-flash-latest.toml | 4 +-- .../openrouter/models/~z-ai/glm-latest.toml | 8 +++--- 7 files changed, 71 insertions(+), 15 deletions(-) create mode 100644 providers/openrouter/models/nex-agi/nex-n2.5-mini.toml create mode 100644 providers/openrouter/models/nex-agi/nex-n2.5-pro.toml diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index b8d53b0809c..7c5828b16e2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.32 -output = 3.96 -cache_read = 0.044 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 854b27b6a72..16f058ed841 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.3 -output = 1.2 -cache_read = 0.006 +input = 0.15 +output = 0.6 +cache_read = 0.003 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/nex-agi/nex-n2.5-mini.toml b/providers/openrouter/models/nex-agi/nex-n2.5-mini.toml new file mode 100644 index 00000000000..3869a240dd3 --- /dev/null +++ b/providers/openrouter/models/nex-agi/nex-n2.5-mini.toml @@ -0,0 +1,28 @@ +name = "Nex-N2.5-Mini" +description = "Multimodal reasoning model for visual analysis, planning, and tool use" +family = "agi" +release_date = "2026-09-08" +last_updated = "2026-09-08" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = false +open_weights = true + +[[reasoning_options]] +type = "effort" +values = ["none", "medium", "high"] + +[cost] +input = 0.025 +output = 0.1 +cache_read = 0.0025 + +[limit] +context = 262_144 +output = 235_929 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/nex-agi/nex-n2.5-pro.toml b/providers/openrouter/models/nex-agi/nex-n2.5-pro.toml new file mode 100644 index 00000000000..b3b848a836e --- /dev/null +++ b/providers/openrouter/models/nex-agi/nex-n2.5-pro.toml @@ -0,0 +1,28 @@ +name = "Nex-N2.5-Pro" +description = "Multimodal reasoning model for visual analysis, planning, and tool use" +family = "agi" +release_date = "2026-09-08" +last_updated = "2026-09-08" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = false +open_weights = true + +[[reasoning_options]] +type = "effort" +values = ["none", "medium", "high"] + +[cost] +input = 0.075 +output = 0.25 +cache_read = 0.015 + +[limit] +context = 262_144 +output = 235_929 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 1db14b9b08b..f88779c2deb 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.558624 -output = 1.675872 -cache_read = 0.018621 +input = 0.638616 +output = 1.915848 +cache_read = 0.021287 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~z-ai/glm-flash-latest.toml b/providers/openrouter/models/~z-ai/glm-flash-latest.toml index ac623f94e52..8f90bbaaf93 100644 --- a/providers/openrouter/models/~z-ai/glm-flash-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-flash-latest.toml @@ -17,11 +17,11 @@ values = ["low", "high", "max"] [cost] input = 0.075 output = 0.25 -cache_read = 0.02 +cache_read = 0.015 [limit] context = 1_310_720 -output = 102_400 +output = 943_718 [modalities] input = ["text", "image", "video"] diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index afa82f5ee86..167f0470b75 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7826 -output = 2.4596 -cache_read = 0.14534 +input = 0.6545 +output = 2.057 +cache_read = 0.107525 [limit] context = 1_310_720 -output = 131_072 +output = 943_718 [modalities] input = ["text"] From 9c941701896ebe52f7a368c54a6df79471d97737 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 04:30:30 +0000 Subject: [PATCH 275/392] chore(sync): update NanoGPT model catalog (#7708) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index a62995b0beb..b4cce19afb3 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.44 -output = 1.32 -cache_read = 0.014 +input = 0.22 +output = 0.66 +cache_read = 0.007 [limit] context = 1_048_576 From 848a1b9ea6ba0fe4f568b2b38b4e694645a0230e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 04:30:38 +0000 Subject: [PATCH 276/392] chore(sync): update Kilo model catalog (#7709) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- providers/kilo/models/~z-ai/glm-flash-latest.toml | 4 ++-- providers/kilo/models/~z-ai/glm-latest.toml | 8 ++++---- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 6caf71dbf01..47b086887a0 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.558624 -output = 1.675872 -cache_read = 0.018621 +input = 0.638616 +output = 1.915848 +cache_read = 0.021287 [limit] context = 1_024_000 diff --git a/providers/kilo/models/~z-ai/glm-flash-latest.toml b/providers/kilo/models/~z-ai/glm-flash-latest.toml index 5035574d3d5..aee70deebb0 100644 --- a/providers/kilo/models/~z-ai/glm-flash-latest.toml +++ b/providers/kilo/models/~z-ai/glm-flash-latest.toml @@ -17,11 +17,11 @@ values = ["low", "high", "max"] [cost] input = 0.075 output = 0.25 -cache_read = 0.02 +cache_read = 0.015 [limit] context = 1_048_576 -output = 102_400 +output = 943_718 [modalities] input = ["text", "image", "video"] diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index a1f67c33d90..3a29b91f006 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.7826 -output = 2.4596 -cache_read = 0.14534 +input = 0.6545 +output = 2.057 +cache_read = 0.107525 [limit] context = 1_048_576 -output = 131_072 +output = 943_718 [modalities] input = ["text"] From 2cea6220cdadc68b8c680077ecd3fe99ab42a14a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 05:27:31 +0000 Subject: [PATCH 277/392] chore(sync): update Eden AI model catalog (#7713) Co-authored-by: opencode-agent[bot] --- .../models/amazon/google.gemma-3-12b-it.toml | 3 +-- .../models/amazon/google.gemma-3-12b-it@us.toml | 3 +-- .../models/amazon/google.gemma-3-27b-it.toml | 3 +-- .../models/amazon/google.gemma-3-27b-it@us.toml | 3 +-- .../models/amazon/google.gemma-3-4b-it.toml | 1 - .../models/amazon/google.gemma-3-4b-it@us.toml | 1 - .../models/amazon/moonshot.kimi-k2-thinking.toml | 5 ++--- .../amazon/openai.gpt-oss-safeguard-20b.toml | 2 -- .../amazon/openai.gpt-oss-safeguard-20b@us.toml | 2 -- .../edenai/models/amazon/zai.glm-4.7-flash.toml | 5 ++++- .../models/amazon/zai.glm-4.7-flash@us.toml | 5 ++++- .../edenai/models/deepinfra/tencent/Hy3.toml | 6 +++--- providers/edenai/models/xai/grok-4.7.toml | 16 ++++++++++++++++ 13 files changed, 33 insertions(+), 22 deletions(-) create mode 100644 providers/edenai/models/xai/grok-4.7.toml diff --git a/providers/edenai/models/amazon/google.gemma-3-12b-it.toml b/providers/edenai/models/amazon/google.gemma-3-12b-it.toml index 76ea0c2327a..3889a7e378e 100644 --- a/providers/edenai/models/amazon/google.gemma-3-12b-it.toml +++ b/providers/edenai/models/amazon/google.gemma-3-12b-it.toml @@ -1,7 +1,6 @@ base_model = "google/gemma-3-12b-it" name = "Gemma 3 12B IT (Amazon Bedrock)" -tool_call = false -structured_output = false +structured_output = true [cost] input = 0.09 diff --git a/providers/edenai/models/amazon/google.gemma-3-12b-it@us.toml b/providers/edenai/models/amazon/google.gemma-3-12b-it@us.toml index 91ebef21c30..332de3d4ed7 100644 --- a/providers/edenai/models/amazon/google.gemma-3-12b-it@us.toml +++ b/providers/edenai/models/amazon/google.gemma-3-12b-it@us.toml @@ -1,7 +1,6 @@ base_model = "google/gemma-3-12b-it" name = "Gemma 3 12B IT (Amazon Bedrock, US)" -tool_call = false -structured_output = false +structured_output = true [cost] input = 0.09 diff --git a/providers/edenai/models/amazon/google.gemma-3-27b-it.toml b/providers/edenai/models/amazon/google.gemma-3-27b-it.toml index b99b2b38ac2..938c2319714 100644 --- a/providers/edenai/models/amazon/google.gemma-3-27b-it.toml +++ b/providers/edenai/models/amazon/google.gemma-3-27b-it.toml @@ -1,7 +1,6 @@ base_model = "google/gemma-3-27b-it" name = "Gemma 3 27B IT (Amazon Bedrock)" -tool_call = false -structured_output = false +structured_output = true [cost] input = 0.23 diff --git a/providers/edenai/models/amazon/google.gemma-3-27b-it@us.toml b/providers/edenai/models/amazon/google.gemma-3-27b-it@us.toml index 4c9f67f300e..a1637470411 100644 --- a/providers/edenai/models/amazon/google.gemma-3-27b-it@us.toml +++ b/providers/edenai/models/amazon/google.gemma-3-27b-it@us.toml @@ -1,7 +1,6 @@ base_model = "google/gemma-3-27b-it" name = "Gemma 3 27B IT (Amazon Bedrock, US)" -tool_call = false -structured_output = false +structured_output = true [cost] input = 0.23 diff --git a/providers/edenai/models/amazon/google.gemma-3-4b-it.toml b/providers/edenai/models/amazon/google.gemma-3-4b-it.toml index f502a9a96cd..536c47c8d73 100644 --- a/providers/edenai/models/amazon/google.gemma-3-4b-it.toml +++ b/providers/edenai/models/amazon/google.gemma-3-4b-it.toml @@ -1,6 +1,5 @@ base_model = "google/gemma-3-4b-it" name = "Gemma 3 4B IT (Amazon Bedrock)" -tool_call = false structured_output = false [cost] diff --git a/providers/edenai/models/amazon/google.gemma-3-4b-it@us.toml b/providers/edenai/models/amazon/google.gemma-3-4b-it@us.toml index 6d658417b4f..2d4e482e0c4 100644 --- a/providers/edenai/models/amazon/google.gemma-3-4b-it@us.toml +++ b/providers/edenai/models/amazon/google.gemma-3-4b-it@us.toml @@ -1,6 +1,5 @@ base_model = "google/gemma-3-4b-it" name = "Gemma 3 4B IT (Amazon Bedrock, US)" -tool_call = false structured_output = false [cost] diff --git a/providers/edenai/models/amazon/moonshot.kimi-k2-thinking.toml b/providers/edenai/models/amazon/moonshot.kimi-k2-thinking.toml index 1fdc4e3a58c..7413aad8307 100644 --- a/providers/edenai/models/amazon/moonshot.kimi-k2-thinking.toml +++ b/providers/edenai/models/amazon/moonshot.kimi-k2-thinking.toml @@ -1,7 +1,6 @@ base_model = "moonshotai/kimi-k2-thinking" name = "Kimi K2 Thinking (Amazon Bedrock)" -tool_call = false -structured_output = false +structured_output = true reasoning_options = [] [cost] @@ -9,4 +8,4 @@ input = 0.6 output = 2.5 [limit] -context = 128_000 +context = 256_000 diff --git a/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b.toml b/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b.toml index 536758dfd0e..c159c3bb673 100644 --- a/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b.toml +++ b/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b.toml @@ -1,7 +1,5 @@ base_model = "openai/gpt-oss-safeguard-20b" name = "GPT OSS Safeguard 20B (Amazon Bedrock)" -tool_call = false -structured_output = false [[reasoning_options]] type = "effort" diff --git a/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b@us.toml b/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b@us.toml index 7992978de7d..98893d21912 100644 --- a/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b@us.toml +++ b/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b@us.toml @@ -1,7 +1,5 @@ base_model = "openai/gpt-oss-safeguard-20b" name = "GPT OSS Safeguard 20B (Amazon Bedrock, US)" -tool_call = false -structured_output = false [[reasoning_options]] type = "effort" diff --git a/providers/edenai/models/amazon/zai.glm-4.7-flash.toml b/providers/edenai/models/amazon/zai.glm-4.7-flash.toml index 2a31f0a6612..bf215607f73 100644 --- a/providers/edenai/models/amazon/zai.glm-4.7-flash.toml +++ b/providers/edenai/models/amazon/zai.glm-4.7-flash.toml @@ -1,8 +1,11 @@ base_model = "zhipuai/glm-4.7-flash" name = "GLM-4.7-Flash (Amazon Bedrock)" -structured_output = false +structured_output = true reasoning_options = [] [cost] input = 0.07 output = 0.4 + +[limit] +context = 203_000 diff --git a/providers/edenai/models/amazon/zai.glm-4.7-flash@us.toml b/providers/edenai/models/amazon/zai.glm-4.7-flash@us.toml index ec6e4df6db1..1aa3c944b18 100644 --- a/providers/edenai/models/amazon/zai.glm-4.7-flash@us.toml +++ b/providers/edenai/models/amazon/zai.glm-4.7-flash@us.toml @@ -1,8 +1,11 @@ base_model = "zhipuai/glm-4.7-flash" name = "GLM-4.7-Flash (Amazon Bedrock, US)" -structured_output = false +structured_output = true reasoning_options = [] [cost] input = 0.07 output = 0.4 + +[limit] +context = 203_000 diff --git a/providers/edenai/models/deepinfra/tencent/Hy3.toml b/providers/edenai/models/deepinfra/tencent/Hy3.toml index c08482653b4..6cb87c36ee1 100644 --- a/providers/edenai/models/deepinfra/tencent/Hy3.toml +++ b/providers/edenai/models/deepinfra/tencent/Hy3.toml @@ -8,9 +8,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.14 -output = 0.58 -cache_read = 0.035 +input = 0.13 +output = 0.53 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/edenai/models/xai/grok-4.7.toml b/providers/edenai/models/xai/grok-4.7.toml new file mode 100644 index 00000000000..65cbf72670b --- /dev/null +++ b/providers/edenai/models/xai/grok-4.7.toml @@ -0,0 +1,16 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 3.2 +output = 9.6 +cache_read = 0.8 From f904d8c5907b90a50ad3774e99c64d29dd6c3afc Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 05:27:39 +0000 Subject: [PATCH 278/392] chore(sync): update Kilo model catalog (#7711) Co-authored-by: opencode-agent[bot] --- .../kilo/models/nex-agi/nex-n2.5-mini.toml | 28 +++++++++++++++++++ .../kilo/models/nex-agi/nex-n2.5-pro.toml | 28 +++++++++++++++++++ 2 files changed, 56 insertions(+) create mode 100644 providers/kilo/models/nex-agi/nex-n2.5-mini.toml create mode 100644 providers/kilo/models/nex-agi/nex-n2.5-pro.toml diff --git a/providers/kilo/models/nex-agi/nex-n2.5-mini.toml b/providers/kilo/models/nex-agi/nex-n2.5-mini.toml new file mode 100644 index 00000000000..04ca31fff1a --- /dev/null +++ b/providers/kilo/models/nex-agi/nex-n2.5-mini.toml @@ -0,0 +1,28 @@ +name = "Nex AGI: Nex-N2.5-Mini" +description = "Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file..." +family = "agi" +release_date = "2026-09-08" +last_updated = "2026-09-08" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "medium", "high"] + +[cost] +input = 0.025 +output = 0.1 +cache_read = 0.0025 + +[limit] +context = 262_144 +output = 235_929 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/kilo/models/nex-agi/nex-n2.5-pro.toml b/providers/kilo/models/nex-agi/nex-n2.5-pro.toml new file mode 100644 index 00000000000..e565d70e36f --- /dev/null +++ b/providers/kilo/models/nex-agi/nex-n2.5-pro.toml @@ -0,0 +1,28 @@ +name = "Nex AGI: Nex-N2.5-Pro" +description = "Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file..." +family = "agi" +release_date = "2026-09-08" +last_updated = "2026-09-08" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "medium", "high"] + +[cost] +input = 0.075 +output = 0.25 +cache_read = 0.015 + +[limit] +context = 262_144 +output = 235_929 + +[modalities] +input = ["text", "image"] +output = ["text"] From 3e8b207f61ac82bff1b9dd5c9575b1f533e8eebb Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 05:27:50 +0000 Subject: [PATCH 279/392] chore(sync): update OpenRouter model catalog (#7712) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/nex-agi/nex-n2.5-mini.toml | 2 +- providers/openrouter/models/nex-agi/nex-n2.5-pro.toml | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/nex-agi/nex-n2.5-mini.toml b/providers/openrouter/models/nex-agi/nex-n2.5-mini.toml index 3869a240dd3..539fc6a6686 100644 --- a/providers/openrouter/models/nex-agi/nex-n2.5-mini.toml +++ b/providers/openrouter/models/nex-agi/nex-n2.5-mini.toml @@ -7,7 +7,7 @@ attachment = true reasoning = true temperature = true tool_call = false -structured_output = false +structured_output = true open_weights = true [[reasoning_options]] diff --git a/providers/openrouter/models/nex-agi/nex-n2.5-pro.toml b/providers/openrouter/models/nex-agi/nex-n2.5-pro.toml index b3b848a836e..0cb11065de4 100644 --- a/providers/openrouter/models/nex-agi/nex-n2.5-pro.toml +++ b/providers/openrouter/models/nex-agi/nex-n2.5-pro.toml @@ -6,8 +6,8 @@ last_updated = "2026-09-08" attachment = true reasoning = true temperature = true -tool_call = false -structured_output = false +tool_call = true +structured_output = true open_weights = true [[reasoning_options]] From 80db4c2dcae857d9c3c4f489c01d86cee7313bcc Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 06:41:25 +0000 Subject: [PATCH 280/392] chore(sync): update NanoGPT model catalog (#7717) Co-authored-by: opencode-agent[bot] --- .../deepseek-v4-flash-vision-exp.toml | 6 +++--- .../models/xiaomi/mimo-v2.6-flash.toml | 19 +++++++++++++++++++ .../xiaomi/mimo-v2.6-pro-ultraspeed.toml | 16 ++++++++++++++++ .../nano-gpt/models/xiaomi/mimo-v2.6-pro.toml | 19 +++++++++++++++++++ 4 files changed, 57 insertions(+), 3 deletions(-) create mode 100644 providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml create mode 100644 providers/nano-gpt/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml create mode 100644 providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index b4cce19afb3..a62995b0beb 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.22 -output = 0.66 -cache_read = 0.007 +input = 0.44 +output = 1.32 +cache_read = 0.014 [limit] context = 1_048_576 diff --git a/providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml b/providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..bf3082ff91f --- /dev/null +++ b/providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,19 @@ +base_model = "xiaomi/mimo-v2.6-flash" +name = "MiMo V2.6 Flash" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 +cache_write = 0 + +[limit] +input = 1_048_576 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..8950d4913de --- /dev/null +++ b/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,16 @@ +base_model = "xiaomi/mimo-v2.6-pro-ultraspeed" +name = "MiMo V2.6 Pro UltraSpeed" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 4.35 +output = 8.7 +cache_read = 0.036 +cache_write = 0 + +[limit] +input = 1_048_576 diff --git a/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml b/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..ea8e65a904c --- /dev/null +++ b/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,19 @@ +base_model = "xiaomi/mimo-v2.6-pro" +name = "MiMo V2.6 Pro" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 +cache_write = 0 + +[limit] +input = 1_048_576 + +[modalities] +input = ["text", "image", "video", "audio"] From d179fe47750299d90db707f4d050e6f6581d5ecf Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 06:41:27 +0000 Subject: [PATCH 281/392] chore(sync): update OpenRouter model catalog (#7716) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- .../openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml | 1 + .../models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- 4 files changed, 8 insertions(+), 7 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..b8d53b0809c 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 1.32 +output = 3.96 +cache_read = 0.044 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 16f058ed841..854b27b6a72 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.15 -output = 0.6 -cache_read = 0.003 +input = 0.3 +output = 1.2 +cache_read = 0.006 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml b/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml index 14e222d894d..35f13bee686 100644 --- a/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml +++ b/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml @@ -8,3 +8,4 @@ output = 1.2 [limit] context = 262_144 +output = 235_929 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 6c267e8f600..8c5f6c380c1 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -21,7 +21,7 @@ values = ["low", "high", "max"] [cost] input = 0.03 -output = 0.8 +output = 1 cache_read = 0.008 [limit] From 39d81f59ca3c8bc150b6c54868e6d3a6d346a440 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 06:41:31 +0000 Subject: [PATCH 282/392] chore(sync): update Kilo model catalog (#7718) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/qwen/qwen3-next-80b-a3b-thinking.toml | 4 ++++ providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml | 2 +- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/providers/kilo/models/qwen/qwen3-next-80b-a3b-thinking.toml b/providers/kilo/models/qwen/qwen3-next-80b-a3b-thinking.toml index 7c76fb082af..5d67b671b7a 100644 --- a/providers/kilo/models/qwen/qwen3-next-80b-a3b-thinking.toml +++ b/providers/kilo/models/qwen/qwen3-next-80b-a3b-thinking.toml @@ -9,3 +9,7 @@ values = ["high"] [cost] input = 0.15 output = 1.2 + +[limit] +context = 262_144 +output = 235_929 diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 0412502ff5d..6200a01f4d0 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -16,7 +16,7 @@ values = ["none", "low", "high", "max"] [cost] input = 0.03 -output = 0.8 +output = 1 cache_read = 0.008 [limit] From a7a1850b68d87a73d7893775aa3749462c07acb3 Mon Sep 17 00:00:00 2001 From: Jack Date: Tue, 22 Sep 2026 15:21:58 +0800 Subject: [PATCH 283/392] chore(opencode): deprecate MiMo V2.5 Free --- providers/opencode/models/mimo-v2.5-free.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/providers/opencode/models/mimo-v2.5-free.toml b/providers/opencode/models/mimo-v2.5-free.toml index 3a1d9a269a6..d799e874914 100644 --- a/providers/opencode/models/mimo-v2.5-free.toml +++ b/providers/opencode/models/mimo-v2.5-free.toml @@ -10,6 +10,7 @@ temperature = true tool_call = true knowledge = "2024-12" open_weights = true +status = "deprecated" [interleaved] field = "reasoning_content" From c167bacf65d44d05a3c777459a7edc7f710c6c36 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 07:30:22 +0000 Subject: [PATCH 284/392] chore(sync): update Kilo model catalog (#7723) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-pro-0813.toml | 1 + .../models/kwaipilot/kat-coder-pro-v2.toml | 24 ------------------- .../models/~deepseek/deepseek-pro-latest.toml | 10 ++++---- 3 files changed, 6 insertions(+), 29 deletions(-) delete mode 100644 providers/kilo/models/kwaipilot/kat-coder-pro-v2.toml diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index 007b654034e..fa4ef3087d9 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -12,3 +12,4 @@ cache_read = 0.044 [limit] context = 1_048_576 +output = 943_718 diff --git a/providers/kilo/models/kwaipilot/kat-coder-pro-v2.toml b/providers/kilo/models/kwaipilot/kat-coder-pro-v2.toml deleted file mode 100644 index 8e2499749ae..00000000000 --- a/providers/kilo/models/kwaipilot/kat-coder-pro-v2.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Kwaipilot: KAT-Coder-Pro V2" -description = "Coding model for repository understanding, refactors, and agentic engineering tasks" -family = "kat-coder" -release_date = "2026-03-27" -last_updated = "2026-03-27" -attachment = false -reasoning = false -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.3 -output = 1.2 -cache_read = 0.06 - -[limit] -context = 262_144 -output = 144_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 47b086887a0..5dc58c6a638 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.638616 -output = 1.915848 -cache_read = 0.021287 +input = 0.624 +output = 2.88 +cache_read = 0.088 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 943_718 [modalities] input = ["text"] From 62558513ae315f477aea08cfea2b017b3020ebcb Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 07:30:37 +0000 Subject: [PATCH 285/392] chore(sync): update OpenRouter model catalog (#7722) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-pro-0813.toml | 7 +++--- .../models/kwaipilot/kat-coder-pro-v2.toml | 24 ------------------- .../models/~deepseek/deepseek-pro-latest.toml | 8 +++---- 3 files changed, 8 insertions(+), 31 deletions(-) delete mode 100644 providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index b8d53b0809c..f96833d38c7 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.32 -output = 3.96 -cache_read = 0.044 +input = 0.624 +output = 2.88 +cache_read = 0.088 [limit] context = 1_048_576 +output = 943_718 diff --git a/providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml b/providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml deleted file mode 100644 index 80b0f00e52d..00000000000 --- a/providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "KAT-Coder-Pro V2" -description = "Coding model for repository understanding, refactors, and agentic engineering tasks" -family = "kat-coder" -release_date = "2026-03-27" -last_updated = "2026-03-27" -attachment = false -reasoning = false -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.3 -output = 1.2 -cache_read = 0.06 - -[limit] -context = 262_144 -output = 144_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index f88779c2deb..1e614023f3e 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.638616 -output = 1.915848 -cache_read = 0.021287 +input = 0.624 +output = 2.88 +cache_read = 0.088 [limit] context = 1_048_576 -output = 384_000 +output = 943_718 [modalities] input = ["text"] From 3406591910d2f2430e766a6fc2f91ff08d71bcc6 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 07:30:44 +0000 Subject: [PATCH 286/392] chore(sync): update DevPass (LLM Gateway) model catalog (#7725) Co-authored-by: opencode-agent[bot] --- providers/llmgateway/models/mimo-v2.6-flash.toml | 13 +++++++++++++ providers/llmgateway/models/mimo-v2.6-pro.toml | 13 +++++++++++++ 2 files changed, 26 insertions(+) create mode 100644 providers/llmgateway/models/mimo-v2.6-flash.toml create mode 100644 providers/llmgateway/models/mimo-v2.6-pro.toml diff --git a/providers/llmgateway/models/mimo-v2.6-flash.toml b/providers/llmgateway/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..a71af774946 --- /dev/null +++ b/providers/llmgateway/models/mimo-v2.6-flash.toml @@ -0,0 +1,13 @@ +base_model = "xiaomi/mimo-v2.6-flash" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[limit] +context = 1_000_000 diff --git a/providers/llmgateway/models/mimo-v2.6-pro.toml b/providers/llmgateway/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..36f93eaabd5 --- /dev/null +++ b/providers/llmgateway/models/mimo-v2.6-pro.toml @@ -0,0 +1,13 @@ +base_model = "xiaomi/mimo-v2.6-pro" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 + +[limit] +context = 1_000_000 From 8e350677b4c9c6f3649c2bdffd51647ea6e5299b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 07:30:48 +0000 Subject: [PATCH 287/392] chore(sync): update LLM Gateway model catalog (#7724) Co-authored-by: opencode-agent[bot] --- .../models/xiaomi/mimo-v2.6-flash.toml | 15 +++++++++++++++ .../models/xiaomi/mimo-v2.6-pro.toml | 15 +++++++++++++++ 2 files changed, 30 insertions(+) create mode 100644 providers/llmgateway-providers/models/xiaomi/mimo-v2.6-flash.toml create mode 100644 providers/llmgateway-providers/models/xiaomi/mimo-v2.6-pro.toml diff --git a/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-flash.toml b/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..718a69cf405 --- /dev/null +++ b/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,15 @@ +base_model = "xiaomi/mimo-v2.6-flash" +name = "MiMo V2.6 Flash (Xiaomi)" +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[limit] +context = 1_000_000 diff --git a/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-pro.toml b/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..3fe108bfc5a --- /dev/null +++ b/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,15 @@ +base_model = "xiaomi/mimo-v2.6-pro" +name = "MiMo V2.6 Pro (Xiaomi)" +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 + +[limit] +context = 1_000_000 From 4fd8087c870d4132757c7a66b1e24b177d6dda3c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 08:34:32 +0000 Subject: [PATCH 288/392] chore(sync): update CrossModel model catalog (#7732) Co-authored-by: opencode-agent[bot] --- .../crossmodel/models/xiaomi/mimo-v2.6-flash.toml | 14 ++++++++++++++ .../crossmodel/models/xiaomi/mimo-v2.6-pro.toml | 14 ++++++++++++++ 2 files changed, 28 insertions(+) create mode 100644 providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml create mode 100644 providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml diff --git a/providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml b/providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..51fd77c6aa8 --- /dev/null +++ b/providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,14 @@ +base_model = "xiaomi/mimo-v2.6-flash" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.16 +output = 0.32 +cache_read = 0.004 +cache_write = 0.16 + +[modalities] +input = ["text", "image", "audio", "video"] diff --git a/providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml b/providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..e0fca43e997 --- /dev/null +++ b/providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,14 @@ +base_model = "xiaomi/mimo-v2.6-pro" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.47 +output = 0.94 +cache_read = 0.005 +cache_write = 0.47 + +[modalities] +input = ["text", "image", "audio", "video"] From c342e16fa8ea3b663ab79f5c421174379a0e6303 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 08:34:51 +0000 Subject: [PATCH 289/392] chore(sync): update NanoGPT model catalog (#7731) Co-authored-by: opencode-agent[bot] --- .../models/abliteration-ai/abliterated-model-large-v2.toml | 4 ++-- .../models/abliteration-ai/abliterated-model-large.toml | 4 ++-- .../nano-gpt/models/abliteration-ai/abliterated-model.toml | 4 ++-- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/nano-gpt/models/abliteration-ai/abliterated-model-large-v2.toml b/providers/nano-gpt/models/abliteration-ai/abliterated-model-large-v2.toml index 2228e8805bd..1a9af392dac 100644 --- a/providers/nano-gpt/models/abliteration-ai/abliterated-model-large-v2.toml +++ b/providers/nano-gpt/models/abliteration-ai/abliterated-model-large-v2.toml @@ -13,9 +13,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 5 +input = 3 output = 5 -cache_read = 0.5 +cache_read = 0.3 [limit] context = 1_000_000 diff --git a/providers/nano-gpt/models/abliteration-ai/abliterated-model-large.toml b/providers/nano-gpt/models/abliteration-ai/abliterated-model-large.toml index 71a9a723881..58ef18e1471 100644 --- a/providers/nano-gpt/models/abliteration-ai/abliterated-model-large.toml +++ b/providers/nano-gpt/models/abliteration-ai/abliterated-model-large.toml @@ -13,9 +13,9 @@ type = "effort" values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] -input = 5 +input = 3 output = 5 -cache_read = 0.5 +cache_read = 0.3 [limit] context = 1_000_000 diff --git a/providers/nano-gpt/models/abliteration-ai/abliterated-model.toml b/providers/nano-gpt/models/abliteration-ai/abliterated-model.toml index ac86f6f5738..46dfe7fcc9f 100644 --- a/providers/nano-gpt/models/abliteration-ai/abliterated-model.toml +++ b/providers/nano-gpt/models/abliteration-ai/abliterated-model.toml @@ -13,9 +13,9 @@ type = "effort" values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] -input = 3 +input = 1 output = 3 -cache_read = 0.3 +cache_read = 0.1 [limit] context = 262_144 From a4c8f9fd66da7eeccae8cb84dae3802993d99bc7 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:28:13 +0000 Subject: [PATCH 290/392] chore(sync): update Kilo model catalog (#7734) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 3 +-- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 2 files changed, 6 insertions(+), 7 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index fa4ef3087d9..01800e18158 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -11,5 +11,4 @@ output = 3.96 cache_read = 0.044 [limit] -context = 1_048_576 -output = 943_718 +context = 1_024_000 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 5dc58c6a638..6caf71dbf01 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.624 -output = 2.88 -cache_read = 0.088 +input = 0.558624 +output = 1.675872 +cache_read = 0.018621 [limit] -context = 1_048_576 -output = 943_718 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] From c44ddd7aa48f99d660e4e549c3fbe129065a94c2 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:28:29 +0000 Subject: [PATCH 291/392] chore(sync): update OpenRouter model catalog (#7735) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 +++---- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 2 files changed, 7 insertions(+), 8 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index f96833d38c7..c387c37a3f7 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,10 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.624 -output = 2.88 -cache_read = 0.088 +input = 0.558624 +output = 1.675872 +cache_read = 0.018621 [limit] context = 1_048_576 -output = 943_718 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 1e614023f3e..1db14b9b08b 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.624 -output = 2.88 -cache_read = 0.088 +input = 0.558624 +output = 1.675872 +cache_read = 0.018621 [limit] context = 1_048_576 -output = 943_718 +output = 384_000 [modalities] input = ["text"] From 12a5301ede55a89a2e0d3c003e8bf16f378d1146 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:29:44 +0000 Subject: [PATCH 292/392] chore(sync): update CrossModel model catalog (#7736) Co-authored-by: opencode-agent[bot] --- .../crossmodel/models/x-ai/grok-4.7.toml | 21 +++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 providers/crossmodel/models/x-ai/grok-4.7.toml diff --git a/providers/crossmodel/models/x-ai/grok-4.7.toml b/providers/crossmodel/models/x-ai/grok-4.7.toml new file mode 100644 index 00000000000..99495b45cae --- /dev/null +++ b/providers/crossmodel/models/x-ai/grok-4.7.toml @@ -0,0 +1,21 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 +cache_write = 2 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 +cache_write = 4 + +[modalities] +input = ["text", "image"] From 4bcee582b0287f80995b9eab05a73d4681f209e8 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 10:28:41 +0000 Subject: [PATCH 293/392] chore(sync): update Kilo model catalog (#7740) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 6caf71dbf01..685f520f471 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.558624 -output = 1.675872 -cache_read = 0.018621 +input = 0.59862 +output = 1.79586 +cache_read = 0.019954 [limit] context = 1_024_000 From 7a6af0dd799013b320d0a8fb38aec01d7a564e5d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 10:29:00 +0000 Subject: [PATCH 294/392] chore(sync): update OpenRouter model catalog (#7739) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index c387c37a3f7..67ed7ee6779 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.558624 -output = 1.675872 -cache_read = 0.018621 +input = 0.59862 +output = 1.79586 +cache_read = 0.019954 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 854b27b6a72..16f058ed841 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.3 -output = 1.2 -cache_read = 0.006 +input = 0.15 +output = 0.6 +cache_read = 0.003 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 1db14b9b08b..7033c741dee 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.558624 -output = 1.675872 -cache_read = 0.018621 +input = 0.59862 +output = 1.79586 +cache_read = 0.019954 [limit] context = 1_048_576 From 614c8224cff67d671f419dd43771601183130573 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 10:29:05 +0000 Subject: [PATCH 295/392] chore(sync): update NanoGPT model catalog (#7741) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index a62995b0beb..b4cce19afb3 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.44 -output = 1.32 -cache_read = 0.014 +input = 0.22 +output = 0.66 +cache_read = 0.007 [limit] context = 1_048_576 From 2f7cfc6e35ccc4b9503c68aa0e8ab4d031483a59 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 11:25:30 +0000 Subject: [PATCH 296/392] chore(sync): update Kilo model catalog (#7743) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 2 +- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index 01800e18158..007b654034e 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -11,4 +11,4 @@ output = 3.96 cache_read = 0.044 [limit] -context = 1_024_000 +context = 1_048_576 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 685f520f471..5dc58c6a638 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.59862 -output = 1.79586 -cache_read = 0.019954 +input = 0.624 +output = 2.88 +cache_read = 0.088 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 943_718 [modalities] input = ["text"] From be72b1cf5fc58e8ede2c3bad9675bc0b435e5adb Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 11:25:35 +0000 Subject: [PATCH 297/392] chore(sync): update OpenRouter model catalog (#7744) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/moonshotai/kimi-k2.7-code.toml | 2 +- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 3 files changed, 8 insertions(+), 8 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 67ed7ee6779..7c5828b16e2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.59862 -output = 1.79586 -cache_read = 0.019954 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/moonshotai/kimi-k2.7-code.toml b/providers/openrouter/models/moonshotai/kimi-k2.7-code.toml index 307383582d9..3fb61dea6a2 100644 --- a/providers/openrouter/models/moonshotai/kimi-k2.7-code.toml +++ b/providers/openrouter/models/moonshotai/kimi-k2.7-code.toml @@ -4,7 +4,7 @@ reasoning_options = [] [cost] input = 0.7062 -output = 3.21 +output = 3.3 cache_read = 0.18 [limit] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 7033c741dee..1e614023f3e 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.59862 -output = 1.79586 -cache_read = 0.019954 +input = 0.624 +output = 2.88 +cache_read = 0.088 [limit] context = 1_048_576 -output = 384_000 +output = 943_718 [modalities] input = ["text"] From 83c0d82263bf118e87ad19bbae94503353109434 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 12:36:50 +0000 Subject: [PATCH 298/392] chore(sync): update Kilo model catalog (#7746) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index 007b654034e..fa4ef3087d9 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -12,3 +12,4 @@ cache_read = 0.044 [limit] context = 1_048_576 +output = 943_718 From e0cb541fcc24796de9f04f6f8e563c11c21a924e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 12:37:08 +0000 Subject: [PATCH 299/392] chore(sync): update OpenRouter model catalog (#7745) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..f96833d38c7 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.624 +output = 2.88 +cache_read = 0.088 [limit] context = 1_048_576 +output = 943_718 From 332109faa1d55b8e16a8da53d74354e5e97b7422 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:28:32 +0000 Subject: [PATCH 300/392] chore(sync): update OpenRouter model catalog (#7750) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 7 +++---- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 4 files changed, 13 insertions(+), 14 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 6f817c5a0a0..141d4b41ca6 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.088606 -output = 0.177212 -cache_read = 0.017721 +input = 0.049 +output = 0.098 +cache_read = 0.0098 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index f96833d38c7..7c5828b16e2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,10 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.624 -output = 2.88 -cache_read = 0.088 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 -output = 943_718 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index e7683af4617..2a431cd5648 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.95526 -output = 1.91052 -cache_read = 0.079605 +input = 0.951432 +output = 1.902864 +cache_read = 0.079286 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 1e614023f3e..9d1738cedb4 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.624 -output = 2.88 -cache_read = 0.088 +input = 0.61908 +output = 1.85724 +cache_read = 0.020636 [limit] context = 1_048_576 -output = 943_718 +output = 384_000 [modalities] input = ["text"] From 7c08237c0e50fa9657fa0ea41c3eae5aeb3e3a1b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:28:51 +0000 Subject: [PATCH 301/392] chore(sync): update Kilo model catalog (#7749) Co-authored-by: opencode-agent[bot] --- .../kilo/models/deepseek/deepseek-v4-pro-0813.toml | 1 - .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 2 files changed, 5 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml index fa4ef3087d9..007b654034e 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro-0813.toml @@ -12,4 +12,3 @@ cache_read = 0.044 [limit] context = 1_048_576 -output = 943_718 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 5dc58c6a638..78a5dc0748a 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.624 -output = 2.88 -cache_read = 0.088 +input = 0.61908 +output = 1.85724 +cache_read = 0.020636 [limit] -context = 1_048_576 -output = 943_718 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] From a8fe72fbf5dbcc280ea7f500bd06fc4cb8e2e594 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 14:29:35 +0000 Subject: [PATCH 302/392] chore(sync): update DevPass (LLM Gateway) model catalog (#7753) Co-authored-by: opencode-agent[bot] --- providers/llmgateway/models/glm-5.3.toml | 2 +- providers/llmgateway/models/inkling-small.toml | 13 +++++++++++++ providers/llmgateway/models/inkling.toml | 13 +++++++++++++ providers/llmgateway/models/muse-glimmer-30b.toml | 10 ++++++++++ .../llmgateway/models/nemotron-3.5-lightning.toml | 9 +++++++++ providers/llmgateway/models/qwen3.8-2.4t-a95b.toml | 13 +++++++++++++ providers/llmgateway/models/step-3.7-flash.toml | 14 ++++++++++++++ 7 files changed, 73 insertions(+), 1 deletion(-) create mode 100644 providers/llmgateway/models/inkling-small.toml create mode 100644 providers/llmgateway/models/inkling.toml create mode 100644 providers/llmgateway/models/muse-glimmer-30b.toml create mode 100644 providers/llmgateway/models/nemotron-3.5-lightning.toml create mode 100644 providers/llmgateway/models/qwen3.8-2.4t-a95b.toml create mode 100644 providers/llmgateway/models/step-3.7-flash.toml diff --git a/providers/llmgateway/models/glm-5.3.toml b/providers/llmgateway/models/glm-5.3.toml index 66cc365baaa..6be69ce3acf 100644 --- a/providers/llmgateway/models/glm-5.3.toml +++ b/providers/llmgateway/models/glm-5.3.toml @@ -2,7 +2,7 @@ base_model = "zhipuai/glm-5.3" [[reasoning_options]] type = "effort" -values = ["low", "high", "max"] +values = ["none", "low", "high", "max"] [cost] input = 1.2 diff --git a/providers/llmgateway/models/inkling-small.toml b/providers/llmgateway/models/inkling-small.toml new file mode 100644 index 00000000000..954d4c25f03 --- /dev/null +++ b/providers/llmgateway/models/inkling-small.toml @@ -0,0 +1,13 @@ +base_model = "thinkingmachines/inkling-small" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.45 +output = 1.2 +cache_read = 0.1 + +[limit] +context = 524_288 diff --git a/providers/llmgateway/models/inkling.toml b/providers/llmgateway/models/inkling.toml new file mode 100644 index 00000000000..1d07005aa33 --- /dev/null +++ b/providers/llmgateway/models/inkling.toml @@ -0,0 +1,13 @@ +base_model = "thinkingmachines/inkling" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.95 +output = 4.05 +cache_read = 0.16 + +[limit] +context = 524_288 diff --git a/providers/llmgateway/models/muse-glimmer-30b.toml b/providers/llmgateway/models/muse-glimmer-30b.toml new file mode 100644 index 00000000000..9d06a10be3b --- /dev/null +++ b/providers/llmgateway/models/muse-glimmer-30b.toml @@ -0,0 +1,10 @@ +base_model = "meta/muse-glimmer-30b" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.04 diff --git a/providers/llmgateway/models/nemotron-3.5-lightning.toml b/providers/llmgateway/models/nemotron-3.5-lightning.toml new file mode 100644 index 00000000000..6e7dba809e3 --- /dev/null +++ b/providers/llmgateway/models/nemotron-3.5-lightning.toml @@ -0,0 +1,9 @@ +base_model = "nvidia/nemotron-3.5-lightning" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.08 +output = 0.2 diff --git a/providers/llmgateway/models/qwen3.8-2.4t-a95b.toml b/providers/llmgateway/models/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..02fe65d078f --- /dev/null +++ b/providers/llmgateway/models/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,13 @@ +base_model = "alibaba/qwen3.8-2.4t-a95b" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 6 +cache_read = 0.25 + +[limit] +context = 1_010_000 diff --git a/providers/llmgateway/models/step-3.7-flash.toml b/providers/llmgateway/models/step-3.7-flash.toml new file mode 100644 index 00000000000..ad0acb372ea --- /dev/null +++ b/providers/llmgateway/models/step-3.7-flash.toml @@ -0,0 +1,14 @@ +base_model = "stepfun/step-3.7-flash" +base_model_omit = ["limit.input"] + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.2 +output = 1.15 +cache_read = 0.04 + +[limit] +context = 262_144 From 6f41b78e33c939a7b9d836fb880a854058ca856b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 14:30:30 +0000 Subject: [PATCH 303/392] chore(sync): update OpenRouter model catalog (#7752) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.3.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- providers/openrouter/models/~z-ai/glm-latest.toml | 8 ++++---- 4 files changed, 14 insertions(+), 14 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 2a431cd5648..799137f52c2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.951432 -output = 1.902864 -cache_read = 0.079286 +input = 0.946386 +output = 1.892772 +cache_read = 0.078866 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index 8b9af97ef26..7c1154ec1bd 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.84 -output = 2.64 -cache_read = 0.156 +input = 0.6538 +output = 2.0548 +cache_read = 0.12142 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 9d1738cedb4..d4c1d5ba628 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.61908 -output = 1.85724 -cache_read = 0.020636 +input = 0.57948 +output = 1.73844 +cache_read = 0.018438 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 167f0470b75..e0e1e1ca329 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.6545 -output = 2.057 -cache_read = 0.107525 +input = 0.6538 +output = 2.0548 +cache_read = 0.12142 [limit] context = 1_310_720 -output = 943_718 +output = 131_072 [modalities] input = ["text"] From 6651aab362256017c2b80910c57b96a9b567b215 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 14:30:53 +0000 Subject: [PATCH 304/392] chore(sync): update Kilo model catalog (#7754) Co-authored-by: opencode-agent[bot] --- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- providers/kilo/models/~z-ai/glm-latest.toml | 8 ++++---- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 78a5dc0748a..27422e128e0 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.61908 -output = 1.85724 -cache_read = 0.020636 +input = 0.57948 +output = 1.73844 +cache_read = 0.018438 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 3a29b91f006..0facfa991b5 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.6545 -output = 2.057 -cache_read = 0.107525 +input = 0.6538 +output = 2.0548 +cache_read = 0.12142 [limit] context = 1_048_576 -output = 943_718 +output = 131_072 [modalities] input = ["text"] From 0609b4f5bddabaad04bf0c56e723dbda8121dc1d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:27:00 +0000 Subject: [PATCH 305/392] chore(sync): update OpenRouter model catalog (#7759) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/aion-labs/aion-2.0.toml | 2 +- providers/openrouter/models/aion-labs/aion-3.0-mini.toml | 2 +- providers/openrouter/models/aion-labs/aion-3.0.toml | 2 +- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/openai/gpt-oss-20b.toml | 8 ++------ .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 6 files changed, 12 insertions(+), 16 deletions(-) diff --git a/providers/openrouter/models/aion-labs/aion-2.0.toml b/providers/openrouter/models/aion-labs/aion-2.0.toml index b54b64899f3..8e499e073cf 100644 --- a/providers/openrouter/models/aion-labs/aion-2.0.toml +++ b/providers/openrouter/models/aion-labs/aion-2.0.toml @@ -16,7 +16,7 @@ output = 1.6 cache_read = 0.2 [limit] -context = 131_072 +context = 1_048_576 output = 32_768 [modalities] diff --git a/providers/openrouter/models/aion-labs/aion-3.0-mini.toml b/providers/openrouter/models/aion-labs/aion-3.0-mini.toml index af968c32457..33c5d843701 100644 --- a/providers/openrouter/models/aion-labs/aion-3.0-mini.toml +++ b/providers/openrouter/models/aion-labs/aion-3.0-mini.toml @@ -16,7 +16,7 @@ output = 1.4 cache_read = 0.18 [limit] -context = 131_072 +context = 1_048_576 output = 32_768 [modalities] diff --git a/providers/openrouter/models/aion-labs/aion-3.0.toml b/providers/openrouter/models/aion-labs/aion-3.0.toml index e25d916b2b3..f16de215471 100644 --- a/providers/openrouter/models/aion-labs/aion-3.0.toml +++ b/providers/openrouter/models/aion-labs/aion-3.0.toml @@ -16,7 +16,7 @@ output = 6 cache_read = 0.75 [limit] -context = 131_072 +context = 1_048_576 output = 32_768 [modalities] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 799137f52c2..a705f18ca62 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.946386 -output = 1.892772 -cache_read = 0.078866 +input = 0.936294 +output = 1.872588 +cache_read = 0.078025 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/openai/gpt-oss-20b.toml b/providers/openrouter/models/openai/gpt-oss-20b.toml index b65dd3718e0..98aeb108997 100644 --- a/providers/openrouter/models/openai/gpt-oss-20b.toml +++ b/providers/openrouter/models/openai/gpt-oss-20b.toml @@ -6,9 +6,5 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.03 -output = 0.13 -cache_read = 0.03 - -[limit] -output = 117_964 +input = 0.018 +output = 0.09 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index d4c1d5ba628..0ea95b2339f 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57948 -output = 1.73844 -cache_read = 0.018438 +input = 0.57156 +output = 1.71468 +cache_read = 0.019052 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] From 8f37270269201a2e8d280327eef9f7e47bc8c37f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:27:04 +0000 Subject: [PATCH 306/392] chore(sync): update Eden AI model catalog (#7757) Co-authored-by: opencode-agent[bot] --- .../infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml | 4 ++-- .../models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml | 4 ++-- providers/edenai/models/ionos/openai/gpt-oss-120b.toml | 4 ++-- providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml | 4 ++-- providers/edenai/models/scaleway/gpt-oss-120b.toml | 4 ++-- providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml | 4 ++-- 6 files changed, 12 insertions(+), 12 deletions(-) diff --git a/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml b/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml index beffb59e2ed..73ebbfe4b80 100644 --- a/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml +++ b/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml @@ -5,8 +5,8 @@ tool_call = false structured_output = false [cost] -input = 0.3447 -output = 0.4596 +input = 0.34389 +output = 0.45852 [limit] context = 100_000 diff --git a/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml b/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml index 11dee7e0f74..92757e9fa56 100644 --- a/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml +++ b/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml @@ -4,5 +4,5 @@ tool_call = false structured_output = false [cost] -input = 0.74685 -output = 0.74685 +input = 0.745095 +output = 0.745095 diff --git a/providers/edenai/models/ionos/openai/gpt-oss-120b.toml b/providers/edenai/models/ionos/openai/gpt-oss-120b.toml index 2bf7b7f0f21..cad05c7d393 100644 --- a/providers/edenai/models/ionos/openai/gpt-oss-120b.toml +++ b/providers/edenai/models/ionos/openai/gpt-oss-120b.toml @@ -8,5 +8,5 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.17235 -output = 0.74685 +input = 0.171945 +output = 0.745095 diff --git a/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml b/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml index d761571f13a..d2f9f4d763d 100644 --- a/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml +++ b/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml @@ -7,8 +7,8 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.4596 -output = 0.9192 +input = 0.45852 +output = 0.91704 [limit] context = 256_000 diff --git a/providers/edenai/models/scaleway/gpt-oss-120b.toml b/providers/edenai/models/scaleway/gpt-oss-120b.toml index dbff07167cc..8f735ccfed6 100644 --- a/providers/edenai/models/scaleway/gpt-oss-120b.toml +++ b/providers/edenai/models/scaleway/gpt-oss-120b.toml @@ -7,8 +7,8 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.17235 -output = 0.6894 +input = 0.171945 +output = 0.68778 [limit] context = 128_000 diff --git a/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml b/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml index 8d733529a6c..55662b4313f 100644 --- a/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml +++ b/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml @@ -3,5 +3,5 @@ name = "Llama-3.3-70B-Instruct (Scaleway)" structured_output = false [cost] -input = 1.0341 -output = 1.0341 +input = 1.03167 +output = 1.03167 From d07f6dce7adfd0e4aa07ceffa9cadd7e059d2455 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:27:20 +0000 Subject: [PATCH 307/392] chore(sync): update Kilo model catalog (#7758) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/aion-labs/aion-2.0.toml | 2 +- providers/kilo/models/aion-labs/aion-3.0-mini.toml | 2 +- providers/kilo/models/aion-labs/aion-3.0.toml | 2 +- providers/kilo/models/openai/gpt-oss-20b.toml | 3 --- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 10 +++++----- 5 files changed, 8 insertions(+), 11 deletions(-) diff --git a/providers/kilo/models/aion-labs/aion-2.0.toml b/providers/kilo/models/aion-labs/aion-2.0.toml index 4bbbba9076b..46cc9f272f7 100644 --- a/providers/kilo/models/aion-labs/aion-2.0.toml +++ b/providers/kilo/models/aion-labs/aion-2.0.toml @@ -19,7 +19,7 @@ output = 1.6 cache_read = 0.2 [limit] -context = 131_072 +context = 1_048_576 output = 32_768 [modalities] diff --git a/providers/kilo/models/aion-labs/aion-3.0-mini.toml b/providers/kilo/models/aion-labs/aion-3.0-mini.toml index 4987c065dc7..84f2979d62c 100644 --- a/providers/kilo/models/aion-labs/aion-3.0-mini.toml +++ b/providers/kilo/models/aion-labs/aion-3.0-mini.toml @@ -19,7 +19,7 @@ output = 1.4 cache_read = 0.18 [limit] -context = 131_072 +context = 1_048_576 output = 32_768 [modalities] diff --git a/providers/kilo/models/aion-labs/aion-3.0.toml b/providers/kilo/models/aion-labs/aion-3.0.toml index caa24ee8921..ac0a56ff9f9 100644 --- a/providers/kilo/models/aion-labs/aion-3.0.toml +++ b/providers/kilo/models/aion-labs/aion-3.0.toml @@ -19,7 +19,7 @@ output = 6 cache_read = 0.75 [limit] -context = 131_072 +context = 1_048_576 output = 32_768 [modalities] diff --git a/providers/kilo/models/openai/gpt-oss-20b.toml b/providers/kilo/models/openai/gpt-oss-20b.toml index fe3d2508db6..98aeb108997 100644 --- a/providers/kilo/models/openai/gpt-oss-20b.toml +++ b/providers/kilo/models/openai/gpt-oss-20b.toml @@ -8,6 +8,3 @@ values = ["low", "medium", "high"] [cost] input = 0.018 output = 0.09 - -[limit] -output = 117_964 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 27422e128e0..9da59d77044 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.57948 -output = 1.73844 -cache_read = 0.018438 +input = 0.57156 +output = 1.71468 +cache_read = 0.019052 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 +output = 384_000 [modalities] input = ["text"] From 173ae60de164b79b830a9cabc8a920c32cf7e0f9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:29:18 +0000 Subject: [PATCH 308/392] chore(sync): update Kilo model catalog (#7760) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/qwen/qwen3.6-27b.toml | 3 +++ providers/kilo/models/tencent/hy3.toml | 6 +++--- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 3 files changed, 9 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/qwen/qwen3.6-27b.toml b/providers/kilo/models/qwen/qwen3.6-27b.toml index 61085517bb0..6ca4eb2df17 100644 --- a/providers/kilo/models/qwen/qwen3.6-27b.toml +++ b/providers/kilo/models/qwen/qwen3.6-27b.toml @@ -9,5 +9,8 @@ values = ["none", "high"] input = 0.45 output = 2.7 +[limit] +output = 262_140 + [modalities] input = ["text", "image", "video"] diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index c3469c1e9dc..e96e3110ab1 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.13 -output = 0.53 -cache_read = 0.033 +input = 0.0825 +output = 0.33 +cache_read = 0.020625 [limit] context = 262_144 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 9da59d77044..6565f5a8e9b 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.57156 -output = 1.71468 -cache_read = 0.019052 +input = 0.56364 +output = 1.69092 +cache_read = 0.018788 [limit] context = 1_024_000 From 9255381af9015a988ca3155e23b6d0742a445f19 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:29:31 +0000 Subject: [PATCH 309/392] chore(sync): update OpenRouter model catalog (#7761) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/qwen/qwen3.6-27b.toml | 9 ++++++--- providers/openrouter/models/tencent/hy3.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 4 files changed, 15 insertions(+), 12 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index a705f18ca62..8ce9eb4af6a 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.936294 -output = 1.872588 -cache_read = 0.078025 +input = 0.927768 +output = 1.855536 +cache_read = 0.077314 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3.6-27b.toml b/providers/openrouter/models/qwen/qwen3.6-27b.toml index 15ca4007021..c188473afdb 100644 --- a/providers/openrouter/models/qwen/qwen3.6-27b.toml +++ b/providers/openrouter/models/qwen/qwen3.6-27b.toml @@ -6,9 +6,12 @@ base_model = "alibaba/qwen3.6-27b" type = "toggle" [cost] -input = 0.3 -output = 2 -cache_read = 0.03 +input = 0.32 +output = 2.7 +cache_read = 0.15 + +[limit] +output = 262_140 [modalities] input = ["text", "image", "video"] diff --git a/providers/openrouter/models/tencent/hy3.toml b/providers/openrouter/models/tencent/hy3.toml index 80ecfdd3aa8..f61fa3175e9 100644 --- a/providers/openrouter/models/tencent/hy3.toml +++ b/providers/openrouter/models/tencent/hy3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.132 -output = 0.528 -cache_read = 0.033 +input = 0.0825 +output = 0.33 +cache_read = 0.020625 [limit] context = 262_144 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 0ea95b2339f..ff538327fb0 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.57156 -output = 1.71468 -cache_read = 0.019052 +input = 0.56364 +output = 1.69092 +cache_read = 0.018788 [limit] context = 1_048_576 From 1ba7a9dff9a2be54e824d7211475b8d9d3b4c5de Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 11:47:14 -0500 Subject: [PATCH 310/392] feat: add Claude Opus 5.5 (#7763) Co-authored-by: rekram1-node --- models/anthropic/claude-opus-5-5.toml | 19 +++++++++++++++++++ .../models/anthropic.claude-opus-5-5.toml | 11 +++++++++++ .../models/au.anthropic.claude-opus-5-5.toml | 13 +++++++++++++ .../models/eu.anthropic.claude-opus-5-5.toml | 13 +++++++++++++ .../global.anthropic.claude-opus-5-5.toml | 12 ++++++++++++ .../models/jp.anthropic.claude-opus-5-5.toml | 13 +++++++++++++ .../models/us.anthropic.claude-opus-5-5.toml | 13 +++++++++++++ .../anthropic/models/claude-opus-5-5.toml | 13 +++++++++++++ .../models/claude-opus-5-5.toml | 13 +++++++++++++ providers/azure/models/claude-opus-5-5.toml | 13 +++++++++++++ .../models/claude-opus-5-5@default.toml | 9 +++++++++ .../models/claude-opus-5-5@default.toml | 12 ++++++++++++ 12 files changed, 154 insertions(+) create mode 100644 models/anthropic/claude-opus-5-5.toml create mode 100644 providers/amazon-bedrock/models/anthropic.claude-opus-5-5.toml create mode 100644 providers/amazon-bedrock/models/au.anthropic.claude-opus-5-5.toml create mode 100644 providers/amazon-bedrock/models/eu.anthropic.claude-opus-5-5.toml create mode 100644 providers/amazon-bedrock/models/global.anthropic.claude-opus-5-5.toml create mode 100644 providers/amazon-bedrock/models/jp.anthropic.claude-opus-5-5.toml create mode 100644 providers/amazon-bedrock/models/us.anthropic.claude-opus-5-5.toml create mode 100644 providers/anthropic/models/claude-opus-5-5.toml create mode 100644 providers/azure-cognitive-services/models/claude-opus-5-5.toml create mode 100644 providers/azure/models/claude-opus-5-5.toml create mode 100644 providers/google-vertex-anthropic/models/claude-opus-5-5@default.toml create mode 100644 providers/google-vertex/models/claude-opus-5-5@default.toml diff --git a/models/anthropic/claude-opus-5-5.toml b/models/anthropic/claude-opus-5-5.toml new file mode 100644 index 00000000000..0f1cfeba4ed --- /dev/null +++ b/models/anthropic/claude-opus-5-5.toml @@ -0,0 +1,19 @@ +name = "Claude Opus 5.5" +description = "Claude model for long-running agentic coding and knowledge work" +family = "claude-opus" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +open_weights = false +knowledge = "2026-06" + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/amazon-bedrock/models/anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..a9684824bf1 --- /dev/null +++ b/providers/amazon-bedrock/models/anthropic.claude-opus-5-5.toml @@ -0,0 +1,11 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/amazon-bedrock/models/au.anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/au.anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..25e28fdb4b6 --- /dev/null +++ b/providers/amazon-bedrock/models/au.anthropic.claude-opus-5-5.toml @@ -0,0 +1,13 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Bedrock regional endpoints carry a 10% pricing premium over global endpoints. +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +name = "Claude Opus 5.5 (AU)" + +[cost] +input = 4.4 +output = 22 +cache_read = 0.22 +cache_write = 5.5 diff --git a/providers/amazon-bedrock/models/eu.anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/eu.anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..8c9d8b97cdd --- /dev/null +++ b/providers/amazon-bedrock/models/eu.anthropic.claude-opus-5-5.toml @@ -0,0 +1,13 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Bedrock regional endpoints carry a 10% pricing premium over global endpoints. +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +name = "Claude Opus 5.5 (EU)" + +[cost] +input = 4.4 +output = 22 +cache_read = 0.22 +cache_write = 5.5 diff --git a/providers/amazon-bedrock/models/global.anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/global.anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..c36a3b86f74 --- /dev/null +++ b/providers/amazon-bedrock/models/global.anthropic.claude-opus-5-5.toml @@ -0,0 +1,12 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +name = "Claude Opus 5.5 (Global)" + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/amazon-bedrock/models/jp.anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/jp.anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..78ad0c8319e --- /dev/null +++ b/providers/amazon-bedrock/models/jp.anthropic.claude-opus-5-5.toml @@ -0,0 +1,13 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Bedrock regional endpoints carry a 10% pricing premium over global endpoints. +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +name = "Claude Opus 5.5 (JP)" + +[cost] +input = 4.4 +output = 22 +cache_read = 0.22 +cache_write = 5.5 diff --git a/providers/amazon-bedrock/models/us.anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/us.anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..f74ececd884 --- /dev/null +++ b/providers/amazon-bedrock/models/us.anthropic.claude-opus-5-5.toml @@ -0,0 +1,13 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Bedrock regional endpoints carry a 10% pricing premium over global endpoints. +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +name = "Claude Opus 5.5 (US)" + +[cost] +input = 4.4 +output = 22 +cache_read = 0.22 +cache_write = 5.5 diff --git a/providers/anthropic/models/claude-opus-5-5.toml b/providers/anthropic/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..7341878caeb --- /dev/null +++ b/providers/anthropic/models/claude-opus-5-5.toml @@ -0,0 +1,13 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[experimental.modes.fast] +cost = { input = 8, output = 40, cache_read = 0.4, cache_write = 10 } +provider = { body = { speed = "fast" }, headers = { anthropic-beta = "fast-mode-2026-02-01" } } diff --git a/providers/azure-cognitive-services/models/claude-opus-5-5.toml b/providers/azure-cognitive-services/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..c4fedc9752f --- /dev/null +++ b/providers/azure-cognitive-services/models/claude-opus-5-5.toml @@ -0,0 +1,13 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[provider] +npm = "@ai-sdk/anthropic" +api = "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" diff --git a/providers/azure/models/claude-opus-5-5.toml b/providers/azure/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..cc062a28b81 --- /dev/null +++ b/providers/azure/models/claude-opus-5-5.toml @@ -0,0 +1,13 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[provider] +npm = "@ai-sdk/anthropic" +api = "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" diff --git a/providers/google-vertex-anthropic/models/claude-opus-5-5@default.toml b/providers/google-vertex-anthropic/models/claude-opus-5-5@default.toml new file mode 100644 index 00000000000..cede54eca95 --- /dev/null +++ b/providers/google-vertex-anthropic/models/claude-opus-5-5@default.toml @@ -0,0 +1,9 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/google-vertex/models/claude-opus-5-5@default.toml b/providers/google-vertex/models/claude-opus-5-5@default.toml new file mode 100644 index 00000000000..8a2a18a47c1 --- /dev/null +++ b/providers/google-vertex/models/claude-opus-5-5@default.toml @@ -0,0 +1,12 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[provider] +npm = "@ai-sdk/google-vertex/anthropic" From 3ede55347c32de6a4cee88e7c68bd31ef76370a2 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 11:47:41 -0500 Subject: [PATCH 311/392] chore(sync): update LLM Gateway model catalog (#7751) Co-authored-by: opencode-agent[bot] --- .../models/deepinfra/glm-5.3.toml | 14 ++++++++++++++ .../models/deepinfra/inkling-small.toml | 16 ++++++++++++++++ .../models/deepinfra/inkling.toml | 17 +++++++++++++++++ .../models/deepinfra/muse-glimmer-30b.toml | 14 ++++++++++++++ .../deepinfra/nemotron-3.5-lightning.toml | 13 +++++++++++++ .../models/deepinfra/qwen3.8-2.4t-a95b.toml | 11 +++++++++++ .../models/deepinfra/qwen3.8-27b.toml | 14 ++++++++++++++ .../models/deepinfra/step-3.7-flash.toml | 17 +++++++++++++++++ .../models/novita/qwen3.8-2.4t-a95b.toml | 14 ++++++++++++++ .../models/novita/step-3.7-flash.toml | 16 ++++++++++++++++ .../models/together-ai/glm-5.3-flash.toml | 15 +++++++++++++++ .../models/together-ai/glm-5.3.toml | 15 +++++++++++++++ .../models/together-ai/inkling.toml | 16 ++++++++++++++++ .../models/together-ai/muse-glimmer-30b.toml | 14 ++++++++++++++ .../models/together-ai/qwen3.8-2.4t-a95b.toml | 15 +++++++++++++++ 15 files changed, 221 insertions(+) create mode 100644 providers/llmgateway-providers/models/deepinfra/glm-5.3.toml create mode 100644 providers/llmgateway-providers/models/deepinfra/inkling-small.toml create mode 100644 providers/llmgateway-providers/models/deepinfra/inkling.toml create mode 100644 providers/llmgateway-providers/models/deepinfra/muse-glimmer-30b.toml create mode 100644 providers/llmgateway-providers/models/deepinfra/nemotron-3.5-lightning.toml create mode 100644 providers/llmgateway-providers/models/deepinfra/qwen3.8-2.4t-a95b.toml create mode 100644 providers/llmgateway-providers/models/deepinfra/qwen3.8-27b.toml create mode 100644 providers/llmgateway-providers/models/deepinfra/step-3.7-flash.toml create mode 100644 providers/llmgateway-providers/models/novita/qwen3.8-2.4t-a95b.toml create mode 100644 providers/llmgateway-providers/models/novita/step-3.7-flash.toml create mode 100644 providers/llmgateway-providers/models/together-ai/glm-5.3-flash.toml create mode 100644 providers/llmgateway-providers/models/together-ai/glm-5.3.toml create mode 100644 providers/llmgateway-providers/models/together-ai/inkling.toml create mode 100644 providers/llmgateway-providers/models/together-ai/muse-glimmer-30b.toml create mode 100644 providers/llmgateway-providers/models/together-ai/qwen3.8-2.4t-a95b.toml diff --git a/providers/llmgateway-providers/models/deepinfra/glm-5.3.toml b/providers/llmgateway-providers/models/deepinfra/glm-5.3.toml new file mode 100644 index 00000000000..49f3f113f39 --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/glm-5.3.toml @@ -0,0 +1,14 @@ +base_model = "zhipuai/glm-5.3" +name = "GLM-5.3 (DeepInfra)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high", "max"] + +[cost] +input = 1.2 +output = 4 +cache_read = 0.2 + +[limit] +context = 1_048_576 diff --git a/providers/llmgateway-providers/models/deepinfra/inkling-small.toml b/providers/llmgateway-providers/models/deepinfra/inkling-small.toml new file mode 100644 index 00000000000..0109639620f --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/inkling-small.toml @@ -0,0 +1,16 @@ +base_model = "thinkingmachines/inkling-small" +name = "Inkling Small (DeepInfra)" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.45 +output = 1.2 +cache_read = 0.1 + +[limit] +context = 524_288 +output = 262_144 diff --git a/providers/llmgateway-providers/models/deepinfra/inkling.toml b/providers/llmgateway-providers/models/deepinfra/inkling.toml new file mode 100644 index 00000000000..41e528e739e --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/inkling.toml @@ -0,0 +1,17 @@ +base_model = "thinkingmachines/inkling" +name = "Inkling (DeepInfra)" +tool_call = false +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.95 +output = 4.05 +cache_read = 0.16 + +[limit] +context = 524_288 +output = 262_144 diff --git a/providers/llmgateway-providers/models/deepinfra/muse-glimmer-30b.toml b/providers/llmgateway-providers/models/deepinfra/muse-glimmer-30b.toml new file mode 100644 index 00000000000..0c26147df21 --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/muse-glimmer-30b.toml @@ -0,0 +1,14 @@ +base_model = "meta/muse-glimmer-30b" +name = "Muse Glimmer 30B (DeepInfra)" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.04 + +[limit] +output = 16_384 diff --git a/providers/llmgateway-providers/models/deepinfra/nemotron-3.5-lightning.toml b/providers/llmgateway-providers/models/deepinfra/nemotron-3.5-lightning.toml new file mode 100644 index 00000000000..aa669cc8be8 --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/nemotron-3.5-lightning.toml @@ -0,0 +1,13 @@ +base_model = "nvidia/nemotron-3.5-lightning" +name = "Nemotron 3.5 Lightning (DeepInfra)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.08 +output = 0.2 + +[limit] +output = 131_072 diff --git a/providers/llmgateway-providers/models/deepinfra/qwen3.8-2.4t-a95b.toml b/providers/llmgateway-providers/models/deepinfra/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..09c8ba4899d --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,11 @@ +base_model = "alibaba/qwen3.8-2.4t-a95b" +name = "Qwen3.8 2.4T A95B (DeepInfra)" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 6 +cache_read = 0.2 diff --git a/providers/llmgateway-providers/models/deepinfra/qwen3.8-27b.toml b/providers/llmgateway-providers/models/deepinfra/qwen3.8-27b.toml new file mode 100644 index 00000000000..3b37e4d93ab --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/qwen3.8-27b.toml @@ -0,0 +1,14 @@ +base_model = "alibaba/qwen3.8-27b" +name = "Qwen3.8 27B (DeepInfra)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.2 +output = 2.5 +cache_read = 0.05 + +[limit] +output = 235_929 diff --git a/providers/llmgateway-providers/models/deepinfra/step-3.7-flash.toml b/providers/llmgateway-providers/models/deepinfra/step-3.7-flash.toml new file mode 100644 index 00000000000..d9f47a5da91 --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/step-3.7-flash.toml @@ -0,0 +1,17 @@ +base_model = "stepfun/step-3.7-flash" +base_model_omit = ["limit.input"] +name = "Step 3.7 Flash (DeepInfra)" +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.2 +output = 1.15 +cache_read = 0.04 + +[limit] +context = 262_144 +output = 32_768 diff --git a/providers/llmgateway-providers/models/novita/qwen3.8-2.4t-a95b.toml b/providers/llmgateway-providers/models/novita/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..6cac460cd05 --- /dev/null +++ b/providers/llmgateway-providers/models/novita/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,14 @@ +base_model = "alibaba/qwen3.8-2.4t-a95b" +name = "Qwen3.8 2.4T A95B (NovitaAI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 6 +cache_read = 0.25 + +[limit] +context = 1_000_000 diff --git a/providers/llmgateway-providers/models/novita/step-3.7-flash.toml b/providers/llmgateway-providers/models/novita/step-3.7-flash.toml new file mode 100644 index 00000000000..ca5843d5fb8 --- /dev/null +++ b/providers/llmgateway-providers/models/novita/step-3.7-flash.toml @@ -0,0 +1,16 @@ +base_model = "stepfun/step-3.7-flash" +base_model_omit = ["limit.input"] +name = "Step 3.7 Flash (NovitaAI)" +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.2 +output = 1.15 +cache_read = 0.04 + +[limit] +context = 262_144 diff --git a/providers/llmgateway-providers/models/together-ai/glm-5.3-flash.toml b/providers/llmgateway-providers/models/together-ai/glm-5.3-flash.toml new file mode 100644 index 00000000000..af0987b5584 --- /dev/null +++ b/providers/llmgateway-providers/models/together-ai/glm-5.3-flash.toml @@ -0,0 +1,15 @@ +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3 Flash (Together AI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.03 + +[limit] +context = 1_048_576 +output = 943_717 diff --git a/providers/llmgateway-providers/models/together-ai/glm-5.3.toml b/providers/llmgateway-providers/models/together-ai/glm-5.3.toml new file mode 100644 index 00000000000..67971580a73 --- /dev/null +++ b/providers/llmgateway-providers/models/together-ai/glm-5.3.toml @@ -0,0 +1,15 @@ +base_model = "zhipuai/glm-5.3" +name = "GLM-5.3 (Together AI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high", "max"] + +[cost] +input = 1.4 +output = 4.4 +cache_read = 0.26 + +[limit] +context = 1_048_576 +output = 943_717 diff --git a/providers/llmgateway-providers/models/together-ai/inkling.toml b/providers/llmgateway-providers/models/together-ai/inkling.toml new file mode 100644 index 00000000000..1d6f4d10c7c --- /dev/null +++ b/providers/llmgateway-providers/models/together-ai/inkling.toml @@ -0,0 +1,16 @@ +base_model = "thinkingmachines/inkling" +name = "Inkling (Together AI)" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 1 +output = 4.05 +cache_read = 0.17 + +[limit] +context = 524_288 +output = 471_859 diff --git a/providers/llmgateway-providers/models/together-ai/muse-glimmer-30b.toml b/providers/llmgateway-providers/models/together-ai/muse-glimmer-30b.toml new file mode 100644 index 00000000000..99056ca7d36 --- /dev/null +++ b/providers/llmgateway-providers/models/together-ai/muse-glimmer-30b.toml @@ -0,0 +1,14 @@ +base_model = "meta/muse-glimmer-30b" +name = "Muse Glimmer 30B (Together AI)" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.35 +output = 1.5 +cache_read = 0.04 + +[limit] +output = 117_964 diff --git a/providers/llmgateway-providers/models/together-ai/qwen3.8-2.4t-a95b.toml b/providers/llmgateway-providers/models/together-ai/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..cd2eecc8a54 --- /dev/null +++ b/providers/llmgateway-providers/models/together-ai/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,15 @@ +base_model = "alibaba/qwen3.8-2.4t-a95b" +name = "Qwen3.8 2.4T A95B (Together AI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.25 + +[limit] +context = 1_010_000 +output = 909_000 From 397169451b894daf7605a3ef07341dad5ed8f062 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 17:24:42 +0000 Subject: [PATCH 312/392] chore(sync): update Kilo model catalog (#7764) Co-authored-by: opencode-agent[bot] --- .../kilo/models/anthropic/claude-opus-5.5.toml | 14 ++++++++++++++ .../kilo/models/~anthropic/claude-opus-latest.toml | 10 +++++----- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 6 +++--- 3 files changed, 22 insertions(+), 8 deletions(-) create mode 100644 providers/kilo/models/anthropic/claude-opus-5.5.toml diff --git a/providers/kilo/models/anthropic/claude-opus-5.5.toml b/providers/kilo/models/anthropic/claude-opus-5.5.toml new file mode 100644 index 00000000000..53b1a0d9fb2 --- /dev/null +++ b/providers/kilo/models/anthropic/claude-opus-5.5.toml @@ -0,0 +1,14 @@ +base_model = "anthropic/claude-opus-5-5" +description = "Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code..." +temperature = true +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/kilo/models/~anthropic/claude-opus-latest.toml b/providers/kilo/models/~anthropic/claude-opus-latest.toml index 7eae0b14770..709f3bb1f1b 100644 --- a/providers/kilo/models/~anthropic/claude-opus-latest.toml +++ b/providers/kilo/models/~anthropic/claude-opus-latest.toml @@ -12,13 +12,13 @@ open_weights = false [[reasoning_options]] type = "effort" -values = ["none", "low", "medium", "high", "xhigh", "max"] +values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 5 -output = 25 -cache_read = 0.5 -cache_write = 6.25 +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 [limit] context = 1_000_000 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 6565f5a8e9b..0d1469b0157 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.56364 -output = 1.69092 -cache_read = 0.018788 +input = 0.55572 +output = 1.66716 +cache_read = 0.018524 [limit] context = 1_024_000 From f310c935bf304ed5f2e2e4fc24e4c61a2802b6ee Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 17:24:48 +0000 Subject: [PATCH 313/392] chore(sync): update Venice model catalog (#7766) Co-authored-by: opencode-agent[bot] --- providers/venice/models/claude-opus-5-5.toml | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 providers/venice/models/claude-opus-5-5.toml diff --git a/providers/venice/models/claude-opus-5-5.toml b/providers/venice/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..ff60fa3a6ef --- /dev/null +++ b/providers/venice/models/claude-opus-5-5.toml @@ -0,0 +1,17 @@ +base_model = "anthropic/claude-opus-5-5" +description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" +release_date = "2026-09-18" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4.8 +output = 24 +cache_read = 0.24 +cache_write = 6 + +[modalities] +input = ["text", "image"] From c8129bec6c110bac7f769b784c19465f11e1a016 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 17:24:50 +0000 Subject: [PATCH 314/392] chore(sync): update OpenRouter model catalog (#7767) Co-authored-by: opencode-agent[bot] --- .../models/anthropic/claude-opus-5.5.toml | 14 ++++++++++++++ .../models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../models/~anthropic/claude-opus-latest.toml | 11 ++++------- .../models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 4 files changed, 25 insertions(+), 14 deletions(-) create mode 100644 providers/openrouter/models/anthropic/claude-opus-5.5.toml diff --git a/providers/openrouter/models/anthropic/claude-opus-5.5.toml b/providers/openrouter/models/anthropic/claude-opus-5.5.toml new file mode 100644 index 00000000000..70709244ff6 --- /dev/null +++ b/providers/openrouter/models/anthropic/claude-opus-5.5.toml @@ -0,0 +1,14 @@ +base_model = "anthropic/claude-opus-5-5" +description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" +temperature = true +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 8ce9eb4af6a..844ff795f34 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.927768 -output = 1.855536 -cache_read = 0.077314 +input = 0.915936 +output = 1.831872 +cache_read = 0.076328 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~anthropic/claude-opus-latest.toml b/providers/openrouter/models/~anthropic/claude-opus-latest.toml index f4dce5e6b4e..8057256dd8c 100644 --- a/providers/openrouter/models/~anthropic/claude-opus-latest.toml +++ b/providers/openrouter/models/~anthropic/claude-opus-latest.toml @@ -12,18 +12,15 @@ tool_call = true structured_output = true open_weights = false -[[reasoning_options]] -type = "toggle" - [[reasoning_options]] type = "effort" values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 5 -output = 25 -cache_read = 0.5 -cache_write = 6.25 +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index ff538327fb0..2a909029289 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.56364 -output = 1.69092 -cache_read = 0.018788 +input = 0.5544 +output = 1.6632 +cache_read = 0.01764 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] From 1bc95312f70441e339d5aa010ad656b9c7706952 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 17:25:07 +0000 Subject: [PATCH 315/392] chore(sync): update Merge Gateway model catalog (#7768) Co-authored-by: opencode-agent[bot] --- .../merge-gateway/models/anthropic/claude-opus-5-5.toml | 9 +++++++++ 1 file changed, 9 insertions(+) create mode 100644 providers/merge-gateway/models/anthropic/claude-opus-5-5.toml diff --git a/providers/merge-gateway/models/anthropic/claude-opus-5-5.toml b/providers/merge-gateway/models/anthropic/claude-opus-5-5.toml new file mode 100644 index 00000000000..5f5b832e6f8 --- /dev/null +++ b/providers/merge-gateway/models/anthropic/claude-opus-5-5.toml @@ -0,0 +1,9 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 From 1023d2a188ba0c65c3535fab3847d1ab45ed1669 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 18:32:21 +0000 Subject: [PATCH 316/392] chore(sync): update OpenRouter model catalog (#7774) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-pro.toml | 6 ++-- .../models/openai/gpt-6-luna-pro.toml | 36 +++++++++++++++++++ .../openrouter/models/openai/gpt-6-luna.toml | 36 +++++++++++++++++++ .../models/openai/gpt-6-sol-pro.toml | 36 +++++++++++++++++++ .../openrouter/models/openai/gpt-6-sol.toml | 36 +++++++++++++++++++ .../models/~deepseek/deepseek-pro-latest.toml | 6 ++-- .../models/~openai/gpt-luna-latest.toml | 16 ++++----- 7 files changed, 158 insertions(+), 14 deletions(-) create mode 100644 providers/openrouter/models/openai/gpt-6-luna-pro.toml create mode 100644 providers/openrouter/models/openai/gpt-6-luna.toml create mode 100644 providers/openrouter/models/openai/gpt-6-sol-pro.toml create mode 100644 providers/openrouter/models/openai/gpt-6-sol.toml diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 844ff795f34..33415e50e77 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.915936 -output = 1.831872 -cache_read = 0.076328 +input = 0.904104 +output = 1.808208 +cache_read = 0.075342 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/openai/gpt-6-luna-pro.toml b/providers/openrouter/models/openai/gpt-6-luna-pro.toml new file mode 100644 index 00000000000..f3f1dd66bc1 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-6-luna-pro.toml @@ -0,0 +1,36 @@ +name = "GPT-6 Luna Pro" +description = "Frontier GPT model for professional reasoning, coding, and multimodal work" +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-6-luna.toml b/providers/openrouter/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..d4a1dfaba03 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-6-luna.toml @@ -0,0 +1,36 @@ +name = "GPT-6 Luna" +description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-6-sol-pro.toml b/providers/openrouter/models/openai/gpt-6-sol-pro.toml new file mode 100644 index 00000000000..a34a53e3bb1 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-6-sol-pro.toml @@ -0,0 +1,36 @@ +name = "GPT-6 Sol Pro" +description = "Frontier GPT model for professional reasoning, coding, and multimodal work" +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-6-sol.toml b/providers/openrouter/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..0994e9d8867 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-6-sol.toml @@ -0,0 +1,36 @@ +name = "GPT-6 Sol" +description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 2a909029289..786d9386078 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5544 -output = 1.6632 -cache_read = 0.01764 +input = 0.54384 +output = 1.63152 +cache_read = 0.017304 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~openai/gpt-luna-latest.toml b/providers/openrouter/models/~openai/gpt-luna-latest.toml index e349af31375..5324cf22735 100644 --- a/providers/openrouter/models/~openai/gpt-luna-latest.toml +++ b/providers/openrouter/models/~openai/gpt-luna-latest.toml @@ -16,17 +16,17 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 0.2 -output = 1.2 -cache_read = 0.02 -cache_write = 0.25 +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 [[cost.tiers]] tier = { type = "context", size = 272_000 } -input = 0.4 -output = 1.8 -cache_read = 0.04 -cache_write = 0.5 +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 [limit] context = 1_050_000 From e5299b8996d2bdc4399001a28b91e42dd198898d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 18:32:24 +0000 Subject: [PATCH 317/392] chore(sync): update NanoGPT model catalog (#7773) Co-authored-by: opencode-agent[bot] --- .../models/anthropic/claude-opus-5.5.toml | 15 +++++++++++++++ 1 file changed, 15 insertions(+) create mode 100644 providers/nano-gpt/models/anthropic/claude-opus-5.5.toml diff --git a/providers/nano-gpt/models/anthropic/claude-opus-5.5.toml b/providers/nano-gpt/models/anthropic/claude-opus-5.5.toml new file mode 100644 index 00000000000..5948dbab4f8 --- /dev/null +++ b/providers/nano-gpt/models/anthropic/claude-opus-5.5.toml @@ -0,0 +1,15 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[limit] +input = 1_000_000 From 18a5916204b7422110327e62536b72d24784c9d7 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 18:32:39 +0000 Subject: [PATCH 318/392] chore(sync): update Kilo model catalog (#7775) Co-authored-by: opencode-agent[bot] --- .../kilo/models/openai/gpt-6-luna-pro.toml | 29 +++++++++++++++++++ providers/kilo/models/openai/gpt-6-luna.toml | 29 +++++++++++++++++++ .../kilo/models/openai/gpt-6-sol-pro.toml | 29 +++++++++++++++++++ providers/kilo/models/openai/gpt-6-sol.toml | 29 +++++++++++++++++++ .../models/~deepseek/deepseek-pro-latest.toml | 10 +++---- .../kilo/models/~openai/gpt-luna-latest.toml | 8 ++--- 6 files changed, 125 insertions(+), 9 deletions(-) create mode 100644 providers/kilo/models/openai/gpt-6-luna-pro.toml create mode 100644 providers/kilo/models/openai/gpt-6-luna.toml create mode 100644 providers/kilo/models/openai/gpt-6-sol-pro.toml create mode 100644 providers/kilo/models/openai/gpt-6-sol.toml diff --git a/providers/kilo/models/openai/gpt-6-luna-pro.toml b/providers/kilo/models/openai/gpt-6-luna-pro.toml new file mode 100644 index 00000000000..3e57329bcf1 --- /dev/null +++ b/providers/kilo/models/openai/gpt-6-luna-pro.toml @@ -0,0 +1,29 @@ +name = "OpenAI: GPT-6 Luna Pro" +description = "GPT-6 Luna Pro is the same underlying model as [GPT-6 Luna](https://openrouter.ai/openai/gpt-6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode" +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/kilo/models/openai/gpt-6-luna.toml b/providers/kilo/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..c6af65eba0a --- /dev/null +++ b/providers/kilo/models/openai/gpt-6-luna.toml @@ -0,0 +1,29 @@ +name = "OpenAI: GPT-6 Luna" +description = "GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic..." +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/kilo/models/openai/gpt-6-sol-pro.toml b/providers/kilo/models/openai/gpt-6-sol-pro.toml new file mode 100644 index 00000000000..f1d263d64fa --- /dev/null +++ b/providers/kilo/models/openai/gpt-6-sol-pro.toml @@ -0,0 +1,29 @@ +name = "OpenAI: GPT-6 Sol Pro" +description = "GPT-6 Sol Pro is the same underlying model as [GPT-6 Sol](https://openrouter.ai/openai/gpt-6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode" +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/kilo/models/openai/gpt-6-sol.toml b/providers/kilo/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..1fe6fe83662 --- /dev/null +++ b/providers/kilo/models/openai/gpt-6-sol.toml @@ -0,0 +1,29 @@ +name = "OpenAI: GPT-6 Sol" +description = "GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional..." +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 0d1469b0157..c3ebc81adc2 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.55572 -output = 1.66716 -cache_read = 0.018524 +input = 0.54384 +output = 1.63152 +cache_read = 0.017304 [limit] -context = 1_024_000 -output = 384_000 +context = 1_048_576 +output = 393_216 [modalities] input = ["text"] diff --git a/providers/kilo/models/~openai/gpt-luna-latest.toml b/providers/kilo/models/~openai/gpt-luna-latest.toml index 6f3028a0af2..d593f0aca35 100644 --- a/providers/kilo/models/~openai/gpt-luna-latest.toml +++ b/providers/kilo/models/~openai/gpt-luna-latest.toml @@ -15,10 +15,10 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 0.2 -output = 1.2 -cache_read = 0.02 -cache_write = 0.25 +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 [limit] context = 1_050_000 From 5473d5b9f0484328e119d5a0215b4105cd5002eb Mon Sep 17 00:00:00 2001 From: R44VC0RP <89211796+R44VC0RP@users.noreply.github.com> Date: Tue, 22 Sep 2026 18:33:19 +0000 Subject: [PATCH 319/392] feat(opencode): add Claude Opus 5.5 and GPT-6 Sol --- models/openai/gpt-6-sol.toml | 21 +++++++++++++++++++ .../opencode/models/claude-opus-5-5.toml | 12 +++++++++++ providers/opencode/models/gpt-6-sol.toml | 19 +++++++++++++++++ 3 files changed, 52 insertions(+) create mode 100644 models/openai/gpt-6-sol.toml create mode 100644 providers/opencode/models/claude-opus-5-5.toml create mode 100644 providers/opencode/models/gpt-6-sol.toml diff --git a/models/openai/gpt-6-sol.toml b/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..824b5d86bed --- /dev/null +++ b/models/openai/gpt-6-sol.toml @@ -0,0 +1,21 @@ +name = "GPT-6 Sol" +description = "OpenAI model for complex coding and agentic workflows" +family = "gpt-sol" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2026-04-20" +open_weights = false + +[limit] +context = 1_050_000 +input = 922_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/opencode/models/claude-opus-5-5.toml b/providers/opencode/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..e39cc456019 --- /dev/null +++ b/providers/opencode/models/claude-opus-5-5.toml @@ -0,0 +1,12 @@ +# Pricing: https://platform.claude.com/docs/en/models/opus-5-5/overview +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[provider] +npm = "@ai-sdk/anthropic" diff --git a/providers/opencode/models/gpt-6-sol.toml b/providers/opencode/models/gpt-6-sol.toml new file mode 100644 index 00000000000..eefd5c4f407 --- /dev/null +++ b/providers/opencode/models/gpt-6-sol.toml @@ -0,0 +1,19 @@ +# Pricing and controls: https://developers.openai.com/api/docs/models/gpt-6-sol +base_model = "openai/gpt-6-sol" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 + +[provider] +npm = "@ai-sdk/openai" From bfafa895d71dff8c2be417f85dc0d0713d6a27b7 Mon Sep 17 00:00:00 2001 From: R44VC0RP <89211796+R44VC0RP@users.noreply.github.com> Date: Tue, 22 Sep 2026 18:34:47 +0000 Subject: [PATCH 320/392] feat(opencode): add GPT-6 Luna --- models/openai/gpt-6-luna.toml | 21 +++++++++++++++++++++ providers/opencode/models/gpt-6-luna.toml | 19 +++++++++++++++++++ 2 files changed, 40 insertions(+) create mode 100644 models/openai/gpt-6-luna.toml create mode 100644 providers/opencode/models/gpt-6-luna.toml diff --git a/models/openai/gpt-6-luna.toml b/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..0a43442c0a5 --- /dev/null +++ b/models/openai/gpt-6-luna.toml @@ -0,0 +1,21 @@ +name = "GPT-6 Luna" +description = "OpenAI's most efficient model for focused, high-volume tasks" +family = "gpt-luna" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2026-05-18" +open_weights = false + +[limit] +context = 1_050_000 +input = 922_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/opencode/models/gpt-6-luna.toml b/providers/opencode/models/gpt-6-luna.toml new file mode 100644 index 00000000000..1263a99bdc4 --- /dev/null +++ b/providers/opencode/models/gpt-6-luna.toml @@ -0,0 +1,19 @@ +# Pricing and controls: https://developers.openai.com/api/docs/models/gpt-6-luna +base_model = "openai/gpt-6-luna" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[provider] +npm = "@ai-sdk/openai" From ef12935def70ea07ba6c98964e94aa15e7148768 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:34:49 -0500 Subject: [PATCH 321/392] chore(sync): update Vercel AI Gateway model catalog (#7765) * chore(sync): update Vercel AI Gateway model catalog * fix(vercel): add Claude Opus 5.5 effort options --------- Co-authored-by: opencode-agent[bot] Co-authored-by: rekram1-node --- .../vercel/models/anthropic/claude-opus-5.5-fast.toml | 9 +++++++++ providers/vercel/models/anthropic/claude-opus-5.5.toml | 8 ++++++++ 2 files changed, 17 insertions(+) create mode 100644 providers/vercel/models/anthropic/claude-opus-5.5-fast.toml create mode 100644 providers/vercel/models/anthropic/claude-opus-5.5.toml diff --git a/providers/vercel/models/anthropic/claude-opus-5.5-fast.toml b/providers/vercel/models/anthropic/claude-opus-5.5-fast.toml new file mode 100644 index 00000000000..9e516c8d185 --- /dev/null +++ b/providers/vercel/models/anthropic/claude-opus-5.5-fast.toml @@ -0,0 +1,9 @@ +base_model = "anthropic/claude-opus-5-5" +name = "Claude Opus 5.5 (Fast)" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 8 +output = 40 +cache_read = 0.4 +cache_write = 10 diff --git a/providers/vercel/models/anthropic/claude-opus-5.5.toml b/providers/vercel/models/anthropic/claude-opus-5.5.toml new file mode 100644 index 00000000000..8c3442bfe4b --- /dev/null +++ b/providers/vercel/models/anthropic/claude-opus-5.5.toml @@ -0,0 +1,8 @@ +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 From bc4f954e23b43c27c590319f1081928d4f22f2a2 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:35:23 -0500 Subject: [PATCH 322/392] chore(sync): update Requesty model catalog (#7769) Co-authored-by: opencode-agent[bot] --- providers/requesty/models/claude-opus-5-5.toml | 15 +++++++++++++++ .../requesty/models/claude-opus-5-5@eu.toml | 16 ++++++++++++++++ 2 files changed, 31 insertions(+) create mode 100644 providers/requesty/models/claude-opus-5-5.toml create mode 100644 providers/requesty/models/claude-opus-5-5@eu.toml diff --git a/providers/requesty/models/claude-opus-5-5.toml b/providers/requesty/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..1c8fca5fbe5 --- /dev/null +++ b/providers/requesty/models/claude-opus-5-5.toml @@ -0,0 +1,15 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/requesty/models/claude-opus-5-5@eu.toml b/providers/requesty/models/claude-opus-5-5@eu.toml new file mode 100644 index 00000000000..eaa19cb4ca0 --- /dev/null +++ b/providers/requesty/models/claude-opus-5-5@eu.toml @@ -0,0 +1,16 @@ +base_model = "anthropic/claude-opus-5-5" +name = "Claude Opus 5.5 (EU)" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 5.5 +output = 27.5 +cache_read = 0.55 +cache_write = 6.875 From 38ac62d15f1c70923d67fe4b622267037f9e5f49 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:39:47 -0500 Subject: [PATCH 323/392] fix: [missing-model] ofox: anthropic/claude-opus-5.5 (#7777) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- .../ofox/models/anthropic/claude-opus-5.5.toml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) create mode 100644 providers/ofox/models/anthropic/claude-opus-5.5.toml diff --git a/providers/ofox/models/anthropic/claude-opus-5.5.toml b/providers/ofox/models/anthropic/claude-opus-5.5.toml new file mode 100644 index 00000000000..23287398bb0 --- /dev/null +++ b/providers/ofox/models/anthropic/claude-opus-5.5.toml @@ -0,0 +1,16 @@ +# Sources: +# https://ofox.ai/models/anthropic/claude-opus-5.5 +# https://platform.claude.com/docs/en/models/opus-5-5/overview +# https://www.anthropic.com/claude-opus-5-5 +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[provider] +npm = "@ai-sdk/anthropic" +api = "https://api.ofox.ai/anthropic/v1" From 5c476477d041eeb90e226971911ea0f8c47d3dee Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:40:36 -0500 Subject: [PATCH 324/392] fix: DeepSeek provider: mark retired aliases `deepseek-v4-flash` and `deepseek-v4-flash-vision-exp` as `status = deprecated` (#7727) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- providers/deepseek/models/deepseek-v4-flash-vision-exp.toml | 3 +-- providers/deepseek/models/deepseek-v4-flash.toml | 3 +-- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/providers/deepseek/models/deepseek-v4-flash-vision-exp.toml b/providers/deepseek/models/deepseek-v4-flash-vision-exp.toml index e1227c7a467..aa1da80696b 100644 --- a/providers/deepseek/models/deepseek-v4-flash-vision-exp.toml +++ b/providers/deepseek/models/deepseek-v4-flash-vision-exp.toml @@ -6,8 +6,7 @@ # https://api-docs.deepseek.com/quick_start/pricing (accessed 2026-09-10) base_model = "deepseek/deepseek-v4.1-flash" name = "DeepSeek V4 Flash Vision Exp" -## mark as deprecated soon -# status = "deprecated" +status = "deprecated" [[reasoning_options]] type = "toggle" diff --git a/providers/deepseek/models/deepseek-v4-flash.toml b/providers/deepseek/models/deepseek-v4-flash.toml index ae7ff2fc79b..3ebe57414da 100644 --- a/providers/deepseek/models/deepseek-v4-flash.toml +++ b/providers/deepseek/models/deepseek-v4-flash.toml @@ -6,8 +6,7 @@ # https://api-docs.deepseek.com/quick_start/pricing (accessed 2026-09-10) base_model = "deepseek/deepseek-v4.1-flash" name = "DeepSeek V4 Flash" -## mark as deprecated soon -# status = "deprecated" +status = "deprecated" [[reasoning_options]] type = "toggle" From b165bd596d27e3fa271e5b859753939fc8962cb2 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:54:44 -0500 Subject: [PATCH 325/392] chore(sync): update Cloudflare AI Gateway model catalog (#7776) Co-authored-by: opencode-agent[bot] --- .../models/xai/grok-4.7.toml | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) create mode 100644 providers/cloudflare-ai-gateway/models/xai/grok-4.7.toml diff --git a/providers/cloudflare-ai-gateway/models/xai/grok-4.7.toml b/providers/cloudflare-ai-gateway/models/xai/grok-4.7.toml new file mode 100644 index 00000000000..48df4f8248b --- /dev/null +++ b/providers/cloudflare-ai-gateway/models/xai/grok-4.7.toml @@ -0,0 +1,19 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 + +[limit] +context = 500_000 From 67cb71f2b5ea59a6d076a70a509cdbe663e152c7 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 14:10:50 -0500 Subject: [PATCH 326/392] feat: add GPT-6 Sol and Luna (#7781) Co-authored-by: rekram1-node --- models/openai/gpt-6-luna.toml | 21 +++++++++++++++++++++ models/openai/gpt-6-sol.toml | 21 +++++++++++++++++++++ providers/azure/models/gpt-6-luna.toml | 17 +++++++++++++++++ providers/azure/models/gpt-6-sol.toml | 17 +++++++++++++++++ providers/openai/models/gpt-6-luna.toml | 24 ++++++++++++++++++++++++ providers/openai/models/gpt-6-sol.toml | 24 ++++++++++++++++++++++++ 6 files changed, 124 insertions(+) create mode 100644 models/openai/gpt-6-luna.toml create mode 100644 models/openai/gpt-6-sol.toml create mode 100644 providers/azure/models/gpt-6-luna.toml create mode 100644 providers/azure/models/gpt-6-sol.toml create mode 100644 providers/openai/models/gpt-6-luna.toml create mode 100644 providers/openai/models/gpt-6-sol.toml diff --git a/models/openai/gpt-6-luna.toml b/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..0a43442c0a5 --- /dev/null +++ b/models/openai/gpt-6-luna.toml @@ -0,0 +1,21 @@ +name = "GPT-6 Luna" +description = "OpenAI's most efficient model for focused, high-volume tasks" +family = "gpt-luna" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2026-05-18" +open_weights = false + +[limit] +context = 1_050_000 +input = 922_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/models/openai/gpt-6-sol.toml b/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..824b5d86bed --- /dev/null +++ b/models/openai/gpt-6-sol.toml @@ -0,0 +1,21 @@ +name = "GPT-6 Sol" +description = "OpenAI model for complex coding and agentic workflows" +family = "gpt-sol" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2026-04-20" +open_weights = false + +[limit] +context = 1_050_000 +input = 922_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/azure/models/gpt-6-luna.toml b/providers/azure/models/gpt-6-luna.toml new file mode 100644 index 00000000000..c105b8ef975 --- /dev/null +++ b/providers/azure/models/gpt-6-luna.toml @@ -0,0 +1,17 @@ +# Availability: https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure-region-availability +# Rates: https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/ +base_model = "openai/gpt-6-luna" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.10 +output = 0.50 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.20 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 diff --git a/providers/azure/models/gpt-6-sol.toml b/providers/azure/models/gpt-6-sol.toml new file mode 100644 index 00000000000..e0ad0d723d0 --- /dev/null +++ b/providers/azure/models/gpt-6-sol.toml @@ -0,0 +1,17 @@ +# Availability: https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure-region-availability +# Rates: https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/ +base_model = "openai/gpt-6-sol" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 2.00 +output = 10.00 +cache_read = 0.20 +cache_write = 2.50 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4.00 +output = 15.00 +cache_read = 0.40 +cache_write = 5.00 diff --git a/providers/openai/models/gpt-6-luna.toml b/providers/openai/models/gpt-6-luna.toml new file mode 100644 index 00000000000..3a132f7fabb --- /dev/null +++ b/providers/openai/models/gpt-6-luna.toml @@ -0,0 +1,24 @@ +# Pricing: https://developers.openai.com/api/docs/models/gpt-6-luna +# Reasoning and modes: https://developers.openai.com/api/docs/guides/latest-model?model=gpt-6-luna +base_model = "openai/gpt-6-luna" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.10 +output = 0.50 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.20 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[experimental.modes.fast] +cost = { input = 0.20, output = 1.00, cache_read = 0.02, cache_write = 0.25 } +provider = { body = { service_tier = "priority" } } + +[experimental.modes.pro] +provider = { body = { reasoning = { mode = "pro" } } } diff --git a/providers/openai/models/gpt-6-sol.toml b/providers/openai/models/gpt-6-sol.toml new file mode 100644 index 00000000000..e141e9653ad --- /dev/null +++ b/providers/openai/models/gpt-6-sol.toml @@ -0,0 +1,24 @@ +# Pricing: https://developers.openai.com/api/docs/models/gpt-6-sol +# Reasoning and modes: https://developers.openai.com/api/docs/guides/latest-model?model=gpt-6-sol +base_model = "openai/gpt-6-sol" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 2.00 +output = 10.00 +cache_read = 0.20 +cache_write = 2.50 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4.00 +output = 15.00 +cache_read = 0.40 +cache_write = 5.00 + +[experimental.modes.fast] +cost = { input = 4.00, output = 20.00, cache_read = 0.40, cache_write = 5.00 } +provider = { body = { service_tier = "priority" } } + +[experimental.modes.pro] +provider = { body = { reasoning = { mode = "pro" } } } From 921b79fcb0452721f464a79cca6e229475bbb277 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 14:11:38 -0500 Subject: [PATCH 327/392] chore(sync): update CoreWeave model catalog (#7762) Co-authored-by: opencode-agent[bot] --- .../wandb/models/google/gemma-4-26B-A4B-it.toml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 providers/wandb/models/google/gemma-4-26B-A4B-it.toml diff --git a/providers/wandb/models/google/gemma-4-26B-A4B-it.toml b/providers/wandb/models/google/gemma-4-26B-A4B-it.toml new file mode 100644 index 00000000000..8014a72bac7 --- /dev/null +++ b/providers/wandb/models/google/gemma-4-26B-A4B-it.toml @@ -0,0 +1,14 @@ +base_model = "google/gemma-4-26b-a4b-it" +name = "Gemma 4 26B A4B" +description = "Gemma 4 26B A4B is a multimodal MoE model with LoRA support and function calling for agentic workflows." + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.1 +output = 0.3 +cache_read = 0.05 + +[limit] +output = 262_144 From 9d18788bfe315796b5902a098b91359fcb7300dd Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 19:23:17 +0000 Subject: [PATCH 328/392] chore(sync): update Kilo model catalog (#7787) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/openai/gpt-6-luna.toml | 19 +------------------ providers/kilo/models/openai/gpt-6-sol.toml | 19 +------------------ .../models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- .../~deepseek/deepseek-v4-flash-latest.toml | 2 +- .../kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- 5 files changed, 10 insertions(+), 44 deletions(-) diff --git a/providers/kilo/models/openai/gpt-6-luna.toml b/providers/kilo/models/openai/gpt-6-luna.toml index c6af65eba0a..9e9940bc126 100644 --- a/providers/kilo/models/openai/gpt-6-luna.toml +++ b/providers/kilo/models/openai/gpt-6-luna.toml @@ -1,14 +1,5 @@ -name = "OpenAI: GPT-6 Luna" +base_model = "openai/gpt-6-luna" description = "GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic..." -family = "gpt" -release_date = "2026-09-22" -last_updated = "2026-09-22" -attachment = true -reasoning = true -temperature = false -tool_call = true -structured_output = true -open_weights = false [[reasoning_options]] type = "effort" @@ -19,11 +10,3 @@ input = 0.1 output = 0.5 cache_read = 0.01 cache_write = 0.125 - -[limit] -context = 1_050_000 -output = 128_000 - -[modalities] -input = ["pdf", "image", "text"] -output = ["text"] diff --git a/providers/kilo/models/openai/gpt-6-sol.toml b/providers/kilo/models/openai/gpt-6-sol.toml index 1fe6fe83662..5960d039080 100644 --- a/providers/kilo/models/openai/gpt-6-sol.toml +++ b/providers/kilo/models/openai/gpt-6-sol.toml @@ -1,14 +1,5 @@ -name = "OpenAI: GPT-6 Sol" +base_model = "openai/gpt-6-sol" description = "GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional..." -family = "gpt" -release_date = "2026-09-22" -last_updated = "2026-09-22" -attachment = true -reasoning = true -temperature = false -tool_call = true -structured_output = true -open_weights = false [[reasoning_options]] type = "effort" @@ -19,11 +10,3 @@ input = 2 output = 10 cache_read = 0.2 cache_write = 2.5 - -[limit] -context = 1_050_000 -output = 128_000 - -[modalities] -input = ["pdf", "image", "text"] -output = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index c3ebc81adc2..7666171face 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.54384 -output = 1.63152 -cache_read = 0.017304 +input = 0.4 +output = 4.3 +cache_read = 0.033 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 6200a01f4d0..0412502ff5d 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -16,7 +16,7 @@ values = ["none", "low", "high", "max"] [cost] input = 0.03 -output = 1 +output = 0.8 cache_read = 0.008 [limit] diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index afff7105d92..366b401a83c 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.5 -output = 7.5 -cache_read = 0.15 +input = 1.4989 +output = 10.758 +cache_read = 0.3 [limit] context = 1_048_576 From d4374392a8ba4fb01ab330a1ad96a260cf78b62c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 19:23:20 +0000 Subject: [PATCH 329/392] chore(sync): update Merge Gateway model catalog (#7784) Co-authored-by: opencode-agent[bot] --- .../models/openai/gpt-6-astra.toml | 1 + .../merge-gateway/models/openai/gpt-6-luna.toml | 17 +++++++++++++++++ .../merge-gateway/models/openai/gpt-6-sol.toml | 17 +++++++++++++++++ 3 files changed, 35 insertions(+) create mode 100644 providers/merge-gateway/models/openai/gpt-6-luna.toml create mode 100644 providers/merge-gateway/models/openai/gpt-6-sol.toml diff --git a/providers/merge-gateway/models/openai/gpt-6-astra.toml b/providers/merge-gateway/models/openai/gpt-6-astra.toml index 223b48abe9e..da7cc6ca051 100644 --- a/providers/merge-gateway/models/openai/gpt-6-astra.toml +++ b/providers/merge-gateway/models/openai/gpt-6-astra.toml @@ -8,6 +8,7 @@ values = ["low", "medium", "high", "xhigh", "max"] input = 10 output = 50 cache_read = 1 +cache_write = 12.5 [modalities] input = ["text", "image"] diff --git a/providers/merge-gateway/models/openai/gpt-6-luna.toml b/providers/merge-gateway/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..7b7f12548e8 --- /dev/null +++ b/providers/merge-gateway/models/openai/gpt-6-luna.toml @@ -0,0 +1,17 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[modalities] +input = ["text", "image"] diff --git a/providers/merge-gateway/models/openai/gpt-6-sol.toml b/providers/merge-gateway/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..e0b3cd3f91b --- /dev/null +++ b/providers/merge-gateway/models/openai/gpt-6-sol.toml @@ -0,0 +1,17 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[modalities] +input = ["text", "image"] From 74e8b5db5ab2c3d0fd136a9a202fb7429b3d5c07 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 19:23:22 +0000 Subject: [PATCH 330/392] chore(sync): update DevPass (LLM Gateway) model catalog (#7790) Co-authored-by: opencode-agent[bot] --- providers/llmgateway/models/claude-opus-5-5.toml | 11 +++++++++++ 1 file changed, 11 insertions(+) create mode 100644 providers/llmgateway/models/claude-opus-5-5.toml diff --git a/providers/llmgateway/models/claude-opus-5-5.toml b/providers/llmgateway/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..85ea7179304 --- /dev/null +++ b/providers/llmgateway/models/claude-opus-5-5.toml @@ -0,0 +1,11 @@ +base_model = "anthropic/claude-opus-5-5" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 From bf1fc1b21f2e22a03e77db8262d6297d71756255 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 19:23:26 +0000 Subject: [PATCH 331/392] chore(sync): update NanoGPT model catalog (#7788) Co-authored-by: opencode-agent[bot] --- providers/nano-gpt/models/lumen-stealth.toml | 23 +++++++++++++++ .../nano-gpt/models/openai/gpt-6-sol-pro.toml | 29 +++++++++++++++++++ .../nano-gpt/models/openai/gpt-6-sol.toml | 15 ++++++++++ 3 files changed, 67 insertions(+) create mode 100644 providers/nano-gpt/models/lumen-stealth.toml create mode 100644 providers/nano-gpt/models/openai/gpt-6-sol-pro.toml create mode 100644 providers/nano-gpt/models/openai/gpt-6-sol.toml diff --git a/providers/nano-gpt/models/lumen-stealth.toml b/providers/nano-gpt/models/lumen-stealth.toml new file mode 100644 index 00000000000..13c7ea58999 --- /dev/null +++ b/providers/nano-gpt/models/lumen-stealth.toml @@ -0,0 +1,23 @@ +name = "Lumen Stealth" +description = "Experimental multimodal model focused on reasoning, creative writing, roleplay, and agentic workflows. Available temporarily for evaluation ahead of public release. During this evaluation, prompts and responses are logged and may be reviewed to evaluate and improve the model. Do not send sensitive or confidential information." +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +tool_call = false +structured_output = false +open_weights = false +reasoning_options = [] + +[cost] +input = 0.05 +output = 0 + +[limit] +context = 200_000 +input = 200_000 +output = 16_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/nano-gpt/models/openai/gpt-6-sol-pro.toml b/providers/nano-gpt/models/openai/gpt-6-sol-pro.toml new file mode 100644 index 00000000000..daa2b93f111 --- /dev/null +++ b/providers/nano-gpt/models/openai/gpt-6-sol-pro.toml @@ -0,0 +1,29 @@ +name = "GPT 6 Sol Pro" +description = "GPT-6 Sol Pro uses the same underlying model as GPT-6 Sol with Pro reasoning mode enabled for higher-quality responses on complex tasks." +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[limit] +context = 1_050_000 +input = 1_050_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/nano-gpt/models/openai/gpt-6-sol.toml b/providers/nano-gpt/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..109a0a67fce --- /dev/null +++ b/providers/nano-gpt/models/openai/gpt-6-sol.toml @@ -0,0 +1,15 @@ +base_model = "openai/gpt-6-sol" +name = "GPT 6 Sol" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[limit] +input = 1_050_000 From 7120b4559017612dd0db8e2babe965bbdfc09021 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 19:23:33 +0000 Subject: [PATCH 332/392] chore(sync): update OpenRouter model catalog (#7791) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/openai/gpt-6-luna.toml | 19 +------------------ .../openrouter/models/openai/gpt-6-sol.toml | 19 +------------------ .../models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- .../~deepseek/deepseek-v4-flash-latest.toml | 2 +- .../models/~moonshotai/kimi-latest.toml | 6 +++--- 6 files changed, 13 insertions(+), 47 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 33415e50e77..74f20945d6e 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.904104 -output = 1.808208 -cache_read = 0.075342 +input = 0.895578 +output = 1.791156 +cache_read = 0.074632 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/openai/gpt-6-luna.toml b/providers/openrouter/models/openai/gpt-6-luna.toml index d4a1dfaba03..be8a0fb9e1b 100644 --- a/providers/openrouter/models/openai/gpt-6-luna.toml +++ b/providers/openrouter/models/openai/gpt-6-luna.toml @@ -1,14 +1,5 @@ -name = "GPT-6 Luna" +base_model = "openai/gpt-6-luna" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" -family = "gpt" -release_date = "2026-09-22" -last_updated = "2026-09-22" -attachment = true -reasoning = true -temperature = false -tool_call = true -structured_output = true -open_weights = false [[reasoning_options]] type = "effort" @@ -26,11 +17,3 @@ input = 0.2 output = 0.75 cache_read = 0.02 cache_write = 0.25 - -[limit] -context = 1_050_000 -output = 128_000 - -[modalities] -input = ["pdf", "image", "text"] -output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-6-sol.toml b/providers/openrouter/models/openai/gpt-6-sol.toml index 0994e9d8867..83f598548a3 100644 --- a/providers/openrouter/models/openai/gpt-6-sol.toml +++ b/providers/openrouter/models/openai/gpt-6-sol.toml @@ -1,14 +1,5 @@ -name = "GPT-6 Sol" +base_model = "openai/gpt-6-sol" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" -family = "gpt" -release_date = "2026-09-22" -last_updated = "2026-09-22" -attachment = true -reasoning = true -temperature = false -tool_call = true -structured_output = true -open_weights = false [[reasoning_options]] type = "effort" @@ -26,11 +17,3 @@ input = 4 output = 15 cache_read = 0.4 cache_write = 5 - -[limit] -context = 1_050_000 -output = 128_000 - -[modalities] -input = ["pdf", "image", "text"] -output = ["text"] diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 786d9386078..3112dca3caa 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.54384 -output = 1.63152 -cache_read = 0.017304 +input = 0.4 +output = 4.3 +cache_read = 0.033 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 8c5f6c380c1..6c267e8f600 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -21,7 +21,7 @@ values = ["low", "high", "max"] [cost] input = 0.03 -output = 1 +output = 0.8 cache_read = 0.008 [limit] diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 0bb8650abd5..0aebd435d56 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.5 -output = 7.5 -cache_read = 0.15 +input = 1.4989 +output = 10.758 +cache_read = 0.3 [limit] context = 1_048_576 From 672754931632e7d652e517606bdf0f1804fbc4ba Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 19:23:39 +0000 Subject: [PATCH 333/392] chore(sync): update LLM Gateway model catalog (#7786) Co-authored-by: opencode-agent[bot] --- .../models/anthropic/claude-opus-5-5.toml | 13 +++++++++++++ 1 file changed, 13 insertions(+) create mode 100644 providers/llmgateway-providers/models/anthropic/claude-opus-5-5.toml diff --git a/providers/llmgateway-providers/models/anthropic/claude-opus-5-5.toml b/providers/llmgateway-providers/models/anthropic/claude-opus-5-5.toml new file mode 100644 index 00000000000..090b539ee3e --- /dev/null +++ b/providers/llmgateway-providers/models/anthropic/claude-opus-5-5.toml @@ -0,0 +1,13 @@ +base_model = "anthropic/claude-opus-5-5" +name = "Claude Opus 5.5 (Anthropic)" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 From 3d6dcd13b178feed787d98a6c3e5efb9cc4f4c9d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 14:59:55 -0500 Subject: [PATCH 334/392] chore(sync): update Vercel AI Gateway model catalog (#7785) * chore(sync): update Vercel AI Gateway model catalog * fix(vercel): list GPT-6 reasoning efforts --------- Co-authored-by: opencode-agent[bot] Co-authored-by: rekram1-node --- .../vercel/models/openai/gpt-6-luna-fast.toml | 19 +++++++++++++++++++ .../vercel/models/openai/gpt-6-luna.toml | 18 ++++++++++++++++++ .../vercel/models/openai/gpt-6-sol-fast.toml | 19 +++++++++++++++++++ providers/vercel/models/openai/gpt-6-sol.toml | 18 ++++++++++++++++++ 4 files changed, 74 insertions(+) create mode 100644 providers/vercel/models/openai/gpt-6-luna-fast.toml create mode 100644 providers/vercel/models/openai/gpt-6-luna.toml create mode 100644 providers/vercel/models/openai/gpt-6-sol-fast.toml create mode 100644 providers/vercel/models/openai/gpt-6-sol.toml diff --git a/providers/vercel/models/openai/gpt-6-luna-fast.toml b/providers/vercel/models/openai/gpt-6-luna-fast.toml new file mode 100644 index 00000000000..5087080071f --- /dev/null +++ b/providers/vercel/models/openai/gpt-6-luna-fast.toml @@ -0,0 +1,19 @@ +base_model = "openai/gpt-6-luna" +name = "GPT-6 Luna (Fast)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.2 +output = 1 +cache_read = 0.02 +cache_write = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 0.4 +output = 1.5 +cache_read = 0.04 +cache_write = 0.5 diff --git a/providers/vercel/models/openai/gpt-6-luna.toml b/providers/vercel/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..c9115379e71 --- /dev/null +++ b/providers/vercel/models/openai/gpt-6-luna.toml @@ -0,0 +1,18 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 diff --git a/providers/vercel/models/openai/gpt-6-sol-fast.toml b/providers/vercel/models/openai/gpt-6-sol-fast.toml new file mode 100644 index 00000000000..8b33bf6106a --- /dev/null +++ b/providers/vercel/models/openai/gpt-6-sol-fast.toml @@ -0,0 +1,19 @@ +base_model = "openai/gpt-6-sol" +name = "GPT-6 Sol (Fast)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.4 +cache_write = 5 + +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 8 +output = 30 +cache_read = 0.8 +cache_write = 10 diff --git a/providers/vercel/models/openai/gpt-6-sol.toml b/providers/vercel/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..39b8ede6517 --- /dev/null +++ b/providers/vercel/models/openai/gpt-6-sol.toml @@ -0,0 +1,18 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 From 38ed0d71b6573ae3423d0a944a40ff50b1278a9d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 20:25:49 +0000 Subject: [PATCH 335/392] chore(sync): update DevPass (LLM Gateway) model catalog (#7798) Co-authored-by: opencode-agent[bot] --- providers/llmgateway/models/gpt-6-luna.toml | 11 +++++++++++ providers/llmgateway/models/gpt-6-sol.toml | 11 +++++++++++ 2 files changed, 22 insertions(+) create mode 100644 providers/llmgateway/models/gpt-6-luna.toml create mode 100644 providers/llmgateway/models/gpt-6-sol.toml diff --git a/providers/llmgateway/models/gpt-6-luna.toml b/providers/llmgateway/models/gpt-6-luna.toml new file mode 100644 index 00000000000..f058ea629ce --- /dev/null +++ b/providers/llmgateway/models/gpt-6-luna.toml @@ -0,0 +1,11 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 diff --git a/providers/llmgateway/models/gpt-6-sol.toml b/providers/llmgateway/models/gpt-6-sol.toml new file mode 100644 index 00000000000..e4694368a0c --- /dev/null +++ b/providers/llmgateway/models/gpt-6-sol.toml @@ -0,0 +1,11 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 From 60b05473b5f603bb3c3e5803b4fc46d525a39b50 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 20:25:54 +0000 Subject: [PATCH 336/392] chore(sync): update LLM Gateway model catalog (#7799) Co-authored-by: opencode-agent[bot] --- .../models/openai/gpt-6-luna.toml | 12 ++++++++++++ .../models/openai/gpt-6-sol.toml | 12 ++++++++++++ 2 files changed, 24 insertions(+) create mode 100644 providers/llmgateway-providers/models/openai/gpt-6-luna.toml create mode 100644 providers/llmgateway-providers/models/openai/gpt-6-sol.toml diff --git a/providers/llmgateway-providers/models/openai/gpt-6-luna.toml b/providers/llmgateway-providers/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..6383b613669 --- /dev/null +++ b/providers/llmgateway-providers/models/openai/gpt-6-luna.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-6-luna" +name = "GPT-6 Luna (OpenAI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 diff --git a/providers/llmgateway-providers/models/openai/gpt-6-sol.toml b/providers/llmgateway-providers/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..88bcf456342 --- /dev/null +++ b/providers/llmgateway-providers/models/openai/gpt-6-sol.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-6-sol" +name = "GPT-6 Sol (OpenAI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 From 95bc3dad0012393187f48814cb1e09f95534c756 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 20:25:57 +0000 Subject: [PATCH 337/392] chore(sync): update NanoGPT model catalog (#7802) Co-authored-by: opencode-agent[bot] --- .../models/{ => nano}/lumen-stealth.toml | 0 .../models/openai/gpt-6-luna-pro.toml | 29 +++++++++++++++++++ .../nano-gpt/models/openai/gpt-6-luna.toml | 15 ++++++++++ .../nano-gpt/models/openai/gpt-6-sol-pro.toml | 2 +- .../nano-gpt/models/openai/gpt-6-sol.toml | 2 +- 5 files changed, 46 insertions(+), 2 deletions(-) rename providers/nano-gpt/models/{ => nano}/lumen-stealth.toml (100%) create mode 100644 providers/nano-gpt/models/openai/gpt-6-luna-pro.toml create mode 100644 providers/nano-gpt/models/openai/gpt-6-luna.toml diff --git a/providers/nano-gpt/models/lumen-stealth.toml b/providers/nano-gpt/models/nano/lumen-stealth.toml similarity index 100% rename from providers/nano-gpt/models/lumen-stealth.toml rename to providers/nano-gpt/models/nano/lumen-stealth.toml diff --git a/providers/nano-gpt/models/openai/gpt-6-luna-pro.toml b/providers/nano-gpt/models/openai/gpt-6-luna-pro.toml new file mode 100644 index 00000000000..cc8bc7d726f --- /dev/null +++ b/providers/nano-gpt/models/openai/gpt-6-luna-pro.toml @@ -0,0 +1,29 @@ +name = "GPT 6 Luna Pro" +description = "GPT-6 Luna Pro uses the same underlying model as GPT-6 Luna with Pro reasoning mode enabled for higher-quality responses on complex tasks." +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.05 +output = 0.25 +cache_read = 0.005 +cache_write = 0.0625 + +[limit] +context = 1_050_000 +input = 1_050_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/nano-gpt/models/openai/gpt-6-luna.toml b/providers/nano-gpt/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..23f8b252eec --- /dev/null +++ b/providers/nano-gpt/models/openai/gpt-6-luna.toml @@ -0,0 +1,15 @@ +base_model = "openai/gpt-6-luna" +name = "GPT 6 Luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.05 +output = 0.25 +cache_read = 0.005 +cache_write = 0.0625 + +[limit] +input = 1_050_000 diff --git a/providers/nano-gpt/models/openai/gpt-6-sol-pro.toml b/providers/nano-gpt/models/openai/gpt-6-sol-pro.toml index daa2b93f111..1720ef7ae10 100644 --- a/providers/nano-gpt/models/openai/gpt-6-sol-pro.toml +++ b/providers/nano-gpt/models/openai/gpt-6-sol-pro.toml @@ -11,7 +11,7 @@ open_weights = false [[reasoning_options]] type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] +values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] input = 2 diff --git a/providers/nano-gpt/models/openai/gpt-6-sol.toml b/providers/nano-gpt/models/openai/gpt-6-sol.toml index 109a0a67fce..ee3ffe25af3 100644 --- a/providers/nano-gpt/models/openai/gpt-6-sol.toml +++ b/providers/nano-gpt/models/openai/gpt-6-sol.toml @@ -3,7 +3,7 @@ name = "GPT 6 Sol" [[reasoning_options]] type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] +values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] input = 2 From c1785c7e6c7be95e3067611da0458fd38644bbb4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 20:26:00 +0000 Subject: [PATCH 338/392] chore(sync): update Kilo model catalog (#7801) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/aion-labs/aion-2.0.toml | 2 +- providers/kilo/models/aion-labs/aion-3.0-mini.toml | 2 +- providers/kilo/models/aion-labs/aion-3.0.toml | 2 +- .../kilo/models/deepseek/deepseek-v4.1-flash.toml | 1 + providers/kilo/models/qwen/qwen3.8-omni-flash.toml | 13 +++++++++++++ .../models/~deepseek/deepseek-flash-latest.toml | 6 +++--- .../kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 7 files changed, 24 insertions(+), 10 deletions(-) create mode 100644 providers/kilo/models/qwen/qwen3.8-omni-flash.toml diff --git a/providers/kilo/models/aion-labs/aion-2.0.toml b/providers/kilo/models/aion-labs/aion-2.0.toml index 46cc9f272f7..4bbbba9076b 100644 --- a/providers/kilo/models/aion-labs/aion-2.0.toml +++ b/providers/kilo/models/aion-labs/aion-2.0.toml @@ -19,7 +19,7 @@ output = 1.6 cache_read = 0.2 [limit] -context = 1_048_576 +context = 131_072 output = 32_768 [modalities] diff --git a/providers/kilo/models/aion-labs/aion-3.0-mini.toml b/providers/kilo/models/aion-labs/aion-3.0-mini.toml index 84f2979d62c..4987c065dc7 100644 --- a/providers/kilo/models/aion-labs/aion-3.0-mini.toml +++ b/providers/kilo/models/aion-labs/aion-3.0-mini.toml @@ -19,7 +19,7 @@ output = 1.4 cache_read = 0.18 [limit] -context = 1_048_576 +context = 131_072 output = 32_768 [modalities] diff --git a/providers/kilo/models/aion-labs/aion-3.0.toml b/providers/kilo/models/aion-labs/aion-3.0.toml index ac0a56ff9f9..caa24ee8921 100644 --- a/providers/kilo/models/aion-labs/aion-3.0.toml +++ b/providers/kilo/models/aion-labs/aion-3.0.toml @@ -19,7 +19,7 @@ output = 6 cache_read = 0.75 [limit] -context = 1_048_576 +context = 131_072 output = 32_768 [modalities] diff --git a/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml b/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml index 2d43303aad0..d755c215bcf 100644 --- a/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml @@ -12,3 +12,4 @@ cache_read = 0.006 [limit] context = 1_048_576 +output = 943_718 diff --git a/providers/kilo/models/qwen/qwen3.8-omni-flash.toml b/providers/kilo/models/qwen/qwen3.8-omni-flash.toml new file mode 100644 index 00000000000..574800249bc --- /dev/null +++ b/providers/kilo/models/qwen/qwen3.8-omni-flash.toml @@ -0,0 +1,13 @@ +base_model = "alibaba/qwen3.8-omni-flash" +description = "Qwen3.8 Omni Flash is an omni-modal reasoning model from Alibaba, the first Qwen model built around agentic capabilities with native audio-video understanding. It is suited for audio-video analysis and summarization,..." +temperature = true +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.15 +output = 0.47 +cache_read = 0.016 diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 2193a017b17..9e869daead9 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.12 -output = 0.48 -cache_read = 0.0036 +input = 0.1 +output = 0.8 +cache_read = 0.1 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 7666171face..6f1d7199aea 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.4 -output = 4.3 -cache_read = 0.033 +input = 0.39996 +output = 1.19988 +cache_read = 0.012726 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] From dadd2f0d0b534dc1652296a4f325287c2814ef36 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 20:26:13 +0000 Subject: [PATCH 339/392] chore(sync): update OpenRouter model catalog (#7800) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/aion-labs/aion-2.0.toml | 2 +- .../models/aion-labs/aion-3.0-mini.toml | 2 +- .../openrouter/models/aion-labs/aion-3.0.toml | 2 +- .../models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../models/deepseek/deepseek-v4.1-flash.toml | 7 ++++--- .../models/qwen/qwen3.8-omni-flash.toml | 16 ++++++++++++++++ .../models/~deepseek/deepseek-flash-latest.toml | 6 +++--- .../models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- 8 files changed, 33 insertions(+), 16 deletions(-) create mode 100644 providers/openrouter/models/qwen/qwen3.8-omni-flash.toml diff --git a/providers/openrouter/models/aion-labs/aion-2.0.toml b/providers/openrouter/models/aion-labs/aion-2.0.toml index 8e499e073cf..b54b64899f3 100644 --- a/providers/openrouter/models/aion-labs/aion-2.0.toml +++ b/providers/openrouter/models/aion-labs/aion-2.0.toml @@ -16,7 +16,7 @@ output = 1.6 cache_read = 0.2 [limit] -context = 1_048_576 +context = 131_072 output = 32_768 [modalities] diff --git a/providers/openrouter/models/aion-labs/aion-3.0-mini.toml b/providers/openrouter/models/aion-labs/aion-3.0-mini.toml index 33c5d843701..af968c32457 100644 --- a/providers/openrouter/models/aion-labs/aion-3.0-mini.toml +++ b/providers/openrouter/models/aion-labs/aion-3.0-mini.toml @@ -16,7 +16,7 @@ output = 1.4 cache_read = 0.18 [limit] -context = 1_048_576 +context = 131_072 output = 32_768 [modalities] diff --git a/providers/openrouter/models/aion-labs/aion-3.0.toml b/providers/openrouter/models/aion-labs/aion-3.0.toml index f16de215471..e25d916b2b3 100644 --- a/providers/openrouter/models/aion-labs/aion-3.0.toml +++ b/providers/openrouter/models/aion-labs/aion-3.0.toml @@ -16,7 +16,7 @@ output = 6 cache_read = 0.75 [limit] -context = 1_048_576 +context = 131_072 output = 32_768 [modalities] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 74f20945d6e..b98c0880f9f 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.895578 -output = 1.791156 -cache_read = 0.074632 +input = 0.887226 +output = 1.774452 +cache_read = 0.073936 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 16f058ed841..f8e214df2a7 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.15 -output = 0.6 -cache_read = 0.003 +input = 0.1 +output = 0.8 +cache_read = 0.1 [limit] context = 1_048_576 +output = 943_718 diff --git a/providers/openrouter/models/qwen/qwen3.8-omni-flash.toml b/providers/openrouter/models/qwen/qwen3.8-omni-flash.toml new file mode 100644 index 00000000000..d5769de2059 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3.8-omni-flash.toml @@ -0,0 +1,16 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +base_model = "alibaba/qwen3.8-omni-flash" +temperature = true +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.15 +output = 0.47 +cache_read = 0.016 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index 56426182daf..9c090dba1b3 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.12 -output = 0.48 -cache_read = 0.0036 +input = 0.1 +output = 0.8 +cache_read = 0.1 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 3112dca3caa..5a32de51cb5 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.4 -output = 4.3 -cache_read = 0.033 +input = 0.39996 +output = 1.19988 +cache_read = 0.012726 [limit] context = 1_048_576 -output = 384_000 +output = 393_216 [modalities] input = ["text"] From 276e02f0972a6b572e9fbacfdea7f7af5b374734 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 21:23:32 +0000 Subject: [PATCH 340/392] chore(sync): update Venice model catalog (#7812) Co-authored-by: opencode-agent[bot] --- .../venice/models/openai-gpt-6-luna.toml | 22 +++++++++++++++++++ providers/venice/models/openai-gpt-6-sol.toml | 22 +++++++++++++++++++ 2 files changed, 44 insertions(+) create mode 100644 providers/venice/models/openai-gpt-6-luna.toml create mode 100644 providers/venice/models/openai-gpt-6-sol.toml diff --git a/providers/venice/models/openai-gpt-6-luna.toml b/providers/venice/models/openai-gpt-6-luna.toml new file mode 100644 index 00000000000..df056dfdc5c --- /dev/null +++ b/providers/venice/models/openai-gpt-6-luna.toml @@ -0,0 +1,22 @@ +base_model = "openai/gpt-6-luna" +description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.125 +output = 0.625 +cache_read = 0.0125 +cache_write = 0.15625 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.25 +output = 0.9375 +cache_read = 0.025 +cache_write = 0.3125 + +[modalities] +input = ["text", "image"] diff --git a/providers/venice/models/openai-gpt-6-sol.toml b/providers/venice/models/openai-gpt-6-sol.toml new file mode 100644 index 00000000000..5a130d788ba --- /dev/null +++ b/providers/venice/models/openai-gpt-6-sol.toml @@ -0,0 +1,22 @@ +base_model = "openai/gpt-6-sol" +description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2.5 +output = 12.5 +cache_read = 0.25 +cache_write = 3.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 5 +output = 18.75 +cache_read = 0.5 +cache_write = 6.25 + +[modalities] +input = ["text", "image"] From 3701e91036b401d801b8e860d836179c2e288d5e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 21:23:37 +0000 Subject: [PATCH 341/392] chore(sync): update OpenRouter model catalog (#7811) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 7 +++---- .../openrouter/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- providers/openrouter/models/~moonshotai/kimi-latest.toml | 4 ++-- 4 files changed, 11 insertions(+), 12 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index b98c0880f9f..ead71f4358c 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.887226 -output = 1.774452 -cache_read = 0.073936 +input = 0.877134 +output = 1.754268 +cache_read = 0.073095 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index f8e214df2a7..16f058ed841 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,10 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.1 -output = 0.8 -cache_read = 0.1 +input = 0.15 +output = 0.6 +cache_read = 0.003 [limit] context = 1_048_576 -output = 943_718 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index 9c090dba1b3..56426182daf 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.1 -output = 0.8 -cache_read = 0.1 +input = 0.12 +output = 0.48 +cache_read = 0.0036 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 0aebd435d56..9cb7f42fe0a 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,8 +20,8 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.4989 -output = 10.758 +input = 1.4 +output = 13 cache_read = 0.3 [limit] From 2bc6cf3c6766209f70f0504df50da0b76cc55f90 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 21:24:23 +0000 Subject: [PATCH 342/392] chore(sync): update Kilo model catalog (#7809) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/deepseek/deepseek-v4.1-flash.toml | 1 - providers/kilo/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- providers/kilo/models/~moonshotai/kimi-latest.toml | 4 ++-- 3 files changed, 5 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml b/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml index d755c215bcf..2d43303aad0 100644 --- a/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml @@ -12,4 +12,3 @@ cache_read = 0.006 [limit] context = 1_048_576 -output = 943_718 diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 9e869daead9..2193a017b17 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.1 -output = 0.8 -cache_read = 0.1 +input = 0.12 +output = 0.48 +cache_read = 0.0036 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 366b401a83c..70e7e6ec8c8 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,8 +15,8 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.4989 -output = 10.758 +input = 1.4 +output = 13 cache_read = 0.3 [limit] From be826a62bb6bbb563d249e00ef1d12557c43c286 Mon Sep 17 00:00:00 2001 From: fwang <83515+fwang@users.noreply.github.com> Date: Tue, 22 Sep 2026 21:53:31 +0000 Subject: [PATCH 343/392] feat(opencode): add Grok 4.7 with 30% off pricing --- providers/opencode/models/grok-4.7.toml | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 providers/opencode/models/grok-4.7.toml diff --git a/providers/opencode/models/grok-4.7.toml b/providers/opencode/models/grok-4.7.toml new file mode 100644 index 00000000000..e121d9b0cd5 --- /dev/null +++ b/providers/opencode/models/grok-4.7.toml @@ -0,0 +1,21 @@ +# Pricing: 30% off OpenCode Go Grok 4.7 rates, including long-context tiers. +base_model = "xai/grok-4.7" +name = "Grok 4.7 (30% Off)" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 1.4 +output = 4.2 +cache_read = 0.35 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.8 +output = 8.4 +cache_read = 0.7 + +[provider] +npm = "@ai-sdk/openai" From f1b0b8c791c2ea32eff83c242cc27c4415c22bb0 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:24:37 +0000 Subject: [PATCH 344/392] chore(sync): update Kilo model catalog (#7817) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/deepseek/deepseek-v4.1-flash.toml | 1 + providers/kilo/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- providers/kilo/models/~z-ai/glm-latest.toml | 6 +++--- 4 files changed, 10 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml b/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml index 2d43303aad0..d755c215bcf 100644 --- a/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml @@ -12,3 +12,4 @@ cache_read = 0.006 [limit] context = 1_048_576 +output = 943_718 diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 2193a017b17..0295282190d 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.12 -output = 0.48 -cache_read = 0.0036 +input = 0.1 +output = 0.5 +cache_read = 0.01 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 70e7e6ec8c8..afff7105d92 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.4 -output = 13 -cache_read = 0.3 +input = 1.5 +output = 7.5 +cache_read = 0.15 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 0facfa991b5..2ce00d3e342 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.6538 -output = 2.0548 -cache_read = 0.12142 +input = 0.5625 +output = 2.5 +cache_read = 0.125 [limit] context = 1_048_576 From e84899e0363bb0e84a6ee04ecaf61883acde210f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:24:48 +0000 Subject: [PATCH 345/392] chore(sync): update OpenRouter model catalog (#7818) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 7 ++++--- .../openrouter/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- providers/openrouter/models/~z-ai/glm-latest.toml | 6 +++--- 5 files changed, 16 insertions(+), 15 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index ead71f4358c..9472d4dda50 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.877134 -output = 1.754268 -cache_read = 0.073095 +input = 0.865302 +output = 1.730604 +cache_read = 0.072109 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 16f058ed841..ee75b6ad088 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.15 -output = 0.6 -cache_read = 0.003 +input = 0.1 +output = 0.5 +cache_read = 0.01 [limit] context = 1_048_576 +output = 943_718 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index 56426182daf..a3bff918037 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.12 -output = 0.48 -cache_read = 0.0036 +input = 0.1 +output = 0.5 +cache_read = 0.01 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 9cb7f42fe0a..0bb8650abd5 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.4 -output = 13 -cache_read = 0.3 +input = 1.5 +output = 7.5 +cache_read = 0.15 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index e0e1e1ca329..9bbf416c3f6 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.6538 -output = 2.0548 -cache_read = 0.12142 +input = 0.5625 +output = 2.5 +cache_read = 0.125 [limit] context = 1_310_720 From 414bfa0469d94a427a86279773518479fb45b6fd Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 23:23:38 +0000 Subject: [PATCH 346/392] chore(sync): update OpenRouter model catalog (#7822) Co-authored-by: opencode-agent[bot] --- .../models/cohere/command-a-plus.toml | 29 +++++++++++++++++++ .../models/deepseek/deepseek-v4-pro.toml | 6 ++-- providers/openrouter/models/z-ai/glm-5.3.toml | 6 ++-- .../models/~moonshotai/kimi-latest.toml | 6 ++-- .../models/~z-ai/glm-flash-latest.toml | 2 +- .../openrouter/models/~z-ai/glm-latest.toml | 6 ++-- 6 files changed, 42 insertions(+), 13 deletions(-) create mode 100644 providers/openrouter/models/cohere/command-a-plus.toml diff --git a/providers/openrouter/models/cohere/command-a-plus.toml b/providers/openrouter/models/cohere/command-a-plus.toml new file mode 100644 index 00000000000..a3ee649020e --- /dev/null +++ b/providers/openrouter/models/cohere/command-a-plus.toml @@ -0,0 +1,29 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +name = "Command A+" +description = "Cohere command model for multilingual enterprise agents, tools, and chat" +family = "command-a" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.3 +output = 1.5 +cache_read = 0.15 + +[limit] +context = 192_000 +output = 64_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 9472d4dda50..19122a2a8b0 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.865302 -output = 1.730604 -cache_read = 0.072109 +input = 0.856776 +output = 1.713552 +cache_read = 0.071398 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index 7c1154ec1bd..5422c468088 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.6538 -output = 2.0548 -cache_read = 0.12142 +input = 0.5614 +output = 1.7644 +cache_read = 0.10426 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 0bb8650abd5..c97ad1a78e9 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.5 -output = 7.5 -cache_read = 0.15 +input = 1.35 +output = 14.33 +cache_read = 0.3 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~z-ai/glm-flash-latest.toml b/providers/openrouter/models/~z-ai/glm-flash-latest.toml index 8f90bbaaf93..c32fca67d4b 100644 --- a/providers/openrouter/models/~z-ai/glm-flash-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-flash-latest.toml @@ -21,7 +21,7 @@ cache_read = 0.015 [limit] context = 1_310_720 -output = 943_718 +output = 131_072 [modalities] input = ["text", "image", "video"] diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 9bbf416c3f6..42058ee6fca 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5625 -output = 2.5 -cache_read = 0.125 +input = 0.5614 +output = 1.7644 +cache_read = 0.10426 [limit] context = 1_310_720 From ead2794b307369126f84794e9d355a05c0ace6fa Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 23:27:59 +0000 Subject: [PATCH 347/392] chore(sync): update Kilo model catalog (#7821) Co-authored-by: opencode-agent[bot] --- .../kilo/models/cohere/command-a-plus.toml | 28 +++++++++++++++++++ .../kilo/models/~moonshotai/kimi-latest.toml | 6 ++-- .../kilo/models/~z-ai/glm-flash-latest.toml | 2 +- providers/kilo/models/~z-ai/glm-latest.toml | 6 ++-- 4 files changed, 35 insertions(+), 7 deletions(-) create mode 100644 providers/kilo/models/cohere/command-a-plus.toml diff --git a/providers/kilo/models/cohere/command-a-plus.toml b/providers/kilo/models/cohere/command-a-plus.toml new file mode 100644 index 00000000000..dcee2ea1432 --- /dev/null +++ b/providers/kilo/models/cohere/command-a-plus.toml @@ -0,0 +1,28 @@ +name = "Cohere: Command A+" +description = "Command A+ is Cohere's flagship model for enterprise agentic workflows. It accepts text and image inputs with a 192K context window, supports native tool calling with strict tool schemas, structured..." +family = "command-a" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.3 +output = 1.5 +cache_read = 0.15 + +[limit] +context = 192_000 +output = 64_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index afff7105d92..5677dcd7e7e 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.5 -output = 7.5 -cache_read = 0.15 +input = 1.35 +output = 14.33 +cache_read = 0.3 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~z-ai/glm-flash-latest.toml b/providers/kilo/models/~z-ai/glm-flash-latest.toml index aee70deebb0..e4ce4e5dfe6 100644 --- a/providers/kilo/models/~z-ai/glm-flash-latest.toml +++ b/providers/kilo/models/~z-ai/glm-flash-latest.toml @@ -21,7 +21,7 @@ cache_read = 0.015 [limit] context = 1_048_576 -output = 943_718 +output = 131_072 [modalities] input = ["text", "image", "video"] diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 2ce00d3e342..d891278d7ef 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5625 -output = 2.5 -cache_read = 0.125 +input = 0.5614 +output = 1.7644 +cache_read = 0.10426 [limit] context = 1_048_576 From 90cc958ca96f1bd0691f0b554eefd6355fc5744b Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 00:58:11 +0000 Subject: [PATCH 348/392] chore(sync): update Vercel AI Gateway model catalog (#7824) Co-authored-by: opencode-agent[bot] --- .../models/google/gemini-2.5-flash-image.toml | 3 +-- .../vercel/models/google/gemini-2.5-flash-lite.toml | 13 +++++++++++-- .../vercel/models/google/gemini-3.7-flash.toml | 1 + .../vercel/models/google/gemini-3.8-flash.toml | 1 + 4 files changed, 14 insertions(+), 4 deletions(-) diff --git a/providers/vercel/models/google/gemini-2.5-flash-image.toml b/providers/vercel/models/google/gemini-2.5-flash-image.toml index 1ea71885c13..115f8bd626c 100644 --- a/providers/vercel/models/google/gemini-2.5-flash-image.toml +++ b/providers/vercel/models/google/gemini-2.5-flash-image.toml @@ -2,7 +2,6 @@ base_model = "google/gemini-2.5-flash-image" name = "Nano Banana (Gemini 2.5 Flash Image)" attachment = false reasoning = false -knowledge = "2025-01" [cost] input = 0.3 @@ -10,4 +9,4 @@ output = 2.5 cache_read = 0.03 [limit] -output = 65_536 +output = 65_535 diff --git a/providers/vercel/models/google/gemini-2.5-flash-lite.toml b/providers/vercel/models/google/gemini-2.5-flash-lite.toml index 9dbd105b8d4..c77c8d9fb11 100644 --- a/providers/vercel/models/google/gemini-2.5-flash-lite.toml +++ b/providers/vercel/models/google/gemini-2.5-flash-lite.toml @@ -1,12 +1,21 @@ base_model = "google/gemini-2.5-flash-lite" -reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 512, max = 24_576 }] name = "Gemini 2.5 Flash Lite" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" +min = 512 +max = 24_576 + [cost] input = 0.1 output = 0.4 cache_read = 0.01 +[limit] +output = 65_535 + [modalities] input = ["text", "image", "pdf"] -output = ["text"] diff --git a/providers/vercel/models/google/gemini-3.7-flash.toml b/providers/vercel/models/google/gemini-3.7-flash.toml index 9a978ed0a3b..c75cf2fdf40 100644 --- a/providers/vercel/models/google/gemini-3.7-flash.toml +++ b/providers/vercel/models/google/gemini-3.7-flash.toml @@ -12,6 +12,7 @@ cache_read = 0.075 [limit] context = 1_000_000 +output = 65_535 [modalities] input = ["text", "image", "pdf"] diff --git a/providers/vercel/models/google/gemini-3.8-flash.toml b/providers/vercel/models/google/gemini-3.8-flash.toml index 32fddda9360..39eb1bf1658 100644 --- a/providers/vercel/models/google/gemini-3.8-flash.toml +++ b/providers/vercel/models/google/gemini-3.8-flash.toml @@ -8,6 +8,7 @@ cache_read = 0.075 [limit] context = 1_000_000 +output = 65_535 [modalities] input = ["text", "image", "pdf"] From 0d39179af05da0f714c4a6024092b18f0a502659 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 00:58:14 +0000 Subject: [PATCH 349/392] chore(sync): update OpenRouter model catalog (#7823) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-flash.toml | 6 +++--- providers/openrouter/models/deepseek/deepseek-v4-pro.toml | 6 +++--- providers/openrouter/models/tencent/hy3.toml | 6 +++--- providers/openrouter/models/z-ai/glm-5.3.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- providers/openrouter/models/~z-ai/glm-latest.toml | 6 +++--- 6 files changed, 19 insertions(+), 19 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml index 141d4b41ca6..6f817c5a0a0 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.049 -output = 0.098 -cache_read = 0.0098 +input = 0.088606 +output = 0.177212 +cache_read = 0.017721 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 19122a2a8b0..e7683af4617 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,9 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.856776 -output = 1.713552 -cache_read = 0.071398 +input = 0.95526 +output = 1.91052 +cache_read = 0.079605 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/tencent/hy3.toml b/providers/openrouter/models/tencent/hy3.toml index f61fa3175e9..80ecfdd3aa8 100644 --- a/providers/openrouter/models/tencent/hy3.toml +++ b/providers/openrouter/models/tencent/hy3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.0825 -output = 0.33 -cache_read = 0.020625 +input = 0.132 +output = 0.528 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index 5422c468088..8b9af97ef26 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5614 -output = 1.7644 -cache_read = 0.10426 +input = 0.84 +output = 2.64 +cache_read = 0.156 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index 5a32de51cb5..3112dca3caa 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,13 +20,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.39996 -output = 1.19988 -cache_read = 0.012726 +input = 0.4 +output = 4.3 +cache_read = 0.033 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 42058ee6fca..9bbf416c3f6 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5614 -output = 1.7644 -cache_read = 0.10426 +input = 0.5625 +output = 2.5 +cache_read = 0.125 [limit] context = 1_310_720 From bd64c9c1d797ca6d681d46aa914cb7b1336444dc Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 00:58:22 +0000 Subject: [PATCH 350/392] chore(sync): update Kilo model catalog (#7825) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/tencent/hy3.toml | 6 +++--- providers/kilo/models/~deepseek/deepseek-pro-latest.toml | 8 ++++---- providers/kilo/models/~z-ai/glm-latest.toml | 6 +++--- 3 files changed, 10 insertions(+), 10 deletions(-) diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index e96e3110ab1..c3469c1e9dc 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.0825 -output = 0.33 -cache_read = 0.020625 +input = 0.13 +output = 0.53 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index 6f1d7199aea..7666171face 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.39996 -output = 1.19988 -cache_read = 0.012726 +input = 0.4 +output = 4.3 +cache_read = 0.033 [limit] context = 1_048_576 -output = 393_216 +output = 384_000 [modalities] input = ["text"] diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index d891278d7ef..2ce00d3e342 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.5614 -output = 1.7644 -cache_read = 0.10426 +input = 0.5625 +output = 2.5 +cache_read = 0.125 [limit] context = 1_048_576 From c4b04c6643cfa9235bff7bdd84f8aefda014eea5 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 01:35:18 +0000 Subject: [PATCH 351/392] chore(sync): update OpenRouter model catalog (#7826) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- providers/openrouter/models/~moonshotai/kimi-latest.toml | 4 ++-- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..b8d53b0809c 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 1.32 +output = 3.96 +cache_read = 0.044 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index c97ad1a78e9..599b9319e64 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,8 +20,8 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.35 -output = 14.33 +input = 1.45 +output = 11.7 cache_read = 0.3 [limit] From cd9335ab37c70978becb7ca96a0da6e4749f9d88 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 01:35:22 +0000 Subject: [PATCH 352/392] chore(sync): update NanoGPT model catalog (#7827) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index b4cce19afb3..a62995b0beb 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.22 -output = 0.66 -cache_read = 0.007 +input = 0.44 +output = 1.32 +cache_read = 0.014 [limit] context = 1_048_576 From faa9cd00f2e52ab6b00210d2637350061fa3b83e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 01:35:28 +0000 Subject: [PATCH 353/392] chore(sync): update Kilo model catalog (#7828) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~moonshotai/kimi-latest.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 5677dcd7e7e..7113c49b465 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,8 +15,8 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.35 -output = 14.33 +input = 1.45 +output = 11.7 cache_read = 0.3 [limit] From 123647e6f9daf344871d6540cd0516629f598b1a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:34:55 +0000 Subject: [PATCH 354/392] chore(sync): update Vercel AI Gateway model catalog (#7829) Co-authored-by: opencode-agent[bot] --- providers/vercel/models/anthropic/claude-opus-4.toml | 3 --- providers/vercel/models/anthropic/claude-sonnet-4.toml | 1 - 2 files changed, 4 deletions(-) diff --git a/providers/vercel/models/anthropic/claude-opus-4.toml b/providers/vercel/models/anthropic/claude-opus-4.toml index a80b0442665..2553fa1f326 100644 --- a/providers/vercel/models/anthropic/claude-opus-4.toml +++ b/providers/vercel/models/anthropic/claude-opus-4.toml @@ -6,6 +6,3 @@ input = 15 output = 75 cache_read = 1.5 cache_write = 18.75 - -[limit] -output = 8_192 diff --git a/providers/vercel/models/anthropic/claude-sonnet-4.toml b/providers/vercel/models/anthropic/claude-sonnet-4.toml index 61e522dfd97..7e189ae395a 100644 --- a/providers/vercel/models/anthropic/claude-sonnet-4.toml +++ b/providers/vercel/models/anthropic/claude-sonnet-4.toml @@ -16,4 +16,3 @@ cache_write = 7.5 [limit] context = 1_000_000 -output = 8_192 From 98852586f2e5f1802c345abebd8772c12fd31a8f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:34:58 +0000 Subject: [PATCH 355/392] chore(sync): update OpenRouter model catalog (#7830) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/~moonshotai/kimi-latest.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 599b9319e64..e135fc2e21e 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,8 +20,8 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.45 -output = 11.7 +input = 1.05 +output = 13 cache_read = 0.3 [limit] From 3d93b844f404d7378880a4ef6e2491d3a11b1f38 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:35:00 +0000 Subject: [PATCH 356/392] chore(sync): update Kilo model catalog (#7831) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~moonshotai/kimi-latest.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 7113c49b465..daf2852a1ed 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,8 +15,8 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.45 -output = 11.7 +input = 1.05 +output = 13 cache_read = 0.3 [limit] From 1dc15b2429516045fae614efcba4c3388e9a7729 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:32:01 +0000 Subject: [PATCH 357/392] chore(sync): update CrossModel model catalog (#7834) Co-authored-by: opencode-agent[bot] --- .../models/anthropic/claude-opus-5-5.toml | 14 +++++++++++++ .../crossmodel/models/openai/gpt-6-luna.toml | 21 +++++++++++++++++++ .../crossmodel/models/openai/gpt-6-sol.toml | 21 +++++++++++++++++++ 3 files changed, 56 insertions(+) create mode 100644 providers/crossmodel/models/anthropic/claude-opus-5-5.toml create mode 100644 providers/crossmodel/models/openai/gpt-6-luna.toml create mode 100644 providers/crossmodel/models/openai/gpt-6-sol.toml diff --git a/providers/crossmodel/models/anthropic/claude-opus-5-5.toml b/providers/crossmodel/models/anthropic/claude-opus-5-5.toml new file mode 100644 index 00000000000..d9abb20946e --- /dev/null +++ b/providers/crossmodel/models/anthropic/claude-opus-5-5.toml @@ -0,0 +1,14 @@ +base_model = "anthropic/claude-opus-5-5" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[modalities] +input = ["text", "image"] diff --git a/providers/crossmodel/models/openai/gpt-6-luna.toml b/providers/crossmodel/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..e91e64b7330 --- /dev/null +++ b/providers/crossmodel/models/openai/gpt-6-luna.toml @@ -0,0 +1,21 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[modalities] +input = ["text", "image"] diff --git a/providers/crossmodel/models/openai/gpt-6-sol.toml b/providers/crossmodel/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..56437cf29c3 --- /dev/null +++ b/providers/crossmodel/models/openai/gpt-6-sol.toml @@ -0,0 +1,21 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 + +[modalities] +input = ["text", "image"] From 446239c4f17c3d6b441d04c5462ff2bcb687a65a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:32:09 +0000 Subject: [PATCH 358/392] chore(sync): update Kilo model catalog (#7833) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~moonshotai/kimi-latest.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index daf2852a1ed..366b401a83c 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,8 +15,8 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.05 -output = 13 +input = 1.4989 +output = 10.758 cache_read = 0.3 [limit] From f6f06181d6887d7c7cd445aafafe4ae900f19299 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:32:14 +0000 Subject: [PATCH 359/392] chore(sync): update OpenRouter model catalog (#7835) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/~moonshotai/kimi-latest.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index e135fc2e21e..0aebd435d56 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,8 +20,8 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.05 -output = 13 +input = 1.4989 +output = 10.758 cache_read = 0.3 [limit] From 1f9dbe7e9eb58e7eb82405fd039159362bb91134 Mon Sep 17 00:00:00 2001 From: Vladimir Glafirov Date: Wed, 23 Sep 2026 05:54:28 +0200 Subject: [PATCH 360/392] feat: add gitlab duo-chat-opus-5-5 model (#7794) * feat: add gitlab duo-chat-opus-5-5 model * fix: use base_model inheritance for duo-chat-opus-5-5 Rewrite as base_model = "anthropic/claude-opus-5-5" plus only the provider-specific delta (name, reasoning_options, zero cost), instead of a full standalone duplicate. This inherits description, family, dates, knowledge (2026-06), booleans, [limit], and [modalities] from the existing lab entry, matching the pattern used by the sibling duo-chat-opus-5 / duo-chat-sonnet-5 files. --- providers/gitlab/models/duo-chat-opus-5-5.toml | 9 +++++++++ 1 file changed, 9 insertions(+) create mode 100644 providers/gitlab/models/duo-chat-opus-5-5.toml diff --git a/providers/gitlab/models/duo-chat-opus-5-5.toml b/providers/gitlab/models/duo-chat-opus-5-5.toml new file mode 100644 index 00000000000..acfba463a94 --- /dev/null +++ b/providers/gitlab/models/duo-chat-opus-5-5.toml @@ -0,0 +1,9 @@ +base_model = "anthropic/claude-opus-5-5" +name = "Agentic Chat (Claude Opus 5.5)" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 From 912dc39011be121800c239a18805f8d568ac7cc9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:55:15 -0500 Subject: [PATCH 361/392] chore(sync): update Requesty model catalog (#7810) Co-authored-by: opencode-agent[bot] --- providers/requesty/models/gpt-6-luna.toml | 19 +++++++++++++++++++ providers/requesty/models/gpt-6-luna@eu.toml | 20 ++++++++++++++++++++ providers/requesty/models/gpt-6-sol.toml | 19 +++++++++++++++++++ providers/requesty/models/gpt-6-sol@eu.toml | 20 ++++++++++++++++++++ 4 files changed, 78 insertions(+) create mode 100644 providers/requesty/models/gpt-6-luna.toml create mode 100644 providers/requesty/models/gpt-6-luna@eu.toml create mode 100644 providers/requesty/models/gpt-6-sol.toml create mode 100644 providers/requesty/models/gpt-6-sol@eu.toml diff --git a/providers/requesty/models/gpt-6-luna.toml b/providers/requesty/models/gpt-6-luna.toml new file mode 100644 index 00000000000..a11e192297d --- /dev/null +++ b/providers/requesty/models/gpt-6-luna.toml @@ -0,0 +1,19 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 diff --git a/providers/requesty/models/gpt-6-luna@eu.toml b/providers/requesty/models/gpt-6-luna@eu.toml new file mode 100644 index 00000000000..b0d9e112fcc --- /dev/null +++ b/providers/requesty/models/gpt-6-luna@eu.toml @@ -0,0 +1,20 @@ +base_model = "openai/gpt-6-luna" +name = "GPT-6 Luna (EU)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.12 +output = 0.6 +cache_read = 0.012 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.24 +output = 0.9 +cache_read = 0.024 diff --git a/providers/requesty/models/gpt-6-sol.toml b/providers/requesty/models/gpt-6-sol.toml new file mode 100644 index 00000000000..fccceb9056e --- /dev/null +++ b/providers/requesty/models/gpt-6-sol.toml @@ -0,0 +1,19 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 2 +output = 10 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 diff --git a/providers/requesty/models/gpt-6-sol@eu.toml b/providers/requesty/models/gpt-6-sol@eu.toml new file mode 100644 index 00000000000..357ef21ce6c --- /dev/null +++ b/providers/requesty/models/gpt-6-sol@eu.toml @@ -0,0 +1,20 @@ +base_model = "openai/gpt-6-sol" +name = "GPT-6 Sol (EU)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 2.4 +output = 12 +cache_read = 0.24 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4.8 +output = 18 +cache_read = 0.48 From c35bce144cb4afd55edc104b7313b0abb06c0137 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:56:31 -0500 Subject: [PATCH 362/392] fix: [missing-model] github-copilot: gpt-6-luna (#7813) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- .../github-copilot/models/gpt-6-luna.toml | 20 +++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 providers/github-copilot/models/gpt-6-luna.toml diff --git a/providers/github-copilot/models/gpt-6-luna.toml b/providers/github-copilot/models/gpt-6-luna.toml new file mode 100644 index 00000000000..bd1d21a407f --- /dev/null +++ b/providers/github-copilot/models/gpt-6-luna.toml @@ -0,0 +1,20 @@ +# Pricing: https://docs.github.com/en/copilot/reference/copilot-billing/models-and-pricing +# GPT-6 Luna rates: ≤272K $0.10/$0.01/$0.125/$0.50; >272K $0.20/$0.02/$0.25/$0.75 per 1M (input/cached/cache write/output) +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 From e68f03d3561839bf645aca27178eff95fb5fe16c Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:56:40 -0500 Subject: [PATCH 363/392] fix: [missing-model] github-copilot: gpt-6-sol (#7814) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- .../github-copilot/models/gpt-6-sol.toml | 20 +++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 providers/github-copilot/models/gpt-6-sol.toml diff --git a/providers/github-copilot/models/gpt-6-sol.toml b/providers/github-copilot/models/gpt-6-sol.toml new file mode 100644 index 00000000000..209b909b0be --- /dev/null +++ b/providers/github-copilot/models/gpt-6-sol.toml @@ -0,0 +1,20 @@ +# Pricing: https://docs.github.com/en/copilot/reference/copilot-billing/models-and-pricing +# GPT-6 Sol rates: ≤272K $2/$0.20/$2.50/$10; >272K $4/$0.40/$5.00/$15 per 1M (input/cached/cache write/output) +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 From c7e9597dc37fadc4a4398123b107539b03d818c3 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:57:25 -0500 Subject: [PATCH 364/392] fix: [missing-model] pioneer: fastino/gliner2.5-multi-v1 (#7793) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- .../models/fastino/gliner2.5-multi-v1.toml | 35 +++++++++++++++++++ 1 file changed, 35 insertions(+) create mode 100644 providers/pioneer/models/fastino/gliner2.5-multi-v1.toml diff --git a/providers/pioneer/models/fastino/gliner2.5-multi-v1.toml b/providers/pioneer/models/fastino/gliner2.5-multi-v1.toml new file mode 100644 index 00000000000..932fc2e0d70 --- /dev/null +++ b/providers/pioneer/models/fastino/gliner2.5-multi-v1.toml @@ -0,0 +1,35 @@ +# Sources: +# - Pioneer live catalog GET https://api.pioneer.ai/base-models (id fastino/gliner2.5-multi-v1: label "GLiNER 2.5 Multi", description "Multilingual boundary NER and span extraction; non-trainable encoder.", context_window 4096, input/output $0.15, cache $0.15, license Apache-2.0, release Aug 2026, is_chat_model true) +# - Model card https://huggingface.co/fastino/gliner2.5-multi-v1 (287M mDeBERTa-v3-base multilingual boundary checkpoint, max_len 4096, Apache-2.0) +# - Release announcement https://fastino.ai/blog/gliner2-5-span-free-information-extraction (released Aug 24, 2026; three Apache-2.0 variants) +# Pioneer is the first-party Fastino inference API (https://pioneer.ai/, https://www.prnewswire.com/news-releases/fastino-launches-pioneer-the-first-agent-for-fine-tuning-and-inference-of-llms-302748105.html); full inline entry follows existing providers/pioneer/models/fastino/* pattern. +name = "GLiNER 2.5 Multi" +description = "Multilingual boundary NER and span extraction; non-trainable encoder." +release_date = "2026-08-24" +last_updated = "2026-08-24" +attachment = false +reasoning = true +temperature = true +tool_call = true +open_weights = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 0.15 +output = 0.15 +cache_read = 0.15 +cache_write = 0.15 + +[limit] +context = 4_096 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] From c3fd2c89b4485817795b54c63c4859e8290374af Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:57:32 -0500 Subject: [PATCH 365/392] fix: [missing-model] ofox: x-ai/grok-4.7 (#7792) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- providers/ofox/models/x-ai/grok-4.7.toml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) create mode 100644 providers/ofox/models/x-ai/grok-4.7.toml diff --git a/providers/ofox/models/x-ai/grok-4.7.toml b/providers/ofox/models/x-ai/grok-4.7.toml new file mode 100644 index 00000000000..b829098d460 --- /dev/null +++ b/providers/ofox/models/x-ai/grok-4.7.toml @@ -0,0 +1,16 @@ +# Ofox: x-ai/grok-4.7 → xAI Grok 4.7 +# Sources: +# - https://api.ofox.ai/v1/models/x-ai/grok-4.7 (prompt $2/M, completion $6/M, cache_read $0.5/M; context 500k; max_completion 65536; supported_parameters includes reasoning) +# - https://ofox.ai/models/x-ai/grok-4.7 (same pricing; max output 66K; reasoning + tools + prompt caching) +# - https://docs.x.ai/developers/models/grok-4.7 (reasoning_effort low|medium|high|xhigh; cannot disable) +# reasoning_options match first-party xAI + relay peers (no none) +base_model = "xai/grok-4.7" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[limit] +output = 65_536 From 3960b9bf712feb58b553e8045edfdfcd0a92efdb Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:57:41 -0500 Subject: [PATCH 366/392] fix: [missing-model] pioneer: fastino/gliguard-PII-multi (#7780) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- .../models/fastino/gliguard-PII-multi.toml | 36 +++++++++++++++++++ 1 file changed, 36 insertions(+) create mode 100644 providers/pioneer/models/fastino/gliguard-PII-multi.toml diff --git a/providers/pioneer/models/fastino/gliguard-PII-multi.toml b/providers/pioneer/models/fastino/gliguard-PII-multi.toml new file mode 100644 index 00000000000..cec904e782a --- /dev/null +++ b/providers/pioneer/models/fastino/gliguard-PII-multi.toml @@ -0,0 +1,36 @@ +# Sources: +# - Pioneer live catalog GET https://api.pioneer.ai/v1/models (id fastino/gliguard-PII-multi: display_name, max_input_tokens 8192, max_tokens 8192, prices 0.15, capabilities all false) +# - Pioneer base-models GET https://api.pioneer.ai/base-models (label GLiNER2-Guardrails-PII-Multi, context_window 8192, license Apache-2.0, release_month Jul 2026) +# - Fastino release blog https://fastino.ai/blog/gliner2-guardrails-pii-multi-safety-moderation-privacy-filtering-small-language-model (released July 8 2026, 0.3B param, single forward pass, Apache 2.0) +# - Fastino model page https://fastino.ai/models/gliner2-guardrails-pii-multi (300M-param multilingual safety moderation + PII detection, open weights) +# - Hugging Face https://huggingface.co/fastino/GLiNER2-Guardrails-PII-Multi (Apache-2.0, 0.3B params, unified safety moderation + PII detection) +name = "GLiNER2-Guardrails-PII-Multi" +description = "A 300M-parameter multilingual model that runs LLM safety moderation and PII detection in a single forward pass." +release_date = "2026-07-08" +last_updated = "2026-07-08" +attachment = false +reasoning = true +temperature = true +tool_call = true +open_weights = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 0.15 +output = 0.15 +cache_read = 0.15 +cache_write = 0.15 + +[limit] +context = 8_192 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] From 7d4d66712d3208d832bd61f3b7b94b4ffc6eb775 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:57:50 -0500 Subject: [PATCH 367/392] fix: [missing-model] ofox: openai/gpt-6-luna (#7804) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- providers/ofox/models/openai/gpt-6-luna.toml | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 providers/ofox/models/openai/gpt-6-luna.toml diff --git a/providers/ofox/models/openai/gpt-6-luna.toml b/providers/ofox/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..990db0bda79 --- /dev/null +++ b/providers/ofox/models/openai/gpt-6-luna.toml @@ -0,0 +1,20 @@ +# Sources: +# - Ofox model page: https://ofox.ai/models/openai/gpt-6-luna +# - Ofox catalog API: https://api.ofox.ai/v2/models/catalog?include=provider_price&search=gpt-6-luna +# - OpenAI model docs (pricing/reasoning): https://developers.openai.com/api/docs/models/gpt-6-luna +# Ofox provider_price override is flat $0.08/$0.40 (+ cache) with no context tier in the catalog, so no [[cost.tiers]]. +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.08 +output = 0.4 +cache_read = 0.008 +cache_write = 0.1 + +[provider] +npm = "@ai-sdk/openai" +api = "https://api.ofox.ai/v1" From d53018c0bf6f9464dcfeb98eae2e948ae781a836 Mon Sep 17 00:00:00 2001 From: Yazan ALBaiz <106471410+yazan-albaiz@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:58:22 +0300 Subject: [PATCH 368/392] feat(amazon-bedrock): add GPT-6 Sol and Luna inference profiles (#7805) --- .../models/global.openai.gpt-6-luna.toml | 24 +++++++++++++++++++ .../models/global.openai.gpt-6-sol.toml | 24 +++++++++++++++++++ .../models/us.openai.gpt-6-luna.toml | 24 +++++++++++++++++++ .../models/us.openai.gpt-6-sol.toml | 24 +++++++++++++++++++ 4 files changed, 96 insertions(+) create mode 100644 providers/amazon-bedrock/models/global.openai.gpt-6-luna.toml create mode 100644 providers/amazon-bedrock/models/global.openai.gpt-6-sol.toml create mode 100644 providers/amazon-bedrock/models/us.openai.gpt-6-luna.toml create mode 100644 providers/amazon-bedrock/models/us.openai.gpt-6-sol.toml diff --git a/providers/amazon-bedrock/models/global.openai.gpt-6-luna.toml b/providers/amazon-bedrock/models/global.openai.gpt-6-luna.toml new file mode 100644 index 00000000000..d541b9a308b --- /dev/null +++ b/providers/amazon-bedrock/models/global.openai.gpt-6-luna.toml @@ -0,0 +1,24 @@ +# Global cross-Region inference profile — served by Bedrock core (Converse), not Mantle, so no [provider] override. +# Profile ID verified via ListInferenceProfiles (ACTIVE) on 2026-09-22; AWS has not published a GPT-6 Luna model card yet. +# Global CRIS pricing = OpenAI API list rates: https://developers.openai.com/api/docs/models/gpt-6-luna +# Converse effort: additionalModelRequestFields.reasoning.effort = none|low|medium|high|xhigh|max (Bedrock's own validation error lists these). +# Live Converse check 2026-09-22 (eu-west-1, global profile): tool call + tool-result round trip; effort low/xhigh/max accepted. +base_model = "openai/gpt-6-luna" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +name = "GPT-6 Luna (Global)" + +[cost] +input = 0.10 +output = 0.50 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { size = 272_000 } +input = 0.20 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[modalities] +input = ["text", "image"] diff --git a/providers/amazon-bedrock/models/global.openai.gpt-6-sol.toml b/providers/amazon-bedrock/models/global.openai.gpt-6-sol.toml new file mode 100644 index 00000000000..fee63c0e68e --- /dev/null +++ b/providers/amazon-bedrock/models/global.openai.gpt-6-sol.toml @@ -0,0 +1,24 @@ +# Global cross-Region inference profile — served by Bedrock core (Converse), not Mantle, so no [provider] override. +# Profile ID verified via ListInferenceProfiles (ACTIVE) on 2026-09-22; AWS has not published a GPT-6 Sol model card yet. +# Global CRIS pricing = OpenAI API list rates: https://developers.openai.com/api/docs/models/gpt-6-sol +# Converse effort: additionalModelRequestFields.reasoning.effort = none|low|medium|high|xhigh|max (Bedrock's validation error on the Luna twin lists these; matches OpenAI-native Sol). +# Live Converse check 2026-09-22 (eu-west-1, global profile): tool call + tool-result round trip; effort low/xhigh/max accepted. +base_model = "openai/gpt-6-sol" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +name = "GPT-6 Sol (Global)" + +[cost] +input = 2.00 +output = 10.00 +cache_read = 0.20 +cache_write = 2.50 + +[[cost.tiers]] +tier = { size = 272_000 } +input = 4.00 +output = 15.00 +cache_read = 0.40 +cache_write = 5.00 + +[modalities] +input = ["text", "image"] diff --git a/providers/amazon-bedrock/models/us.openai.gpt-6-luna.toml b/providers/amazon-bedrock/models/us.openai.gpt-6-luna.toml new file mode 100644 index 00000000000..0d59417b8e6 --- /dev/null +++ b/providers/amazon-bedrock/models/us.openai.gpt-6-luna.toml @@ -0,0 +1,24 @@ +# US cross-Region inference profile — served by Bedrock core (Converse), not Mantle, so no [provider] override. +# Profile ID verified via ListInferenceProfiles (ACTIVE) on 2026-09-22; AWS has not published a GPT-6 Luna model card yet. +# Geo CRIS pricing = OpenAI API list rates + AWS's 10% fee (as stated on the GPT-6 Astra card): https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html +# Converse effort: additionalModelRequestFields.reasoning.effort = none|low|medium|high|xhigh|max (Bedrock's own validation error lists these). +# Live Converse check 2026-09-22 (eu-west-1, global profile): tool call + tool-result round trip; effort low/xhigh/max accepted. +base_model = "openai/gpt-6-luna" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +name = "GPT-6 Luna (US)" + +[cost] +input = 0.11 +output = 0.55 +cache_read = 0.011 +cache_write = 0.1375 + +[[cost.tiers]] +tier = { size = 272_000 } +input = 0.22 +output = 0.825 +cache_read = 0.022 +cache_write = 0.275 + +[modalities] +input = ["text", "image"] diff --git a/providers/amazon-bedrock/models/us.openai.gpt-6-sol.toml b/providers/amazon-bedrock/models/us.openai.gpt-6-sol.toml new file mode 100644 index 00000000000..9cb9cff8731 --- /dev/null +++ b/providers/amazon-bedrock/models/us.openai.gpt-6-sol.toml @@ -0,0 +1,24 @@ +# US cross-Region inference profile — served by Bedrock core (Converse), not Mantle, so no [provider] override. +# Profile ID verified via ListInferenceProfiles (ACTIVE) on 2026-09-22; AWS has not published a GPT-6 Sol model card yet. +# Geo CRIS pricing = OpenAI API list rates + AWS's 10% fee (as stated on the GPT-6 Astra card): https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html +# Converse effort: additionalModelRequestFields.reasoning.effort = none|low|medium|high|xhigh|max (Bedrock's validation error on the Luna twin lists these; matches OpenAI-native Sol). +# Live Converse check 2026-09-22 (eu-west-1, global profile): tool call + tool-result round trip; effort low/xhigh/max accepted. +base_model = "openai/gpt-6-sol" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +name = "GPT-6 Sol (US)" + +[cost] +input = 2.20 +output = 11.00 +cache_read = 0.22 +cache_write = 2.75 + +[[cost.tiers]] +tier = { size = 272_000 } +input = 4.40 +output = 16.50 +cache_read = 0.44 +cache_write = 5.50 + +[modalities] +input = ["text", "image"] From cc75f6ae0a9d7a5f26c79676201e1d1cb8f0e9ef Mon Sep 17 00:00:00 2001 From: David Strouk Date: Wed, 23 Sep 2026 06:58:28 +0300 Subject: [PATCH 369/392] feat(kenari): add deepseek-v4-1-flash (#7755) Live on https://kenari.id/v1/models as deepseek-v4-1-flash (context 1M, input text+image, reasoning low/high/max, tool_call). Uses base_model deepseek/deepseek-v4.1-flash, cost 0/0 per Kenari subscription convention. Co-authored-by: davidstrouk --- providers/kenari/models/deepseek-v4-1-flash.toml | 11 +++++++++++ 1 file changed, 11 insertions(+) create mode 100644 providers/kenari/models/deepseek-v4-1-flash.toml diff --git a/providers/kenari/models/deepseek-v4-1-flash.toml b/providers/kenari/models/deepseek-v4-1-flash.toml new file mode 100644 index 00000000000..2a775a81e1f --- /dev/null +++ b/providers/kenari/models/deepseek-v4-1-flash.toml @@ -0,0 +1,11 @@ +# Source: https://kenari.id/v1/models (id deepseek-v4-1-flash, reasoning low/high/max) +# Kenari convention: subscription pricing, cost 0/0 like other Kenari entries. +base_model = "deepseek/deepseek-v4.1-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 From 1ec5c576bb1c1a375a5ec4819c696cdb95afb6b2 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:58:45 -0500 Subject: [PATCH 370/392] fix: [missing-model] github-copilot: claude-opus-5.5 (#7815) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- .../github-copilot/models/claude-opus-5.5.toml | 13 +++++++++++++ 1 file changed, 13 insertions(+) create mode 100644 providers/github-copilot/models/claude-opus-5.5.toml diff --git a/providers/github-copilot/models/claude-opus-5.5.toml b/providers/github-copilot/models/claude-opus-5.5.toml new file mode 100644 index 00000000000..6e3f31961a6 --- /dev/null +++ b/providers/github-copilot/models/claude-opus-5.5.toml @@ -0,0 +1,13 @@ +# Sources: +# - https://github.blog/changelog/2026-09-22-claude-opus-5-5-is-now-available-in-github-copilot/ +# - https://docs.github.com/en/copilot/reference/copilot-billing/models-and-pricing +# - https://www.anthropic.com/claude-opus-5-5 +# - https://platform.claude.com/docs/en/models/opus-5-5/overview +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 From 80f9e2b41a4fde3e7bc39315ade9c53a75138793 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:59:18 -0500 Subject: [PATCH 371/392] fix: [missing-model] ofox: openai/gpt-6-sol (#7803) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- providers/ofox/models/openai/gpt-6-sol.toml | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 providers/ofox/models/openai/gpt-6-sol.toml diff --git a/providers/ofox/models/openai/gpt-6-sol.toml b/providers/ofox/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..c686a7a19a1 --- /dev/null +++ b/providers/ofox/models/openai/gpt-6-sol.toml @@ -0,0 +1,20 @@ +# Sources: +# - Ofox model page: https://ofox.ai/models/openai/gpt-6-sol +# - Ofox catalog API: https://api.ofox.ai/v2/models/catalog?include=provider_price&search=gpt-6-sol +# - OpenAI model docs (reasoning effort + list pricing): https://developers.openai.com/api/docs/models/gpt-6-sol +# Ofox provider_price is the -20% promo override ($1.6/$8 + cache vs $2/$10 list); flat with no >=272k context tier in the catalog, so no [[cost.tiers]]. +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 1.6 +output = 8 +cache_read = 0.16 +cache_write = 2 + +[provider] +npm = "@ai-sdk/openai" +api = "https://api.ofox.ai/v1" From 3487ff2546c77c2ed30bfa677a8cbb72708aee7c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 22:59:25 -0500 Subject: [PATCH 372/392] chore(sync): update DigitalOcean model catalog (#7789) Co-authored-by: opencode-agent[bot] --- .../models/anthropic-claude-opus-5.5.toml | 22 +++++++++++++++++++ .../models/openai-gpt-6-luna.toml | 20 +++++++++++++++++ .../digitalocean/models/openai-gpt-6-sol.toml | 20 +++++++++++++++++ 3 files changed, 62 insertions(+) create mode 100644 providers/digitalocean/models/anthropic-claude-opus-5.5.toml create mode 100644 providers/digitalocean/models/openai-gpt-6-luna.toml create mode 100644 providers/digitalocean/models/openai-gpt-6-sol.toml diff --git a/providers/digitalocean/models/anthropic-claude-opus-5.5.toml b/providers/digitalocean/models/anthropic-claude-opus-5.5.toml new file mode 100644 index 00000000000..23c000bcf3d --- /dev/null +++ b/providers/digitalocean/models/anthropic-claude-opus-5.5.toml @@ -0,0 +1,22 @@ +base_model = "anthropic/claude-opus-5-5" +name = "Anthropic Claude Opus 5.5" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 8 +output = 30 +cache_read = 0.4 +cache_write = 10 + +[modalities] +input = ["text", "image"] diff --git a/providers/digitalocean/models/openai-gpt-6-luna.toml b/providers/digitalocean/models/openai-gpt-6-luna.toml new file mode 100644 index 00000000000..27a362b9081 --- /dev/null +++ b/providers/digitalocean/models/openai-gpt-6-luna.toml @@ -0,0 +1,20 @@ +base_model = "openai/gpt-6-luna" +name = "OpenAI GPT-6 Luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 + +[modalities] +input = ["text", "image"] diff --git a/providers/digitalocean/models/openai-gpt-6-sol.toml b/providers/digitalocean/models/openai-gpt-6-sol.toml new file mode 100644 index 00000000000..8c434d8b143 --- /dev/null +++ b/providers/digitalocean/models/openai-gpt-6-sol.toml @@ -0,0 +1,20 @@ +base_model = "openai/gpt-6-sol" +name = "OpenAI GPT-6 Sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 + +[modalities] +input = ["text", "image"] From 1c243c995f740d9b9c7c17608d678edb2367ef6e Mon Sep 17 00:00:00 2001 From: Yihan Yan Date: Wed, 23 Sep 2026 12:01:03 +0800 Subject: [PATCH 373/392] fix(xiaomi): MiMo V2.6 open weights and input modalities (#7719) * fix(xiaomi): mark MiMo V2.6 Flash/Pro open-weight and drop unsupported pdf input UltraSpeed remains closed-weight. Align input modalities to text/image/audio/video. * docs(xiaomi): link MiMo V2.6 Flash/Pro Hugging Face weights * fix(xiaomi): mark MiMo V2.6 Pro-UltraSpeed open-weight with Pro-RL weights Same parameters as Pro; link the shared Hugging Face release. * fix(xiaomi): drop incorrect MiMo V2.6 knowledge cutoff The 2026-09-22 value was the release date, not a knowledge cutoff. Omit until a real cutoff is known. --- models/xiaomi/mimo-v2.6-flash.toml | 9 ++++++--- models/xiaomi/mimo-v2.6-pro-ultraspeed.toml | 8 ++++++-- models/xiaomi/mimo-v2.6-pro.toml | 9 ++++++--- providers/xiaomi/models/mimo-v2.6-flash.toml | 5 ++--- providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml | 4 ++-- providers/xiaomi/models/mimo-v2.6-pro.toml | 5 ++--- 6 files changed, 24 insertions(+), 16 deletions(-) diff --git a/models/xiaomi/mimo-v2.6-flash.toml b/models/xiaomi/mimo-v2.6-flash.toml index f065c84b627..2ac6276589a 100644 --- a/models/xiaomi/mimo-v2.6-flash.toml +++ b/models/xiaomi/mimo-v2.6-flash.toml @@ -7,13 +7,16 @@ attachment = true reasoning = true temperature = true tool_call = true -knowledge = "2026-09-22" -open_weights = false +open_weights = true [limit] context = 1_048_576 output = 131_072 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["text", "image", "audio", "video"] output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Flash-RL" diff --git a/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml index a4ef4e05975..82d134ed28a 100644 --- a/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml +++ b/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -7,12 +7,16 @@ attachment = true reasoning = true temperature = true tool_call = true -open_weights = false +open_weights = true [limit] context = 1_048_576 output = 131_072 [modalities] -input = ["text", "image", "video", "audio"] +input = ["text", "image", "audio", "video"] output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Pro-RL" diff --git a/models/xiaomi/mimo-v2.6-pro.toml b/models/xiaomi/mimo-v2.6-pro.toml index fea580df360..fc53e943157 100644 --- a/models/xiaomi/mimo-v2.6-pro.toml +++ b/models/xiaomi/mimo-v2.6-pro.toml @@ -7,13 +7,16 @@ attachment = true reasoning = true temperature = true tool_call = true -knowledge = "2026-09-22" -open_weights = false +open_weights = true [limit] context = 1_048_576 output = 131_072 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["text", "image", "audio", "video"] output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Pro-RL" diff --git a/providers/xiaomi/models/mimo-v2.6-flash.toml b/providers/xiaomi/models/mimo-v2.6-flash.toml index 338e97b8d17..8a98ed174da 100644 --- a/providers/xiaomi/models/mimo-v2.6-flash.toml +++ b/providers/xiaomi/models/mimo-v2.6-flash.toml @@ -9,8 +9,7 @@ attachment = true reasoning = true temperature = true tool_call = true -knowledge = "2026-09-22" -open_weights = false +open_weights = true [[reasoning_options]] type = "toggle" @@ -25,7 +24,7 @@ context = 1_048_576 output = 131_072 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["text", "image", "audio", "video"] output = ["text"] [interleaved] diff --git a/providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml b/providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml index f1d91260627..400e80cc7c5 100644 --- a/providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml +++ b/providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml @@ -9,7 +9,7 @@ attachment = true reasoning = true temperature = true tool_call = true -open_weights = false +open_weights = true [[reasoning_options]] type = "toggle" @@ -24,7 +24,7 @@ context = 1_048_576 output = 131_072 [modalities] -input = ["text", "image", "video", "audio"] +input = ["text", "image", "audio", "video"] output = ["text"] [interleaved] diff --git a/providers/xiaomi/models/mimo-v2.6-pro.toml b/providers/xiaomi/models/mimo-v2.6-pro.toml index 8debc33d06f..d3b71b59d4c 100644 --- a/providers/xiaomi/models/mimo-v2.6-pro.toml +++ b/providers/xiaomi/models/mimo-v2.6-pro.toml @@ -9,8 +9,7 @@ attachment = true reasoning = true temperature = true tool_call = true -knowledge = "2026-09-22" -open_weights = false +open_weights = true [[reasoning_options]] type = "toggle" @@ -25,7 +24,7 @@ context = 1_048_576 output = 131_072 [modalities] -input = ["text", "image", "audio", "video", "pdf"] +input = ["text", "image", "audio", "video"] output = ["text"] [interleaved] From 975556366c20f5980a197a7a7da84e781314dd53 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 23:01:55 -0500 Subject: [PATCH 374/392] chore(sync): update Cloudflare Workers AI model catalog (#7707) Co-authored-by: opencode-agent[bot] --- .../cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml b/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml index fc4739a9fbe..0dcc22de116 100644 --- a/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml @@ -3,7 +3,10 @@ # https://developers.cloudflare.com/workers-ai/models/gpt-oss-20b/sync-input.json (accessed 2026-06-25) base_model = "openai/gpt-oss-20b" description = "Open-weight GPT model for self-hosted reasoning and instruction-following workloads" -reasoning_options = [] + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 0.2 From 83ca66bff6a341b30720836510d4b905b8bc766c Mon Sep 17 00:00:00 2001 From: Abdul Munim Date: Wed, 23 Sep 2026 06:02:01 +0200 Subject: [PATCH 375/392] cline-pass: add missing mimo-v2.6-flash, mimo-v2.6-pro, muse-spark-1.3-contributor (#7729) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(cline-pass): add mimo-v2.6 and muse-spark-1.3 models Live clinePass catalog lists 14 models; three slugs were absent from providers/cline-pass/models/cline-pass/. All three serve requests on api.cline.bot (probe-verified 2026-09-22); nonexistent slugs 404. muse-spark-1.3-contributor carries Meta reference rates because Cline publishes no ClinePass rate for it. * fix(cline-pass): follow review — wire-cited effort, drop muse cost mimo-v2.6-* reasoning_options now list exactly the seven levels the ClinePass endpoint accepts (probe 2026-09-22: seven levels -> 200, unknown level -> 500), cited in a leading wire comment. muse-spark-1.3-contributor drops [cost]: Cline publishes no ClinePass rate for the slug. * fix(cline-pass): revert mimo effort to same-host peer set Effect probe (reasoning_tokens, 3 trials/level) shows minimal/max do not differentiate on this host, so declare the mimo-v2.5* peer set per review; wire comment keeps the full acceptance evidence. --- .../models/cline-pass/mimo-v2.6-flash.toml | 16 ++++++++++++++++ .../models/cline-pass/mimo-v2.6-pro.toml | 16 ++++++++++++++++ .../cline-pass/muse-spark-1.3-contributor.toml | 8 ++++++++ 3 files changed, 40 insertions(+) create mode 100644 providers/cline-pass/models/cline-pass/mimo-v2.6-flash.toml create mode 100644 providers/cline-pass/models/cline-pass/mimo-v2.6-pro.toml create mode 100644 providers/cline-pass/models/cline-pass/muse-spark-1.3-contributor.toml diff --git a/providers/cline-pass/models/cline-pass/mimo-v2.6-flash.toml b/providers/cline-pass/models/cline-pass/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..612d086a7f4 --- /dev/null +++ b/providers/cline-pass/models/cline-pass/mimo-v2.6-flash.toml @@ -0,0 +1,16 @@ +# Model ID cline-pass/mimo-v2.6-flash; reference pricing from the Cline model +# catalog https://api.cline.bot/api/v1/ai/cline/models (accessed 2026-09-22): +# $0.14 / $0.28 / $0.0028 per 1M tokens (input/output/cache read). +# Wire evidence (2026-09-22): POST /api/v1/chat/completions accepts +# reasoning_effort = none|low|medium|high|xhigh|minimal|max (HTTP 200) and +# rejects unknown levels (HTTP 500). An effect probe (3 trials per level, +# reasoning_tokens per response) shows no measurable difference for +# minimal/max, so this file declares the established same-host peer set +# used by cline-pass/mimo-v2.5* (none, low, medium, high, xhigh). +base_model = "xiaomi/mimo-v2.6-flash" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 diff --git a/providers/cline-pass/models/cline-pass/mimo-v2.6-pro.toml b/providers/cline-pass/models/cline-pass/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..d42aa0606d7 --- /dev/null +++ b/providers/cline-pass/models/cline-pass/mimo-v2.6-pro.toml @@ -0,0 +1,16 @@ +# Model ID cline-pass/mimo-v2.6-pro; reference pricing from the Cline model +# catalog https://api.cline.bot/api/v1/ai/cline/models (accessed 2026-09-22): +# $0.435 / $0.87 / $0.0036 per 1M tokens (input/output/cache read). +# Wire evidence (2026-09-22): POST /api/v1/chat/completions accepts +# reasoning_effort = none|low|medium|high|xhigh|minimal|max (HTTP 200) and +# rejects unknown levels (HTTP 500). An effect probe (reasoning_tokens per +# response) shows no measurable difference for minimal/max, so this file +# declares the established same-host peer set used by cline-pass/mimo-v2.5-pro +# (none, low, medium, high, xhigh). +base_model = "xiaomi/mimo-v2.6-pro" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 diff --git a/providers/cline-pass/models/cline-pass/muse-spark-1.3-contributor.toml b/providers/cline-pass/models/cline-pass/muse-spark-1.3-contributor.toml new file mode 100644 index 00000000000..a5f354380fd --- /dev/null +++ b/providers/cline-pass/models/cline-pass/muse-spark-1.3-contributor.toml @@ -0,0 +1,8 @@ +# Model ID cline-pass/muse-spark-1.3-contributor. Cline publishes no +# ClinePass reference rate for this model: docs.cline.bot/getting-started/ +# clinepass has no row for it (accessed 2026-09-22) and the model is absent +# from api.cline.bot/api/v1/ai/cline/models, so no [cost] block is declared. +base_model = "meta/muse-spark-1.3" +name = "Muse Spark 1.3 Contributor" + +reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high", "xhigh"] }] From 552f3b99c6595cd6dc579be2a19e5faef9be5e57 Mon Sep 17 00:00:00 2001 From: Tianning Li <51225681+lit26@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:03:44 +0800 Subject: [PATCH 376/392] feat(stepfun): add step-5-preview to StepFun providers (#7535) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds the Step 5 Preview lab entry and serves it from all four StepFun hosts: StepFun (China), StepFun (Global), and both Step Plan channels. Specs from the official model page: 1M context / input / output, text + image + video input, text output, tool calling, JSON Mode + JSON Schema, prompt caching, and low/medium/high reasoning effort via Chat `reasoning_effort`, Messages `output_config.effort`, and Responses `reasoning.effort`. Pricing: China converts the CNY list price (7 / 20 / 0.35 per 1M) at the same 7.2973 CNY/USD rate the other StepFun entries use; Global uses the published USD price (1.00 / 2.70 / 0.05) directly, so that entry is a real file rather than a symlink into providers/stepfun. Step Plan entries carry no cost block, matching their siblings — that channel bills against plan credit, not per token. Provider comments also updated: step-5-preview added to the effort matrix, and the Responses API now serves step-5-preview alongside step-3.7-flash. Sources: https://platform.stepfun.com/docs/zh/guides/models/step-5-preview https://platform.stepfun.ai/docs/en/guides/models/step-5-preview https://platform.stepfun.com/docs/zh/guides/pricing/details https://platform.stepfun.ai/docs/en/guides/pricing/details https://platform.stepfun.com/docs/zh/step-plan/integrations/reasoning-api Co-authored-by: Tianning Li --- .../models/step-5-preview.toml | 11 ++++++++++ providers/stepfun-ai-step-plan/provider.toml | 8 ++++---- providers/stepfun/models/step-5-preview.toml | 20 +++++++++++++++++++ providers/stepfun/provider.toml | 7 ++++--- 4 files changed, 39 insertions(+), 7 deletions(-) create mode 100644 providers/stepfun-ai-step-plan/models/step-5-preview.toml create mode 100644 providers/stepfun/models/step-5-preview.toml diff --git a/providers/stepfun-ai-step-plan/models/step-5-preview.toml b/providers/stepfun-ai-step-plan/models/step-5-preview.toml new file mode 100644 index 00000000000..674a616d307 --- /dev/null +++ b/providers/stepfun-ai-step-plan/models/step-5-preview.toml @@ -0,0 +1,11 @@ +# Chat `reasoning_effort` and Messages `output_config.effort` accept +# low/medium/high (accessed 2026-09-20). +# https://platform.stepfun.ai/docs/en/step-plan/integrations/reasoning-api +base_model = "stepfun/step-5-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/stepfun-ai-step-plan/provider.toml b/providers/stepfun-ai-step-plan/provider.toml index 15e9fb3b704..f357f0a530f 100644 --- a/providers/stepfun-ai-step-plan/provider.toml +++ b/providers/stepfun-ai-step-plan/provider.toml @@ -1,12 +1,12 @@ name = "StepFun Step Plan (Global)" env = ["STEPFUN_API_KEY"] npm = "@ai-sdk/openai-compatible" -# Reasoning HTTP format (accessed 2026-06-25): +# Reasoning HTTP format (accessed 2026-09-20): # Step Plan exposes POST /step_plan/v1/chat/completions with top-level # `reasoning_effort` and POST /step_plan/v1/messages with -# `output_config.effort`. step-3.7-flash accepts low/medium/high; -# step-3.5-flash and step-3.5-flash-2603 accept low/high. No plan Responses -# endpoint is listed. +# `output_config.effort`. step-5-preview and step-3.7-flash accept +# low/medium/high; step-3.5-flash and step-3.5-flash-2603 accept low/high. +# No plan Responses endpoint is listed. # Source: # https://platform.stepfun.ai/docs/en/step-plan/integrations/reasoning-api doc = "https://platform.stepfun.ai/docs/en/step-plan/integrations/reasoning-api" diff --git a/providers/stepfun/models/step-5-preview.toml b/providers/stepfun/models/step-5-preview.toml new file mode 100644 index 00000000000..94aa0f9f5f1 --- /dev/null +++ b/providers/stepfun/models/step-5-preview.toml @@ -0,0 +1,20 @@ +# Chat `reasoning_effort`, Messages `output_config.effort`, and Responses +# `reasoning.effort` accept low/medium/high (accessed 2026-09-20). +# https://platform.stepfun.com/docs/zh/guides/models/step-5-preview +# Cost converted from CNY list price (¥7 / ¥20 / ¥0.35 per 1M tokens) at the +# same 7.2973 CNY/USD rate used by the other stepfun entries (accessed +# 2026-09-20). +# https://platform.stepfun.com/docs/zh/guides/pricing/details +base_model = "stepfun/step-5-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.959 +output = 2.741 +cache_read = 0.048 diff --git a/providers/stepfun/provider.toml b/providers/stepfun/provider.toml index 0de8764f300..b687e0750f1 100644 --- a/providers/stepfun/provider.toml +++ b/providers/stepfun/provider.toml @@ -1,11 +1,12 @@ name = "StepFun (China)" env = ["STEPFUN_API_KEY"] npm = "@ai-sdk/openai-compatible" -# Reasoning HTTP format (accessed 2026-06-25): +# Reasoning HTTP format (accessed 2026-09-20): # POST /v1/chat/completions uses top-level `reasoning_effort`; POST /v1/messages # uses `output_config.effort`; POST /v1/responses uses `reasoning.effort`. -# Values are low/medium/high for step-3.7-flash; step-3.5-flash-2603 accepts -# low/high. Responses supports only step-3.7-flash. Chat returns reasoning at +# Values are low/medium/high for step-5-preview and step-3.7-flash; +# step-3.5-flash-2603 accepts low/high. Responses supports only +# step-5-preview and step-3.7-flash. Chat returns reasoning at # `choices[].message.reasoning` or streamed `choices[].delta.reasoning`; # `reasoning_format` is general (default) or deepseek-style, the latter using # `reasoning_content`. Responses streams `response.reasoning_text.delta`. From 1222d431c02455357c5b2cd31e613e41e0f06a9f Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 23:09:32 -0500 Subject: [PATCH 377/392] fix: add supported Kimi K3 Bedrock profiles (#7795) Co-authored-by: Aiden Cline --- .../models/global.moonshotai.kimi-k3.toml | 22 +++++++++++++++++++ .../models/us.moonshotai.kimi-k3.toml | 22 +++++++++++++++++++ 2 files changed, 44 insertions(+) create mode 100644 providers/amazon-bedrock/models/global.moonshotai.kimi-k3.toml create mode 100644 providers/amazon-bedrock/models/us.moonshotai.kimi-k3.toml diff --git a/providers/amazon-bedrock/models/global.moonshotai.kimi-k3.toml b/providers/amazon-bedrock/models/global.moonshotai.kimi-k3.toml new file mode 100644 index 00000000000..05556e9c377 --- /dev/null +++ b/providers/amazon-bedrock/models/global.moonshotai.kimi-k3.toml @@ -0,0 +1,22 @@ +# Sources: https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-moonshot-ai-kimi-k3.html +# https://docs.aws.amazon.com/bedrock/latest/userguide/conversation-inference.html +# Global CRIS standard pricing; cache_write is the 30-minute rate. +# Reasoning is always on; no Bedrock Converse effort control is documented or verified. +# Live Converse 2026-09-23: reasoning_effort accepted invalid strings, numbers, +# and objects exactly like an unrelated unknown field; low/max probes showed no +# verified effect. The bare model ID was rejected; an inference profile is required. +# Global and US profiles returned reasoningContent. +# Bedrock supports text and image input, but not the base model's video input. +base_model = "moonshotai/kimi-k3" +name = "Kimi K3 (Global)" +reasoning_options = [] +interleaved = true + +[cost] +input = 3.00 +output = 15.00 +cache_read = 0.30 +cache_write = 3.75 + +[modalities] +input = ["text", "image"] diff --git a/providers/amazon-bedrock/models/us.moonshotai.kimi-k3.toml b/providers/amazon-bedrock/models/us.moonshotai.kimi-k3.toml new file mode 100644 index 00000000000..9c724cee821 --- /dev/null +++ b/providers/amazon-bedrock/models/us.moonshotai.kimi-k3.toml @@ -0,0 +1,22 @@ +# Sources: https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-moonshot-ai-kimi-k3.html +# https://docs.aws.amazon.com/bedrock/latest/userguide/conversation-inference.html +# US CRIS standard pricing; cache_write is the 30-minute rate. +# Reasoning is always on; no Bedrock Converse effort control is documented or verified. +# Live Converse 2026-09-23: reasoning_effort accepted invalid strings, numbers, +# and objects exactly like an unrelated unknown field; low/max probes showed no +# verified effect. The bare model ID was rejected; an inference profile is required. +# Global and US profiles returned reasoningContent. +# Bedrock supports text and image input, but not the base model's video input. +base_model = "moonshotai/kimi-k3" +name = "Kimi K3 (US)" +reasoning_options = [] +interleaved = true + +[cost] +input = 3.30 +output = 16.50 +cache_read = 0.33 +cache_write = 4.125 + +[modalities] +input = ["text", "image"] From 07fb48aae70f62ae6d63b7a7121347aa88171380 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 22 Sep 2026 23:13:28 -0500 Subject: [PATCH 378/392] fix: ClinePass thinking-effort levels are wrong for all models: registry shows none/low/medium/high/xhigh, but DeepSeek V4.1 Flash officially supports off/low/high/max (#7534) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- .../cline-pass/models/cline-pass/deepseek-v4-flash.toml | 6 +++++- .../cline-pass/models/cline-pass/deepseek-v4-pro.toml | 6 +++++- .../cline-pass/models/cline-pass/deepseek-v4.1-flash.toml | 6 ++++-- providers/cline-pass/models/cline-pass/glm-5.2.toml | 6 +++++- providers/cline-pass/models/cline-pass/kimi-k2.6.toml | 5 ++++- .../cline-pass/models/cline-pass/kimi-k2.7-code.toml | 5 ++++- providers/cline-pass/models/cline-pass/kimi-k3.toml | 4 +++- providers/cline-pass/models/cline-pass/mimo-v2.5-pro.toml | 5 ++++- providers/cline-pass/models/cline-pass/mimo-v2.5.toml | 5 ++++- providers/cline-pass/models/cline-pass/minimax-m3.toml | 5 ++++- providers/cline-pass/models/cline-pass/qwen3.7-max.toml | 6 +++++- providers/cline-pass/models/cline-pass/qwen3.7-plus.toml | 6 +++++- providers/cline-pass/models/cline-pass/qwen3.8-max.toml | 8 ++++---- 13 files changed, 56 insertions(+), 17 deletions(-) diff --git a/providers/cline-pass/models/cline-pass/deepseek-v4-flash.toml b/providers/cline-pass/models/cline-pass/deepseek-v4-flash.toml index a402b4d533b..044e118d9af 100644 --- a/providers/cline-pass/models/cline-pass/deepseek-v4-flash.toml +++ b/providers/cline-pass/models/cline-pass/deepseek-v4-flash.toml @@ -1,5 +1,9 @@ +# https://docs.cline.bot/getting-started/clinepass +# https://api-docs.deepseek.com/guides/thinking_mode/ (OpenAI: thinking.type + reasoning_effort low|high|max) +# Effort: reasoning_effort = none|low|high|max (none disables thinking; DeepSeek Flash graded set) +# Legacy Flash slug; lab serves V4.1-Flash. Same reasoning surface as deepseek-v4.1-flash. base_model = "deepseek/deepseek-v4-flash" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] [cost] input = 0.14 diff --git a/providers/cline-pass/models/cline-pass/deepseek-v4-pro.toml b/providers/cline-pass/models/cline-pass/deepseek-v4-pro.toml index 3fb0bd31762..191909977a6 100644 --- a/providers/cline-pass/models/cline-pass/deepseek-v4-pro.toml +++ b/providers/cline-pass/models/cline-pass/deepseek-v4-pro.toml @@ -1,5 +1,9 @@ +# https://docs.cline.bot/getting-started/clinepass +# https://api-docs.deepseek.com/guides/thinking_mode/ (OpenAI: thinking.type + reasoning_effort) +# First-party DeepSeek Pro: effort high|max (lab providers/deepseek/models/deepseek-v4-pro.toml) +# Effort: reasoning_effort = none|high|max (none disables thinking) base_model = "deepseek/deepseek-v4-pro" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "effort", values = ["none", "high", "max"] }] [cost] input = 1.74 diff --git a/providers/cline-pass/models/cline-pass/deepseek-v4.1-flash.toml b/providers/cline-pass/models/cline-pass/deepseek-v4.1-flash.toml index cdc87207c83..453a58da360 100644 --- a/providers/cline-pass/models/cline-pass/deepseek-v4.1-flash.toml +++ b/providers/cline-pass/models/cline-pass/deepseek-v4.1-flash.toml @@ -1,9 +1,11 @@ # https://api.cline.bot/api/v1/ai/cline/recommended-models (clinePass id cline-pass/deepseek-v4.1-flash, accessed 2026-09-11) # https://docs.cline.bot/getting-started/clinepass (ClinePass reference pricing cites DeepSeek API pricing for Flash quota) # https://api-docs.deepseek.com/quick_start/pricing (DeepSeek-V4.1-Flash off-peak USD/MTok; ClinePass docs table still lists deepseek-v4-flash only) -# Reasoning effort values match other ClinePass DeepSeek entries on this host. +# https://api-docs.deepseek.com/guides/thinking_mode/ (OpenAI: thinking.type + reasoning_effort low|high|max; medium/xhigh map to high) +# Effort: reasoning_effort = none|low|high|max (none disables thinking; matches DeepSeek Flash graded set + off) +# ClinePass is an OpenAI-compatible relay; do not advertise GPT-style medium/xhigh. base_model = "deepseek/deepseek-v4.1-flash" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] [cost] input = 0.15 diff --git a/providers/cline-pass/models/cline-pass/glm-5.2.toml b/providers/cline-pass/models/cline-pass/glm-5.2.toml index b3cac6301a6..7458eed5caa 100644 --- a/providers/cline-pass/models/cline-pass/glm-5.2.toml +++ b/providers/cline-pass/models/cline-pass/glm-5.2.toml @@ -1,5 +1,9 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Zhipu GLM-5.2: effective effort high|max; none|minimal skip thinking +# (providers/zhipuai/models/glm-5.2.toml; https://docs.bigmodel.cn/cn/guide/capabilities/thinking) +# Effort: reasoning_effort = none|high|max base_model = "zhipuai/glm-5.2" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "effort", values = ["none", "high", "max"] }] [cost] input = 1.4 diff --git a/providers/cline-pass/models/cline-pass/kimi-k2.6.toml b/providers/cline-pass/models/cline-pass/kimi-k2.6.toml index 373fb552c69..fc3fb04b0db 100644 --- a/providers/cline-pass/models/cline-pass/kimi-k2.6.toml +++ b/providers/cline-pass/models/cline-pass/kimi-k2.6.toml @@ -1,5 +1,8 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Moonshot K2.6: thinking on/off only (providers/moonshotai/models/kimi-k2.6.toml) +# Toggle: thinking / reasoning enabled|disabled (OpenAI-compatible relay; no graded effort on lab) base_model = "moonshotai/kimi-k2.6" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 0.95 diff --git a/providers/cline-pass/models/cline-pass/kimi-k2.7-code.toml b/providers/cline-pass/models/cline-pass/kimi-k2.7-code.toml index 7340bf79c7a..9fc9a2f9a5a 100644 --- a/providers/cline-pass/models/cline-pass/kimi-k2.7-code.toml +++ b/providers/cline-pass/models/cline-pass/kimi-k2.7-code.toml @@ -1,5 +1,8 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Moonshot K2.7 Code: always-on reasoning, no caller control +# (providers/moonshotai/models/kimi-k2.7-code.toml; openrouter peer also []) base_model = "moonshotai/kimi-k2.7-code" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [] [cost] input = 0.95 diff --git a/providers/cline-pass/models/cline-pass/kimi-k3.toml b/providers/cline-pass/models/cline-pass/kimi-k3.toml index 856a9963ae3..2b9dee1d87c 100644 --- a/providers/cline-pass/models/cline-pass/kimi-k3.toml +++ b/providers/cline-pass/models/cline-pass/kimi-k3.toml @@ -1,7 +1,9 @@ # https://docs.cline.bot/getting-started/clinepass (accessed 2026-07-22) # Model ID cline-pass/kimi-k3; reference pricing $3 / $15 / $0.30 per 1M tokens (input/output/cache read). +# Lab Moonshot K3: thinking.type + output_config.effort low|high|max (providers/moonshotai/models/kimi-k3.toml) +# Effort: reasoning_effort = none|low|high|max (none disables; no medium/xhigh) base_model = "moonshotai/kimi-k3" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] [cost] input = 3 diff --git a/providers/cline-pass/models/cline-pass/mimo-v2.5-pro.toml b/providers/cline-pass/models/cline-pass/mimo-v2.5-pro.toml index 4b15e595d0b..8ff3d62b6dc 100644 --- a/providers/cline-pass/models/cline-pass/mimo-v2.5-pro.toml +++ b/providers/cline-pass/models/cline-pass/mimo-v2.5-pro.toml @@ -1,5 +1,8 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Xiaomi MiMo-V2.5-Pro: thinking on/off only (providers/xiaomi/models/mimo-v2.5-pro.toml) +# Toggle: thinking / reasoning enabled|disabled (no graded effort on lab) base_model = "xiaomi/mimo-v2.5-pro" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 1.74 diff --git a/providers/cline-pass/models/cline-pass/mimo-v2.5.toml b/providers/cline-pass/models/cline-pass/mimo-v2.5.toml index a05c72ca039..3a9f90ed0bc 100644 --- a/providers/cline-pass/models/cline-pass/mimo-v2.5.toml +++ b/providers/cline-pass/models/cline-pass/mimo-v2.5.toml @@ -1,5 +1,8 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Xiaomi MiMo-V2.5: thinking on/off only (providers/xiaomi/models/mimo-v2.5.toml) +# Toggle: thinking / reasoning enabled|disabled (no graded effort on lab) base_model = "xiaomi/mimo-v2.5" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 0.14 diff --git a/providers/cline-pass/models/cline-pass/minimax-m3.toml b/providers/cline-pass/models/cline-pass/minimax-m3.toml index 1367e07bc0e..599f9fc9ec1 100644 --- a/providers/cline-pass/models/cline-pass/minimax-m3.toml +++ b/providers/cline-pass/models/cline-pass/minimax-m3.toml @@ -1,5 +1,8 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab MiniMax-M3: thinking on/off only (providers/minimax/models/MiniMax-M3.toml) +# Toggle: thinking / reasoning enabled|disabled (no graded effort on lab) base_model = "minimax/MiniMax-M3" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 0.3 diff --git a/providers/cline-pass/models/cline-pass/qwen3.7-max.toml b/providers/cline-pass/models/cline-pass/qwen3.7-max.toml index 7ec1495bb20..302e667ede8 100644 --- a/providers/cline-pass/models/cline-pass/qwen3.7-max.toml +++ b/providers/cline-pass/models/cline-pass/qwen3.7-max.toml @@ -1,5 +1,9 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Alibaba Qwen3.7 Max: enable_thinking + thinking_budget (providers/alibaba/models/qwen3.7-max.toml) +# ClinePass OpenAI-compatible Chat Completions path: peers (OpenRouter) expose on/off only, not budget_tokens. +# Toggle: enable_thinking / reasoning enabled|disabled base_model = "alibaba/qwen3.7-max" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 2.5 diff --git a/providers/cline-pass/models/cline-pass/qwen3.7-plus.toml b/providers/cline-pass/models/cline-pass/qwen3.7-plus.toml index 55e28b335eb..4c0d4b2c041 100644 --- a/providers/cline-pass/models/cline-pass/qwen3.7-plus.toml +++ b/providers/cline-pass/models/cline-pass/qwen3.7-plus.toml @@ -1,5 +1,9 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Alibaba Qwen3.7 Plus: enable_thinking + thinking_budget (providers/alibaba/models/qwen3.7-plus.toml) +# ClinePass OpenAI-compatible Chat Completions path: peers (OpenRouter) expose on/off only, not budget_tokens. +# Toggle: enable_thinking / reasoning enabled|disabled base_model = "alibaba/qwen3.7-plus" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 0.4 diff --git a/providers/cline-pass/models/cline-pass/qwen3.8-max.toml b/providers/cline-pass/models/cline-pass/qwen3.8-max.toml index a72991d846b..6a9d910be85 100644 --- a/providers/cline-pass/models/cline-pass/qwen3.8-max.toml +++ b/providers/cline-pass/models/cline-pass/qwen3.8-max.toml @@ -1,11 +1,11 @@ # https://docs.cline.bot/getting-started/clinepass (accessed 2026-08-24) # Model ID cline-pass/qwen3.8-max; reference pricing $2 / $6 / $0.25 per 1M tokens (input/output/cache read), cache write $2.5. +# Lab Alibaba Qwen3.8 Max: enable_thinking + reasoning_effort low|medium|xhigh +# (providers/alibaba/models/qwen3.8-max.toml; https://docs.qwencloud.com/developer-guides/text-generation/thinking) +# Effort includes none for off (hybrid model); no high (lab maps high→xhigh; native graded set is low|medium|xhigh) base_model = "alibaba/qwen3.8-max" structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high", "xhigh"] +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "xhigh"] }] [cost] input = 2 From 76689e21b85fc1b87fe6fbcac0a5d4aeb61b0c23 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:31:20 +0000 Subject: [PATCH 379/392] chore(sync): update NanoGPT model catalog (#7839) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 6 +++--- providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml | 3 --- providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml | 3 --- 3 files changed, 3 insertions(+), 9 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index a62995b0beb..b4cce19afb3 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.44 -output = 1.32 -cache_read = 0.014 +input = 0.22 +output = 0.66 +cache_read = 0.007 [limit] context = 1_048_576 diff --git a/providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml b/providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml index bf3082ff91f..79ac140f7b6 100644 --- a/providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml +++ b/providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml @@ -14,6 +14,3 @@ cache_write = 0 [limit] input = 1_048_576 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml b/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml index ea8e65a904c..a99c234165d 100644 --- a/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml +++ b/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml @@ -14,6 +14,3 @@ cache_write = 0 [limit] input = 1_048_576 - -[modalities] -input = ["text", "image", "video", "audio"] From 7d4b261ad404a5211432478df74e9e1a01a917e4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:31:24 +0000 Subject: [PATCH 380/392] chore(sync): update OpenRouter model catalog (#7840) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml | 3 --- providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml | 3 --- .../openrouter/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- 5 files changed, 9 insertions(+), 15 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index b8d53b0809c..7c5828b16e2 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.32 -output = 3.96 -cache_read = 0.044 +input = 0.66 +output = 1.98 +cache_read = 0.022 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index ee75b6ad088..c689935f095 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.1 -output = 0.5 -cache_read = 0.01 +input = 0.094 +output = 0.6 +cache_read = 0.05 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml b/providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml index 3e52e783fa7..e1bb054f013 100644 --- a/providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml +++ b/providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml @@ -11,6 +11,3 @@ type = "toggle" input = 0.14 output = 0.28 cache_read = 0.0028 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml b/providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml index 2f40e81f221..7f938318372 100644 --- a/providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml +++ b/providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml @@ -11,6 +11,3 @@ type = "toggle" input = 0.435 output = 0.87 cache_read = 0.0036 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index a3bff918037..2ca0c23e6a3 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.1 -output = 0.5 -cache_read = 0.01 +input = 0.094 +output = 0.6 +cache_read = 0.05 [limit] context = 1_048_576 From e4eb305bc0faa591cbf275f65001629e85e0211a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:31:32 +0000 Subject: [PATCH 381/392] chore(sync): update EmpirioLabs AI model catalog (#7837) Co-authored-by: opencode-agent[bot] --- providers/empiriolabs/models/mimo-v2-6-flash.toml | 3 --- providers/empiriolabs/models/mimo-v2-6-pro.toml | 3 --- 2 files changed, 6 deletions(-) diff --git a/providers/empiriolabs/models/mimo-v2-6-flash.toml b/providers/empiriolabs/models/mimo-v2-6-flash.toml index ab796f4d7ef..14d640b4db5 100644 --- a/providers/empiriolabs/models/mimo-v2-6-flash.toml +++ b/providers/empiriolabs/models/mimo-v2-6-flash.toml @@ -12,6 +12,3 @@ cache_read = 0.7 [limit] context = 1_000_000 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/empiriolabs/models/mimo-v2-6-pro.toml b/providers/empiriolabs/models/mimo-v2-6-pro.toml index 04dcc7ceb0e..54b2f7069c3 100644 --- a/providers/empiriolabs/models/mimo-v2-6-pro.toml +++ b/providers/empiriolabs/models/mimo-v2-6-pro.toml @@ -12,6 +12,3 @@ cache_read = 2.175 [limit] context = 1_000_000 - -[modalities] -input = ["text", "image", "video", "audio"] From 59eda218803092a288b792a51e5e1ce081480f3c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:31:35 +0000 Subject: [PATCH 382/392] chore(sync): update Kilo model catalog (#7838) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/xiaomi/mimo-v2.6-flash.toml | 3 --- providers/kilo/models/xiaomi/mimo-v2.6-pro.toml | 3 --- providers/kilo/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- 3 files changed, 3 insertions(+), 9 deletions(-) diff --git a/providers/kilo/models/xiaomi/mimo-v2.6-flash.toml b/providers/kilo/models/xiaomi/mimo-v2.6-flash.toml index b4165fb24ad..0a1c6e4fe63 100644 --- a/providers/kilo/models/xiaomi/mimo-v2.6-flash.toml +++ b/providers/kilo/models/xiaomi/mimo-v2.6-flash.toml @@ -10,6 +10,3 @@ values = ["none", "high"] input = 0.14 output = 0.28 cache_read = 0.0028 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/kilo/models/xiaomi/mimo-v2.6-pro.toml b/providers/kilo/models/xiaomi/mimo-v2.6-pro.toml index e3e0a47689b..1285f109afd 100644 --- a/providers/kilo/models/xiaomi/mimo-v2.6-pro.toml +++ b/providers/kilo/models/xiaomi/mimo-v2.6-pro.toml @@ -10,6 +10,3 @@ values = ["none", "high"] input = 0.435 output = 0.87 cache_read = 0.0036 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 0295282190d..48b2c256919 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.1 -output = 0.5 -cache_read = 0.01 +input = 0.094 +output = 0.6 +cache_read = 0.05 [limit] context = 1_048_576 From 7d87c4fc34b195d8d5fd449903054f666fbb3d24 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:31:48 +0000 Subject: [PATCH 383/392] chore(sync): update CrossModel model catalog (#7841) Co-authored-by: opencode-agent[bot] --- providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml | 3 --- providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml | 3 --- 2 files changed, 6 deletions(-) diff --git a/providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml b/providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml index 51fd77c6aa8..5fdd8d29025 100644 --- a/providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml +++ b/providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml @@ -9,6 +9,3 @@ input = 0.16 output = 0.32 cache_read = 0.004 cache_write = 0.16 - -[modalities] -input = ["text", "image", "audio", "video"] diff --git a/providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml b/providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml index e0fca43e997..1dbc817d0fa 100644 --- a/providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml +++ b/providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml @@ -9,6 +9,3 @@ input = 0.47 output = 0.94 cache_read = 0.005 cache_write = 0.47 - -[modalities] -input = ["text", "image", "audio", "video"] From 1c94dc631474bfb5c37251758f2a280ae3a3a91a Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:27:40 +0000 Subject: [PATCH 384/392] chore(sync): update Kilo model catalog (#7845) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/qwen/qwen3.8-2.4t-a95b.toml | 1 + providers/kilo/models/~deepseek/deepseek-flash-latest.toml | 4 ++-- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/providers/kilo/models/qwen/qwen3.8-2.4t-a95b.toml b/providers/kilo/models/qwen/qwen3.8-2.4t-a95b.toml index 39ad2aa079f..c79bdfce589 100644 --- a/providers/kilo/models/qwen/qwen3.8-2.4t-a95b.toml +++ b/providers/kilo/models/qwen/qwen3.8-2.4t-a95b.toml @@ -16,3 +16,4 @@ cache_write = 2.5 [limit] context = 1_000_000 +output = 262_144 diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 48b2c256919..8923039f0f4 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.094 +input = 0.079 output = 0.6 -cache_read = 0.05 +cache_read = 0.06 [limit] context = 1_048_576 From b3318a67977cea8231db5fa772f1ba1900ded7f9 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:27:44 +0000 Subject: [PATCH 385/392] chore(sync): update Eden AI model catalog (#7847) Co-authored-by: opencode-agent[bot] --- .../models/amazon/moonshotai.kimi-k2.5.toml | 1 - .../models/anthropic/claude-opus-5-5.toml | 12 ++++++++++++ .../databricks-deepseek-v4-flash-0731.toml | 2 +- .../databricks-deepseek-v4-pro-0813.toml | 6 +++--- .../models/databricks/databricks-inkling.toml | 6 +++--- providers/edenai/models/openai/gpt-6-luna.toml | 18 ++++++++++++++++++ providers/edenai/models/openai/gpt-6-sol.toml | 18 ++++++++++++++++++ 7 files changed, 55 insertions(+), 8 deletions(-) create mode 100644 providers/edenai/models/anthropic/claude-opus-5-5.toml create mode 100644 providers/edenai/models/openai/gpt-6-luna.toml create mode 100644 providers/edenai/models/openai/gpt-6-sol.toml diff --git a/providers/edenai/models/amazon/moonshotai.kimi-k2.5.toml b/providers/edenai/models/amazon/moonshotai.kimi-k2.5.toml index 45d5d0a17cb..c64d911d778 100644 --- a/providers/edenai/models/amazon/moonshotai.kimi-k2.5.toml +++ b/providers/edenai/models/amazon/moonshotai.kimi-k2.5.toml @@ -1,6 +1,5 @@ base_model = "moonshotai/kimi-k2.5" name = "Kimi K2.5 (Amazon Bedrock)" -structured_output = false reasoning_options = [] [cost] diff --git a/providers/edenai/models/anthropic/claude-opus-5-5.toml b/providers/edenai/models/anthropic/claude-opus-5-5.toml new file mode 100644 index 00000000000..8d8ff11a7d2 --- /dev/null +++ b/providers/edenai/models/anthropic/claude-opus-5-5.toml @@ -0,0 +1,12 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/edenai/models/databricks/databricks-deepseek-v4-flash-0731.toml b/providers/edenai/models/databricks/databricks-deepseek-v4-flash-0731.toml index d4d7d35c658..ec6f7cacc88 100644 --- a/providers/edenai/models/databricks/databricks-deepseek-v4-flash-0731.toml +++ b/providers/edenai/models/databricks/databricks-deepseek-v4-flash-0731.toml @@ -9,5 +9,5 @@ values = ["none", "low", "high", "max"] [cost] input = 0.14 output = 0.28 -cache_read = 0.028 +cache_read = 0.014 cache_write = 0.14 diff --git a/providers/edenai/models/databricks/databricks-deepseek-v4-pro-0813.toml b/providers/edenai/models/databricks/databricks-deepseek-v4-pro-0813.toml index a25d6b3e95b..a293c8ac08b 100644 --- a/providers/edenai/models/databricks/databricks-deepseek-v4-pro-0813.toml +++ b/providers/edenai/models/databricks/databricks-deepseek-v4-pro-0813.toml @@ -7,7 +7,7 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.31999 -output = 3.95997 -cache_read = 0.13202 +input = 1.32 +output = 3.959999 +cache_read = 0.132 cache_write = 1.31999 diff --git a/providers/edenai/models/databricks/databricks-inkling.toml b/providers/edenai/models/databricks/databricks-inkling.toml index 177f6bd3f2d..eef08eb01d0 100644 --- a/providers/edenai/models/databricks/databricks-inkling.toml +++ b/providers/edenai/models/databricks/databricks-inkling.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "minimal", "low", "medium", "high", "max"] [cost] -input = 1.00002 -output = 4.04999 -cache_read = 0.17003 +input = 1 +output = 4.05 +cache_read = 0.1 cache_write = 1.00002 [limit] diff --git a/providers/edenai/models/openai/gpt-6-luna.toml b/providers/edenai/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..9a9537cbf02 --- /dev/null +++ b/providers/edenai/models/openai/gpt-6-luna.toml @@ -0,0 +1,18 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 diff --git a/providers/edenai/models/openai/gpt-6-sol.toml b/providers/edenai/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..94f694d64de --- /dev/null +++ b/providers/edenai/models/openai/gpt-6-sol.toml @@ -0,0 +1,18 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 From d149193b3a0397cdb37b4105a26a543ad79ea2a7 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:27:46 +0000 Subject: [PATCH 386/392] chore(sync): update OpenRouter model catalog (#7846) Co-authored-by: opencode-agent[bot] --- providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml | 4 ++-- providers/openrouter/models/qwen/qwen3.8-2.4t-a95b.toml | 1 + .../openrouter/models/~deepseek/deepseek-flash-latest.toml | 4 ++-- 3 files changed, 5 insertions(+), 4 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index c689935f095..e0411f29de8 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.094 +input = 0.079 output = 0.6 -cache_read = 0.05 +cache_read = 0.06 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3.8-2.4t-a95b.toml b/providers/openrouter/models/qwen/qwen3.8-2.4t-a95b.toml index 74b5d1276f6..abacdd68c88 100644 --- a/providers/openrouter/models/qwen/qwen3.8-2.4t-a95b.toml +++ b/providers/openrouter/models/qwen/qwen3.8-2.4t-a95b.toml @@ -18,3 +18,4 @@ cache_read = 0.25 [limit] context = 1_048_576 +output = 262_144 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index 2ca0c23e6a3..aeddf9446b1 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.094 +input = 0.079 output = 0.6 -cache_read = 0.05 +cache_read = 0.06 [limit] context = 1_048_576 From f380ec9a313fa04cacf315c2d78402aea21a416e Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:41:58 +0000 Subject: [PATCH 387/392] chore(sync): update NanoGPT model catalog (#7848) Co-authored-by: opencode-agent[bot] --- .../models/deepseek/deepseek-v4-flash-vision-exp.toml | 6 +++--- providers/nano-gpt/models/nano/lumen-stealth.toml | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index b4cce19afb3..a62995b0beb 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.22 -output = 0.66 -cache_read = 0.007 +input = 0.44 +output = 1.32 +cache_read = 0.014 [limit] context = 1_048_576 diff --git a/providers/nano-gpt/models/nano/lumen-stealth.toml b/providers/nano-gpt/models/nano/lumen-stealth.toml index 13c7ea58999..9b27918c763 100644 --- a/providers/nano-gpt/models/nano/lumen-stealth.toml +++ b/providers/nano-gpt/models/nano/lumen-stealth.toml @@ -16,7 +16,7 @@ output = 0 [limit] context = 200_000 input = 200_000 -output = 16_000 +output = 100_000 [modalities] input = ["text", "image"] From ceec55e82ddeb5028bf335903b8d85b8833b180c Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:42:04 +0000 Subject: [PATCH 388/392] chore(sync): update OpenRouter model catalog (#7849) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4-pro-0813.toml | 6 +++--- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- providers/openrouter/models/qwen/qwen3.8-2.4t-a95b.toml | 1 - .../openrouter/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- 4 files changed, 9 insertions(+), 10 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..b8d53b0809c 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 1.32 +output = 3.96 +cache_read = 0.044 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index e0411f29de8..0e2589390ee 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.079 -output = 0.6 -cache_read = 0.06 +input = 0.04 +output = 0.64 +cache_read = 0.016 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/qwen/qwen3.8-2.4t-a95b.toml b/providers/openrouter/models/qwen/qwen3.8-2.4t-a95b.toml index abacdd68c88..74b5d1276f6 100644 --- a/providers/openrouter/models/qwen/qwen3.8-2.4t-a95b.toml +++ b/providers/openrouter/models/qwen/qwen3.8-2.4t-a95b.toml @@ -18,4 +18,3 @@ cache_read = 0.25 [limit] context = 1_048_576 -output = 262_144 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index aeddf9446b1..893b093ef16 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.079 -output = 0.6 -cache_read = 0.06 +input = 0.04 +output = 0.64 +cache_read = 0.016 [limit] context = 1_048_576 From 812f7a1be29cb6da43494bae51c0ab3fcb517ad3 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:42:17 +0000 Subject: [PATCH 389/392] chore(sync): update Kilo model catalog (#7850) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/qwen/qwen3.8-2.4t-a95b.toml | 1 - providers/kilo/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- 2 files changed, 3 insertions(+), 4 deletions(-) diff --git a/providers/kilo/models/qwen/qwen3.8-2.4t-a95b.toml b/providers/kilo/models/qwen/qwen3.8-2.4t-a95b.toml index c79bdfce589..39ad2aa079f 100644 --- a/providers/kilo/models/qwen/qwen3.8-2.4t-a95b.toml +++ b/providers/kilo/models/qwen/qwen3.8-2.4t-a95b.toml @@ -16,4 +16,3 @@ cache_write = 2.5 [limit] context = 1_000_000 -output = 262_144 diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 8923039f0f4..4173b146120 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.079 -output = 0.6 -cache_read = 0.06 +input = 0.04 +output = 0.64 +cache_read = 0.016 [limit] context = 1_048_576 From 519f233069fc0749386b97d41c3373c9b7de470d Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 07:31:19 +0000 Subject: [PATCH 390/392] chore(sync): update OpenRouter model catalog (#7855) Co-authored-by: opencode-agent[bot] --- .../openrouter/models/deepseek/deepseek-v4.1-flash.toml | 6 +++--- .../openrouter/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- providers/openrouter/models/~moonshotai/kimi-latest.toml | 6 +++--- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 0e2589390ee..ffb1c4c680d 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.04 -output = 0.64 -cache_read = 0.016 +input = 0.039 +output = 0.6 +cache_read = 0.038 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index 893b093ef16..c649cf1995c 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.04 -output = 0.64 -cache_read = 0.016 +input = 0.039 +output = 0.6 +cache_read = 0.038 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index 0aebd435d56..4c30be833f9 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.4989 -output = 10.758 -cache_read = 0.3 +input = 1.49 +output = 14.5 +cache_read = 0.21 [limit] context = 1_048_576 From e3468982b4d22c9bf86e739d606de38f4fbf55f4 Mon Sep 17 00:00:00 2001 From: "opencode-agent[bot]" <219766164+opencode-agent[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 07:31:28 +0000 Subject: [PATCH 391/392] chore(sync): update Kilo model catalog (#7854) Co-authored-by: opencode-agent[bot] --- providers/kilo/models/~deepseek/deepseek-flash-latest.toml | 6 +++--- providers/kilo/models/~moonshotai/kimi-latest.toml | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 4173b146120..6b3302e301e 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.04 -output = 0.64 -cache_read = 0.016 +input = 0.039 +output = 0.6 +cache_read = 0.038 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 366b401a83c..163a53ae425 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.4989 -output = 10.758 -cache_read = 0.3 +input = 1.49 +output = 14.5 +cache_read = 0.21 [limit] context = 1_048_576 From b02a7650c0adc30c1685e2ff093231e84cb02572 Mon Sep 17 00:00:00 2001 From: Wynand Huizinga Date: Wed, 23 Sep 2026 09:34:50 +0200 Subject: [PATCH 392/392] fix(nebul): mark Kimi-K3 non-reasoning to match supports_reasoning = false --- packages/core/src/sync/providers/nebul.ts | 19 ++++++++------- packages/core/test/nebul.test.ts | 16 +++++++++++++ .../nebul/models/moonshotai/Kimi-K3.toml | 23 +++++++++---------- sync.md | 1 + 4 files changed, 39 insertions(+), 20 deletions(-) diff --git a/packages/core/src/sync/providers/nebul.ts b/packages/core/src/sync/providers/nebul.ts index 4c61e9e98e7..0207352d7f2 100644 --- a/packages/core/src/sync/providers/nebul.ts +++ b/packages/core/src/sync/providers/nebul.ts @@ -110,25 +110,28 @@ export const nebul = { : existing?.cost; const limit = info.max_input_tokens != null ? { context: info.max_input_tokens } : existing?.limit; if (existing === undefined && (baseModel === undefined || cost === undefined || limit === undefined)) return undefined; + // A hand-authored reasoning = false marks a served ID whose lab model reasons + // but which this host runs with thinking disabled (the catalog reports + // supports_reasoning = false and no reasoning_efforts). Keep the override and + // suppress the control/trace machinery entirely: no reasoning_options to + // require, and no interleaved side channel when no traces are returned. + const reasoningDisabled = existing?.reasoning === false; // Fail closed rather than emitting no reasoning_options: a reasoner with // neither advertised efforts nor authored options would sync as an empty // entry (no caller control). The runner keeps the file and lists it in the // skipped notice so the options can be hand-authored. - const isReasoner = baseModel !== undefined + const isReasoner = !reasoningDisabled && (baseModel !== undefined ? modelMetadata(baseModel).reasoning === true - : existing?.reasoning === true; + : existing?.reasoning === true); if (isReasoner && (info.reasoning_efforts ?? []).length === 0 && existing?.reasoning_options === undefined) { throw new MissingReasoningOptionsError( id, `${id} is a reasoning model, but Nebul advertises no reasoning_efforts and the catalog entry has no reasoning_options; hand-author them`, ); } - const values = { - interleaved: existing?.interleaved, - reasoning_options: buildReasoningOptions(entry, existing), - cost, - limit, - }; + const values = reasoningDisabled + ? { reasoning: false, interleaved: undefined, reasoning_options: undefined, cost, limit } + : { interleaved: existing?.interleaved, reasoning_options: buildReasoningOptions(entry, existing), cost, limit }; if (baseModel !== undefined) { return { id, diff --git a/packages/core/test/nebul.test.ts b/packages/core/test/nebul.test.ts index 63399f45bc1..f716af6b2cd 100644 --- a/packages/core/test/nebul.test.ts +++ b/packages/core/test/nebul.test.ts @@ -81,6 +81,22 @@ test("keeps authored effort sets when the host advertises none", () => { expect(translated?.model.reasoning_options).toEqual(authored); }); +test("keeps an authored reasoning = false override for a lab reasoner the host serves without thinking", () => { + const existing = { + base_model: "moonshotai/kimi-k3", + reasoning: false, + reasoning_options: [{ type: "effort" as const, values: ["low"] }], + interleaved: { field: "reasoning_content" as const }, + } as ExistingModel; + const translated = nebul.translateModel( + nebulEntry("moonshotai/Kimi-K3", { reasoning_efforts: undefined }), + context(existing), + ); + expect(translated?.model.reasoning).toBe(false); + expect(translated?.model.reasoning_options).toBeUndefined(); + expect(translated?.model.interleaved).toBeUndefined(); +}); + test("fails closed when a reasoner advertises no efforts and none are authored", () => { const entry = nebulEntry("zai-org/GLM-5.3", { reasoning_efforts: [] }); expect(() => nebul.translateModel(entry, context(undefined))).toThrow(MissingReasoningOptionsError); diff --git a/providers/nebul/models/moonshotai/Kimi-K3.toml b/providers/nebul/models/moonshotai/Kimi-K3.toml index 2c90aaf6ce6..85a66cacacc 100644 --- a/providers/nebul/models/moonshotai/Kimi-K3.toml +++ b/providers/nebul/models/moonshotai/Kimi-K3.toml @@ -1,17 +1,16 @@ -# Reasoning control per Nebul's docs: GET /v1/model/info `reasoning_efforts` is the -# per-model source of truth for "the reasoning_effort values each model meaningfully -# accepts" (https://docs.nebul.io/docs/inference-api/models/model-catalog). It -# advertises none for this model — the live response also marks this served ID with -# supports_reasoning = false — and the Chat Completions API documents no on/off -# toggle (https://docs.nebul.io/docs/inference-api/models/chat-completions), so this -# host exposes no caller control here. -# Trace field: reasoning_content per https://docs.nebul.io/docs/inference-api/advanced-topics/reasoning +# Reasoning on this host: GET /v1/model/info marks this served ID with +# supports_reasoning = false and exposes no reasoning_efforts for it +# (https://api.inference.nebul.io/v1/model/info, checked 2026-09-22). Per +# Nebul's docs that catalog is the source of truth for reasoning-capable +# models, and the reasoning guide does not list Kimi among the +# trace-returning families +# (https://docs.nebul.io/docs/inference-api/advanced-topics/reasoning). +# Nebul serves this ID with thinking disabled — no reasoning controls and no +# reasoning_content traces — so override the inherited lab default with +# reasoning = false and declare no reasoning_options / interleaved. base_model = "moonshotai/kimi-k3" -reasoning_options = [] - -[interleaved] -field = "reasoning_content" +reasoning = false [cost] input = 4.73 diff --git a/sync.md b/sync.md index 01d484bcdeb..356da1af5a5 100644 --- a/sync.md +++ b/sync.md @@ -294,6 +294,7 @@ Nebul is implemented in `packages/core/src/sync/providers/nebul.ts`. - Whole-catalog faults fail closed: an empty response, or one where nothing matches the chat-model filter (`model_type: "llm"`, `mode: "chat"`), throws in `parseModels` before any file is written or deleted — mirroring the ingest-side guards. - Per-model robustness is the inverse: an existing entry survives transient null pricing, a missing context limit, or a served alias that no longer resolves to lab metadata, keeping its authored `base_model`/`cost`/`limit`; only brand-new models require a fully-priced, resolvable source entry. Embeddings and rerankers are filtered by mode/model_type; specialized document-OCR models (name-scoped) and entries the host flags via `display_tags` (`Guard Model`, `Content Safety`, `Private`, `Internal`) are out of catalog scope; entries naming a successor via `model_info.superseded_by_model_name` are excluded as superseded. All of these skip silently. - `reasoning_options` come from the endpoint's per-model `reasoning_efforts` list, Nebul's only documented reasoning control. A reasoner with neither advertised efforts nor authored controls fails sync for manual authoring rather than writing an empty control set. +- An authored `reasoning = false` is the escape hatch for a served ID whose lab model reasons but which this host runs with thinking disabled (catalog reports `supports_reasoning = false` and no `reasoning_efforts`). The override is kept, the missing-reasoning-options guard does not fire, and any authored `reasoning_options`/`interleaved` are dropped from the synced file. - `interleaved`, `status`, and other fields the endpoint does not expose are preserved from existing files; the reasoning trace channel (`reasoning_content` vs `message.reasoning` vs inline ``) is family-specific per Nebul's docs. - `BASE_MODEL_ALIASES` in the module bridges served IDs to models.dev's canonical metadata naming where they differ (e.g. `mistralai/Mistral-Medium-3.5-128B` → `mistral/mistral-medium-2604`). Canonical paths are provider-agnostic — often release-date slugs rather than the host's internal model name — so an alias does not mean the wrong model is referenced.