From 84a8e1b7b3347b23b7b156b1210b5aad1311c777 Mon Sep 17 00:00:00 2001 From: K-3-LT <76082878+K-3-LT@users.noreply.github.com> Date: Tue, 15 Sep 2026 14:35:26 +0800 Subject: [PATCH 1/2] feat: update IteraCompute model catalog --- .../deepseek/deepseek-v4-flash-0731.toml | 16 ++++++++++ .../models/deepseek/deepseek-v4-pro-0813.toml | 16 ++++++++++ .../iteracompute/ornith-1.5-35b-a3b.toml | 31 ------------------- .../models/iteracompute/qwen3.8-27b.toml | 24 -------------- .../models/minimax/minimax-m3.toml | 19 ++++++++++++ .../models/moonshotai/kimi-k3.toml | 20 ++++++++++++ .../models/ornith-ai/ornith-1.5-35b-a3b.toml | 22 +++++++++++++ .../models/qwen/qwen3.8-2.4t-a95b.toml | 19 ++++++++++++ .../iteracompute/models/qwen/qwen3.8-27b.toml | 19 ++++++++++++ .../models/z-ai/glm-5.3-flash.toml | 20 ++++++++++++ .../iteracompute/models/z-ai/glm-5.3.toml | 17 ++++++++++ 11 files changed, 168 insertions(+), 55 deletions(-) create mode 100644 providers/iteracompute/models/deepseek/deepseek-v4-flash-0731.toml create mode 100644 providers/iteracompute/models/deepseek/deepseek-v4-pro-0813.toml delete mode 100644 providers/iteracompute/models/iteracompute/ornith-1.5-35b-a3b.toml delete mode 100644 providers/iteracompute/models/iteracompute/qwen3.8-27b.toml create mode 100644 providers/iteracompute/models/minimax/minimax-m3.toml create mode 100644 providers/iteracompute/models/moonshotai/kimi-k3.toml create mode 100644 providers/iteracompute/models/ornith-ai/ornith-1.5-35b-a3b.toml create mode 100644 providers/iteracompute/models/qwen/qwen3.8-2.4t-a95b.toml create mode 100644 providers/iteracompute/models/qwen/qwen3.8-27b.toml create mode 100644 providers/iteracompute/models/z-ai/glm-5.3-flash.toml create mode 100644 providers/iteracompute/models/z-ai/glm-5.3.toml diff --git a/providers/iteracompute/models/deepseek/deepseek-v4-flash-0731.toml b/providers/iteracompute/models/deepseek/deepseek-v4-flash-0731.toml new file mode 100644 index 00000000000..e90d275db10 --- /dev/null +++ b/providers/iteracompute/models/deepseek/deepseek-v4-flash-0731.toml @@ -0,0 +1,16 @@ +# Sources (accessed 2026-09-15): +# https://api.iteracompute.com/v1/models (model ID, availability, capabilities, limits, and pricing) +# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) +# Reasoning is mandatory and the public catalog exposes no caller-selectable reasoning control. +# Prices are in USD per million tokens. +base_model = "deepseek/deepseek-v4-flash-0731" +reasoning_options = [] + +[cost] +input = 0.35 +output = 1.30 +cache_read = 0.03 + +[limit] +context = 970_000 +output = 393_216 diff --git a/providers/iteracompute/models/deepseek/deepseek-v4-pro-0813.toml b/providers/iteracompute/models/deepseek/deepseek-v4-pro-0813.toml new file mode 100644 index 00000000000..a7681a06e75 --- /dev/null +++ b/providers/iteracompute/models/deepseek/deepseek-v4-pro-0813.toml @@ -0,0 +1,16 @@ +# Sources (accessed 2026-09-15): +# https://api.iteracompute.com/v1/models (model ID, availability, capabilities, limits, and pricing) +# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) +# Reasoning is mandatory and the public catalog exposes no caller-selectable reasoning control. +# Prices are in USD per million tokens. +base_model = "deepseek/deepseek-v4-pro-0813" +reasoning_options = [] + +[cost] +input = 1.10 +output = 3.50 +cache_read = 0.11 + +[limit] +context = 1_048_576 +output = 393_216 diff --git a/providers/iteracompute/models/iteracompute/ornith-1.5-35b-a3b.toml b/providers/iteracompute/models/iteracompute/ornith-1.5-35b-a3b.toml deleted file mode 100644 index 59d4eb1b5ea..00000000000 --- a/providers/iteracompute/models/iteracompute/ornith-1.5-35b-a3b.toml +++ /dev/null @@ -1,31 +0,0 @@ -# Sources (accessed 2026-08-28): -# https://api.iteracompute.com/v1/models (model ID, availability, limits, and pricing) -# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) -# https://huggingface.co/ornith-ai/Ornith-1.5-35B-A3B (upstream capabilities) -# -# IteraCompute serves the language-only NVFP4 build with a 327,680-token -# combined window, up to 262,144 input tokens, and up to 65,536 output tokens. -# Reasoning toggle wire path: reasoning_effort = "none" disables thinking; -# any accepted non-none effort enables it. Responses stream reasoning through -# the reasoning_content delta field. -# Prices are in USD per million tokens. -base_model = "deepreinforce/ornith-1.5-35b-a3b" -attachment = false - -[interleaved] -field = "reasoning_content" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.30 -output = 3.00 - -[limit] -context = 327_680 -input = 262_144 -output = 65_536 - -[modalities] -input = ["text"] diff --git a/providers/iteracompute/models/iteracompute/qwen3.8-27b.toml b/providers/iteracompute/models/iteracompute/qwen3.8-27b.toml deleted file mode 100644 index 8eb4170a5f0..00000000000 --- a/providers/iteracompute/models/iteracompute/qwen3.8-27b.toml +++ /dev/null @@ -1,24 +0,0 @@ -# Sources (accessed 2026-09-01): -# https://api.iteracompute.com/v1/models (model ID, availability, and input/output pricing) -# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) -# https://huggingface.co/gittensor-model-hub/Qwen3.8-27B-NVFP4-RTX5090 (served checkpoint and image input) -# -# IteraCompute serves Qwen3.8 27B through its OpenAI-compatible API. -# The NVFP4 deployment accepts text and image input and extends the combined -# serving window to 327,680 tokens with up to 262,144 input and 65,536 output tokens. -# Prices are in USD per million tokens. -# Effort: reasoning_effort = none|low|medium|xhigh (xhigh is the default). -base_model = "alibaba/qwen3.8-27b" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "xhigh"] }] - -[cost] -input = 0.30 -output = 2.50 - -[limit] -context = 327_680 -input = 262_144 -output = 65_536 - -[modalities] -input = ["text", "image"] diff --git a/providers/iteracompute/models/minimax/minimax-m3.toml b/providers/iteracompute/models/minimax/minimax-m3.toml new file mode 100644 index 00000000000..10f663ec30d --- /dev/null +++ b/providers/iteracompute/models/minimax/minimax-m3.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-15): +# https://api.iteracompute.com/v1/models (model ID, availability, capabilities, limits, and pricing) +# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) +# Reasoning is mandatory and the public catalog exposes no caller-selectable reasoning control. +# Prices are in USD per million tokens. +base_model = "minimax/MiniMax-M3" +reasoning_options = [] + +[cost] +input = 0.40 +output = 1.60 +cache_read = 0.08 + +[limit] +context = 1_048_576 +output = 524_288 + +[modalities] +input = ["text", "image"] diff --git a/providers/iteracompute/models/moonshotai/kimi-k3.toml b/providers/iteracompute/models/moonshotai/kimi-k3.toml new file mode 100644 index 00000000000..b6055d6489f --- /dev/null +++ b/providers/iteracompute/models/moonshotai/kimi-k3.toml @@ -0,0 +1,20 @@ +# Sources (accessed 2026-09-15): +# https://api.iteracompute.com/v1/models (model ID, availability, capabilities, limits, and pricing) +# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) +# Reasoning is mandatory and the public catalog exposes no caller-selectable reasoning control. +# Prices are in USD per million tokens. +base_model = "moonshotai/kimi-k3" +structured_output = false +reasoning_options = [] + +[cost] +input = 3.10 +output = 15.50 +cache_read = 0.31 + +[limit] +context = 1_048_576 +output = 999_999 + +[modalities] +input = ["text", "image"] diff --git a/providers/iteracompute/models/ornith-ai/ornith-1.5-35b-a3b.toml b/providers/iteracompute/models/ornith-ai/ornith-1.5-35b-a3b.toml new file mode 100644 index 00000000000..e0ee4a079e5 --- /dev/null +++ b/providers/iteracompute/models/ornith-ai/ornith-1.5-35b-a3b.toml @@ -0,0 +1,22 @@ +# Sources (accessed 2026-09-15): +# https://api.iteracompute.com/v1/models (canonical model ID, availability, capabilities, limits, and pricing) +# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) +# Toggle wire path: reasoning_effort = "none" disables thinking; responses expose reasoning_content. +# Prices are in USD per million tokens. +base_model = "deepreinforce/ornith-1.5-35b-a3b" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.30 +output = 3.00 +cache_read = 0.03 + +[limit] +context = 327_680 +input = 262_144 +output = 65_536 diff --git a/providers/iteracompute/models/qwen/qwen3.8-2.4t-a95b.toml b/providers/iteracompute/models/qwen/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..1499c45a580 --- /dev/null +++ b/providers/iteracompute/models/qwen/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-15): +# https://api.iteracompute.com/v1/models (model ID, availability, capabilities, limits, and pricing) +# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) +# Prices are in USD per million tokens. +base_model = "alibaba/qwen3.8-2.4t-a95b" +attachment = true +reasoning_options = [{ type = "toggle" }] + +[cost] +input = 1.95 +output = 5.95 +cache_read = 0.20 + +[limit] +context = 970_000 +output = 131_072 + +[modalities] +input = ["text", "image"] diff --git a/providers/iteracompute/models/qwen/qwen3.8-27b.toml b/providers/iteracompute/models/qwen/qwen3.8-27b.toml new file mode 100644 index 00000000000..bb7bc16c39d --- /dev/null +++ b/providers/iteracompute/models/qwen/qwen3.8-27b.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-15): +# https://api.iteracompute.com/v1/models (canonical model ID, availability, capabilities, limits, and pricing) +# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) +# Prices are in USD per million tokens. +base_model = "alibaba/qwen3.8-27b" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "xhigh"] }] + +[cost] +input = 0.30 +output = 2.50 +cache_read = 0.03 + +[limit] +context = 327_680 +input = 262_144 +output = 65_536 + +[modalities] +input = ["text", "image"] diff --git a/providers/iteracompute/models/z-ai/glm-5.3-flash.toml b/providers/iteracompute/models/z-ai/glm-5.3-flash.toml new file mode 100644 index 00000000000..2f3eea506f0 --- /dev/null +++ b/providers/iteracompute/models/z-ai/glm-5.3-flash.toml @@ -0,0 +1,20 @@ +# Sources (accessed 2026-09-15): +# https://api.iteracompute.com/v1/models (model ID, availability, capabilities, limits, pricing, and capacity) +# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) +# Effort wire path: reasoning.effort = low|high|max. Reasoning is mandatory. +# Prices are in USD per million tokens. +base_model = "zhipuai/glm-5.3-flash" +structured_output = false +reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] + +[cost] +input = 0.12 +output = 0.48 +cache_read = 0.02 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image"] diff --git a/providers/iteracompute/models/z-ai/glm-5.3.toml b/providers/iteracompute/models/z-ai/glm-5.3.toml new file mode 100644 index 00000000000..04a97272db5 --- /dev/null +++ b/providers/iteracompute/models/z-ai/glm-5.3.toml @@ -0,0 +1,17 @@ +# Sources (accessed 2026-09-15): +# https://api.iteracompute.com/v1/models (model ID, availability, capabilities, limits, and pricing) +# https://iteracompute.com/docs.html (OpenAI-compatible API documentation) +# Reasoning is mandatory and the public catalog exposes no caller-selectable reasoning control. +# Prices are in USD per million tokens. +base_model = "zhipuai/glm-5.3" +structured_output = false +reasoning_options = [] + +[cost] +input = 1.50 +output = 6.20 +cache_read = 0.23 + +[limit] +context = 1_048_576 +output = 131_072 From ffc5b282e835e77f3b87b924436fa5d0d317eba2 Mon Sep 17 00:00:00 2001 From: Tao Date: Wed, 16 Sep 2026 16:50:12 +0800 Subject: [PATCH 2/2] fix: sync IteraCompute prices with production catalog --- .../models/deepseek/deepseek-v4-flash-0731.toml | 6 +++--- .../iteracompute/models/deepseek/deepseek-v4-pro-0813.toml | 2 +- providers/iteracompute/models/minimax/minimax-m3.toml | 4 ++-- providers/iteracompute/models/moonshotai/kimi-k3.toml | 6 +++--- providers/iteracompute/models/z-ai/glm-5.3-flash.toml | 6 +++--- providers/iteracompute/models/z-ai/glm-5.3.toml | 6 +++--- 6 files changed, 15 insertions(+), 15 deletions(-) diff --git a/providers/iteracompute/models/deepseek/deepseek-v4-flash-0731.toml b/providers/iteracompute/models/deepseek/deepseek-v4-flash-0731.toml index e90d275db10..de987dfd5fe 100644 --- a/providers/iteracompute/models/deepseek/deepseek-v4-flash-0731.toml +++ b/providers/iteracompute/models/deepseek/deepseek-v4-flash-0731.toml @@ -7,9 +7,9 @@ base_model = "deepseek/deepseek-v4-flash-0731" reasoning_options = [] [cost] -input = 0.35 -output = 1.30 -cache_read = 0.03 +input = 0.34 +output = 1.05 +cache_read = 0.035 [limit] context = 970_000 diff --git a/providers/iteracompute/models/deepseek/deepseek-v4-pro-0813.toml b/providers/iteracompute/models/deepseek/deepseek-v4-pro-0813.toml index a7681a06e75..0199acb7930 100644 --- a/providers/iteracompute/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/iteracompute/models/deepseek/deepseek-v4-pro-0813.toml @@ -8,7 +8,7 @@ reasoning_options = [] [cost] input = 1.10 -output = 3.50 +output = 3.30 cache_read = 0.11 [limit] diff --git a/providers/iteracompute/models/minimax/minimax-m3.toml b/providers/iteracompute/models/minimax/minimax-m3.toml index 10f663ec30d..f9d9908e3f6 100644 --- a/providers/iteracompute/models/minimax/minimax-m3.toml +++ b/providers/iteracompute/models/minimax/minimax-m3.toml @@ -7,8 +7,8 @@ base_model = "minimax/MiniMax-M3" reasoning_options = [] [cost] -input = 0.40 -output = 1.60 +input = 0.29 +output = 1.20 cache_read = 0.08 [limit] diff --git a/providers/iteracompute/models/moonshotai/kimi-k3.toml b/providers/iteracompute/models/moonshotai/kimi-k3.toml index b6055d6489f..4e4af1d1240 100644 --- a/providers/iteracompute/models/moonshotai/kimi-k3.toml +++ b/providers/iteracompute/models/moonshotai/kimi-k3.toml @@ -8,9 +8,9 @@ structured_output = false reasoning_options = [] [cost] -input = 3.10 -output = 15.50 -cache_read = 0.31 +input = 3.00 +output = 14.90 +cache_read = 0.29 [limit] context = 1_048_576 diff --git a/providers/iteracompute/models/z-ai/glm-5.3-flash.toml b/providers/iteracompute/models/z-ai/glm-5.3-flash.toml index 2f3eea506f0..19503c99421 100644 --- a/providers/iteracompute/models/z-ai/glm-5.3-flash.toml +++ b/providers/iteracompute/models/z-ai/glm-5.3-flash.toml @@ -8,9 +8,9 @@ structured_output = false reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] [cost] -input = 0.12 -output = 0.48 -cache_read = 0.02 +input = 0.14 +output = 0.49 +cache_read = 0.03 [limit] context = 1_048_576 diff --git a/providers/iteracompute/models/z-ai/glm-5.3.toml b/providers/iteracompute/models/z-ai/glm-5.3.toml index 04a97272db5..2ba1f3cca79 100644 --- a/providers/iteracompute/models/z-ai/glm-5.3.toml +++ b/providers/iteracompute/models/z-ai/glm-5.3.toml @@ -8,9 +8,9 @@ structured_output = false reasoning_options = [] [cost] -input = 1.50 -output = 6.20 -cache_read = 0.23 +input = 1.20 +output = 3.50 +cache_read = 0.26 [limit] context = 1_048_576