diff --git a/.github/workflows/ci-fixer.yml b/.github/workflows/ci-fixer.yml index 9d1205a126c..5a94d2a3db2 100644 --- a/.github/workflows/ci-fixer.yml +++ b/.github/workflows/ci-fixer.yml @@ -160,7 +160,7 @@ jobs: Failed log excerpt: EOF cat "$LOG_FILE" - } | opencode run --agent ci-fixer -m opencode/grok-4.5 | tee "$RESPONSE_FILE" + } | opencode run --agent ci-fixer -m opencode/muse-spark-1.3 --variant xhigh | tee "$RESPONSE_FILE" - name: Check changed paths if: steps.budget.outputs.run == 'true' && steps.budget-cache.outputs.cache-hit != 'true' diff --git a/.github/workflows/issue-fixer.yml b/.github/workflows/issue-fixer.yml index 17a2ce9a544..4e1f144f164 100644 --- a/.github/workflows/issue-fixer.yml +++ b/.github/workflows/issue-fixer.yml @@ -60,7 +60,7 @@ jobs: + "If it is a feature request, a request to track a new kind of information, a question, or any miscellaneous non-catalog-data request, do not edit files. Respond briefly that it needs maintainer review and no automated fix was opened." ' "$ISSUE_FILE" > "$PROMPT_FILE" - opencode run --agent issue-fixer -m opencode/grok-4.5 --format json < "$PROMPT_FILE" | tee "$EVENTS_FILE" + opencode run --agent issue-fixer -m opencode/muse-spark-1.3 --variant xhigh --format json < "$PROMPT_FILE" | tee "$EVENTS_FILE" if ! jq -ers 'map(select(.type == "text") | .part.text) | last | select(length > 0)' "$EVENTS_FILE" > "$RESPONSE_FILE"; then echo "Issue fixer did not produce a final response." >&2 diff --git a/.github/workflows/opencode.yml b/.github/workflows/opencode.yml index 24c819769a9..2e06c6beccb 100644 --- a/.github/workflows/opencode.yml +++ b/.github/workflows/opencode.yml @@ -27,4 +27,5 @@ jobs: env: OPENCODE_API_KEY: ${{ secrets.OPENCODE_API_KEY }} with: - model: opencode/grok-4.5 + model: opencode/muse-spark-1.3 + variant: xhigh diff --git a/.github/workflows/pr-reviewer.yml b/.github/workflows/pr-reviewer.yml index 48a5c853311..fe824e9d48d 100644 --- a/.github/workflows/pr-reviewer.yml +++ b/.github/workflows/pr-reviewer.yml @@ -78,7 +78,7 @@ jobs: export PR_REVIEW_READY_FILE rm -f "$PR_REVIEW_READY_FILE" - opencode run --agent pr-reviewer -m opencode/grok-4.5 --format json <<'EOF' | tee "$EVENTS_FILE" + opencode run --agent pr-reviewer -m opencode/muse-spark-1.3 --variant xhigh --format json <<'EOF' | tee "$EVENTS_FILE" Review this pull request using the trusted reviewer instructions. Start with `.pr-review/pull-request.json`, `.pr-review/diff.patch`, `AGENTS.md`, and the contributing guidance in `README.md`. Read `sync.md`, the reasoning-options audit guide, schema code, and nearby base-revision files when relevant to the changed files. Use only the read, glob, grep, and mark-pr-ready tools. Return only the final review comment in the agent's required output format. Never include progress narration or passed-check summaries. EOF diff --git a/README.md b/README.md index f26ba19ab83..8a6c27e736a 100644 --- a/README.md +++ b/README.md @@ -22,6 +22,17 @@ You can access this data through an API. curl https://models.dev/api.json ``` +Specialized model types are omitted by default. Filter by one or more +comma-separated model types, or use `all` for the complete catalog: + +```bash +curl "https://models.dev/api.json?type=decision" +curl "https://models.dev/api.json?type=all" +``` + +The currently supported model type is `decision`. The `type` parameter is also +available on `models.json`, `catalog.json`, and `model-schema.json`. + Use the **Model ID** field to do a lookup on any model; it's the identifier used by [AI SDK](https://ai-sdk.dev/). Provider-agnostic model metadata is available separately: @@ -270,6 +281,7 @@ Models must conform to the following schema, as defined in `packages/core/src/sc **Model Schema:** - `name`: String — Display name of the model +- `type` _(optional)_: String — Specialized model behavior; currently supports `decision` - `attachment`: Boolean — Supports file attachments - `reasoning`: Boolean — Supports reasoning / chain-of-thought - `tool_call`: Boolean - Supports tool calling diff --git a/bun.lock b/bun.lock index 2ba689570d3..2e4ee0e36cd 100644 --- a/bun.lock +++ b/bun.lock @@ -24,6 +24,9 @@ }, "packages/function": { "name": "@models.dev/function", + "dependencies": { + "@models.dev/core": "workspace:*", + }, "devDependencies": { "@cloudflare/workers-types": "4.20250522.0", "@tsconfig/bun": "catalog:", diff --git a/models/anthropic/claude-opus-5-5.toml b/models/anthropic/claude-opus-5-5.toml new file mode 100644 index 00000000000..0f1cfeba4ed --- /dev/null +++ b/models/anthropic/claude-opus-5-5.toml @@ -0,0 +1,19 @@ +name = "Claude Opus 5.5" +description = "Claude model for long-running agentic coding and knowledge work" +family = "claude-opus" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +open_weights = false +knowledge = "2026-06" + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/models/deepseek/deepseek-v4-flash-vision-exp.toml b/models/deepseek/deepseek-v4-flash-vision-exp.toml index ef5426b119c..18d7744f7c8 100644 --- a/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -1,14 +1,16 @@ +# Open weights (MIT): https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp name = "DeepSeek V4 Flash Vision Exp" description = "Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work" family = "deepseek-flash" release_date = "2026-08-21" -last_updated = "2026-08-21" +last_updated = "2026-09-01" attachment = true reasoning = true temperature = true tool_call = true structured_output = true -open_weights = false +open_weights = true +license = "MIT" [limit] context = 1_000_000 @@ -17,3 +19,7 @@ output = 384_000 [modalities] input = ["text", "image"] output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp" diff --git a/models/deepseek/deepseek-v4.1-flash.toml b/models/deepseek/deepseek-v4.1-flash.toml index aac72588623..68b2de09bb4 100644 --- a/models/deepseek/deepseek-v4.1-flash.toml +++ b/models/deepseek/deepseek-v4.1-flash.toml @@ -1,3 +1,4 @@ +# Open weights (MIT): https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash name = "DeepSeek V4.1 Flash" description = "DeepSeek V4.1 Flash model for reasoning and agentic coding" family = "deepseek-flash" @@ -18,4 +19,8 @@ output = 384_000 [modalities] input = ["text", "image"] -output = ["text"] \ No newline at end of file +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash" diff --git a/models/google/gemma-4-12b-it.toml b/models/google/gemma-4-12b-it.toml new file mode 100644 index 00000000000..af2b5f0d680 --- /dev/null +++ b/models/google/gemma-4-12b-it.toml @@ -0,0 +1,23 @@ +name = "Gemma 4 12B IT" +description = "Compact Gemma 4 instruction model for open, self-hosted chat and reasoning" +family = "gemma" +release_date = "2026-06-09" +last_updated = "2026-06-09" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/google/gemma-4-12B-it" diff --git a/models/mixedbread/toast-1.toml b/models/mixedbread/toast-1.toml new file mode 100644 index 00000000000..775acb667f6 --- /dev/null +++ b/models/mixedbread/toast-1.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-19): +# https://www.mixedbread.com/docs/agent/models +# https://www.mixedbread.com/docs/agent/chat-completions +name = "Toast 1" +description = "Specialized search model for knowledge-intensive questions, multi-step retrieval, and evidence synthesis" +release_date = "2026-08-13" +last_updated = "2026-08-13" +attachment = false +reasoning = false +tool_call = true +open_weights = false + +[limit] +context = 131_000 +output = 4_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/models/nex-agi/nex-n2-pro.toml b/models/nex-agi/nex-n2-pro.toml new file mode 100644 index 00000000000..3b134613e5f --- /dev/null +++ b/models/nex-agi/nex-n2-pro.toml @@ -0,0 +1,23 @@ +# https://huggingface.co/nex-agi/Nex-N2-Pro +name = "Nex-N2-Pro" +description = "Open agentic MoE model (397B total, 17B active) for coding, tool use, and research workflows" +release_date = "2026-06-02" +last_updated = "2026-06-02" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[limit] +context = 262_144 +output = 262_144 + +[modalities] +input = ["text", "image"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/nex-agi/Nex-N2-Pro" diff --git a/models/openai/gpt-6-luna.toml b/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..0a43442c0a5 --- /dev/null +++ b/models/openai/gpt-6-luna.toml @@ -0,0 +1,21 @@ +name = "GPT-6 Luna" +description = "OpenAI's most efficient model for focused, high-volume tasks" +family = "gpt-luna" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2026-05-18" +open_weights = false + +[limit] +context = 1_050_000 +input = 922_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/models/openai/gpt-6-sol.toml b/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..824b5d86bed --- /dev/null +++ b/models/openai/gpt-6-sol.toml @@ -0,0 +1,21 @@ +name = "GPT-6 Sol" +description = "OpenAI model for complex coding and agentic workflows" +family = "gpt-sol" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +knowledge = "2026-04-20" +open_weights = false + +[limit] +context = 1_050_000 +input = 922_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/models/quiverai/arrow-2-telos.toml b/models/quiverai/arrow-2-telos.toml new file mode 100644 index 00000000000..ae0e17b52d6 --- /dev/null +++ b/models/quiverai/arrow-2-telos.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-19): +# https://docs.quiver.ai/developers/models +# https://docs.quiver.ai/developers/models/arrow-2-telos +name = "Arrow 2 Telos" +description = "High-fidelity SVG generation model for complex vector work and long-context refinement" +release_date = "2026-09-16" +last_updated = "2026-09-16" +attachment = true +reasoning = true +tool_call = true +open_weights = false + +[limit] +context = 1_050_000 +output = 65_536 + +[modalities] +input = ["text", "image"] +output = ["text", "image"] diff --git a/models/quiverai/arrow-2.toml b/models/quiverai/arrow-2.toml new file mode 100644 index 00000000000..5d73a411794 --- /dev/null +++ b/models/quiverai/arrow-2.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-19): +# https://docs.quiver.ai/developers/models +# https://docs.quiver.ai/developers/models/arrow-2 +name = "Arrow 2" +description = "Fast SVG generation model for creation, vectorization, editing, and animation" +release_date = "2026-09-16" +last_updated = "2026-09-16" +attachment = true +reasoning = true +tool_call = true +open_weights = false + +[limit] +context = 131_072 +output = 65_536 + +[modalities] +input = ["text", "image"] +output = ["text", "image"] diff --git a/models/stepfun/step-5-preview.toml b/models/stepfun/step-5-preview.toml new file mode 100644 index 00000000000..e4718be0938 --- /dev/null +++ b/models/stepfun/step-5-preview.toml @@ -0,0 +1,20 @@ +name = "Step 5 Preview" +description = "StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window" +# release_date is StepFun's API catalog first-appearance timestamp (created=1789564127, +# 2026-09-16 13:08 UTC) — no dated launch announcement was found on the model docs page. +release_date = "2026-09-16" +last_updated = "2026-09-20" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[limit] +context = 1_000_000 +input = 1_000_000 +output = 1_000_000 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/models/typesafe/jev-latest.toml b/models/typesafe/jev-latest.toml index 9b2f3a75896..6b54d1bd25f 100644 --- a/models/typesafe/jev-latest.toml +++ b/models/typesafe/jev-latest.toml @@ -4,6 +4,7 @@ # - https://typesafe.ai/blog/introducing-system-one-models-and-jev # The 64k context limit covers the state plus all questions in one request. name = "Jev" +type = "decision" description = "System One model for fast, typed probabilistic decisions over text or structured state" release_date = "2026-09-15" last_updated = "2026-09-15" diff --git a/models/vivgrid/viv-fast.toml b/models/vivgrid/viv-fast.toml new file mode 100644 index 00000000000..9bed45cd435 --- /dev/null +++ b/models/vivgrid/viv-fast.toml @@ -0,0 +1,26 @@ +# Sources: +# - https://docs.vivgrid.com/models/viv-fast +# - https://docs.vivgrid.com/api/model-api +name = "Viv Fast" +description = "Fast coding model" +release_date = "2026-09-21" +last_updated = "2026-09-21" +reasoning = true +knowledge = "2026-09" +temperature = false +open_weights = false +tool_call = true +structured_output = true +attachment = true + +[limit] +context = 1_000_000 +output = 256_000 + +[modalities] +input = ["text", "image"] +output = ["text"] + +[[links]] +label = "Model docs" +url = "https://docs.vivgrid.com/models/viv-fast" diff --git a/models/xai/grok-4.7.toml b/models/xai/grok-4.7.toml new file mode 100644 index 00000000000..7872d4c7219 --- /dev/null +++ b/models/xai/grok-4.7.toml @@ -0,0 +1,21 @@ +# Sources: https://docs.x.ai/developers/models/grok-4.7, https://docs.x.ai/developers/models +name = "Grok 4.7" +description = "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects" +family = "grok" +knowledge = "2026-05" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[limit] +context = 500_000 +output = 500_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/models/xai/grok-imagine-image-quality.toml b/models/xai/grok-imagine-image-quality.toml new file mode 100644 index 00000000000..46eb1539995 --- /dev/null +++ b/models/xai/grok-imagine-image-quality.toml @@ -0,0 +1,26 @@ +# Sources: +# - https://docs.x.ai/docs/models +# - https://docs.x.ai/developers/models/grok-imagine-image-quality +# - https://docs.x.ai/developers/pricing +# - https://docs.x.ai/docs/guides/image-generation +# Pricing: $0.05/image (1K), $0.07/image (2K); image input $0.01/image (not token-based; no [cost] authored) +# Release: dated alias grok-imagine-image-quality-20260403; also aliased as grok-imagine-image-quality-latest, grok-imagine-image-pro + +name = "Grok Imagine Image Quality" +description = "Higher-fidelity Grok Imagine image model for prompt-driven generation, editing, and visual design workflows" +family = "grok" +release_date = "2026-04-03" +last_updated = "2026-04-03" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 16_000 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["image"] diff --git a/models/xai/grok-voice-stt-1.0.toml b/models/xai/grok-voice-stt-1.0.toml new file mode 100644 index 00000000000..725c96c69d7 --- /dev/null +++ b/models/xai/grok-voice-stt-1.0.toml @@ -0,0 +1,17 @@ +name = "Grok Voice STT 1.0" +description = "Grok Voice STT 1.0 is xAI's speech-to-text model. It supports transcription with word-level timestamps, optional speaker diarization, and multichannel audio." +release_date = "2026-08-04" +last_updated = "2026-08-04" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 15_000 +output = 15_000 + +[modalities] +input = ["audio"] +output = ["text"] diff --git a/models/xai/grok-voice-tts-1.0.toml b/models/xai/grok-voice-tts-1.0.toml new file mode 100644 index 00000000000..b519e016a19 --- /dev/null +++ b/models/xai/grok-voice-tts-1.0.toml @@ -0,0 +1,17 @@ +name = "Grok Voice TTS 1.0" +description = "Convert text into spoken audio with a single API call. The API supports a rich set of expressive voices, inline speech tags for fine-grained delivery control, and output formats from high-fidelity MP3 to telephony-optimized μ-law." +release_date = "2026-07-31" +last_updated = "2026-07-31" +attachment = false +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 15_000 +output = 15_000 + +[modalities] +input = ["text"] +output = ["audio"] diff --git a/models/xiaomi/mimo-v2.6-flash.toml b/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..2ac6276589a --- /dev/null +++ b/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,22 @@ +name = "MiMo-V2.6-Flash" +description = "MiMo Flash model for multimodal coding agents and long-context automation" +family = "mimo" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +open_weights = true + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "audio", "video"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Flash-RL" diff --git a/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..82d134ed28a --- /dev/null +++ b/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,22 @@ +name = "MiMo-V2.6-Pro-UltraSpeed" +description = "MiMo pro model for strong multimodal reasoning and agent execution" +family = "mimo" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +open_weights = true + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "audio", "video"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Pro-RL" diff --git a/models/xiaomi/mimo-v2.6-pro.toml b/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..fc53e943157 --- /dev/null +++ b/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,22 @@ +name = "MiMo-V2.6-Pro" +description = "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution" +family = "mimo" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +open_weights = true + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "audio", "video"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Pro-RL" diff --git a/models/zhipuai/glm-image.toml b/models/zhipuai/glm-image.toml new file mode 100644 index 00000000000..b3f3746cb2e --- /dev/null +++ b/models/zhipuai/glm-image.toml @@ -0,0 +1,22 @@ +name = "GLM-Image" +description = "GLM-Image is an image generation model adopts a hybrid autoregressive + diffusion decoder architecture. In general image generation quality, GLM‑Image aligns with mainstream latent diffusion approaches, but it shows significant advantages in text-rendering and knowledge‑intensive generation scenarios. It performs especially well in tasks requiring precise semantic understanding and complex information expression, while maintaining strong capabilities in high‑fidelity and fine‑grained detail generation. In addition to text‑to‑image generation, GLM‑Image also supports a rich set of image‑to‑image tasks including image editing, style transfer, identity‑preserving generation, and multi‑subject consistency." +release_date = "2026-01-19" +last_updated = "2026-01-19" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = true +license = "MIT" + +[limit] +context = 10_240 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["image"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/zai-org/GLM-Image" diff --git a/packages/core/src/filter.ts b/packages/core/src/filter.ts new file mode 100644 index 00000000000..3cba07d6cae --- /dev/null +++ b/packages/core/src/filter.ts @@ -0,0 +1,83 @@ +export const MODEL_TYPES = ["decision"] as const; + +export type ModelTypeValue = (typeof MODEL_TYPES)[number]; +export type ModelTypeFilter = "default" | "all" | ModelTypeValue[]; + +interface TypedModel { + type?: ModelTypeValue; +} + +interface TypedProvider { + models: Record; +} + +export class InvalidModelTypeError extends Error { + constructor(value: string) { + super(`Invalid type value: ${value}`); + this.name = "InvalidModelTypeError"; + } +} + +export function parseModelTypes(value: string | null): ModelTypeFilter { + if (value === null || value === "") return "default"; + if (value === "all") return "all"; + + const types = value.split(","); + if ( + types.length === 0 || + types.includes("all") || + types.some((type) => !MODEL_TYPES.includes(type as ModelTypeValue)) + ) { + throw new InvalidModelTypeError(value); + } + + return [...new Set(types)] as ModelTypeValue[]; +} + +function includesModel(model: TypedModel, filter: ModelTypeFilter) { + if (filter === "all") return true; + if (filter === "default") return model.type === undefined; + return model.type !== undefined && filter.includes(model.type); +} + +export function filterProvidersByModelType< + T extends Record, +>(providers: T, filter: ModelTypeFilter): T { + if (filter === "all") return providers; + + return Object.fromEntries( + Object.entries(providers).flatMap(([providerID, provider]) => { + const models = Object.fromEntries( + Object.entries(provider.models).filter(([, model]) => + includesModel(model, filter), + ), + ); + return Object.keys(models).length === 0 + ? [] + : [[providerID, { ...provider, models }]]; + }), + ) as T; +} + +export function filterModelsByModelType>( + models: T, + filter: ModelTypeFilter, +): T { + if (filter === "all") return models; + return Object.fromEntries( + Object.entries(models).filter(([, model]) => includesModel(model, filter)), + ) as T; +} + +export function filterCatalogByModelType< + TProviders extends Record, + TModels extends Record, +>( + catalog: { providers: TProviders; models: TModels }, + filter: ModelTypeFilter, +): { providers: TProviders; models: TModels } { + return { + providers: filterProvidersByModelType(catalog.providers, filter), + models: filterModelsByModelType(catalog.models, filter), + }; +} diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 69f4e893dcb..b8be1ed0f80 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -2,3 +2,4 @@ export * from "./schema.js"; export * from "./generate.js"; export * from "./describe.js"; export * from "./family.js"; +export * from "./filter.js"; diff --git a/packages/core/src/schema.ts b/packages/core/src/schema.ts index c6c8b8f38b2..5c2d7b06681 100644 --- a/packages/core/src/schema.ts +++ b/packages/core/src/schema.ts @@ -1,6 +1,7 @@ import { z } from "zod"; import { ModelFamily } from "./family"; +import { MODEL_TYPES } from "./filter"; type JsonValue = | string @@ -148,6 +149,8 @@ const DateString = z const Modality = z.enum(["text", "audio", "image", "video", "pdf"]); +export const ModelType = z.enum(MODEL_TYPES); + const Modalities = z .object({ input: z.array(Modality), @@ -219,6 +222,7 @@ export const BenchmarkResult = z const ModelMetadataBase = z.object({ id: z.string(), + type: ModelType.optional(), name: z.string().min(1, "Model name cannot be empty"), description: z.string().min(1, "Model description cannot be empty"), family: ModelFamily.optional(), @@ -245,6 +249,7 @@ export type ModelMetadata = z.infer; const ModelBase = z.object({ id: z.string(), + type: ModelType.optional(), name: z.string().min(1, "Model name cannot be empty"), description: z.string().min(1, "Model description cannot be empty"), family: ModelFamily.optional(), diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index 725423e1b2e..05889f5314f 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -1014,6 +1014,7 @@ export function formatToml(model: z.infer) { if ("base_model_omit" in model && model.base_model_omit !== undefined) { lines.push(`base_model_omit = [${model.base_model_omit.map(quote).join(", ")}]`); } + if (model.type !== undefined) lines.push(`type = ${quote(model.type)}`); if (model.name !== undefined) lines.push(`name = ${quote(model.name)}`); if (model.description !== undefined) lines.push(`description = ${quote(model.description)}`); if (model.family !== undefined) lines.push(`family = ${quote(model.family)}`); diff --git a/packages/core/src/sync/providers/aiand.ts b/packages/core/src/sync/providers/aiand.ts index 9545bd91e0f..860bbc50944 100644 --- a/packages/core/src/sync/providers/aiand.ts +++ b/packages/core/src/sync/providers/aiand.ts @@ -31,6 +31,28 @@ const FeedReasoningOption = z.union([ CatalogReasoningOption, ]); +const PRICE = z.number().nonnegative(); +const FeedCostFields = { + input: PRICE, + output: PRICE, + reasoning: PRICE.optional(), + cache_read: PRICE.optional(), + cache_write: PRICE.optional(), + input_audio: PRICE.optional(), + output_audio: PRICE.optional(), +}; +const FeedCostTier = z + .object({ + ...FeedCostFields, + tier: z + .object({ + type: z.literal("context").default("context"), + size: z.number().int().nonnegative(), + }) + .passthrough(), + }) + .passthrough(); + export const AiandModel = z .object({ id: z.string().min(1), @@ -45,13 +67,7 @@ export const AiandModel = z temperature: z.boolean(), tool_call: z.boolean(), structured_output: z.boolean().optional(), - cost: z - .object({ - input: z.number().nonnegative(), - output: z.number().nonnegative(), - cache_read: z.number().nonnegative().optional(), - }) - .passthrough(), + cost: z.object({ ...FeedCostFields, tiers: z.array(FeedCostTier).optional() }).passthrough(), limit: z .object({ context: z.number().int().positive(), @@ -182,12 +198,14 @@ export function buildAiandModel( cost: { input: model.cost.input, output: model.cost.output, - reasoning: authored?.cost?.reasoning, - cache_read: model.cost.cache_read, - cache_write: authored?.cost?.cache_write, - input_audio: authored?.cost?.input_audio, - output_audio: authored?.cost?.output_audio, - tiers: authored?.cost?.tiers, + reasoning: model.cost.reasoning ?? authored?.cost?.reasoning, + cache_read: model.cost.cache_read ?? authored?.cost?.cache_read, + cache_write: model.cost.cache_write ?? authored?.cost?.cache_write, + input_audio: model.cost.input_audio ?? authored?.cost?.input_audio, + output_audio: model.cost.output_audio ?? authored?.cost?.output_audio, + // Unlike auxiliary prices, tiers are a complete source assertion: an + // omitted list means flat pricing and clears stale authored tiers. + tiers: model.cost.tiers, }, // Absence means active on the feed, and the feed owns deprecation: a // curated alpha/beta survives omission, a curated deprecated does not, diff --git a/packages/core/src/sync/providers/cloudflare-workers-ai.ts b/packages/core/src/sync/providers/cloudflare-workers-ai.ts index 252e9ae9cba..0ed0e391513 100644 --- a/packages/core/src/sync/providers/cloudflare-workers-ai.ts +++ b/packages/core/src/sync/providers/cloudflare-workers-ai.ts @@ -1,16 +1,25 @@ import { z } from "zod"; -import { readdirSync } from "node:fs"; +import { readFileSync, readdirSync } from "node:fs"; import path from "node:path"; -import type { ExistingModel, SyncedModel, SyncProvider } from "../index.js"; -import { - buildOpenRouterModel, - OpenRouterModel, - OpenRouterResponse, -} from "./openrouter.js"; +import type { ExistingModel, SyncedFullModel, SyncedModel, SyncProvider } from "../index.js"; +import { MissingReasoningOptionsError } from "../missing-reasoning-options.js"; +import { buildOpenRouterModel, OpenRouterModel } from "./openrouter.js"; const API_BASE = "https://api.cloudflare.com/client/v4/accounts"; -const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models"); +const TOGGLE_HEADER = "# Toggle: chat_template_kwargs.enable_thinking = true|false\n"; +// Models with a verified enable_thinking control. DeepSeek/Kimi use other wire paths; +// their authored comments are preserved below, never inferred from generic schemas. +// These are exact model IDs, not a default for future models or whole publishers. +const ENABLE_THINKING_MODELS = new Set([ + "@cf/google/gemma-4-26b-a4b-it", + "@cf/nvidia/nemotron-3-120b-a12b", + "@cf/qwen/qwen3.8-27b", + "@cf/zai-org/glm-4.7-flash", + "@cf/zai-org/glm-5.2", +]); +const ROOT_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", ".."); +const MODELS_DIR = path.join(ROOT_DIR, "models"); const metadataFilesByPublisher = new Map(); const METADATA_PUBLISHERS: Record = { "deepseek-ai": "deepseek", @@ -24,37 +33,50 @@ const METADATA_PUBLISHERS: Record = { "zai-org": "zhipuai", }; -const CloudflareOpenRouterResponse = z.object({ - result: z.union([OpenRouterResponse, z.array(OpenRouterModel)]).optional(), - result_info: z.object({ - page: z.number().optional(), - total_pages: z.number().optional(), - }).passthrough().optional(), -}).passthrough(); +const WorkersAiReasoning = OpenRouterModel.shape.reasoning.unwrap().partial({ mandatory: true }); +const WorkersAiModel = OpenRouterModel.extend({ reasoning: WorkersAiReasoning.optional() }); +type WorkersAiModel = z.infer; const CloudflareModel = z.object({ - id: z.string(), - name: z.string(), - created: z.number(), + id: z.string().trim().min(1), + name: z.string().trim().min(1), + created: z.number().int().nonnegative(), hugging_face_id: z.string().nullable().optional(), - context_length: z.number(), - max_output_length: z.number().nullable().optional(), - input_modalities: z.array(z.string()).optional(), - output_modalities: z.array(z.string()).optional(), - pricing: z.object({ - prompt: z.string(), - completion: z.string(), - internal_reasoning: z.string().optional(), - input_cache_read: z.string().optional(), - input_cache_write: z.string().optional(), + context_length: z.number().int().positive(), + max_output_length: z.number().int().positive().nullable().optional(), + input_modalities: z.array(z.string().min(1)).min(1).optional(), + output_modalities: z.array(z.string().min(1)).min(1).optional(), + pricing: OpenRouterModel.shape.pricing.extend({ + prompt: z.string().refine(validPrice), + completion: z.string().refine(validPrice), }), supported_features: z.array(z.string()).optional(), supported_sampling_parameters: z.array(z.string()).optional(), + // Validate reasoning separately so malformed controls do not discard the model. + reasoning: z.unknown().optional(), }).passthrough(); const CloudflareResponse = z.object({ - data: z.array(CloudflareModel), -}).passthrough(); + data: z.array(z.unknown()).optional(), + result: z.union([ + z.array(z.unknown()), + z.object({ data: z.array(z.unknown()) }), + ]).optional(), + success: z.literal(true).optional(), + result_info: z.object({ + total_pages: z.number().int().positive().optional(), + }).optional(), +}).refine((response) => response.data !== undefined || response.result !== undefined, { + message: "Cloudflare Workers AI response did not include model data", +}); + +function modelRows(response: z.infer) { + return response.data ?? (Array.isArray(response.result) ? response.result : response.result!.data); +} + +function validPrice(value: string) { + return value.trim() !== "" && Number.isFinite(Number(value)) && Number(value) >= 0; +} type CloudflareModel = z.infer; @@ -62,6 +84,8 @@ export const cloudflareWorkersAi = { id: "cloudflare-workers-ai", name: "Cloudflare Workers AI", modelsDir: "providers/cloudflare-workers-ai/models", + deleteMissing: false, + authoritativeHeaders: true, async fetchModels() { const accountID = process.env.CLOUDFLARE_WORKERS_AI_SYNC_ACCOUNT_ID; const token = process.env.CLOUDFLARE_WORKERS_AI_SYNC_API_TOKEN; @@ -72,50 +96,104 @@ export const cloudflareWorkersAi = { } const first = await fetchPage(accountID, token, 1); - const models = parseCloudflareModels(first); - const pageInfo = CloudflareOpenRouterResponse.safeParse(first).success - ? CloudflareOpenRouterResponse.parse(first).result_info - : undefined; - - for (let page = 2; page <= (pageInfo?.total_pages ?? 1); page++) { - models.push(...parseCloudflareModels(await fetchPage(accountID, token, page))); + if (first === undefined) throw new Error("Cloudflare Workers AI search returned no usable models"); + const rows = modelRows(first); + for (let page = 2; page <= (first.result_info?.total_pages ?? 1); page++) { + const response = await fetchPage(accountID, token, page); + // Keep successful pages; deleteMissing: false retains models on failed pages. + if (response !== undefined) rows.push(...modelRows(response)); } - return { data: models }; + return { data: rows }; }, parseModels(raw) { - return parseCloudflareModels(raw); + const models = parseCloudflareModels(raw); + if (models.length === 0) throw new Error("Cloudflare Workers AI search returned no usable models"); + return models; }, translateModel(model, context) { - const normalized = normalizeModel(model); - const id = normalized.id.replace(/^workers-ai\//, ""); + const id = model.id; + const existing = context.existing(id); + if (existing === undefined && hasReasoning(model) && workersAiReasoningOptions(model) === undefined) { + throw new MissingReasoningOptionsError(id, "Workers AI Search does not specify concrete reasoning controls; manual authoring is needed"); + } + const translated = buildWorkersAiModel(model, existing); + // Only read paths already present in the sync runner's catalogue map. + const header = existing === undefined ? "" : modelHeader(path.resolve(ROOT_DIR, this.modelsDir, `${id}.toml`)); + if (header === undefined) return undefined; + // Remove only the exact generated line; preserve every other leading comment. + const preservedHeader = header.replace(TOGGLE_HEADER, ""); + // Respect curated model-specific wire instructions, including Qwen's split lines. + const hasToggleWire = /\b(?:thinking\.type|chat_template_kwargs\.thinking|enable_thinking)\b/.test(preservedHeader); + const toggleHeader = ENABLE_THINKING_MODELS.has(id) ? TOGGLE_HEADER : ""; + if (translated.reasoning_options?.some((option) => option.type === "toggle") && !hasToggleWire && !toggleHeader) { + if (existing === undefined) { + throw new MissingReasoningOptionsError(id, "Workers AI toggle wire path needs manual verification"); + } + console.warn(`Keeping catalogue reasoning for ${id}: Workers AI toggle wire path is unknown`); + // Keep the reasoning boundary small: other usable properties still update. + return { + id, + model: buildWorkersAiModel({ ...model, reasoning: undefined }, existing), + header, + }; + } return { id, - model: buildWorkersAiModel(normalized, context.existing(id)), + model: translated, + header: (translated.reasoning_options?.some((option) => option.type === "toggle") && !hasToggleWire ? toggleHeader : "") + + preservedHeader, }; }, -} satisfies SyncProvider; +} satisfies SyncProvider; + +function hasReasoning(model: WorkersAiModel) { + return model.supported_parameters.includes("reasoning") || model.supported_parameters.includes("include_reasoning") + || model.reasoning?.mandatory !== undefined || model.reasoning?.supported_efforts !== undefined + || model.reasoning?.supports_max_tokens === true; +} + +function modelHeader(file: string) { + try { + const lines = readFileSync(file, "utf8").split("\n"); + const firstKey = lines.findIndex((line) => line.trim() !== "" && !line.trim().startsWith("#")); + return lines.slice(0, firstKey === -1 ? lines.length : firstKey).join("\n") + "\n"; + } catch (error) { + console.warn(`Skipping Workers AI model with an unreadable catalogue header: ${file}`, error); + return undefined; + } +} export function buildWorkersAiModel( - model: z.infer, + model: WorkersAiModel, existing: ExistingModel | undefined, ): SyncedModel { + const reasoningOptions = workersAiReasoningOptions(model); + const reasoning = reasoningOptions !== undefined + ? true + : existing?.reasoning ?? hasReasoning(model); const source = { ...model, + reasoning: undefined, + supported_parameters: [ + ...model.supported_parameters.filter((parameter) => !["reasoning", "include_reasoning"].includes(parameter)), + ...(reasoning ? ["reasoning"] : []), + ], name: existing?.name ?? model.name, top_provider: { ...model.top_provider, max_completion_tokens: existing?.limit?.output ?? model.top_provider.max_completion_tokens, }, }; - const synced = { - ...buildOpenRouterModel( - source, - existing, - existing?.base_model ?? resolveCloudflareBaseModel(model), - ), - reasoning_options: existing?.reasoning_options, - }; + // The shared builder uses these options when source.reasoning is omitted. + const existingWithReasoningOptions = reasoningOptions === undefined + ? existing + : { ...existing, reasoning_options: reasoningOptions }; + const synced = buildOpenRouterModel( + source, + existingWithReasoningOptions, + existing?.base_model ?? resolveCloudflareBaseModel(model), + ); if ("base_model" in synced) return synced; return { ...synced, @@ -129,7 +207,50 @@ export function buildWorkersAiModel( }; } -export function resolveCloudflareBaseModel(model: z.infer) { +function workersAiReasoningOptions({ id, reasoning }: WorkersAiModel): SyncedFullModel["reasoning_options"] { + // A null gateway allowlist does not identify concrete model controls. + // Preserve the catalogue instead of expanding it to every effort in the schema. + if (reasoning === undefined || reasoning.supported_efforts === null) return undefined; + + const options: NonNullable = []; + const efforts = reasoning.supported_efforts; + if (efforts?.length === 0) return undefined; + if (efforts === undefined && reasoning.supports_max_tokens !== true) { + // GLM-4.7-Flash's verified ConfigAPI/Search shape is { mandatory: false, + // default_enabled: true }; its only control is enable_thinking. The generic + // input schema's low/medium/high enum is not a model capability. + // Creator: https://huggingface.co/zai-org/GLM-4.7-Flash/blob/main/chat_template.jinja + // Require the explicit off-capability flag; absence of efforts alone says nothing. + return id === "@cf/zai-org/glm-4.7-flash" && reasoning.mandatory === false + ? [{ type: "toggle" }] + : undefined; + } + // Without either an explicit mandatory flag or a named off setting, a partial + // effort list cannot tell us whether replacing the catalogue would lose a toggle. + if (reasoning.mandatory === undefined && !efforts?.includes("none")) return undefined; + + if (reasoning.mandatory === false && !efforts?.includes("none")) { + options.push({ type: "toggle" }); + } + + const values = reasoning.mandatory ? efforts?.filter((value) => value !== "none") : efforts; + // One mandatory effective effort offers no caller choice. Unlike missing or + // empty metadata, a concrete singleton establishes this explicitly. + const fixedEffort = reasoning.mandatory === true && values?.length === 1; + if (values?.length && !fixedEffort) { + options.push({ type: "effort", values: [...values] }); + } + + if (reasoning.supports_max_tokens === true) { + // Explicit reasoning.max_tokens support, never inferred from output limits. + options.push({ type: "budget_tokens" }); + } + + // Empty control metadata never clears authored controls; a known fixed effort can. + return options.length > 0 || fixedEffort ? options : undefined; +} + +export function resolveCloudflareBaseModel(model: WorkersAiModel) { const [, publisher] = model.id.replace(/^workers-ai\//, "").split("/"); if (publisher === undefined) return undefined; @@ -163,39 +284,61 @@ async function fetchPage(accountID: string, token: string, page: number) { url.searchParams.set("per_page", "1000"); url.searchParams.set("page", String(page)); - const response = await fetch(url, { - headers: { Authorization: `Bearer ${token}` }, - }); - if (!response.ok) { - throw new Error( - `Cloudflare Workers AI models request failed: ${response.status} ${response.statusText}${await responseDetails(response)}`, - ); + for (let attempt = 1; attempt <= 4; attempt++) { + try { + const response = await fetch(url, { + headers: { Authorization: `Bearer ${token}` }, + signal: AbortSignal.timeout(30_000), + }); + if (!response.ok) { + throw new Error(`${response.status} ${response.statusText}${await responseDetails(response)}`); + } + return CloudflareResponse.parse(await response.json()); + } catch (error) { + console.warn( + `Workers AI search page ${page}, attempt ${attempt}/4 failed: ${error instanceof Error ? error.message : String(error)}`, + ); + if (attempt < 4) await Bun.sleep(30_000); + } } - return response.json(); } -function parseCloudflareModels(raw: unknown): CloudflareModel[] { - const cloudflare = CloudflareResponse.safeParse(raw); - if (cloudflare.success) return cloudflare.data.data; - - const direct = OpenRouterResponse.safeParse(raw); - if (direct.success) return direct.data.data.map((model) => CloudflareModel.parse(model)); - - const wrapped = CloudflareOpenRouterResponse.parse(raw); - if (wrapped.result === undefined) { - throw new Error("Cloudflare Workers AI response did not include model data"); - } - const models = Array.isArray(wrapped.result) ? wrapped.result : wrapped.result.data; - return models.map((model) => CloudflareModel.parse(model)); +function parseCloudflareModels(raw: unknown): WorkersAiModel[] { + const response = CloudflareResponse.parse(raw); + return modelRows(response).flatMap((row) => { + const parsed = CloudflareModel.safeParse(row); + if (!parsed.success) { + console.warn(`Skipping invalid Workers AI model: ${parsed.error.message}`); + return []; + } + try { + return [normalizeModel(parsed.data)]; + } catch (error) { + console.warn(`Skipping invalid Workers AI model ${parsed.data.id}: ${error instanceof Error ? error.message : String(error)}`); + return []; + } + }); } function normalizeModel(model: CloudflareModel) { - if ("architecture" in model && "top_provider" in model && "supported_parameters" in model) { - return OpenRouterModel.parse(model); + const parsed = WorkersAiReasoning.safeParse(model.reasoning); + const reasoning = parsed.success ? parsed.data : undefined; + if (!parsed.success && model.reasoning != null) { + console.warn(`Ignoring invalid Workers AI reasoning for ${model.id}: ${parsed.error.message}`); + } + const id = model.id.replace(/^workers-ai\//, ""); + const normalizedID = id.startsWith("@cf/") ? id : `@cf/${id}`; + if ("architecture" in model || "top_provider" in model || "supported_parameters" in model) { + const normalized = WorkersAiModel.parse({ ...model, id: normalizedID, reasoning }); + z.number().int().positive().nullable().parse(normalized.top_provider.max_completion_tokens); + if (normalized.architecture.input_modalities.length === 0 || normalized.architecture.output_modalities.length === 0) { + throw new Error("Model modalities must not be empty"); + } + return normalized; } - return OpenRouterModel.parse({ - id: model.id.startsWith("@cf/") ? model.id : `@cf/${model.id.replace(/^@cf\//, "")}`, + return WorkersAiModel.parse({ + id: normalizedID, name: model.name, created: model.created, hugging_face_id: model.hugging_face_id ?? null, @@ -214,6 +357,7 @@ function normalizeModel(model: CloudflareModel) { ...model.supported_sampling_parameters ?? [], ...model.supported_features ?? [], ], + reasoning, }); } diff --git a/packages/core/src/sync/providers/crossmodel.ts b/packages/core/src/sync/providers/crossmodel.ts index e405a811a8f..0a1f3d067fe 100644 --- a/packages/core/src/sync/providers/crossmodel.ts +++ b/packages/core/src/sync/providers/crossmodel.ts @@ -193,8 +193,9 @@ export function buildCrossModel( // CrossModel serves threshold-tiered pricing. The lowest-threshold tier is the // headline [cost]; every higher tier maps to a [[cost.tiers]] context band // (threshold -> tier size), so tier pricing stays fresh on each sync instead of - // being frozen at hand-authored values. Fall back to the existing tiers only - // when the API reports none. + // being frozen at hand-authored values. When the API reports usable pricing, + // its tier list is authoritative; fall back to the existing cost only when the + // source pricing is absent or unusable. const tiers = [...(model.pricing?.tiers ?? [])].sort( (a, b) => (a.threshold ?? 0) - (b.threshold ?? 0), ); @@ -210,7 +211,7 @@ export function buildCrossModel( .filter((entry): entry is NonNullable => entry !== undefined); const cost = base !== undefined - ? { ...base, tiers: contextTiers.length > 0 ? contextTiers : existing?.cost?.tiers } + ? { ...base, tiers: contextTiers.length > 0 ? contextTiers : undefined } : existing?.cost; // Every served model reports a context window; without one (and no existing diff --git a/packages/core/src/sync/providers/nebul.ts b/packages/core/src/sync/providers/nebul.ts index 4c61e9e98e7..0207352d7f2 100644 --- a/packages/core/src/sync/providers/nebul.ts +++ b/packages/core/src/sync/providers/nebul.ts @@ -110,25 +110,28 @@ export const nebul = { : existing?.cost; const limit = info.max_input_tokens != null ? { context: info.max_input_tokens } : existing?.limit; if (existing === undefined && (baseModel === undefined || cost === undefined || limit === undefined)) return undefined; + // A hand-authored reasoning = false marks a served ID whose lab model reasons + // but which this host runs with thinking disabled (the catalog reports + // supports_reasoning = false and no reasoning_efforts). Keep the override and + // suppress the control/trace machinery entirely: no reasoning_options to + // require, and no interleaved side channel when no traces are returned. + const reasoningDisabled = existing?.reasoning === false; // Fail closed rather than emitting no reasoning_options: a reasoner with // neither advertised efforts nor authored options would sync as an empty // entry (no caller control). The runner keeps the file and lists it in the // skipped notice so the options can be hand-authored. - const isReasoner = baseModel !== undefined + const isReasoner = !reasoningDisabled && (baseModel !== undefined ? modelMetadata(baseModel).reasoning === true - : existing?.reasoning === true; + : existing?.reasoning === true); if (isReasoner && (info.reasoning_efforts ?? []).length === 0 && existing?.reasoning_options === undefined) { throw new MissingReasoningOptionsError( id, `${id} is a reasoning model, but Nebul advertises no reasoning_efforts and the catalog entry has no reasoning_options; hand-author them`, ); } - const values = { - interleaved: existing?.interleaved, - reasoning_options: buildReasoningOptions(entry, existing), - cost, - limit, - }; + const values = reasoningDisabled + ? { reasoning: false, interleaved: undefined, reasoning_options: undefined, cost, limit } + : { interleaved: existing?.interleaved, reasoning_options: buildReasoningOptions(entry, existing), cost, limit }; if (baseModel !== undefined) { return { id, diff --git a/packages/core/src/sync/providers/requesty.ts b/packages/core/src/sync/providers/requesty.ts index 00f6dbcc990..d8b0bdf4de5 100644 --- a/packages/core/src/sync/providers/requesty.ts +++ b/packages/core/src/sync/providers/requesty.ts @@ -184,8 +184,9 @@ function reasoningOptions( } function buildCost(model: RequestyModel): SyncedFullModel["cost"] { - const input = model.input_price; - const output = model.output_price; + const base = model.pricing?.[0] ?? model; + const input = base.input_price; + const output = base.output_price; if (input == null || output == null) return undefined; const tiers = (model.pricing ?? []).slice(1).map((band) => ({ @@ -198,8 +199,8 @@ function buildCost(model: RequestyModel): SyncedFullModel["cost"] { return { input: pricePerMillion(input), output: pricePerMillion(output), - cache_read: chargedPricePerMillion(model.cached_price), - cache_write: chargedPricePerMillion(model.caching_price), + cache_read: chargedPricePerMillion(base.cached_price), + cache_write: chargedPricePerMillion(base.caching_price), tiers: tiers.length > 0 ? tiers : undefined, }; } diff --git a/packages/core/src/sync/providers/vercel.ts b/packages/core/src/sync/providers/vercel.ts index fdf8be631dd..e7800f77e32 100644 --- a/packages/core/src/sync/providers/vercel.ts +++ b/packages/core/src/sync/providers/vercel.ts @@ -237,17 +237,75 @@ function price(value: string | undefined) { : undefined; } +function tieredPrice(value: string | undefined, tiers: z.infer[] | undefined) { + const base = price(value); + const normalized = (tiers ?? []) + .map((tier, index, values) => ({ + start: tier.min ?? (index === 0 ? 0 : values[index - 1]?.max ?? 0), + cost: price(tier.cost), + })) + .filter((tier): tier is { start: number; cost: number } => tier.cost !== undefined) + .sort((a, b) => a.start - b.start); + + return { + base: normalized[0]?.cost ?? base, + thresholds: normalized.map((tier) => tier.start).filter((start) => start > 0), + at(threshold: number) { + return normalized.findLast((tier) => tier.start <= threshold)?.cost ?? base; + }, + }; +} + function buildCost(pricing: VercelModel["pricing"], existing?: ExistingModel["cost"]) { - const input = price(pricing?.input_tiers?.[0]?.cost ?? pricing?.input); - const output = price(pricing?.output_tiers?.[0]?.cost ?? pricing?.output); + const hasPricingTiers = [ + pricing?.input_tiers, + pricing?.output_tiers, + pricing?.input_cache_read_tiers, + pricing?.input_cache_write_tiers, + ].some((tiers) => (tiers?.length ?? 0) > 0); + const inputPrice = tieredPrice(pricing?.input, pricing?.input_tiers); + const outputPrice = tieredPrice(pricing?.output, pricing?.output_tiers); + const cacheReadPrice = tieredPrice(pricing?.input_cache_read, pricing?.input_cache_read_tiers); + const cacheWritePrice = tieredPrice(pricing?.input_cache_write, pricing?.input_cache_write_tiers); + const input = inputPrice.base; + const output = outputPrice.base; if (input === undefined || output === undefined) return undefined; + + const thresholds = new Set([ + ...inputPrice.thresholds, + ...outputPrice.thresholds, + ...cacheReadPrice.thresholds, + ...cacheWritePrice.thresholds, + ]); + const tiers: NonNullable["tiers"]> = []; + let previous = { + input, + output, + cache_read: cacheReadPrice.base, + cache_write: cacheWritePrice.base, + }; + for (const size of [...thresholds].sort((a, b) => a - b)) { + const tierInput = inputPrice.at(size); + const tierOutput = outputPrice.at(size); + if (tierInput === undefined || tierOutput === undefined) continue; + const current = { + input: tierInput, + output: tierOutput, + cache_read: cacheReadPrice.at(size), + cache_write: cacheWritePrice.at(size), + }; + if (JSON.stringify(current) === JSON.stringify(previous)) continue; + tiers.push({ tier: { type: "context", size }, ...current }); + previous = current; + } + return { input, output, reasoning: existing?.reasoning, - cache_read: price(pricing?.input_cache_read_tiers?.[0]?.cost ?? pricing?.input_cache_read), - cache_write: price(pricing?.input_cache_write_tiers?.[0]?.cost ?? pricing?.input_cache_write), - tiers: existing?.tiers, + cache_read: cacheReadPrice.base, + cache_write: cacheWritePrice.base, + tiers: hasPricingTiers ? (tiers.length > 0 ? tiers : undefined) : existing?.tiers, }; } @@ -291,6 +349,7 @@ function sameVercelModel(current: ExistingModel, desired: SyncedModel) { [current.cost?.output, desiredModel.cost?.output, true], [current.cost?.cache_read, desiredModel.cost?.cache_read, true], [current.cost?.cache_write, desiredModel.cost?.cache_write, true], + [current.cost?.tiers, desiredModel.cost?.tiers], [current.limit?.context, desiredModel.limit?.context], [current.limit?.input, desiredModel.limit?.input], [current.limit?.output, desiredModel.limit?.output], diff --git a/packages/core/test/aiand.test.ts b/packages/core/test/aiand.test.ts index d4e0c74c1fb..9e531806203 100644 --- a/packages/core/test/aiand.test.ts +++ b/packages/core/test/aiand.test.ts @@ -173,13 +173,67 @@ test("skippedNotice stays silent on a clean sync", () => { test("authored-only cost fields ride along; feed prices are authoritative", () => { const built = buildAiandModel( aiandModel(), - { cost: { input: 9, output: 9, cache_write: 0.5 }, limit: { context: 1, input: 128_000, output: 1 } }, + { + cost: { input: 9, output: 9, reasoning: 0.75, cache_write: 0.5, input_audio: 1.25 }, + limit: { context: 1, input: 128_000, output: 1 }, + }, null, ); - expect(built.cost).toMatchObject({ input: 0.15, output: 0.25, cache_write: 0.5 }); + expect(built.cost).toMatchObject({ + input: 0.15, + output: 0.25, + reasoning: 0.75, + cache_read: 0.08, + cache_write: 0.5, + input_audio: 1.25, + }); expect(built.limit).toEqual({ context: 1_048_576, input: 128_000, output: 384_000 }); }); +test("feed cost tiers and auxiliary prices replace authored values", () => { + const tiers = [{ + tier: { type: "context" as const, size: 200_000 }, + input: 0.3, + output: 0.5, + cache_read: 0.16, + }]; + const model = AiandModel.parse({ + ...aiandModel(), + cost: { + ...aiandModel().cost, + reasoning: 0.4, + cache_write: 0.12, + output_audio: 2, + tiers, + }, + }); + const built = buildAiandModel(model, { + cost: { + input: 9, + output: 9, + reasoning: 9, + cache_write: 9, + output_audio: 9, + tiers: [{ tier: { type: "context", size: 100_000 }, input: 9, output: 9 }], + }, + }, null); + + expect(built.cost).toMatchObject({ reasoning: 0.4, cache_write: 0.12, output_audio: 2 }); + expect(built.cost?.tiers).toEqual(tiers); +}); + +test("omitted feed cost tiers clear authored tiers", () => { + const built = buildAiandModel(aiandModel(), { + cost: { + input: 9, + output: 9, + tiers: [{ tier: { type: "context", size: 200_000 }, input: 9, output: 9 }], + }, + }, null); + + expect(built.cost?.tiers).toBeUndefined(); +}); + test("a non-reasoning feed model omits reasoning_options entirely", () => { const built = buildAiandModel( aiandModel({ reasoning: false, reasoning_options: undefined }), diff --git a/packages/core/test/cloudflare-workers-ai.test.ts b/packages/core/test/cloudflare-workers-ai.test.ts new file mode 100644 index 00000000000..e15416cb7ec --- /dev/null +++ b/packages/core/test/cloudflare-workers-ai.test.ts @@ -0,0 +1,56 @@ +import { expect, test } from "bun:test"; + +import { cloudflareWorkersAi } from "../src/sync/providers/cloudflare-workers-ai.js"; + +test("preserves OpenRouter pricing overrides as context tiers", () => { + const [source] = cloudflareWorkersAi.parseModels({ + result: { + data: [{ + id: "workers-ai/@cf/test/context-priced", + name: "Context Priced", + created: 1_782_777_600, + hugging_face_id: null, + knowledge_cutoff: null, + context_length: 256_000, + architecture: { + input_modalities: ["text"], + output_modalities: ["text"], + }, + pricing: { + prompt: "0.000001", + completion: "0.000002", + overrides: [{ + min_prompt_tokens: 128_000, + prompt: "0.000003", + completion: "0.000004", + }], + }, + top_provider: { + context_length: 256_000, + max_completion_tokens: 8_192, + }, + supported_parameters: [], + }], + }, + }); + + expect(source?.pricing.overrides).toEqual([{ + min_prompt_tokens: 128_000, + prompt: "0.000003", + completion: "0.000004", + }]); + + const translated = cloudflareWorkersAi.translateModel(source!, { + existing: () => undefined, + authored: () => undefined, + }); + expect(translated.model.cost).toEqual({ + input: 1, + output: 2, + tiers: [{ + tier: { type: "context", size: 128_000 }, + input: 3, + output: 4, + }], + }); +}); diff --git a/packages/core/test/filter.test.ts b/packages/core/test/filter.test.ts new file mode 100644 index 00000000000..ea14f1588fb --- /dev/null +++ b/packages/core/test/filter.test.ts @@ -0,0 +1,103 @@ +import { describe, expect, test } from "bun:test"; + +import { + filterCatalogByModelType, + generateCatalog, + InvalidModelTypeError, + parseModelTypes, +} from "../src/index.js"; +import type { ModelMetadata, Provider } from "../src/index.js"; +import path from "node:path"; + +describe("model type filtering", () => { + test("defaults to untyped models and supports specific types and all", () => { + expect(parseModelTypes(null)).toBe("default"); + expect(parseModelTypes("")).toBe("default"); + expect(parseModelTypes("decision")).toEqual(["decision"]); + expect(parseModelTypes("all")).toBe("all"); + }); + + test("rejects unknown types and combining all with a type", () => { + expect(() => parseModelTypes("unknown")).toThrow(InvalidModelTypeError); + expect(() => parseModelTypes("all,decision")).toThrow( + InvalidModelTypeError, + ); + }); + + test("omits typed models by default", () => { + const catalog = fixture(); + const filtered = filterCatalogByModelType(catalog, "default"); + + expect(Object.keys(filtered.models)).toEqual(["standard"]); + expect(Object.keys(filtered.providers.example!.models)).toEqual([ + "standard", + ]); + expect(filtered.providers.decisionOnly).toBeUndefined(); + }); + + test("filters canonical and provider models by requested type", () => { + const catalog = fixture(); + const filtered = filterCatalogByModelType(catalog, ["decision"]); + + expect(Object.keys(filtered.models)).toEqual(["decision"]); + expect(Object.keys(filtered.providers.example!.models)).toEqual([ + "decision", + ]); + expect(Object.keys(filtered.providers.decisionOnly!.models)).toEqual([ + "decision", + ]); + expect(filterCatalogByModelType(catalog, "all")).toEqual(catalog); + }); + + test("every repository Jev model inherits decision and is omitted by default", async () => { + const root = path.join(import.meta.dir, "..", "..", ".."); + const catalog = await generateCatalog(root); + const jevModels = Object.values(catalog.providers).flatMap((provider) => + Object.values(provider.models).filter((model) => + model.id.toLowerCase().includes("jev"), + ), + ); + + expect(jevModels.length).toBeGreaterThan(0); + expect(jevModels.every((model) => model.type === "decision")).toBe(true); + + const defaults = filterCatalogByModelType(catalog, "default"); + expect( + Object.values(defaults.providers).some((provider) => + Object.values(provider.models).some((model) => model.type !== undefined), + ), + ).toBe(false); + expect(defaults.models["typesafe/jev-latest"]).toBeUndefined(); + expect(defaults.providers.vivgrid?.models.jev).toBeUndefined(); + + const decisions = filterCatalogByModelType(catalog, ["decision"]); + expect(decisions.models["typesafe/jev-latest"]?.type).toBe("decision"); + expect( + Object.values(decisions.providers).flatMap((provider) => + Object.values(provider.models), + ).length, + ).toBe(jevModels.length); + }); +}); + +function fixture() { + const standard = model("standard-model"); + const decision = model("decision-model", "decision"); + return { + models: { standard, decision }, + providers: { + example: { + id: "example", + models: { standard, decision }, + } as unknown as Provider, + decisionOnly: { + id: "decision-only", + models: { decision }, + } as unknown as Provider, + }, + }; +} + +function model(id: string, type?: "decision") { + return { id, type } as unknown as ModelMetadata; +} diff --git a/packages/core/test/nebul.test.ts b/packages/core/test/nebul.test.ts index 63399f45bc1..f716af6b2cd 100644 --- a/packages/core/test/nebul.test.ts +++ b/packages/core/test/nebul.test.ts @@ -81,6 +81,22 @@ test("keeps authored effort sets when the host advertises none", () => { expect(translated?.model.reasoning_options).toEqual(authored); }); +test("keeps an authored reasoning = false override for a lab reasoner the host serves without thinking", () => { + const existing = { + base_model: "moonshotai/kimi-k3", + reasoning: false, + reasoning_options: [{ type: "effort" as const, values: ["low"] }], + interleaved: { field: "reasoning_content" as const }, + } as ExistingModel; + const translated = nebul.translateModel( + nebulEntry("moonshotai/Kimi-K3", { reasoning_efforts: undefined }), + context(existing), + ); + expect(translated?.model.reasoning).toBe(false); + expect(translated?.model.reasoning_options).toBeUndefined(); + expect(translated?.model.interleaved).toBeUndefined(); +}); + test("fails closed when a reasoner advertises no efforts and none are authored", () => { const entry = nebulEntry("zai-org/GLM-5.3", { reasoning_efforts: [] }); expect(() => nebul.translateModel(entry, context(undefined))).toThrow(MissingReasoningOptionsError); diff --git a/packages/core/test/requesty.test.ts b/packages/core/test/requesty.test.ts index 597b6de65c8..52eaec2ccde 100644 --- a/packages/core/test/requesty.test.ts +++ b/packages/core/test/requesty.test.ts @@ -49,3 +49,81 @@ test.each(["claude-fable-5.1", "claude-fable-5.1@eu"])( }); }, ); + +test("uses the first Requesty pricing band as the base cost", () => { + const model = buildRequestyModel(RequestyModel.parse({ + id: "requesty-priced-model", + created: Date.parse("2026-09-01") / 1_000, + description: "Requesty description", + context_window: 1_000_000, + max_output_tokens: 128_000, + input_price: 0.000001, + output_price: 0.000002, + cached_price: 0.000003, + caching_price: 0.000004, + pricing: [ + { + prompt_tokens_threshold: 0, + input_price: 0.00001, + output_price: 0.00005, + cached_price: 0.00000025, + caching_price: 0.0000125, + }, + { + prompt_tokens_threshold: 200_000, + input_price: 0.00002, + output_price: 0.000075, + cached_price: 0.0000005, + caching_price: 0.000025, + }, + ], + })); + + expect(model.cost).toEqual({ + input: 10, + output: 50, + cache_read: 0.25, + cache_write: 12.5, + tiers: [{ + tier: { type: "context", size: 200_000 }, + input: 20, + output: 75, + cache_read: 0.5, + cache_write: 25, + }], + }); +}); + +test("uses Requesty pricing bands when top-level prices are null", () => { + const model = buildRequestyModel(RequestyModel.parse({ + id: "requesty-priced-model", + created: Date.parse("2026-09-01") / 1_000, + description: "Requesty description", + context_window: 1_000_000, + max_output_tokens: 128_000, + input_price: null, + output_price: null, + pricing: [ + { + prompt_tokens_threshold: 0, + input_price: 0.00001, + output_price: 0.00005, + }, + { + prompt_tokens_threshold: 200_000, + input_price: null, + output_price: null, + }, + ], + })); + + expect(model.cost).toEqual({ + input: 10, + output: 50, + tiers: [{ + tier: { type: "context", size: 200_000 }, + input: 10, + output: 50, + }], + }); +}); diff --git a/packages/core/test/schema.test.ts b/packages/core/test/schema.test.ts index f343c50b615..97a6d602032 100644 --- a/packages/core/test/schema.test.ts +++ b/packages/core/test/schema.test.ts @@ -70,6 +70,12 @@ describe("model schema", () => { expect(AuthoredModel.safeParse(model).success).toBe(false); }); + test("accepts decision model types", () => { + const model = baseModel({ type: "decision" }); + + expect(AuthoredModel.safeParse(model).success).toBe(true); + }); + test("accepts calendar-valid model dates", () => { for (const field of dateFields) { for (const value of [ diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 3557ba66127..3da3fe65f8f 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -449,6 +449,38 @@ test("syncs CrossModel's structured-output capability", () => { }); }); +test("clears stale CrossModel context tiers only when source pricing is usable", () => { + const existing: ExistingModel = { + base_model: "alibaba/qwen3.8-max", + cost: { + input: 9, + output: 27, + tiers: [ + { + tier: { type: "context", size: 200_000 }, + input: 18, + output: 54, + }, + ], + }, + }; + + const authoritative = buildCrossModel(crossModelModel(), existing); + const absent = buildCrossModel(crossModelModel({ pricing: undefined }), existing); + const unusable = buildCrossModel( + crossModelModel({ + pricing: { + tiers: [{ threshold: 0, input_micro_per_1m: 1_880_000 }], + }, + }), + existing, + ); + + expect(authoritative?.cost).toEqual({ input: 1.88, output: 5.63 }); + expect(absent?.cost).toEqual(existing.cost); + expect(unusable?.cost).toEqual(existing.cost); +}); + test("parses CrossModel's nullable reasoning controls", () => { const parsed = CrossModelResponse.parse({ data: [ @@ -2806,6 +2838,28 @@ test("resolves Eden AI aliases to the model they point at", () => { ).toBe("anthropic/claude-opus-5"); }); +test("preserves model type when formatting synced TOML", () => { + const content = formatToml({ + id: "typesafe/jev-latest", + type: "decision", + name: "Jev", + description: "System One model for typed decisions", + release_date: "2026-09-15", + last_updated: "2026-09-15", + attachment: false, + reasoning: false, + tool_call: false, + open_weights: false, + limit: { context: 64_000, output: 0 }, + modalities: { input: ["text"], output: ["text"] }, + }); + + expect(Bun.TOML.parse(content)).toMatchObject({ + type: "decision", + name: "Jev", + }); +}); + test("formats interleaved as a root field before reasoning option tables", () => { const content = formatToml({ id: "example/model", @@ -4401,7 +4455,7 @@ test("retains Merge Gateway models missing from an API-key-scoped response", () expect(mergeGateway.deleteMissing).toBe(false); }); -test("parses Vercel pricing tiers with an implicit zero minimum", () => { +test("translates Vercel pricing tiers with an implicit zero minimum", () => { const [model] = vercel.parseModels({ data: [{ id: "openai/gpt-5.6-luna", @@ -4418,14 +4472,36 @@ test("parses Vercel pricing tiers with an implicit zero minimum", () => { { cost: "0.0000001", max: 272_000 }, { cost: "0.0000002", min: 272_000 }, ], + input_tiers: [ + { cost: "0.000001", max: 272_000 }, + { cost: "0.000002", min: 272_000 }, + ], + output_tiers: [ + { cost: "0.000006", max: 272_000 }, + { cost: "0.000009", min: 272_000 }, + ], }, }], }); expect(model).toBeDefined(); - expect(buildVercelModel(model!, undefined)).toMatchObject({ - cost: { input: 1, output: 6, cache_read: 0.1 }, + const synced = buildVercelModel(model!, undefined); + expect(synced).toMatchObject({ + cost: { + input: 1, + output: 6, + cache_read: 0.1, + tiers: [{ + tier: { type: "context", size: 272_000 }, + input: 2, + output: 9, + cache_read: 0.2, + }], + }, }); + expect(vercel.sameModel?.({ + cost: { input: 1, output: 6, cache_read: 0.1 }, + }, synced)).toBe(false); }); test("Vercel factored models inherit temperature from base metadata", () => { diff --git a/packages/function/package.json b/packages/function/package.json index 1d1c7c9f5c1..e1fd3d8260a 100644 --- a/packages/function/package.json +++ b/packages/function/package.json @@ -3,6 +3,9 @@ "name": "@models.dev/function", "private": true, "type": "module", + "dependencies": { + "@models.dev/core": "workspace:*" + }, "devDependencies": { "@cloudflare/workers-types": "4.20250522.0", "@tsconfig/bun": "catalog:" diff --git a/packages/function/src/worker.ts b/packages/function/src/worker.ts index 9ccccdd42e4..d616805d8f9 100644 --- a/packages/function/src/worker.ts +++ b/packages/function/src/worker.ts @@ -1,3 +1,13 @@ +import { + filterCatalogByModelType, + filterModelsByModelType, + filterProvidersByModelType, + InvalidModelTypeError, + MODEL_TYPES, + parseModelTypes, +} from "@models.dev/core/src/filter.js"; +import type { ModelTypeValue } from "@models.dev/core/src/filter.js"; + export interface Env { ASSETS: any; PosthogToken: string; @@ -64,11 +74,8 @@ export default { } if (url.pathname === "/model-schema.json") { - const apiUrl = new URL(url); - apiUrl.pathname = "/_api.json"; - const apiResponse = await env.ASSETS.fetch( - new Request(apiUrl.toString(), request), - ); + const apiResponse = await catalogResponse(url, request, env, "api"); + if (!apiResponse.ok) return apiResponse; const providers = (await apiResponse.json()) as Record< string, { models: Record } @@ -102,11 +109,11 @@ export default { } if (url.pathname === "/api.json") { - url.pathname = "/_api.json"; + return catalogResponse(url, request, env, "api"); } else if (url.pathname === "/models.json") { - url.pathname = "/_models.json"; + return catalogResponse(url, request, env, "models"); } else if (url.pathname === "/catalog.json") { - url.pathname = "/_catalog.json"; + return catalogResponse(url, request, env, "catalog"); } else if ( url.pathname === "/" || url.pathname === "/index.html" || @@ -143,6 +150,81 @@ export default { }, }; +type CatalogEndpoint = "api" | "models" | "catalog"; + +async function catalogResponse( + url: URL, + request: Request, + env: Env, + endpoint: CatalogEndpoint, +) { + let filter; + try { + filter = parseModelTypes(url.searchParams.get("type")); + } catch (error) { + if (!(error instanceof InvalidModelTypeError)) throw error; + return Response.json( + { + error: error.message, + allowed: [...MODEL_TYPES, "all"], + }, + { + status: 400, + headers: { "Access-Control-Allow-Origin": "*" }, + }, + ); + } + + const assetUrl = new URL(url); + const suffix = filter === "default" + ? "" + : filter === "all" + ? "-all" + : filter.length === 1 + ? `-${filter[0]}` + : undefined; + assetUrl.pathname = `/_${endpoint}${suffix ?? "-all"}.json`; + assetUrl.search = ""; + const assetResponse = await env.ASSETS.fetch( + new Request(assetUrl.toString(), request), + ); + if (!assetResponse.ok || suffix !== undefined) return assetResponse; + + const value = await assetResponse.json(); + const filtered = endpoint === "api" + ? filterProvidersByModelType( + value as Record, + filter, + ) + : endpoint === "models" + ? filterModelsByModelType( + value as Record, + filter, + ) + : filterCatalogByModelType( + value as { + providers: Record; + models: Record; + }, + filter, + ); + + const headers = new Headers(assetResponse.headers); + headers.delete("Content-Length"); + headers.delete("ETag"); + headers.set("Content-Type", "application/json"); + headers.set("Cache-Control", "public, max-age=3600"); + return new Response(JSON.stringify(filtered), { headers }); +} + +interface CatalogModel { + type?: ModelTypeValue; +} + +interface CatalogProvider { + models: Record; +} + function isHtmlRoute(pathname: string) { return ( pathname === "/models" || diff --git a/packages/function/test/worker.test.ts b/packages/function/test/worker.test.ts new file mode 100644 index 00000000000..26d9e0fb04a --- /dev/null +++ b/packages/function/test/worker.test.ts @@ -0,0 +1,138 @@ +import { describe, expect, test } from "bun:test"; + +import worker, { type Env } from "../src/worker.js"; + +const textModel = { + id: "text-model", + modalities: { input: ["text"], output: ["text"] }, +}; +const decisionModel = { + id: "decision-model", + type: "decision", + modalities: { input: ["text"], output: ["text"] }, +}; +const providers = { + example: { + id: "example", + models: { text: textModel, decision: decisionModel }, + }, +}; +const models = { text: textModel, decision: decisionModel }; + +describe("catalog API model type filtering", () => { + test("omits typed models from api.json by default", async () => { + const response = await request("/api.json"); + const body = await response.json(); + + expect(Object.keys(body.example.models)).toEqual(["text"]); + }); + + test("omits typed models from models.json by default", async () => { + const response = await request("/models.json"); + const body = await response.json(); + + expect(Object.keys(body)).toEqual(["text"]); + }); + + test("omits typed models from catalog.json by default", async () => { + const response = await request("/catalog.json"); + const body = await response.json(); + + expect(Object.keys(body.models)).toEqual(["text"]); + expect(Object.keys(body.providers.example.models)).toEqual(["text"]); + }); + + test("returns explicitly requested decision models", async () => { + const response = await request("/catalog.json?type=decision"); + const body = await response.json(); + + expect(Object.keys(body.models)).toEqual(["decision"]); + expect(Object.keys(body.providers.example.models)).toEqual(["decision"]); + }); + + test("returns the complete static catalog for all", async () => { + const response = await request("/models.json?type=all"); + const body = await response.json(); + + expect(Object.keys(body)).toEqual(["text", "decision"]); + }); + + test("omits typed models from model-schema.json by default", async () => { + const response = await request("/model-schema.json"); + const body = await response.json(); + + expect(body.$defs.Model.enum).toEqual(["example/text"]); + }); + + test("includes typed models in model-schema.json when requested", async () => { + const response = await request("/model-schema.json?type=all"); + const body = await response.json(); + + expect(body.$defs.Model.enum).toEqual([ + "example/decision", + "example/text", + ]); + }); + + test("rejects unknown model types", async () => { + const response = await request("/api.json?type=unknown"); + + expect(response.status).toBe(400); + }); +}); + +async function request(path: string) { + const env = { + ASSETS: { + fetch(input: Request) { + const pathname = new URL(input.url).pathname; + if (pathname === "/_api.json") { + return Response.json({ + example: { ...providers.example, models: { text: textModel } }, + }); + } + if (pathname === "/_api-all.json") return Response.json(providers); + if (pathname === "/_api-decision.json") { + return Response.json({ + example: { ...providers.example, models: { decision: decisionModel } }, + }); + } + if (pathname === "/_models.json") { + return Response.json({ text: textModel }); + } + if (pathname === "/_models-all.json") return Response.json(models); + if (pathname === "/_models-decision.json") { + return Response.json({ decision: decisionModel }); + } + if (pathname === "/_catalog.json") { + return Response.json({ + providers: { + example: { ...providers.example, models: { text: textModel } }, + }, + models: { text: textModel }, + }); + } + if (pathname === "/_catalog-all.json") { + return Response.json({ providers, models }); + } + if (pathname === "/_catalog-decision.json") { + return Response.json({ + providers: { + example: { ...providers.example, models: { decision: decisionModel } }, + }, + models: { decision: decisionModel }, + }); + } + return new Response(null, { status: 404 }); + }, + }, + } as unknown as Env; + + return worker.fetch( + new Request(`https://models.dev${path}`, { + headers: { "user-agent": "test" }, + }), + env, + { waitUntil() {} } as unknown as ExecutionContext, + ); +} diff --git a/packages/sdk/src/client.ts b/packages/sdk/src/client.ts index f63aa5c3660..ee3848e4e3f 100644 --- a/packages/sdk/src/client.ts +++ b/packages/sdk/src/client.ts @@ -1,5 +1,5 @@ import { ModelsDevError } from "./error.js" -import type { Catalog, ModelMetadataMap, ProviderMap } from "./types.js" +import type { Catalog, ModelMetadataMap, ModelType, ProviderMap } from "./types.js" /** Accepted anywhere headers can be passed. Same shapes as the standard `HeadersInit`. */ export type HeadersInput = Headers | Record | Array<[string, string]> @@ -21,6 +21,8 @@ export interface RequestOptions { readonly signal?: AbortSignal /** Extra headers for this request. Overrides client-level headers. */ readonly headers?: HeadersInput + /** Specialized model types to include. Omit for standard models; use `"all"` for the complete catalog. */ + readonly modelTypes?: "all" | readonly ModelType[] } /** @@ -41,7 +43,15 @@ export function make(options: ClientOptions = {}) { let response: Response try { - response = await fetch(new URL(path, base), { + const url = new URL(path, base) + const modelTypes = requestOptions?.modelTypes + if (modelTypes === "all") { + url.searchParams.set("type", "all") + } else if (modelTypes && modelTypes.length > 0) { + url.searchParams.set("type", modelTypes.join(",")) + } + + response = await fetch(url, { method: "GET", headers, signal: requestOptions?.signal, diff --git a/packages/sdk/src/effect.ts b/packages/sdk/src/effect.ts index cd1780d182e..24827173e48 100644 --- a/packages/sdk/src/effect.ts +++ b/packages/sdk/src/effect.ts @@ -1,4 +1,4 @@ // Effect-native client. Requires the optional peer dependency `effect`. export * as Models from "./effect/client.js" -export { ModelsDevError, type ClientOptions, type ModelsClient } from "./effect/client.js" +export { ModelsDevError, type ClientOptions, type ModelsClient, type RequestOptions } from "./effect/client.js" export type * from "./types.js" diff --git a/packages/sdk/src/effect/client.ts b/packages/sdk/src/effect/client.ts index 0eb98835d55..ea005cf8f82 100644 --- a/packages/sdk/src/effect/client.ts +++ b/packages/sdk/src/effect/client.ts @@ -1,6 +1,6 @@ import { Context, Effect, Layer, Schema } from "effect" import { HttpClient, HttpClientResponse } from "effect/unstable/http" -import type { Catalog, ModelMetadataMap, ProviderMap } from "../types.js" +import type { Catalog, ModelMetadataMap, ModelType, ProviderMap } from "../types.js" /** The only error in the failure channel of client methods. Wraps the underlying `HttpClientError` as `cause`. */ export class ModelsDevError extends Schema.TaggedErrorClass()("ModelsDevError", { @@ -14,6 +14,11 @@ export interface ClientOptions { readonly headers?: Record } +export interface RequestOptions { + /** Specialized model types to include. Omit for standard models; use `"all"` for the complete catalog. */ + readonly modelTypes?: "all" | readonly ModelType[] +} + /** * Creates a stateless models.dev client on top of the `HttpClient` service * from the environment (`FetchHttpClient.layer`, `NodeHttpClient.layer`, or a @@ -26,9 +31,17 @@ export const make = (options?: ClientOptions) => const baseUrl = options?.baseUrl ?? "https://models.dev" const base = baseUrl.endsWith("/") ? baseUrl : baseUrl + "/" - const get = (path: string): Effect.Effect => - http - .get(new URL(path, base), { + const get = (path: string, requestOptions?: RequestOptions): Effect.Effect => { + const url = new URL(path, base) + const modelTypes = requestOptions?.modelTypes + if (modelTypes === "all") { + url.searchParams.set("type", "all") + } else if (modelTypes && modelTypes.length > 0) { + url.searchParams.set("type", modelTypes.join(",")) + } + + return http + .get(url, { headers: options?.headers, }) .pipe( @@ -37,14 +50,15 @@ export const make = (options?: ClientOptions) => Effect.map((data) => data as A), Effect.mapError((cause) => new ModelsDevError({ cause })), ) + } return { /** All providers with their models, pricing, and limits (`/api.json`). */ - providers: () => get("api.json"), + providers: (options?: RequestOptions) => get("api.json", options), /** Provider-agnostic model metadata (`/models.json`). */ - models: () => get("models.json"), + models: (options?: RequestOptions) => get("models.json", options), /** Providers and model metadata in a single request (`/catalog.json`). */ - catalog: () => get("catalog.json"), + catalog: (options?: RequestOptions) => get("catalog.json", options), } }) diff --git a/packages/sdk/src/types.ts b/packages/sdk/src/types.ts index 92999e2dea7..250551d5e7a 100644 --- a/packages/sdk/src/types.ts +++ b/packages/sdk/src/types.ts @@ -80,6 +80,9 @@ export interface ModelCost extends Cost { /** Input/output data types a model supports. */ export type Modality = "text" | "audio" | "image" | "video" | "pdf" +/** A model's specialized behavioral contract. Omitted for standard generative models. */ +export type ModelType = "decision" + export interface Modalities { input: Modality[] output: Modality[] @@ -143,6 +146,7 @@ export interface BenchmarkResult { export interface ModelMetadata { /** Canonical model ID, e.g. "anthropic/claude-opus-4-6". */ id: string + type?: ModelType name: string description: string family?: ModelFamily @@ -208,6 +212,7 @@ export interface ModelProviderConfig { export interface Model { /** Provider-scoped model ID, e.g. "claude-opus-4-6". */ id: string + type?: ModelType name: string description: string family?: ModelFamily diff --git a/packages/sdk/test/client.test.ts b/packages/sdk/test/client.test.ts index 3f898ef9157..61bc8fda104 100644 --- a/packages/sdk/test/client.test.ts +++ b/packages/sdk/test/client.test.ts @@ -40,6 +40,17 @@ test("models() and catalog() hit their endpoints", async () => { expect(calls.map((call) => call.url.href)).toEqual(["https://models.dev/models.json", "https://models.dev/catalog.json"]) }) +test("catalog endpoints encode model type filters", async () => { + const { calls, fetch } = stub({}) + const client = Models.make({ fetch }) + await client.providers({ modelTypes: ["decision"] }) + await client.models({ modelTypes: "all" }) + expect(calls.map((call) => call.url.href)).toEqual([ + "https://models.dev/api.json?type=decision", + "https://models.dev/models.json?type=all", + ]) +}) + test("baseUrl with subpath is preserved, with or without trailing slash", async () => { const { calls, fetch } = stub({}) await Models.make({ fetch, baseUrl: "https://example.com/mirror" }).providers() diff --git a/packages/sdk/test/effect.test.ts b/packages/sdk/test/effect.test.ts index 943acedbeb7..025e4a190e3 100644 --- a/packages/sdk/test/effect.test.ts +++ b/packages/sdk/test/effect.test.ts @@ -42,6 +42,20 @@ test("models() and catalog() hit their endpoints, baseUrl subpath preserved", as ]) }) +test("catalog endpoints encode model type filters", async () => { + const { requests, layer } = stub({}) + const program = Effect.gen(function* () { + const client = yield* Models.make() + yield* client.providers({ modelTypes: ["decision"] }) + yield* client.catalog({ modelTypes: "all" }) + }) + await program.pipe(Effect.provide(layer), Effect.runPromise) + expect(requests.map((request) => request.url)).toEqual([ + "https://models.dev/api.json?type=decision", + "https://models.dev/catalog.json?type=all", + ]) +}) + test("custom headers are sent", async () => { const { requests, layer } = stub({}) const program = Effect.gen(function* () { diff --git a/packages/web/script/build.ts b/packages/web/script/build.ts index f387bccd848..8866aa8e435 100755 --- a/packages/web/script/build.ts +++ b/packages/web/script/build.ts @@ -1,6 +1,11 @@ #!/usr/bin/env bun import { RenderedPages, Providers, Models, renderDocument } from "../src/render"; +import { + filterCatalogByModelType, + MODEL_TYPES, + type ModelTypeFilter, +} from "@models.dev/core"; import fs from "fs/promises"; import path from "path"; @@ -74,15 +79,18 @@ for (const [route, rendered] of RenderedPages) { await Bun.write(filePath, renderDocument(template, rendered)); } -await Bun.write("./dist/api.json", JSON.stringify(Providers)); -await Bun.write( - "./dist/catalog.json", - JSON.stringify({ models: Models, providers: Providers }), -); -await Bun.write("./dist/models.json", JSON.stringify(Models)); - -await fs.rename("./dist/api.json", "./dist/_api.json"); -await fs.rename("./dist/catalog.json", "./dist/_catalog.json"); -await fs.rename("./dist/models.json", "./dist/_models.json"); +const catalog = { models: Models, providers: Providers }; +const variants: Array<[suffix: string, filter: ModelTypeFilter]> = [ + ["", "default"], + ["-all", "all"], + ...MODEL_TYPES.map((type) => [`-${type}`, [type]] as const), +]; + +for (const [suffix, filter] of variants) { + const filtered = filterCatalogByModelType(catalog, filter); + await Bun.write(`./dist/_api${suffix}.json`, JSON.stringify(filtered.providers)); + await Bun.write(`./dist/_models${suffix}.json`, JSON.stringify(filtered.models)); + await Bun.write(`./dist/_catalog${suffix}.json`, JSON.stringify(filtered)); +} await fs.rm("./dist/index.html", { force: true }); diff --git a/packages/web/src/render.tsx b/packages/web/src/render.tsx index 93c73c6e66e..34632f57b1b 100644 --- a/packages/web/src/render.tsx +++ b/packages/web/src/render.tsx @@ -1497,21 +1497,22 @@ function HelpDialog() {

API

You can access provider data, provider-agnostic model metadata, or the - combined catalog through JSON endpoints. + combined catalog through JSON endpoints. Specialized model types are + omitted by default; the site includes all model types.

Logos

diff --git a/packages/web/src/server.ts b/packages/web/src/server.ts index 5adb9e9fc60..f43e6ce830a 100644 --- a/packages/web/src/server.ts +++ b/packages/web/src/server.ts @@ -1,5 +1,12 @@ import Index from "../index.html"; import { getRenderedPage, Models, Providers, renderDocument } from "./render"; +import { + filterCatalogByModelType, + filterModelsByModelType, + filterProvidersByModelType, + InvalidModelTypeError, + parseModelTypes, +} from "@models.dev/core"; import path from "path"; const assetPort = Number(Bun.env.ASSET_PORT ?? 16000); @@ -99,30 +106,37 @@ Bun.serve({ }, }); }, - "/api.json": () => - Response.json(Providers, { - headers: { - "Cache-Control": "public, max-age=3600", - }, - }), - "/models.json": () => - Response.json(Models, { - headers: { - "Cache-Control": "public, max-age=3600", - }, - }), - "/catalog.json": () => - Response.json( - { models: Models, providers: Providers }, - { - headers: { - "Cache-Control": "public, max-age=3600", - }, - }, - ), + "/api.json": (req) => catalogResponse(req, "api"), + "/models.json": (req) => catalogResponse(req, "models"), + "/catalog.json": (req) => catalogResponse(req, "catalog"), }, }); +function catalogResponse(req: Request, endpoint: "api" | "models" | "catalog") { + let filter; + try { + filter = parseModelTypes(new URL(req.url).searchParams.get("type")); + } catch (error) { + if (!(error instanceof InvalidModelTypeError)) throw error; + return Response.json({ error: error.message }, { status: 400 }); + } + + const value = endpoint === "api" + ? filterProvidersByModelType(Providers, filter) + : endpoint === "models" + ? filterModelsByModelType(Models, filter) + : filterCatalogByModelType( + { models: Models, providers: Providers }, + filter, + ); + + return Response.json(value, { + headers: { + "Cache-Control": "public, max-age=3600", + }, + }); +} + const server = Bun.serve({ development: true, hostname: "0.0.0.0", diff --git a/providers/aihubmix/models/claude-fable-5-1.toml b/providers/aihubmix/models/claude-fable-5-1.toml new file mode 100644 index 00000000000..b29ef102523 --- /dev/null +++ b/providers/aihubmix/models/claude-fable-5-1.toml @@ -0,0 +1,18 @@ +# Toggle: $.thinking.type = "enabled"|"disabled"|"adaptive" on the Anthropic-compatible +# /v1/messages path; $.enable_thinking = true|false on /v1/chat/completions. +# Effort: $.output_config.effort on /v1/messages; $.reasoning_effort on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message +# Pricing, context window, max output and the effort levels come from +# https://aihubmix.com/api/v1/models?type=llm (accessed 2026-09-18), which lists this +# model on both faces. Cache read is a quarter of Claude Fable 5's rate; input, output +# and cache write are unchanged from it, matching the sibling entry in this directory. +base_model = "anthropic/claude-fable-5-1" +structured_output = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +interleaved = true + +[cost] +input = 11 +output = 55 +cache_read = 0.275 +cache_write = 13.75 diff --git a/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml b/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml new file mode 100644 index 00000000000..db3b1b7c1e9 --- /dev/null +++ b/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml @@ -0,0 +1,24 @@ +# Toggle: +# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11); +# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path. +# Effort: low|high|max +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "deepseek/deepseek-v4-flash-0731" +name = "DeepSeek V4 Flash 0731 Fast" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.28 +output = 1.4 +cache_read = 0.07 diff --git a/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml b/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml new file mode 100644 index 00000000000..3cd40d48314 --- /dev/null +++ b/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml @@ -0,0 +1,23 @@ +# Toggle: +# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11); +# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path. +# Effort: low|high|max +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "deepseek/deepseek-v4-flash-vision-exp" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.155 +output = 0.62 +cache_read = 0.0031 diff --git a/providers/aihubmix/models/gemini-3.1-flash-image.toml b/providers/aihubmix/models/gemini-3.1-flash-image.toml new file mode 100644 index 00000000000..1c51818cb89 --- /dev/null +++ b/providers/aihubmix/models/gemini-3.1-flash-image.toml @@ -0,0 +1,17 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-image +# Effort: reasoning_effort = minimal|high on /v1/chat/completions. +# Gemini native: generationConfig.thinkingConfig.thinkingLevel uses the corresponding uppercase levels. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.1-flash-image" + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.5 +output = 3 diff --git a/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml b/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml new file mode 100644 index 00000000000..14141e81cc6 --- /dev/null +++ b/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml @@ -0,0 +1,18 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite-image +# Effort: reasoning_effort = minimal|high on /v1/chat/completions. +# Gemini native: generationConfig.thinkingConfig.thinkingLevel uses the corresponding uppercase levels. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.1-flash-lite-image" + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.25 +output = 1.5 +cache_read = 0.025 diff --git a/providers/aihubmix/models/gemini-3.5-flash-lite.toml b/providers/aihubmix/models/gemini-3.5-flash-lite.toml new file mode 100644 index 00000000000..19c43bfae2f --- /dev/null +++ b/providers/aihubmix/models/gemini-3.5-flash-lite.toml @@ -0,0 +1,19 @@ +# Sources (accessed 2026-09-20): +# AIHubMix catalog row model_id=gemini-3.5-flash-lite publishes +# pricing={"input":0.3,"output":2.499999,"cache_read":0.03}. +# Preserve the host's exact published price; Google's 2.50 is a different host. +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://ai.google.dev/gemini-api/docs/thinking +# Effort: reasoning_effort = minimal|low|medium|high on /v1/chat/completions. +# Gemini native: generationConfig.thinkingConfig.thinkingLevel uses the corresponding uppercase levels. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.5-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0.3 +output = 2.499999 +cache_read = 0.03 diff --git a/providers/aihubmix/models/gemini-3.6-flash.toml b/providers/aihubmix/models/gemini-3.6-flash.toml new file mode 100644 index 00000000000..d9c9722442d --- /dev/null +++ b/providers/aihubmix/models/gemini-3.6-flash.toml @@ -0,0 +1,14 @@ +# Effort: minimal|low|medium|high +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.6-flash" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 1.5 +output = 7.5 +cache_read = 0.15 diff --git a/providers/aihubmix/models/gemini-3.8-flash.toml b/providers/aihubmix/models/gemini-3.8-flash.toml new file mode 100644 index 00000000000..9657ff10785 --- /dev/null +++ b/providers/aihubmix/models/gemini-3.8-flash.toml @@ -0,0 +1,14 @@ +# Effort: low|medium|high +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.8-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 diff --git a/providers/aihubmix/models/gpt-6-astra.toml b/providers/aihubmix/models/gpt-6-astra.toml new file mode 100644 index 00000000000..b005bff58b8 --- /dev/null +++ b/providers/aihubmix/models/gpt-6-astra.toml @@ -0,0 +1,22 @@ +# Effort: low|medium|high|xhigh|max +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "openai/gpt-6-astra" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 20 +output = 75 +cache_read = 2 +cache_write = 25 diff --git a/providers/aihubmix/models/hy3.toml b/providers/aihubmix/models/hy3.toml new file mode 100644 index 00000000000..a5f55cbba88 --- /dev/null +++ b/providers/aihubmix/models/hy3.toml @@ -0,0 +1,20 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://huggingface.co/tencent/Hy3 +# Effort: reasoning_effort = none|low|high on /v1/chat/completions. +# Off is effort=none; AIHubMix no_think is normalized to none, so no separate toggle. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "tencent/hy3" +structured_output = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high"] + +[cost] +input = 0.1562 +output = 0.6248 +cache_read = 0.03905 diff --git a/providers/aihubmix/models/hy4-preview.toml b/providers/aihubmix/models/hy4-preview.toml new file mode 100644 index 00000000000..dc0f2ad6ac7 --- /dev/null +++ b/providers/aihubmix/models/hy4-preview.toml @@ -0,0 +1,28 @@ +# Off is effort=none; graded levels — no toggle. The same off elsewhere: +# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11); +# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path. +# Effort: none|high +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# Budget: +# integer $.reasoning.max_tokens on /v1/chat/completions; $.thinking.budget_tokens >= 1024 on /v1/messages; +# integer $.generationConfig.thinkingConfig.thinkingBudget on the Gemini path (-1 dynamic, 0 off where supported); +# the Responses path carries effort but has no reasoning-token budget field. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "tencent/hy4-preview" +structured_output = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.845 +output = 2.535 +cache_read = 0.04225 diff --git a/providers/aihubmix/models/longcat-2.0.toml b/providers/aihubmix/models/longcat-2.0.toml new file mode 100644 index 00000000000..4f0b9ae311d --- /dev/null +++ b/providers/aihubmix/models/longcat-2.0.toml @@ -0,0 +1,17 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://github.com/meituan-longcat/LongCat-2.0 +# Toggle: enable_thinking = true|false on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "meituan/longcat-2.0" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.7746 +output = 3.0984 +cache_read = 0.015492 diff --git a/providers/aihubmix/models/minimax-m3.toml b/providers/aihubmix/models/minimax-m3.toml new file mode 100644 index 00000000000..2103f7586f0 --- /dev/null +++ b/providers/aihubmix/models/minimax-m3.toml @@ -0,0 +1,17 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://platform.minimaxi.com/docs/api-reference/text-chat-openai.md +# Toggle: enable_thinking = true|false on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "minimax/MiniMax-M3" +structured_output = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.288 +output = 1.152 diff --git a/providers/aihubmix/models/muse-spark-1.1.toml b/providers/aihubmix/models/muse-spark-1.1.toml new file mode 100644 index 00000000000..f0c00f709bf --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.1.toml @@ -0,0 +1,20 @@ +# Sources (accessed 2026-09-20): +# AIHubMix catalog row model_id=muse-spark-1.1 declares +# input_modalities="text,image,video,audio,pdf". Audio is a host-catalog override +# relative to the lab entry; this is a catalog declaration, not an inference test. +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://dev.meta.ai/docs/models/ +# Effort: reasoning_effort = minimal|low|medium|high|xhigh on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "meta/muse-spark-1.1" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[cost] +input = 1.375 +output = 4.675 + +[modalities] +input = ["text", "image", "video", "audio", "pdf"] diff --git a/providers/aihubmix/models/muse-spark-1.2.toml b/providers/aihubmix/models/muse-spark-1.2.toml new file mode 100644 index 00000000000..01d1fa8c78a --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.2.toml @@ -0,0 +1,14 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://dev.meta.ai/docs/overview/ +# Effort: reasoning_effort = minimal|low|medium|high|xhigh on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "meta/muse-spark-1.2" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[cost] +input = 1.375 +output = 4.675 diff --git a/providers/aihubmix/models/muse-spark-1.3.toml b/providers/aihubmix/models/muse-spark-1.3.toml new file mode 100644 index 00000000000..21edfd6c678 --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.3.toml @@ -0,0 +1,14 @@ +# Effort: minimal|low|medium|high|xhigh +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "meta/muse-spark-1.3" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[cost] +input = 1.375 +output = 4.675 +cache_read = 0.165 diff --git a/providers/aihubmix/models/ox-alpha.toml b/providers/aihubmix/models/ox-alpha.toml new file mode 100644 index 00000000000..800e20046f6 --- /dev/null +++ b/providers/aihubmix/models/ox-alpha.toml @@ -0,0 +1,23 @@ +# Toggle: +# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11); +# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path. +# Effort: low|high|max +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "zhipuai/glm-5.3-flash" +name = "Ox Alpha" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/qwen3.8-flash.toml b/providers/aihubmix/models/qwen3.8-flash.toml new file mode 100644 index 00000000000..69e8766006b --- /dev/null +++ b/providers/aihubmix/models/qwen3.8-flash.toml @@ -0,0 +1,31 @@ +# Toggle: +# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11); +# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path. +# Effort: low|medium|xhigh +# $.reasoning_effort on /v1/chat/completions (alias $.reasoning.effort, which is also the Responses field); +# $.output_config.effort on /v1/messages, subject to model support. +# Budget: +# integer $.reasoning.max_tokens on /v1/chat/completions; $.thinking.budget_tokens >= 1024 on /v1/messages; +# integer $.generationConfig.thinkingConfig.thinkingBudget on the Gemini path (-1 dynamic, 0 off where supported); +# the Responses path carries effort but has no reasoning-token budget field. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "alibaba/qwen3.8-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1126 +output = 0.380025 +cache_read = 0.014075 +cache_write = 0.175937 diff --git a/providers/aihubmix/models/qwen3.8-omni-flash.toml b/providers/aihubmix/models/qwen3.8-omni-flash.toml new file mode 100644 index 00000000000..8dea0f811e5 --- /dev/null +++ b/providers/aihubmix/models/qwen3.8-omni-flash.toml @@ -0,0 +1,23 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://www.alibabacloud.com/help/en/model-studio/qwen-omni +# Toggle: enable_thinking = true|false on /v1/chat/completions. +# Effort: reasoning_effort = low|medium|xhigh on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "alibaba/qwen3.8-omni-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 0.1126 +output = 0.380025 +cache_read = 0.014075 +cache_write = 0.175937 diff --git a/providers/aihubmix/models/step-3.7-flash.toml b/providers/aihubmix/models/step-3.7-flash.toml new file mode 100644 index 00000000000..d6dcca8c632 --- /dev/null +++ b/providers/aihubmix/models/step-3.7-flash.toml @@ -0,0 +1,18 @@ +# Sources (accessed 2026-09-20): +# https://aihubmix.com/api/v1/models?type=llm (USD per million tokens) +# https://platform.stepfun.com/docs/zh/guides/models/step-3.7-flash +# Effort: reasoning_effort = low|medium|high on /v1/chat/completions. +# https://docs.aihubmix.com/cn/api/unified-inference +base_model = "stepfun/step-3.7-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 0.22 +output = 1.32 +cache_read = 0.044 diff --git a/providers/alibaba-cn/models/deepseek-v4.1-flash.toml b/providers/alibaba-cn/models/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..6fe0ab81592 --- /dev/null +++ b/providers/alibaba-cn/models/deepseek-v4.1-flash.toml @@ -0,0 +1,27 @@ +# CNY→USD rate: 6.721845, 2026-09-18, source: https://open.er-api.com/v6/latest/USD +# Sources (accessed 2026-09-18): +# https://bailian.console.aliyun.com/cn-beijing?tab=model#/model-market/detail/deepseek-v4.1-flash?serviceSite=asia-pacific-china +# https://help.aliyun.com/zh/model-studio/deepseek-api — capability table, +# reasoning_effort ladder, max_tokens default (393_216, shared with thinking) +# https://help.aliyun.com/zh/model-studio/billing-for-model-studio — Beijing +# peak/off-peak pricing: CNY 2/1 input, 8/4 output per 1M tokens with context +# cache discount; peak price converted to USD using the rate above (off-peak +# is half). +# Hybrid reasoning: enable_thinking = true|false toggles thinking; +# reasoning_effort = low|high|max (default high) — low is only supported on +# v4.1-flash / v4-flash-0731 / v4-pro-0813, and v4.1-flash is the current +# low-tier bearer. reasoning_content streams in deltas. Responses API +# supported (Beijing and Singapore only). Limits and modalities are identical +# to the deepseek lab entry (1M context, 384_000 output, text+image input), +# so they are inherited and not restated. +base_model = "deepseek/deepseek-v4.1-flash" + +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "high", "max"] }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.29754 +output = 1.19015 +cache_read = 0.01488 diff --git a/providers/alibaba-cn/models/qwen3-max.toml b/providers/alibaba-cn/models/qwen3-max.toml index ffb4eb4582b..6f9229234b8 100644 --- a/providers/alibaba-cn/models/qwen3-max.toml +++ b/providers/alibaba-cn/models/qwen3-max.toml @@ -1,23 +1,49 @@ +# Sources (accessed 2026-09-22): +# https://help.aliyun.com/en/model-studio/model-pricing (China (Beijing) USD list prices) +# https://help.aliyun.com/en/model-studio/model-qwen3-max (alias, capabilities, and limits) +# https://help.aliyun.com/en/model-studio/deep-thinking +# qwen3-max currently aliases qwen3-max-2026-01-23. +# Toggle: enable_thinking true|false +# Budget: thinking_budget (integer reasoning tokens, max 81,920) name = "Qwen3 Max" description = "Flagship Qwen model for complex reasoning, coding, and agentic workflows" family = "qwen" -release_date = "2025-09-23" -last_updated = "2026-09-11" +release_date = "2026-01-23" +last_updated = "2026-09-22" attachment = false -reasoning = false +reasoning = true +structured_output = true temperature = true knowledge = "2025-04" tool_call = true open_weights = false +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" +max = 81_920 + +[interleaved] +field = "reasoning_content" + [cost] -input = 1.291 -output = 7.749 +input = 0.359 +output = 1.434 +reasoning = 1.434 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.574 +output = 2.294 +reasoning = 2.294 [[cost.tiers]] tier = { type = "context", size = 128_000 } -input = 2.153 -output = 12.915 +input = 1.004 +output = 4.014 +reasoning = 4.014 [limit] context = 262_144 diff --git a/providers/alibaba-cn/models/qwen3.5-flash.toml b/providers/alibaba-cn/models/qwen3.5-flash.toml index 51651ed2e75..4e8481bd07f 100644 --- a/providers/alibaba-cn/models/qwen3.5-flash.toml +++ b/providers/alibaba-cn/models/qwen3.5-flash.toml @@ -1,8 +1,14 @@ +# Sources (accessed 2026-09-22): +# https://help.aliyun.com/en/model-studio/model-pricing (China (Beijing) USD list prices) +# https://help.aliyun.com/en/model-studio/qwen3-5-flash (capabilities and limits) +# https://help.aliyun.com/en/model-studio/deep-thinking +# Toggle: enable_thinking true|false +# Budget: thinking_budget (integer reasoning tokens, max 81,920) name = "Qwen3.5 Flash" description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" family = "qwen" release_date = "2026-02-23" -last_updated = "2026-09-11" +last_updated = "2026-09-22" attachment = true reasoning = true structured_output = true @@ -18,16 +24,25 @@ type = "toggle" type = "budget_tokens" max = 81_920 +[interleaved] +field = "reasoning_content" + [cost] -input = 0.172 -output = 1.033 -reasoning = 1.033 +input = 0.029 +output = 0.287 +reasoning = 0.287 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.115 +output = 1.147 +reasoning = 1.147 [[cost.tiers]] tier = { type = "context", size = 256_000 } -input = 0.689 -output = 4.133 -reasoning = 4.133 +input = 0.172 +output = 1.72 +reasoning = 1.72 [limit] context = 1_000_000 diff --git a/providers/alibaba-cn/models/qwen3.5-plus.toml b/providers/alibaba-cn/models/qwen3.5-plus.toml index 56105ed505b..429ae670c66 100644 --- a/providers/alibaba-cn/models/qwen3.5-plus.toml +++ b/providers/alibaba-cn/models/qwen3.5-plus.toml @@ -1,10 +1,17 @@ +# Sources (accessed 2026-09-22): +# https://help.aliyun.com/en/model-studio/model-pricing (China (Beijing) USD list prices) +# https://help.aliyun.com/en/model-studio/qwen3-5-plus (capabilities and limits) +# https://help.aliyun.com/en/model-studio/deep-thinking +# Toggle: enable_thinking true|false +# Budget: thinking_budget (integer reasoning tokens, max 81,920) name = "Qwen3.5 Plus" description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" family = "qwen" release_date = "2026-02-16" -last_updated = "2026-09-11" -attachment = false +last_updated = "2026-09-22" +attachment = true reasoning = true +structured_output = true temperature = true knowledge = "2025-04" tool_call = true @@ -17,16 +24,25 @@ type = "toggle" type = "budget_tokens" max = 81_920 +[interleaved] +field = "reasoning_content" + [cost] +input = 0.115 +output = 0.688 +reasoning = 0.688 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } input = 0.287 -output = 1.722 -reasoning = 1.722 +output = 1.72 +reasoning = 1.72 [[cost.tiers]] tier = { type = "context", size = 256_000 } -input = 1.148 -output = 6.888 -reasoning = 6.888 +input = 0.573 +output = 3.44 +reasoning = 3.44 [limit] context = 1_000_000 diff --git a/providers/amazon-bedrock/models/anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..a9684824bf1 --- /dev/null +++ b/providers/amazon-bedrock/models/anthropic.claude-opus-5-5.toml @@ -0,0 +1,11 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/amazon-bedrock/models/au.anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/au.anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..25e28fdb4b6 --- /dev/null +++ b/providers/amazon-bedrock/models/au.anthropic.claude-opus-5-5.toml @@ -0,0 +1,13 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Bedrock regional endpoints carry a 10% pricing premium over global endpoints. +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +name = "Claude Opus 5.5 (AU)" + +[cost] +input = 4.4 +output = 22 +cache_read = 0.22 +cache_write = 5.5 diff --git a/providers/amazon-bedrock/models/eu.anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/eu.anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..8c9d8b97cdd --- /dev/null +++ b/providers/amazon-bedrock/models/eu.anthropic.claude-opus-5-5.toml @@ -0,0 +1,13 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Bedrock regional endpoints carry a 10% pricing premium over global endpoints. +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +name = "Claude Opus 5.5 (EU)" + +[cost] +input = 4.4 +output = 22 +cache_read = 0.22 +cache_write = 5.5 diff --git a/providers/amazon-bedrock/models/global.anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/global.anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..c36a3b86f74 --- /dev/null +++ b/providers/amazon-bedrock/models/global.anthropic.claude-opus-5-5.toml @@ -0,0 +1,12 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +name = "Claude Opus 5.5 (Global)" + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/amazon-bedrock/models/global.moonshotai.kimi-k3.toml b/providers/amazon-bedrock/models/global.moonshotai.kimi-k3.toml new file mode 100644 index 00000000000..05556e9c377 --- /dev/null +++ b/providers/amazon-bedrock/models/global.moonshotai.kimi-k3.toml @@ -0,0 +1,22 @@ +# Sources: https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-moonshot-ai-kimi-k3.html +# https://docs.aws.amazon.com/bedrock/latest/userguide/conversation-inference.html +# Global CRIS standard pricing; cache_write is the 30-minute rate. +# Reasoning is always on; no Bedrock Converse effort control is documented or verified. +# Live Converse 2026-09-23: reasoning_effort accepted invalid strings, numbers, +# and objects exactly like an unrelated unknown field; low/max probes showed no +# verified effect. The bare model ID was rejected; an inference profile is required. +# Global and US profiles returned reasoningContent. +# Bedrock supports text and image input, but not the base model's video input. +base_model = "moonshotai/kimi-k3" +name = "Kimi K3 (Global)" +reasoning_options = [] +interleaved = true + +[cost] +input = 3.00 +output = 15.00 +cache_read = 0.30 +cache_write = 3.75 + +[modalities] +input = ["text", "image"] diff --git a/providers/amazon-bedrock/models/global.openai.gpt-6-luna.toml b/providers/amazon-bedrock/models/global.openai.gpt-6-luna.toml new file mode 100644 index 00000000000..d541b9a308b --- /dev/null +++ b/providers/amazon-bedrock/models/global.openai.gpt-6-luna.toml @@ -0,0 +1,24 @@ +# Global cross-Region inference profile — served by Bedrock core (Converse), not Mantle, so no [provider] override. +# Profile ID verified via ListInferenceProfiles (ACTIVE) on 2026-09-22; AWS has not published a GPT-6 Luna model card yet. +# Global CRIS pricing = OpenAI API list rates: https://developers.openai.com/api/docs/models/gpt-6-luna +# Converse effort: additionalModelRequestFields.reasoning.effort = none|low|medium|high|xhigh|max (Bedrock's own validation error lists these). +# Live Converse check 2026-09-22 (eu-west-1, global profile): tool call + tool-result round trip; effort low/xhigh/max accepted. +base_model = "openai/gpt-6-luna" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +name = "GPT-6 Luna (Global)" + +[cost] +input = 0.10 +output = 0.50 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { size = 272_000 } +input = 0.20 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[modalities] +input = ["text", "image"] diff --git a/providers/amazon-bedrock/models/global.openai.gpt-6-sol.toml b/providers/amazon-bedrock/models/global.openai.gpt-6-sol.toml new file mode 100644 index 00000000000..fee63c0e68e --- /dev/null +++ b/providers/amazon-bedrock/models/global.openai.gpt-6-sol.toml @@ -0,0 +1,24 @@ +# Global cross-Region inference profile — served by Bedrock core (Converse), not Mantle, so no [provider] override. +# Profile ID verified via ListInferenceProfiles (ACTIVE) on 2026-09-22; AWS has not published a GPT-6 Sol model card yet. +# Global CRIS pricing = OpenAI API list rates: https://developers.openai.com/api/docs/models/gpt-6-sol +# Converse effort: additionalModelRequestFields.reasoning.effort = none|low|medium|high|xhigh|max (Bedrock's validation error on the Luna twin lists these; matches OpenAI-native Sol). +# Live Converse check 2026-09-22 (eu-west-1, global profile): tool call + tool-result round trip; effort low/xhigh/max accepted. +base_model = "openai/gpt-6-sol" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +name = "GPT-6 Sol (Global)" + +[cost] +input = 2.00 +output = 10.00 +cache_read = 0.20 +cache_write = 2.50 + +[[cost.tiers]] +tier = { size = 272_000 } +input = 4.00 +output = 15.00 +cache_read = 0.40 +cache_write = 5.00 + +[modalities] +input = ["text", "image"] diff --git a/providers/amazon-bedrock/models/jp.anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/jp.anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..78ad0c8319e --- /dev/null +++ b/providers/amazon-bedrock/models/jp.anthropic.claude-opus-5-5.toml @@ -0,0 +1,13 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Bedrock regional endpoints carry a 10% pricing premium over global endpoints. +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +name = "Claude Opus 5.5 (JP)" + +[cost] +input = 4.4 +output = 22 +cache_read = 0.22 +cache_write = 5.5 diff --git a/providers/amazon-bedrock/models/us.anthropic.claude-opus-5-5.toml b/providers/amazon-bedrock/models/us.anthropic.claude-opus-5-5.toml new file mode 100644 index 00000000000..f74ececd884 --- /dev/null +++ b/providers/amazon-bedrock/models/us.anthropic.claude-opus-5-5.toml @@ -0,0 +1,13 @@ +# Sources: https://platform.claude.com/docs/en/build-with-claude/claude-in-amazon-bedrock +# https://platform.claude.com/docs/en/about-claude/pricing +# Bedrock regional endpoints carry a 10% pricing premium over global endpoints. +# Effort: additionalModelRequestFields.output_config.effort (Converse); output_config.effort (InvokeModel). +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +name = "Claude Opus 5.5 (US)" + +[cost] +input = 4.4 +output = 22 +cache_read = 0.22 +cache_write = 5.5 diff --git a/providers/amazon-bedrock/models/us.moonshotai.kimi-k3.toml b/providers/amazon-bedrock/models/us.moonshotai.kimi-k3.toml new file mode 100644 index 00000000000..9c724cee821 --- /dev/null +++ b/providers/amazon-bedrock/models/us.moonshotai.kimi-k3.toml @@ -0,0 +1,22 @@ +# Sources: https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-moonshot-ai-kimi-k3.html +# https://docs.aws.amazon.com/bedrock/latest/userguide/conversation-inference.html +# US CRIS standard pricing; cache_write is the 30-minute rate. +# Reasoning is always on; no Bedrock Converse effort control is documented or verified. +# Live Converse 2026-09-23: reasoning_effort accepted invalid strings, numbers, +# and objects exactly like an unrelated unknown field; low/max probes showed no +# verified effect. The bare model ID was rejected; an inference profile is required. +# Global and US profiles returned reasoningContent. +# Bedrock supports text and image input, but not the base model's video input. +base_model = "moonshotai/kimi-k3" +name = "Kimi K3 (US)" +reasoning_options = [] +interleaved = true + +[cost] +input = 3.30 +output = 16.50 +cache_read = 0.33 +cache_write = 4.125 + +[modalities] +input = ["text", "image"] diff --git a/providers/amazon-bedrock/models/us.openai.gpt-6-luna.toml b/providers/amazon-bedrock/models/us.openai.gpt-6-luna.toml new file mode 100644 index 00000000000..0d59417b8e6 --- /dev/null +++ b/providers/amazon-bedrock/models/us.openai.gpt-6-luna.toml @@ -0,0 +1,24 @@ +# US cross-Region inference profile — served by Bedrock core (Converse), not Mantle, so no [provider] override. +# Profile ID verified via ListInferenceProfiles (ACTIVE) on 2026-09-22; AWS has not published a GPT-6 Luna model card yet. +# Geo CRIS pricing = OpenAI API list rates + AWS's 10% fee (as stated on the GPT-6 Astra card): https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html +# Converse effort: additionalModelRequestFields.reasoning.effort = none|low|medium|high|xhigh|max (Bedrock's own validation error lists these). +# Live Converse check 2026-09-22 (eu-west-1, global profile): tool call + tool-result round trip; effort low/xhigh/max accepted. +base_model = "openai/gpt-6-luna" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +name = "GPT-6 Luna (US)" + +[cost] +input = 0.11 +output = 0.55 +cache_read = 0.011 +cache_write = 0.1375 + +[[cost.tiers]] +tier = { size = 272_000 } +input = 0.22 +output = 0.825 +cache_read = 0.022 +cache_write = 0.275 + +[modalities] +input = ["text", "image"] diff --git a/providers/amazon-bedrock/models/us.openai.gpt-6-sol.toml b/providers/amazon-bedrock/models/us.openai.gpt-6-sol.toml new file mode 100644 index 00000000000..9cb9cff8731 --- /dev/null +++ b/providers/amazon-bedrock/models/us.openai.gpt-6-sol.toml @@ -0,0 +1,24 @@ +# US cross-Region inference profile — served by Bedrock core (Converse), not Mantle, so no [provider] override. +# Profile ID verified via ListInferenceProfiles (ACTIVE) on 2026-09-22; AWS has not published a GPT-6 Sol model card yet. +# Geo CRIS pricing = OpenAI API list rates + AWS's 10% fee (as stated on the GPT-6 Astra card): https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html +# Converse effort: additionalModelRequestFields.reasoning.effort = none|low|medium|high|xhigh|max (Bedrock's validation error on the Luna twin lists these; matches OpenAI-native Sol). +# Live Converse check 2026-09-22 (eu-west-1, global profile): tool call + tool-result round trip; effort low/xhigh/max accepted. +base_model = "openai/gpt-6-sol" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +name = "GPT-6 Sol (US)" + +[cost] +input = 2.20 +output = 11.00 +cache_read = 0.22 +cache_write = 2.75 + +[[cost.tiers]] +tier = { size = 272_000 } +input = 4.40 +output = 16.50 +cache_read = 0.44 +cache_write = 5.50 + +[modalities] +input = ["text", "image"] diff --git a/providers/anthropic/models/claude-opus-5-5.toml b/providers/anthropic/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..7341878caeb --- /dev/null +++ b/providers/anthropic/models/claude-opus-5-5.toml @@ -0,0 +1,13 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[experimental.modes.fast] +cost = { input = 8, output = 40, cache_read = 0.4, cache_write = 10 } +provider = { body = { speed = "fast" }, headers = { anthropic-beta = "fast-mode-2026-02-01" } } diff --git a/providers/azure-cognitive-services/models/claude-opus-5-5.toml b/providers/azure-cognitive-services/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..c4fedc9752f --- /dev/null +++ b/providers/azure-cognitive-services/models/claude-opus-5-5.toml @@ -0,0 +1,13 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[provider] +npm = "@ai-sdk/anthropic" +api = "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" diff --git a/providers/azure/models/claude-opus-5-5.toml b/providers/azure/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..cc062a28b81 --- /dev/null +++ b/providers/azure/models/claude-opus-5-5.toml @@ -0,0 +1,13 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[provider] +npm = "@ai-sdk/anthropic" +api = "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" diff --git a/providers/azure/models/gpt-6-luna.toml b/providers/azure/models/gpt-6-luna.toml new file mode 100644 index 00000000000..c105b8ef975 --- /dev/null +++ b/providers/azure/models/gpt-6-luna.toml @@ -0,0 +1,17 @@ +# Availability: https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure-region-availability +# Rates: https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/ +base_model = "openai/gpt-6-luna" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.10 +output = 0.50 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.20 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 diff --git a/providers/azure/models/gpt-6-sol.toml b/providers/azure/models/gpt-6-sol.toml new file mode 100644 index 00000000000..e0ad0d723d0 --- /dev/null +++ b/providers/azure/models/gpt-6-sol.toml @@ -0,0 +1,17 @@ +# Availability: https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure-region-availability +# Rates: https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/ +base_model = "openai/gpt-6-sol" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 2.00 +output = 10.00 +cache_read = 0.20 +cache_write = 2.50 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4.00 +output = 15.00 +cache_read = 0.40 +cache_write = 5.00 diff --git a/providers/cline-pass/models/cline-pass/deepseek-v4-flash.toml b/providers/cline-pass/models/cline-pass/deepseek-v4-flash.toml index a402b4d533b..044e118d9af 100644 --- a/providers/cline-pass/models/cline-pass/deepseek-v4-flash.toml +++ b/providers/cline-pass/models/cline-pass/deepseek-v4-flash.toml @@ -1,5 +1,9 @@ +# https://docs.cline.bot/getting-started/clinepass +# https://api-docs.deepseek.com/guides/thinking_mode/ (OpenAI: thinking.type + reasoning_effort low|high|max) +# Effort: reasoning_effort = none|low|high|max (none disables thinking; DeepSeek Flash graded set) +# Legacy Flash slug; lab serves V4.1-Flash. Same reasoning surface as deepseek-v4.1-flash. base_model = "deepseek/deepseek-v4-flash" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] [cost] input = 0.14 diff --git a/providers/cline-pass/models/cline-pass/deepseek-v4-pro.toml b/providers/cline-pass/models/cline-pass/deepseek-v4-pro.toml index 3fb0bd31762..191909977a6 100644 --- a/providers/cline-pass/models/cline-pass/deepseek-v4-pro.toml +++ b/providers/cline-pass/models/cline-pass/deepseek-v4-pro.toml @@ -1,5 +1,9 @@ +# https://docs.cline.bot/getting-started/clinepass +# https://api-docs.deepseek.com/guides/thinking_mode/ (OpenAI: thinking.type + reasoning_effort) +# First-party DeepSeek Pro: effort high|max (lab providers/deepseek/models/deepseek-v4-pro.toml) +# Effort: reasoning_effort = none|high|max (none disables thinking) base_model = "deepseek/deepseek-v4-pro" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "effort", values = ["none", "high", "max"] }] [cost] input = 1.74 diff --git a/providers/cline-pass/models/cline-pass/deepseek-v4.1-flash.toml b/providers/cline-pass/models/cline-pass/deepseek-v4.1-flash.toml index cdc87207c83..453a58da360 100644 --- a/providers/cline-pass/models/cline-pass/deepseek-v4.1-flash.toml +++ b/providers/cline-pass/models/cline-pass/deepseek-v4.1-flash.toml @@ -1,9 +1,11 @@ # https://api.cline.bot/api/v1/ai/cline/recommended-models (clinePass id cline-pass/deepseek-v4.1-flash, accessed 2026-09-11) # https://docs.cline.bot/getting-started/clinepass (ClinePass reference pricing cites DeepSeek API pricing for Flash quota) # https://api-docs.deepseek.com/quick_start/pricing (DeepSeek-V4.1-Flash off-peak USD/MTok; ClinePass docs table still lists deepseek-v4-flash only) -# Reasoning effort values match other ClinePass DeepSeek entries on this host. +# https://api-docs.deepseek.com/guides/thinking_mode/ (OpenAI: thinking.type + reasoning_effort low|high|max; medium/xhigh map to high) +# Effort: reasoning_effort = none|low|high|max (none disables thinking; matches DeepSeek Flash graded set + off) +# ClinePass is an OpenAI-compatible relay; do not advertise GPT-style medium/xhigh. base_model = "deepseek/deepseek-v4.1-flash" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] [cost] input = 0.15 diff --git a/providers/cline-pass/models/cline-pass/glm-5.2.toml b/providers/cline-pass/models/cline-pass/glm-5.2.toml index b3cac6301a6..7458eed5caa 100644 --- a/providers/cline-pass/models/cline-pass/glm-5.2.toml +++ b/providers/cline-pass/models/cline-pass/glm-5.2.toml @@ -1,5 +1,9 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Zhipu GLM-5.2: effective effort high|max; none|minimal skip thinking +# (providers/zhipuai/models/glm-5.2.toml; https://docs.bigmodel.cn/cn/guide/capabilities/thinking) +# Effort: reasoning_effort = none|high|max base_model = "zhipuai/glm-5.2" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "effort", values = ["none", "high", "max"] }] [cost] input = 1.4 diff --git a/providers/cline-pass/models/cline-pass/kimi-k2.6.toml b/providers/cline-pass/models/cline-pass/kimi-k2.6.toml index 373fb552c69..fc3fb04b0db 100644 --- a/providers/cline-pass/models/cline-pass/kimi-k2.6.toml +++ b/providers/cline-pass/models/cline-pass/kimi-k2.6.toml @@ -1,5 +1,8 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Moonshot K2.6: thinking on/off only (providers/moonshotai/models/kimi-k2.6.toml) +# Toggle: thinking / reasoning enabled|disabled (OpenAI-compatible relay; no graded effort on lab) base_model = "moonshotai/kimi-k2.6" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 0.95 diff --git a/providers/cline-pass/models/cline-pass/kimi-k2.7-code.toml b/providers/cline-pass/models/cline-pass/kimi-k2.7-code.toml index 7340bf79c7a..9fc9a2f9a5a 100644 --- a/providers/cline-pass/models/cline-pass/kimi-k2.7-code.toml +++ b/providers/cline-pass/models/cline-pass/kimi-k2.7-code.toml @@ -1,5 +1,8 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Moonshot K2.7 Code: always-on reasoning, no caller control +# (providers/moonshotai/models/kimi-k2.7-code.toml; openrouter peer also []) base_model = "moonshotai/kimi-k2.7-code" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [] [cost] input = 0.95 diff --git a/providers/cline-pass/models/cline-pass/kimi-k3.toml b/providers/cline-pass/models/cline-pass/kimi-k3.toml index 856a9963ae3..2b9dee1d87c 100644 --- a/providers/cline-pass/models/cline-pass/kimi-k3.toml +++ b/providers/cline-pass/models/cline-pass/kimi-k3.toml @@ -1,7 +1,9 @@ # https://docs.cline.bot/getting-started/clinepass (accessed 2026-07-22) # Model ID cline-pass/kimi-k3; reference pricing $3 / $15 / $0.30 per 1M tokens (input/output/cache read). +# Lab Moonshot K3: thinking.type + output_config.effort low|high|max (providers/moonshotai/models/kimi-k3.toml) +# Effort: reasoning_effort = none|low|high|max (none disables; no medium/xhigh) base_model = "moonshotai/kimi-k3" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] [cost] input = 3 diff --git a/providers/cline-pass/models/cline-pass/mimo-v2.5-pro.toml b/providers/cline-pass/models/cline-pass/mimo-v2.5-pro.toml index 4b15e595d0b..8ff3d62b6dc 100644 --- a/providers/cline-pass/models/cline-pass/mimo-v2.5-pro.toml +++ b/providers/cline-pass/models/cline-pass/mimo-v2.5-pro.toml @@ -1,5 +1,8 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Xiaomi MiMo-V2.5-Pro: thinking on/off only (providers/xiaomi/models/mimo-v2.5-pro.toml) +# Toggle: thinking / reasoning enabled|disabled (no graded effort on lab) base_model = "xiaomi/mimo-v2.5-pro" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 1.74 diff --git a/providers/cline-pass/models/cline-pass/mimo-v2.5.toml b/providers/cline-pass/models/cline-pass/mimo-v2.5.toml index a05c72ca039..3a9f90ed0bc 100644 --- a/providers/cline-pass/models/cline-pass/mimo-v2.5.toml +++ b/providers/cline-pass/models/cline-pass/mimo-v2.5.toml @@ -1,5 +1,8 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Xiaomi MiMo-V2.5: thinking on/off only (providers/xiaomi/models/mimo-v2.5.toml) +# Toggle: thinking / reasoning enabled|disabled (no graded effort on lab) base_model = "xiaomi/mimo-v2.5" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 0.14 diff --git a/providers/cline-pass/models/cline-pass/mimo-v2.6-flash.toml b/providers/cline-pass/models/cline-pass/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..612d086a7f4 --- /dev/null +++ b/providers/cline-pass/models/cline-pass/mimo-v2.6-flash.toml @@ -0,0 +1,16 @@ +# Model ID cline-pass/mimo-v2.6-flash; reference pricing from the Cline model +# catalog https://api.cline.bot/api/v1/ai/cline/models (accessed 2026-09-22): +# $0.14 / $0.28 / $0.0028 per 1M tokens (input/output/cache read). +# Wire evidence (2026-09-22): POST /api/v1/chat/completions accepts +# reasoning_effort = none|low|medium|high|xhigh|minimal|max (HTTP 200) and +# rejects unknown levels (HTTP 500). An effect probe (3 trials per level, +# reasoning_tokens per response) shows no measurable difference for +# minimal/max, so this file declares the established same-host peer set +# used by cline-pass/mimo-v2.5* (none, low, medium, high, xhigh). +base_model = "xiaomi/mimo-v2.6-flash" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 diff --git a/providers/cline-pass/models/cline-pass/mimo-v2.6-pro.toml b/providers/cline-pass/models/cline-pass/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..d42aa0606d7 --- /dev/null +++ b/providers/cline-pass/models/cline-pass/mimo-v2.6-pro.toml @@ -0,0 +1,16 @@ +# Model ID cline-pass/mimo-v2.6-pro; reference pricing from the Cline model +# catalog https://api.cline.bot/api/v1/ai/cline/models (accessed 2026-09-22): +# $0.435 / $0.87 / $0.0036 per 1M tokens (input/output/cache read). +# Wire evidence (2026-09-22): POST /api/v1/chat/completions accepts +# reasoning_effort = none|low|medium|high|xhigh|minimal|max (HTTP 200) and +# rejects unknown levels (HTTP 500). An effect probe (reasoning_tokens per +# response) shows no measurable difference for minimal/max, so this file +# declares the established same-host peer set used by cline-pass/mimo-v2.5-pro +# (none, low, medium, high, xhigh). +base_model = "xiaomi/mimo-v2.6-pro" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 diff --git a/providers/cline-pass/models/cline-pass/minimax-m3.toml b/providers/cline-pass/models/cline-pass/minimax-m3.toml index 1367e07bc0e..599f9fc9ec1 100644 --- a/providers/cline-pass/models/cline-pass/minimax-m3.toml +++ b/providers/cline-pass/models/cline-pass/minimax-m3.toml @@ -1,5 +1,8 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab MiniMax-M3: thinking on/off only (providers/minimax/models/MiniMax-M3.toml) +# Toggle: thinking / reasoning enabled|disabled (no graded effort on lab) base_model = "minimax/MiniMax-M3" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 0.3 diff --git a/providers/cline-pass/models/cline-pass/muse-spark-1.3-contributor.toml b/providers/cline-pass/models/cline-pass/muse-spark-1.3-contributor.toml new file mode 100644 index 00000000000..a5f354380fd --- /dev/null +++ b/providers/cline-pass/models/cline-pass/muse-spark-1.3-contributor.toml @@ -0,0 +1,8 @@ +# Model ID cline-pass/muse-spark-1.3-contributor. Cline publishes no +# ClinePass reference rate for this model: docs.cline.bot/getting-started/ +# clinepass has no row for it (accessed 2026-09-22) and the model is absent +# from api.cline.bot/api/v1/ai/cline/models, so no [cost] block is declared. +base_model = "meta/muse-spark-1.3" +name = "Muse Spark 1.3 Contributor" + +reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high", "xhigh"] }] diff --git a/providers/cline-pass/models/cline-pass/qwen3.7-max.toml b/providers/cline-pass/models/cline-pass/qwen3.7-max.toml index 7ec1495bb20..302e667ede8 100644 --- a/providers/cline-pass/models/cline-pass/qwen3.7-max.toml +++ b/providers/cline-pass/models/cline-pass/qwen3.7-max.toml @@ -1,5 +1,9 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Alibaba Qwen3.7 Max: enable_thinking + thinking_budget (providers/alibaba/models/qwen3.7-max.toml) +# ClinePass OpenAI-compatible Chat Completions path: peers (OpenRouter) expose on/off only, not budget_tokens. +# Toggle: enable_thinking / reasoning enabled|disabled base_model = "alibaba/qwen3.7-max" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 2.5 diff --git a/providers/cline-pass/models/cline-pass/qwen3.7-plus.toml b/providers/cline-pass/models/cline-pass/qwen3.7-plus.toml index 55e28b335eb..4c0d4b2c041 100644 --- a/providers/cline-pass/models/cline-pass/qwen3.7-plus.toml +++ b/providers/cline-pass/models/cline-pass/qwen3.7-plus.toml @@ -1,5 +1,9 @@ +# https://docs.cline.bot/getting-started/clinepass +# Lab Alibaba Qwen3.7 Plus: enable_thinking + thinking_budget (providers/alibaba/models/qwen3.7-plus.toml) +# ClinePass OpenAI-compatible Chat Completions path: peers (OpenRouter) expose on/off only, not budget_tokens. +# Toggle: enable_thinking / reasoning enabled|disabled base_model = "alibaba/qwen3.7-plus" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +reasoning_options = [{ type = "toggle" }] [cost] input = 0.4 diff --git a/providers/cline-pass/models/cline-pass/qwen3.8-max.toml b/providers/cline-pass/models/cline-pass/qwen3.8-max.toml index a72991d846b..6a9d910be85 100644 --- a/providers/cline-pass/models/cline-pass/qwen3.8-max.toml +++ b/providers/cline-pass/models/cline-pass/qwen3.8-max.toml @@ -1,11 +1,11 @@ # https://docs.cline.bot/getting-started/clinepass (accessed 2026-08-24) # Model ID cline-pass/qwen3.8-max; reference pricing $2 / $6 / $0.25 per 1M tokens (input/output/cache read), cache write $2.5. +# Lab Alibaba Qwen3.8 Max: enable_thinking + reasoning_effort low|medium|xhigh +# (providers/alibaba/models/qwen3.8-max.toml; https://docs.qwencloud.com/developer-guides/text-generation/thinking) +# Effort includes none for off (hybrid model); no high (lab maps high→xhigh; native graded set is low|medium|xhigh) base_model = "alibaba/qwen3.8-max" structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high", "xhigh"] +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "xhigh"] }] [cost] input = 2 diff --git a/providers/opper/models/xai/grok-4.6.toml b/providers/cloudflare-ai-gateway/models/xai/grok-4.7.toml similarity index 76% rename from providers/opper/models/xai/grok-4.6.toml rename to providers/cloudflare-ai-gateway/models/xai/grok-4.7.toml index 1f8386fd483..48df4f8248b 100644 --- a/providers/opper/models/xai/grok-4.6.toml +++ b/providers/cloudflare-ai-gateway/models/xai/grok-4.7.toml @@ -1,4 +1,4 @@ -base_model = "xai/grok-4.6" +base_model = "xai/grok-4.7" [[reasoning_options]] type = "effort" @@ -15,5 +15,5 @@ input = 4 output = 12 cache_read = 1 -[interleaved] -field = "reasoning_content" +[limit] +context = 500_000 diff --git a/providers/cloudflare-workers-ai/models/@cf/aisingapore/gemma-sea-lion-v4-27b-it.toml b/providers/cloudflare-workers-ai/models/@cf/aisingapore/gemma-sea-lion-v4-27b-it.toml index 8976b92b55a..3fe191a4d99 100644 --- a/providers/cloudflare-workers-ai/models/@cf/aisingapore/gemma-sea-lion-v4-27b-it.toml +++ b/providers/cloudflare-workers-ai/models/@cf/aisingapore/gemma-sea-lion-v4-27b-it.toml @@ -1,23 +1,7 @@ +base_model = "aisingapore/gemma-sea-lion-v4-27b-it" name = "Gemma Sea Lion V4 27B It" -description = "Open Gemma instruction model for efficient chat and self-hosted deployments" -family = "gemma" -release_date = "2025-09-23" -last_updated = "2025-09-23" -attachment = false -reasoning = false -temperature = true -tool_call = false structured_output = false -open_weights = true [cost] input = 0.351 output = 0.555 - -[limit] -context = 128_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml index b72f1489e71..48cdf638dfe 100644 --- a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml +++ b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-flash-0731.toml @@ -1,13 +1,14 @@ -# Toggle: thinking.type = enabled|disabled. -# Effort: reasoning_effort = high|max. +# Effort: reasoning_effort = none|low|high|max; none disables thinking. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# Context/output capped to Cloudflare Workers AI context window (1,048,576). +# Search reports 1,310,720, exceeding AI Gateway max_completion_tokens. +# This dated release adds a distinct low effort; the older preview does not. +# Creator: https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731#chat-template base_model = "deepseek/deepseek-v4-flash-0731" -[[reasoning_options]] -type = "toggle" - [[reasoning_options]] type = "effort" -values = ["high", "max"] +values = ["none", "low", "high", "max"] [cost] input = 0.44 diff --git a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-pro-0813.toml b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-pro-0813.toml index b28c06ffbe1..32c7c6708b0 100644 --- a/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-pro-0813.toml +++ b/providers/cloudflare-workers-ai/models/@cf/deepseek-ai/deepseek-v4-pro-0813.toml @@ -1,8 +1,12 @@ -# Toggle: thinking.type = enabled|disabled. -# Effort: reasoning_effort = high|max. +# Effort: reasoning_effort = none|low|high|max; none disables thinking. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# This dated release adds a distinct low effort; the older preview does not. +# Creator: https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813#chat-template base_model = "deepseek/deepseek-v4-pro-0813" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["high", "max"] }] +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high", "max"] [cost] input = 1.32 diff --git a/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml b/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml index 817334b2cee..2c6a8e4386f 100644 --- a/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml +++ b/providers/cloudflare-workers-ai/models/@cf/google/gemma-4-26b-a4b-it.toml @@ -1,11 +1,15 @@ -# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and -# `chat_template_kwargs.enable_thinking = true|false`; no budget is documented. -# https://developers.cloudflare.com/workers-ai/models/gemma-4-26b-a4b-it/sync-input.json (accessed 2026-06-25) +# Effort: reasoning_effort = none|high; none disables thinking. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# Per-model Search config normalizes low/medium to high; these are aliases, +# not distinct levels despite the generic sync-input.json effort enum. base_model = "google/gemma-4-26b-a4b-it" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] interleaved = true +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + [cost] input = 0.1 output = 0.3 diff --git a/providers/cloudflare-workers-ai/models/@cf/ibm-granite/granite-4.0-h-micro.toml b/providers/cloudflare-workers-ai/models/@cf/ibm-granite/granite-4.0-h-micro.toml index a43a661e427..1aa972501cf 100644 --- a/providers/cloudflare-workers-ai/models/@cf/ibm-granite/granite-4.0-h-micro.toml +++ b/providers/cloudflare-workers-ai/models/@cf/ibm-granite/granite-4.0-h-micro.toml @@ -1,14 +1,7 @@ +base_model = "ibm/granite-4-h-micro" name = "Granite 4.0 H Micro" description = "Efficient model for low-latency assistance, extraction, and routine automation" -family = "granite" -release_date = "2025-10-07" -last_updated = "2025-10-07" -attachment = false -reasoning = false -temperature = true -tool_call = true structured_output = false -open_weights = true [cost] input = 0.017 @@ -17,7 +10,3 @@ output = 0.112 [limit] context = 131_000 output = 131_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/cloudflare-workers-ai/models/@cf/mistralai/mistral-small-3.1-24b-instruct.toml b/providers/cloudflare-workers-ai/models/@cf/mistralai/mistral-small-3.1-24b-instruct.toml index e456a0e8625..7ed149c081e 100644 --- a/providers/cloudflare-workers-ai/models/@cf/mistralai/mistral-small-3.1-24b-instruct.toml +++ b/providers/cloudflare-workers-ai/models/@cf/mistralai/mistral-small-3.1-24b-instruct.toml @@ -1,23 +1,15 @@ +base_model = "mistral/mistral-small-3-1-24b-instruct-2503" name = "Mistral Small 3.1 24B Instruct" description = "Efficient Mistral model for fast chat, extraction, and production assistants" -family = "mistral-small" -release_date = "2025-03-18" -last_updated = "2025-03-18" attachment = false -reasoning = false -temperature = true -tool_call = true structured_output = false -open_weights = true [cost] input = 0.351 output = 0.555 [limit] -context = 128_000 output = 128_000 [modalities] input = ["text"] -output = ["text"] diff --git a/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml b/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml index 3c4c22a8305..6148d7218ca 100644 --- a/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml +++ b/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.6.toml @@ -1,9 +1,13 @@ -# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and -# `chat_template_kwargs.thinking = true|false`; no budget is documented. -# https://developers.cloudflare.com/workers-ai/models/kimi-k2.6/sync-input.json (accessed 2026-06-25) +# Effort: reasoning_effort = none|high; none selects instant mode. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# Per-model Search config normalizes low/medium to high; these are aliases, +# not distinct levels despite the generic sync-input.json effort enum. base_model = "moonshotai/kimi-k2.6" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] [interleaved] field = "reasoning_content" diff --git a/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.7-code.toml b/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.7-code.toml index a06c8bc9a29..43a8509e3a6 100644 --- a/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.7-code.toml +++ b/providers/cloudflare-workers-ai/models/@cf/moonshotai/kimi-k2.7-code.toml @@ -1,6 +1,8 @@ +# Thinking is mandatory; high is the only effective effort, so aliases add no caller control. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). base_model = "moonshotai/kimi-k2.7-code" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] temperature = true +reasoning_options = [] [cost] input = 0.95 diff --git a/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml b/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml index 554decfc2e3..a75a99e22b1 100644 --- a/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/nvidia/nemotron-3-120b-a12b.toml @@ -5,9 +5,15 @@ base_model = "nvidia/nemotron-3-super-120b-a12b" name = "Nemotron 3 Super 120B" structured_output = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] interleaved = true +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.5 output = 1.5 diff --git a/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml b/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml index fc4739a9fbe..0dcc22de116 100644 --- a/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/openai/gpt-oss-20b.toml @@ -3,7 +3,10 @@ # https://developers.cloudflare.com/workers-ai/models/gpt-oss-20b/sync-input.json (accessed 2026-06-25) base_model = "openai/gpt-oss-20b" description = "Open-weight GPT model for self-hosted reasoning and instruction-following workloads" -reasoning_options = [] + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 0.2 diff --git a/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3.8-27b.toml b/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3.8-27b.toml index 307aa38fb79..f491dd473a3 100644 --- a/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3.8-27b.toml +++ b/providers/cloudflare-workers-ai/models/@cf/qwen/qwen3.8-27b.toml @@ -5,7 +5,13 @@ base_model = "alibaba/qwen3.8-27b" description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "xhigh"] }] + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] [cost] input = 0.45 diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml index 197fa7c94b5..37f0d8a203e 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-4.7-flash.toml @@ -1,32 +1,21 @@ -# Native `/ai/run` accepts `reasoning_effort = low|medium|high` and -# `chat_template_kwargs.enable_thinking = true|false`; no budget is documented. -# https://developers.cloudflare.com/workers-ai/models/glm-4.7-flash/sync-input.json (accessed 2026-06-25) - -name = "GLM-4.7-Flash" +# Toggle: chat_template_kwargs.enable_thinking = true|false (default: true). +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# Workers AI's per-model config exposes only the binary thinking toggle; +# the generic sync-input.json low/medium/high enum is not an effort selector here. +# Creator template: https://huggingface.co/zai-org/GLM-4.7-Flash/blob/main/chat_template.jinja +base_model = "zhipuai/glm-4.7-flash" description = "Efficient GLM model for fast reasoning, coding, and agent workflows" -family = "glm-flash" -release_date = "2026-01-19" -last_updated = "2026-01-19" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true structured_output = true -knowledge = "2025-04" -open_weights = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.0605 output = 0.4 [limit] context = 131_072 -output = 131_072 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.2.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.2.toml index 5642ab6e4ea..48337ad55f9 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.2.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.2.toml @@ -1,14 +1,13 @@ +# Effort: reasoning_effort = none|high|max; none disables thinking. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). # Context and output limits verified against Cloudflare Workers AI's OpenAI-compatible endpoint. # https://developers.cloudflare.com/workers-ai/models/glm-5.2/ base_model = "zhipuai/glm-5.2" name = "Glm 5.2" -[[reasoning_options]] -type = "toggle" - [[reasoning_options]] type = "effort" -values = ["low", "medium", "high"] +values = ["none", "high", "max"] [cost] input = 1.4 diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml index 000ebf8a49a..13c5aacae4c 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3-flash.toml @@ -1,18 +1,14 @@ +# Effort: reasoning_effort = low|high|max; thinking is mandatory. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). # Context/output capped to Cloudflare Workers AI context window (1,048,576). -# Catalog/search previously reported 1,310,720, which exceeds AI Gateway max_completion_tokens. -# https://developers.cloudflare.com/workers-ai/models/glm-5.3-flash/ +# Search reports 1,310,720, exceeding AI Gateway max_completion_tokens. +base_model = "zhipuai/glm-5.3-flash" name = "Glm 5.3 Flash" description = "GLM vision model for visual reasoning, documents, and multimodal agents" -family = "glm-flash" -release_date = "2026-08-26" -last_updated = "2026-08-26" -attachment = true -reasoning = true -temperature = true -tool_call = true -structured_output = true -open_weights = true -reasoning_options = [] + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] [cost] input = 0.15 @@ -25,4 +21,3 @@ output = 1_048_576 [modalities] input = ["text", "image"] -output = ["text"] diff --git a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml index a5f20bbaf80..456f1fb0549 100644 --- a/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml +++ b/providers/cloudflare-workers-ai/models/@cf/zai-org/glm-5.3.toml @@ -1,7 +1,14 @@ +# Effort: reasoning_effort = low|high|max; thinking is mandatory. +# Source: Workers AI /ai/models/search?format=openrouter (2026-09-18). +# Context/output capped to Cloudflare Workers AI context window (1,048,576). +# Search reports 1,310,720, exceeding AI Gateway max_completion_tokens. base_model = "zhipuai/glm-5.3" name = "Glm 5.3" description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] [cost] input = 1.4 @@ -10,4 +17,4 @@ cache_read = 0.26 [limit] context = 1_310_720 -output = 1_310_720 +output = 1_048_576 diff --git a/providers/coralbricks/models/deepseek-v4.1-flash-fast-fp4.toml b/providers/coralbricks/models/deepseek-v4.1-flash-fast-fp4.toml new file mode 100644 index 00000000000..9c63cb4a2db --- /dev/null +++ b/providers/coralbricks/models/deepseek-v4.1-flash-fast-fp4.toml @@ -0,0 +1,25 @@ +# CoralBricks pricing per https://www.coralbricks.ai/pricing and https://www.coralbricks.ai/api/public/models (accessed 2026-09-21): +# input $0.30, output $1.20, cache write $0.09 per 1M tokens; cached input is free. +# Effort: reasoning_effort = none|low|high|max. Reasoning is opt-in: with no reasoning_effort, or none, the model +# answers without reasoning. thinking.type is ignored by this endpoint. +# Verified live against POST https://inference.coralbricks.ai/v1/chat/completions (2026-09-21): low, high and max +# return HTTP 200 with reasoning_content; none and no field return 0 reasoning tokens. +base_model = "deepseek/deepseek-v4.1-flash" +name = "DeepSeek V4.1 Flash FP4" +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] +interleaved = true +attachment = false + +[limit] +context = 1_048_576 +output = 131_072 + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0 +cache_write = 0.09 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/coralbricks/models/glm-5.3-flash-fp4.toml b/providers/coralbricks/models/glm-5.3-flash-fp4.toml index a7836490ddd..2c52bfdcb00 100644 --- a/providers/coralbricks/models/glm-5.3-flash-fp4.toml +++ b/providers/coralbricks/models/glm-5.3-flash-fp4.toml @@ -12,6 +12,7 @@ output = 131_072 input = 0.15 output = 0.5 cache_read = 0 +cache_write = 0.23 [modalities] input = ["text", "image", "video"] diff --git a/providers/coralbricks/models/glm-5.3-fp4.toml b/providers/coralbricks/models/glm-5.3-fp4.toml index 2488afc97cb..90e669e9d54 100644 --- a/providers/coralbricks/models/glm-5.3-fp4.toml +++ b/providers/coralbricks/models/glm-5.3-fp4.toml @@ -12,3 +12,4 @@ output = 131_072 input = 1.12 output = 4.4 cache_read = 0 +cache_write = 1.68 diff --git a/providers/coralbricks/models/gpt-oss-120b.toml b/providers/coralbricks/models/gpt-oss-120b.toml index 8fd3f30e11e..80b28acc1b5 100644 --- a/providers/coralbricks/models/gpt-oss-120b.toml +++ b/providers/coralbricks/models/gpt-oss-120b.toml @@ -8,3 +8,4 @@ interleaved = true input = 0.12 output = 0.6 cache_read = 0 +cache_write = 0.18 diff --git a/providers/coralbricks/models/kimi-k3.toml b/providers/coralbricks/models/kimi-k3.toml deleted file mode 100644 index 77c2aabdc7b..00000000000 --- a/providers/coralbricks/models/kimi-k3.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "moonshotai/kimi-k3" - -reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] -interleaved = true - -[cost] -input = 3.0 -output = 15.0 -cache_read = 0 diff --git a/providers/cortecs/models/qwen3-32b.toml b/providers/cortecs/models/qwen3-32b.toml index d6ad27de61b..32225e598f0 100644 --- a/providers/cortecs/models/qwen3-32b.toml +++ b/providers/cortecs/models/qwen3-32b.toml @@ -4,9 +4,8 @@ structured_output = true reasoning_options = [] [cost] -input = 0.089 -output = 0.312 +input = 0.179 +output = 0.697 [limit] -context = 32_000 -output = 32_000 +context = 16_384 diff --git a/providers/crossmodel/models/anthropic/claude-opus-5-5.toml b/providers/crossmodel/models/anthropic/claude-opus-5-5.toml new file mode 100644 index 00000000000..d9abb20946e --- /dev/null +++ b/providers/crossmodel/models/anthropic/claude-opus-5-5.toml @@ -0,0 +1,14 @@ +base_model = "anthropic/claude-opus-5-5" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[modalities] +input = ["text", "image"] diff --git a/providers/crossmodel/models/openai/gpt-6-luna.toml b/providers/crossmodel/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..e91e64b7330 --- /dev/null +++ b/providers/crossmodel/models/openai/gpt-6-luna.toml @@ -0,0 +1,21 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[modalities] +input = ["text", "image"] diff --git a/providers/crossmodel/models/openai/gpt-6-sol.toml b/providers/crossmodel/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..56437cf29c3 --- /dev/null +++ b/providers/crossmodel/models/openai/gpt-6-sol.toml @@ -0,0 +1,21 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 + +[modalities] +input = ["text", "image"] diff --git a/providers/crossmodel/models/qwen/qwen3.8-omni-flash.toml b/providers/crossmodel/models/qwen/qwen3.8-omni-flash.toml new file mode 100644 index 00000000000..b70235cd8f0 --- /dev/null +++ b/providers/crossmodel/models/qwen/qwen3.8-omni-flash.toml @@ -0,0 +1,17 @@ +base_model = "alibaba/qwen3.8-omni-flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.13 +output = 0.43 +cache_read = 0.016 +cache_write = 0.13 diff --git a/providers/crossmodel/models/x-ai/grok-4.7.toml b/providers/crossmodel/models/x-ai/grok-4.7.toml new file mode 100644 index 00000000000..99495b45cae --- /dev/null +++ b/providers/crossmodel/models/x-ai/grok-4.7.toml @@ -0,0 +1,21 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 +cache_write = 2 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 +cache_write = 4 + +[modalities] +input = ["text", "image"] diff --git a/providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml b/providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..5fdd8d29025 --- /dev/null +++ b/providers/crossmodel/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,11 @@ +base_model = "xiaomi/mimo-v2.6-flash" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.16 +output = 0.32 +cache_read = 0.004 +cache_write = 0.16 diff --git a/providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml b/providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..1dbc817d0fa --- /dev/null +++ b/providers/crossmodel/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,11 @@ +base_model = "xiaomi/mimo-v2.6-pro" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.47 +output = 0.94 +cache_read = 0.005 +cache_write = 0.47 diff --git a/providers/deepinfra/models/tencent/Hy3.toml b/providers/deepinfra/models/tencent/Hy3.toml index 141b04fc694..743a6959ec1 100644 --- a/providers/deepinfra/models/tencent/Hy3.toml +++ b/providers/deepinfra/models/tencent/Hy3.toml @@ -3,9 +3,9 @@ structured_output = true reasoning_options = [] [cost] -input = 0.14 -output = 0.58 -cache_read = 0.035 +input = 0.13 +output = 0.53 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/deepseek/models/deepseek-v4-flash-vision-exp.toml b/providers/deepseek/models/deepseek-v4-flash-vision-exp.toml index e1227c7a467..aa1da80696b 100644 --- a/providers/deepseek/models/deepseek-v4-flash-vision-exp.toml +++ b/providers/deepseek/models/deepseek-v4-flash-vision-exp.toml @@ -6,8 +6,7 @@ # https://api-docs.deepseek.com/quick_start/pricing (accessed 2026-09-10) base_model = "deepseek/deepseek-v4.1-flash" name = "DeepSeek V4 Flash Vision Exp" -## mark as deprecated soon -# status = "deprecated" +status = "deprecated" [[reasoning_options]] type = "toggle" diff --git a/providers/deepseek/models/deepseek-v4-flash.toml b/providers/deepseek/models/deepseek-v4-flash.toml index ae7ff2fc79b..3ebe57414da 100644 --- a/providers/deepseek/models/deepseek-v4-flash.toml +++ b/providers/deepseek/models/deepseek-v4-flash.toml @@ -6,8 +6,7 @@ # https://api-docs.deepseek.com/quick_start/pricing (accessed 2026-09-10) base_model = "deepseek/deepseek-v4.1-flash" name = "DeepSeek V4 Flash" -## mark as deprecated soon -# status = "deprecated" +status = "deprecated" [[reasoning_options]] type = "toggle" diff --git a/providers/deepseek/models/deepseek-v4-pro.toml b/providers/deepseek/models/deepseek-v4-pro.toml index 4d5aff0281b..6a81c8584d2 100644 --- a/providers/deepseek/models/deepseek-v4-pro.toml +++ b/providers/deepseek/models/deepseek-v4-pro.toml @@ -1,18 +1,23 @@ +# Toggle: thinking.type = enabled|disabled on /chat/completions. +# Effort: reasoning_effort = low|high|max; default high. +# Anthropic: output_config.effort = low|high|max; budget ignored. +# The 2026-08-13 V4-Pro GA announcement explicitly introduces all three levels; +# the pricing page maps API model deepseek-v4-pro to DeepSeek-V4-Pro-0813. +# https://api-docs.deepseek.com/news/news260813/ (accessed 2026-09-20) +# https://api-docs.deepseek.com/guides/thinking_mode/ (accessed 2026-09-20) +# https://api-docs.deepseek.com/quick_start/pricing/ (accessed 2026-09-20) # Reasoning tokens are billed at the output rate (no separate CoT price). # `completion_tokens_details.reasoning_tokens` is a subset of completion_tokens. # https://api-docs.deepseek.com/quick_start/pricing/ (accessed 2026-08-12) base_model = "deepseek/deepseek-v4-pro-0813" name = "DeepSeek V4 Pro" -# OpenAI: `thinking.type = enabled|disabled`, `reasoning_effort = high|max`. -# Anthropic: `thinking.type`, `output_config.effort = high|max`; budget ignored. -# https://api-docs.deepseek.com/api/create-chat-completion (accessed 2026-06-25) [[reasoning_options]] type = "toggle" [[reasoning_options]] type = "effort" -values = ["high", "max"] +values = ["low", "high", "max"] [interleaved] field = "reasoning_content" diff --git a/providers/deepseek/provider.toml b/providers/deepseek/provider.toml index 11731946bfd..76274ae9de4 100644 --- a/providers/deepseek/provider.toml +++ b/providers/deepseek/provider.toml @@ -2,11 +2,11 @@ name = "DeepSeek" env = ["DEEPSEEK_API_KEY"] npm = "@ai-sdk/openai-compatible" # OpenAI Chat is POST `/chat/completions`: `thinking.type = enabled|disabled` -# and `reasoning_effort = low|high|max`. Flash maps low→low; Pro maps low→high; -# xhigh maps to high (flash) or max (pro). +# and `reasoning_effort = low|high|max`. Both current models map minimal→low, +# medium/xhigh→high, and ultra→max; low/high/max remain distinct native levels. # Anthropic Messages is POST `/anthropic/v1/messages`: `thinking.type` and # `output_config.effort = low|high|max`; `thinking.budget_tokens` is ignored. -# https://api-docs.deepseek.com/guides/thinking_mode/ (accessed 2026-08-02) +# https://api-docs.deepseek.com/guides/thinking_mode/ (accessed 2026-09-20) # https://api-docs.deepseek.com/guides/anthropic_api (accessed 2026-06-25) doc = "https://api-docs.deepseek.com/quick_start/pricing" api = "https://api.deepseek.com" diff --git a/providers/digitalocean/models/anthropic-claude-opus-5.5.toml b/providers/digitalocean/models/anthropic-claude-opus-5.5.toml new file mode 100644 index 00000000000..23c000bcf3d --- /dev/null +++ b/providers/digitalocean/models/anthropic-claude-opus-5.5.toml @@ -0,0 +1,22 @@ +base_model = "anthropic/claude-opus-5-5" +name = "Anthropic Claude Opus 5.5" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 8 +output = 30 +cache_read = 0.4 +cache_write = 10 + +[modalities] +input = ["text", "image"] diff --git a/providers/digitalocean/models/deepseek-3.2.toml b/providers/digitalocean/models/deepseek-3.2.toml index 86a0601e001..d8d92fa42fa 100644 --- a/providers/digitalocean/models/deepseek-3.2.toml +++ b/providers/digitalocean/models/deepseek-3.2.toml @@ -18,9 +18,9 @@ type = "effort" values = ["none", "low", "medium", "high"] [cost] -input = 0.25 -output = 0.8 -cache_read = 0.075 +input = 0.5 +output = 1.6 +cache_read = 0.15 [limit] context = 163_840 diff --git a/providers/digitalocean/models/deepseek-4-flash.toml b/providers/digitalocean/models/deepseek-4-flash.toml index 2e033308f91..143346a359b 100644 --- a/providers/digitalocean/models/deepseek-4-flash.toml +++ b/providers/digitalocean/models/deepseek-4-flash.toml @@ -10,9 +10,9 @@ tool_call = true open_weights = false [cost] -input = 0.0679 -output = 0.168 -cache_read = 0.0168 +input = 0.14 +output = 0.28 +cache_read = 0.028 [limit] context = 1_048_576 diff --git a/providers/digitalocean/models/deepseek-v4-flash-0731.toml b/providers/digitalocean/models/deepseek-v4-flash-0731.toml index 74ce3b0ae91..2e9e64c5632 100644 --- a/providers/digitalocean/models/deepseek-v4-flash-0731.toml +++ b/providers/digitalocean/models/deepseek-v4-flash-0731.toml @@ -5,9 +5,9 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.08 -output = 0.252 -cache_read = 0.0252 +input = 0.14 +output = 0.28 +cache_read = 0.028 [limit] context = 1_048_576 diff --git a/providers/digitalocean/models/deepseek-v4-pro.toml b/providers/digitalocean/models/deepseek-v4-pro.toml index 8a4edb98729..b4ebe0207c1 100644 --- a/providers/digitalocean/models/deepseek-v4-pro.toml +++ b/providers/digitalocean/models/deepseek-v4-pro.toml @@ -19,9 +19,9 @@ type = "effort" values = ["low", "medium", "high", "xhigh"] [cost] -input = 0.87 -output = 1.74 -cache_read = 0.174 +input = 1.74 +output = 3.48 +cache_read = 0.348 [limit] context = 1_048_576 diff --git a/providers/digitalocean/models/glm-5.2.toml b/providers/digitalocean/models/glm-5.2.toml index 3f01602c179..eb3b3d1b78d 100644 --- a/providers/digitalocean/models/glm-5.2.toml +++ b/providers/digitalocean/models/glm-5.2.toml @@ -8,9 +8,9 @@ type = "effort" values = ["medium", "high", "xhigh"] [cost] -input = 0.7 -output = 2.2 -cache_read = 0.105 +input = 1.4 +output = 4.4 +cache_read = 0.21 [limit] context = 262_144 diff --git a/providers/digitalocean/models/glm-5.3.toml b/providers/digitalocean/models/glm-5.3.toml index 9ba7811a008..6172a61a5ab 100644 --- a/providers/digitalocean/models/glm-5.3.toml +++ b/providers/digitalocean/models/glm-5.3.toml @@ -6,9 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.95 -output = 3.4 -cache_read = 0.2 +input = 1.4 +output = 4.4 +cache_read = 0.26 [limit] context = 1_048_576 diff --git a/providers/digitalocean/models/kimi-k3.toml b/providers/digitalocean/models/kimi-k3.toml index 7533a7423e2..62d0f7b7404 100644 --- a/providers/digitalocean/models/kimi-k3.toml +++ b/providers/digitalocean/models/kimi-k3.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.55 -output = 12.95 -cache_read = 0.285 +input = 3 +output = 15 +cache_read = 0.3 [modalities] input = ["text", "image"] diff --git a/providers/digitalocean/models/mimo-v2.5-pro.toml b/providers/digitalocean/models/mimo-v2.5-pro.toml index 20f6f6db3ea..365235fb39f 100644 --- a/providers/digitalocean/models/mimo-v2.5-pro.toml +++ b/providers/digitalocean/models/mimo-v2.5-pro.toml @@ -9,9 +9,9 @@ type = "effort" values = ["none", "high"] [cost] -input = 0.4 -output = 1.5 -cache_read = 0.08 +input = 0.8 +output = 3 +cache_read = 0.16 [limit] context = 262_144 diff --git a/providers/digitalocean/models/openai-gpt-6-luna.toml b/providers/digitalocean/models/openai-gpt-6-luna.toml new file mode 100644 index 00000000000..27a362b9081 --- /dev/null +++ b/providers/digitalocean/models/openai-gpt-6-luna.toml @@ -0,0 +1,20 @@ +base_model = "openai/gpt-6-luna" +name = "OpenAI GPT-6 Luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 + +[modalities] +input = ["text", "image"] diff --git a/providers/digitalocean/models/openai-gpt-6-sol.toml b/providers/digitalocean/models/openai-gpt-6-sol.toml new file mode 100644 index 00000000000..8c434d8b143 --- /dev/null +++ b/providers/digitalocean/models/openai-gpt-6-sol.toml @@ -0,0 +1,20 @@ +base_model = "openai/gpt-6-sol" +name = "OpenAI GPT-6 Sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 + +[modalities] +input = ["text", "image"] diff --git a/providers/digitalocean/models/openai-gpt-oss-120b.toml b/providers/digitalocean/models/openai-gpt-oss-120b.toml index fc45d3aac8b..4af4fef37ad 100644 --- a/providers/digitalocean/models/openai-gpt-oss-120b.toml +++ b/providers/digitalocean/models/openai-gpt-oss-120b.toml @@ -19,8 +19,8 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.055 -output = 0.385 +input = 0.1 +output = 0.7 cache_read = 0.02 [limit] diff --git a/providers/edenai/models/amazon/google.gemma-3-12b-it.toml b/providers/edenai/models/amazon/google.gemma-3-12b-it.toml index 76ea0c2327a..3889a7e378e 100644 --- a/providers/edenai/models/amazon/google.gemma-3-12b-it.toml +++ b/providers/edenai/models/amazon/google.gemma-3-12b-it.toml @@ -1,7 +1,6 @@ base_model = "google/gemma-3-12b-it" name = "Gemma 3 12B IT (Amazon Bedrock)" -tool_call = false -structured_output = false +structured_output = true [cost] input = 0.09 diff --git a/providers/edenai/models/amazon/google.gemma-3-12b-it@us.toml b/providers/edenai/models/amazon/google.gemma-3-12b-it@us.toml index 91ebef21c30..332de3d4ed7 100644 --- a/providers/edenai/models/amazon/google.gemma-3-12b-it@us.toml +++ b/providers/edenai/models/amazon/google.gemma-3-12b-it@us.toml @@ -1,7 +1,6 @@ base_model = "google/gemma-3-12b-it" name = "Gemma 3 12B IT (Amazon Bedrock, US)" -tool_call = false -structured_output = false +structured_output = true [cost] input = 0.09 diff --git a/providers/edenai/models/amazon/google.gemma-3-27b-it.toml b/providers/edenai/models/amazon/google.gemma-3-27b-it.toml index b99b2b38ac2..938c2319714 100644 --- a/providers/edenai/models/amazon/google.gemma-3-27b-it.toml +++ b/providers/edenai/models/amazon/google.gemma-3-27b-it.toml @@ -1,7 +1,6 @@ base_model = "google/gemma-3-27b-it" name = "Gemma 3 27B IT (Amazon Bedrock)" -tool_call = false -structured_output = false +structured_output = true [cost] input = 0.23 diff --git a/providers/edenai/models/amazon/google.gemma-3-27b-it@us.toml b/providers/edenai/models/amazon/google.gemma-3-27b-it@us.toml index 4c9f67f300e..a1637470411 100644 --- a/providers/edenai/models/amazon/google.gemma-3-27b-it@us.toml +++ b/providers/edenai/models/amazon/google.gemma-3-27b-it@us.toml @@ -1,7 +1,6 @@ base_model = "google/gemma-3-27b-it" name = "Gemma 3 27B IT (Amazon Bedrock, US)" -tool_call = false -structured_output = false +structured_output = true [cost] input = 0.23 diff --git a/providers/edenai/models/amazon/google.gemma-3-4b-it.toml b/providers/edenai/models/amazon/google.gemma-3-4b-it.toml index f502a9a96cd..536c47c8d73 100644 --- a/providers/edenai/models/amazon/google.gemma-3-4b-it.toml +++ b/providers/edenai/models/amazon/google.gemma-3-4b-it.toml @@ -1,6 +1,5 @@ base_model = "google/gemma-3-4b-it" name = "Gemma 3 4B IT (Amazon Bedrock)" -tool_call = false structured_output = false [cost] diff --git a/providers/edenai/models/amazon/google.gemma-3-4b-it@us.toml b/providers/edenai/models/amazon/google.gemma-3-4b-it@us.toml index 6d658417b4f..2d4e482e0c4 100644 --- a/providers/edenai/models/amazon/google.gemma-3-4b-it@us.toml +++ b/providers/edenai/models/amazon/google.gemma-3-4b-it@us.toml @@ -1,6 +1,5 @@ base_model = "google/gemma-3-4b-it" name = "Gemma 3 4B IT (Amazon Bedrock, US)" -tool_call = false structured_output = false [cost] diff --git a/providers/edenai/models/amazon/moonshot.kimi-k2-thinking.toml b/providers/edenai/models/amazon/moonshot.kimi-k2-thinking.toml index 1fdc4e3a58c..7413aad8307 100644 --- a/providers/edenai/models/amazon/moonshot.kimi-k2-thinking.toml +++ b/providers/edenai/models/amazon/moonshot.kimi-k2-thinking.toml @@ -1,7 +1,6 @@ base_model = "moonshotai/kimi-k2-thinking" name = "Kimi K2 Thinking (Amazon Bedrock)" -tool_call = false -structured_output = false +structured_output = true reasoning_options = [] [cost] @@ -9,4 +8,4 @@ input = 0.6 output = 2.5 [limit] -context = 128_000 +context = 256_000 diff --git a/providers/edenai/models/amazon/moonshotai.kimi-k2.5.toml b/providers/edenai/models/amazon/moonshotai.kimi-k2.5.toml index 45d5d0a17cb..c64d911d778 100644 --- a/providers/edenai/models/amazon/moonshotai.kimi-k2.5.toml +++ b/providers/edenai/models/amazon/moonshotai.kimi-k2.5.toml @@ -1,6 +1,5 @@ base_model = "moonshotai/kimi-k2.5" name = "Kimi K2.5 (Amazon Bedrock)" -structured_output = false reasoning_options = [] [cost] diff --git a/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b.toml b/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b.toml index 536758dfd0e..c159c3bb673 100644 --- a/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b.toml +++ b/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b.toml @@ -1,7 +1,5 @@ base_model = "openai/gpt-oss-safeguard-20b" name = "GPT OSS Safeguard 20B (Amazon Bedrock)" -tool_call = false -structured_output = false [[reasoning_options]] type = "effort" diff --git a/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b@us.toml b/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b@us.toml index 7992978de7d..98893d21912 100644 --- a/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b@us.toml +++ b/providers/edenai/models/amazon/openai.gpt-oss-safeguard-20b@us.toml @@ -1,7 +1,5 @@ base_model = "openai/gpt-oss-safeguard-20b" name = "GPT OSS Safeguard 20B (Amazon Bedrock, US)" -tool_call = false -structured_output = false [[reasoning_options]] type = "effort" diff --git a/providers/edenai/models/amazon/zai.glm-4.7-flash.toml b/providers/edenai/models/amazon/zai.glm-4.7-flash.toml index 2a31f0a6612..bf215607f73 100644 --- a/providers/edenai/models/amazon/zai.glm-4.7-flash.toml +++ b/providers/edenai/models/amazon/zai.glm-4.7-flash.toml @@ -1,8 +1,11 @@ base_model = "zhipuai/glm-4.7-flash" name = "GLM-4.7-Flash (Amazon Bedrock)" -structured_output = false +structured_output = true reasoning_options = [] [cost] input = 0.07 output = 0.4 + +[limit] +context = 203_000 diff --git a/providers/edenai/models/amazon/zai.glm-4.7-flash@us.toml b/providers/edenai/models/amazon/zai.glm-4.7-flash@us.toml index ec6e4df6db1..1aa3c944b18 100644 --- a/providers/edenai/models/amazon/zai.glm-4.7-flash@us.toml +++ b/providers/edenai/models/amazon/zai.glm-4.7-flash@us.toml @@ -1,8 +1,11 @@ base_model = "zhipuai/glm-4.7-flash" name = "GLM-4.7-Flash (Amazon Bedrock, US)" -structured_output = false +structured_output = true reasoning_options = [] [cost] input = 0.07 output = 0.4 + +[limit] +context = 203_000 diff --git a/providers/edenai/models/anthropic/claude-opus-5-5.toml b/providers/edenai/models/anthropic/claude-opus-5-5.toml new file mode 100644 index 00000000000..8d8ff11a7d2 --- /dev/null +++ b/providers/edenai/models/anthropic/claude-opus-5-5.toml @@ -0,0 +1,12 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/edenai/models/cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct.toml b/providers/edenai/models/cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct.toml index 33e52d4c7a4..be84a18d071 100644 --- a/providers/edenai/models/cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct.toml +++ b/providers/edenai/models/cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct.toml @@ -1,7 +1,7 @@ base_model = "alibaba/qwen2.5-coder-32b-instruct" name = "Qwen2.5-Coder-32B-Instruct (Cloudflare)" tool_call = false -structured_output = false +structured_output = true [cost] input = 0.66 diff --git a/providers/edenai/models/databricks/databricks-deepseek-v4-flash-0731.toml b/providers/edenai/models/databricks/databricks-deepseek-v4-flash-0731.toml index d4d7d35c658..ec6f7cacc88 100644 --- a/providers/edenai/models/databricks/databricks-deepseek-v4-flash-0731.toml +++ b/providers/edenai/models/databricks/databricks-deepseek-v4-flash-0731.toml @@ -9,5 +9,5 @@ values = ["none", "low", "high", "max"] [cost] input = 0.14 output = 0.28 -cache_read = 0.028 +cache_read = 0.014 cache_write = 0.14 diff --git a/providers/edenai/models/databricks/databricks-deepseek-v4-pro-0813.toml b/providers/edenai/models/databricks/databricks-deepseek-v4-pro-0813.toml index a25d6b3e95b..a293c8ac08b 100644 --- a/providers/edenai/models/databricks/databricks-deepseek-v4-pro-0813.toml +++ b/providers/edenai/models/databricks/databricks-deepseek-v4-pro-0813.toml @@ -7,7 +7,7 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 1.31999 -output = 3.95997 -cache_read = 0.13202 +input = 1.32 +output = 3.959999 +cache_read = 0.132 cache_write = 1.31999 diff --git a/providers/edenai/models/databricks/databricks-inkling.toml b/providers/edenai/models/databricks/databricks-inkling.toml index 177f6bd3f2d..eef08eb01d0 100644 --- a/providers/edenai/models/databricks/databricks-inkling.toml +++ b/providers/edenai/models/databricks/databricks-inkling.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "minimal", "low", "medium", "high", "max"] [cost] -input = 1.00002 -output = 4.04999 -cache_read = 0.17003 +input = 1 +output = 4.05 +cache_read = 0.1 cache_write = 1.00002 [limit] diff --git a/providers/edenai/models/deepinfra/tencent/Hy3.toml b/providers/edenai/models/deepinfra/tencent/Hy3.toml index c08482653b4..6cb87c36ee1 100644 --- a/providers/edenai/models/deepinfra/tencent/Hy3.toml +++ b/providers/edenai/models/deepinfra/tencent/Hy3.toml @@ -8,9 +8,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.14 -output = 0.58 -cache_read = 0.035 +input = 0.13 +output = 0.53 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/edenai/models/deepseek/deepseek-v4-pro.toml b/providers/edenai/models/deepseek/deepseek-v4-pro.toml index 043e1bb1f41..05723a127e4 100644 --- a/providers/edenai/models/deepseek/deepseek-v4-pro.toml +++ b/providers/edenai/models/deepseek/deepseek-v4-pro.toml @@ -2,7 +2,7 @@ base_model = "deepseek/deepseek-v4-pro" [[reasoning_options]] type = "effort" -values = ["none", "high", "max"] +values = ["none", "low", "high", "max"] [cost] input = 0.66 diff --git a/providers/edenai/models/flexai/DeepSeek-V4-Flash-0731.toml b/providers/edenai/models/flexai/DeepSeek-V4-Flash-0731.toml index 8a3564408ab..1e1e8029798 100644 --- a/providers/edenai/models/flexai/DeepSeek-V4-Flash-0731.toml +++ b/providers/edenai/models/flexai/DeepSeek-V4-Flash-0731.toml @@ -12,4 +12,4 @@ input = 0.065 output = 0.18 [limit] -context = 786_432 +context = 1_048_576 diff --git a/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml b/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml index d5d18a5d5f0..73ebbfe4b80 100644 --- a/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml +++ b/providers/edenai/models/infomaniak/mistralai/Ministral-3-14B-Instruct-2512.toml @@ -5,8 +5,8 @@ tool_call = false structured_output = false [cost] -input = 0.34443 -output = 0.45924 +input = 0.34389 +output = 0.45852 [limit] context = 100_000 diff --git a/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml b/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml index fc91f75adc5..92757e9fa56 100644 --- a/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml +++ b/providers/edenai/models/ionos/meta-llama/Llama-3.3-70B-Instruct.toml @@ -4,5 +4,5 @@ tool_call = false structured_output = false [cost] -input = 0.746265 -output = 0.746265 +input = 0.745095 +output = 0.745095 diff --git a/providers/edenai/models/ionos/openai/gpt-oss-120b.toml b/providers/edenai/models/ionos/openai/gpt-oss-120b.toml index 488fd48dab9..cad05c7d393 100644 --- a/providers/edenai/models/ionos/openai/gpt-oss-120b.toml +++ b/providers/edenai/models/ionos/openai/gpt-oss-120b.toml @@ -8,5 +8,5 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.172215 -output = 0.746265 +input = 0.171945 +output = 0.745095 diff --git a/providers/edenai/models/mistral/mistral-large-2512.toml b/providers/edenai/models/mistral/mistral-large-2512.toml index 96902738ba4..6944f065750 100644 --- a/providers/edenai/models/mistral/mistral-large-2512.toml +++ b/providers/edenai/models/mistral/mistral-large-2512.toml @@ -2,9 +2,9 @@ base_model = "mistral/mistral-large-2512" structured_output = true [cost] -input = 0.55 -output = 1.65 -cache_read = 0.055 +input = 0.5 +output = 1.5 +cache_read = 0.05 [modalities] input = ["text", "image", "pdf"] diff --git a/providers/edenai/models/openai/gpt-6-luna.toml b/providers/edenai/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..9a9537cbf02 --- /dev/null +++ b/providers/edenai/models/openai/gpt-6-luna.toml @@ -0,0 +1,18 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 diff --git a/providers/edenai/models/openai/gpt-6-sol.toml b/providers/edenai/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..94f694d64de --- /dev/null +++ b/providers/edenai/models/openai/gpt-6-sol.toml @@ -0,0 +1,18 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 diff --git a/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml b/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml index 2236e082a15..d2f9f4d763d 100644 --- a/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml +++ b/providers/edenai/models/scaleway/deepseek-v4-flash-0731.toml @@ -7,8 +7,8 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.45924 -output = 0.91848 +input = 0.45852 +output = 0.91704 [limit] context = 256_000 diff --git a/providers/edenai/models/scaleway/gpt-oss-120b.toml b/providers/edenai/models/scaleway/gpt-oss-120b.toml index c78acf705d7..8f735ccfed6 100644 --- a/providers/edenai/models/scaleway/gpt-oss-120b.toml +++ b/providers/edenai/models/scaleway/gpt-oss-120b.toml @@ -7,8 +7,8 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.172215 -output = 0.68886 +input = 0.171945 +output = 0.68778 [limit] context = 128_000 diff --git a/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml b/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml index b6133a15c15..55662b4313f 100644 --- a/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml +++ b/providers/edenai/models/scaleway/llama-3.3-70b-instruct.toml @@ -3,5 +3,5 @@ name = "Llama-3.3-70B-Instruct (Scaleway)" structured_output = false [cost] -input = 1.03329 -output = 1.03329 +input = 1.03167 +output = 1.03167 diff --git a/providers/edenai/models/tensorx/moonshotai/kimi-k2.5.toml b/providers/edenai/models/tensorx/moonshotai/kimi-k2.5.toml index 87d3cc945d7..fc327494824 100644 --- a/providers/edenai/models/tensorx/moonshotai/kimi-k2.5.toml +++ b/providers/edenai/models/tensorx/moonshotai/kimi-k2.5.toml @@ -1,6 +1,5 @@ base_model = "moonshotai/kimi-k2.5" name = "Kimi K2.5 (TensorX)" -structured_output = false reasoning_options = [] [cost] diff --git a/providers/edenai/models/xai/grok-4.7.toml b/providers/edenai/models/xai/grok-4.7.toml new file mode 100644 index 00000000000..65cbf72670b --- /dev/null +++ b/providers/edenai/models/xai/grok-4.7.toml @@ -0,0 +1,16 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 3.2 +output = 9.6 +cache_read = 0.8 diff --git a/providers/empiriolabs/models/mimo-v2-6-flash.toml b/providers/empiriolabs/models/mimo-v2-6-flash.toml new file mode 100644 index 00000000000..14d640b4db5 --- /dev/null +++ b/providers/empiriolabs/models/mimo-v2-6-flash.toml @@ -0,0 +1,14 @@ +base_model = "xiaomi/mimo-v2.6-flash" +name = "MiMo V2.6 Flash" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.7 +output = 1.4 +cache_read = 0.7 + +[limit] +context = 1_000_000 diff --git a/providers/empiriolabs/models/mimo-v2-6-pro-ultraspeed.toml b/providers/empiriolabs/models/mimo-v2-6-pro-ultraspeed.toml new file mode 100644 index 00000000000..e69d20993d0 --- /dev/null +++ b/providers/empiriolabs/models/mimo-v2-6-pro-ultraspeed.toml @@ -0,0 +1,14 @@ +base_model = "xiaomi/mimo-v2.6-pro-ultraspeed" +name = "MiMo V2.6 Pro UltraSpeed" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 21.75 +output = 43.5 +cache_read = 21.75 + +[limit] +context = 1_000_000 diff --git a/providers/empiriolabs/models/mimo-v2-6-pro.toml b/providers/empiriolabs/models/mimo-v2-6-pro.toml new file mode 100644 index 00000000000..54b2f7069c3 --- /dev/null +++ b/providers/empiriolabs/models/mimo-v2-6-pro.toml @@ -0,0 +1,14 @@ +base_model = "xiaomi/mimo-v2.6-pro" +name = "MiMo V2.6 Pro" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 2.175 +output = 4.35 +cache_read = 2.175 + +[limit] +context = 1_000_000 diff --git a/providers/empiriolabs/models/step-5-preview.toml b/providers/empiriolabs/models/step-5-preview.toml new file mode 100644 index 00000000000..c4ca4d28659 --- /dev/null +++ b/providers/empiriolabs/models/step-5-preview.toml @@ -0,0 +1,16 @@ +base_model = "stepfun/step-5-preview" +base_model_omit = ["limit.input"] +temperature = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1 +output = 2.7 +cache_read = 0.05 + +[limit] +context = 1_024_000 +output = 131_072 diff --git a/providers/github-copilot/models/claude-opus-5.5.toml b/providers/github-copilot/models/claude-opus-5.5.toml new file mode 100644 index 00000000000..6e3f31961a6 --- /dev/null +++ b/providers/github-copilot/models/claude-opus-5.5.toml @@ -0,0 +1,13 @@ +# Sources: +# - https://github.blog/changelog/2026-09-22-claude-opus-5-5-is-now-available-in-github-copilot/ +# - https://docs.github.com/en/copilot/reference/copilot-billing/models-and-pricing +# - https://www.anthropic.com/claude-opus-5-5 +# - https://platform.claude.com/docs/en/models/opus-5-5/overview +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/github-copilot/models/gpt-6-luna.toml b/providers/github-copilot/models/gpt-6-luna.toml new file mode 100644 index 00000000000..bd1d21a407f --- /dev/null +++ b/providers/github-copilot/models/gpt-6-luna.toml @@ -0,0 +1,20 @@ +# Pricing: https://docs.github.com/en/copilot/reference/copilot-billing/models-and-pricing +# GPT-6 Luna rates: ≤272K $0.10/$0.01/$0.125/$0.50; >272K $0.20/$0.02/$0.25/$0.75 per 1M (input/cached/cache write/output) +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 diff --git a/providers/github-copilot/models/gpt-6-sol.toml b/providers/github-copilot/models/gpt-6-sol.toml new file mode 100644 index 00000000000..209b909b0be --- /dev/null +++ b/providers/github-copilot/models/gpt-6-sol.toml @@ -0,0 +1,20 @@ +# Pricing: https://docs.github.com/en/copilot/reference/copilot-billing/models-and-pricing +# GPT-6 Sol rates: ≤272K $2/$0.20/$2.50/$10; >272K $4/$0.40/$5.00/$15 per 1M (input/cached/cache write/output) +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 diff --git a/providers/github-copilot/models/grok-4.7.toml b/providers/github-copilot/models/grok-4.7.toml new file mode 100644 index 00000000000..25d6bf546e2 --- /dev/null +++ b/providers/github-copilot/models/grok-4.7.toml @@ -0,0 +1,24 @@ +# Sources: +# - https://github.blog/changelog/2026-09-21-grok-4-7-is-now-available-in-github-copilot/ +# - https://docs.github.com/en/copilot/reference/copilot-billing/models-and-pricing +# - https://docs.github.com/en/copilot/reference/ai-models/supported-models +# - https://docs.x.ai/developers/models/grok-4.7 +# - https://docs.x.ai/developers/pricing +base_model = "xai/grok-4.7" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 + +[limit] +context = 500_000 +input = 372_000 +output = 128_000 diff --git a/providers/gitlab/models/duo-chat-opus-5-5.toml b/providers/gitlab/models/duo-chat-opus-5-5.toml new file mode 100644 index 00000000000..acfba463a94 --- /dev/null +++ b/providers/gitlab/models/duo-chat-opus-5-5.toml @@ -0,0 +1,9 @@ +base_model = "anthropic/claude-opus-5-5" +name = "Agentic Chat (Claude Opus 5.5)" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 diff --git a/providers/google-vertex-anthropic/models/claude-opus-5-5@default.toml b/providers/google-vertex-anthropic/models/claude-opus-5-5@default.toml new file mode 100644 index 00000000000..cede54eca95 --- /dev/null +++ b/providers/google-vertex-anthropic/models/claude-opus-5-5@default.toml @@ -0,0 +1,9 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/google-vertex/models/claude-opus-5-5@default.toml b/providers/google-vertex/models/claude-opus-5-5@default.toml new file mode 100644 index 00000000000..8a2a18a47c1 --- /dev/null +++ b/providers/google-vertex/models/claude-opus-5-5@default.toml @@ -0,0 +1,12 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[provider] +npm = "@ai-sdk/google-vertex/anthropic" diff --git a/providers/huggingface/models/tencent/Hy4-preview.toml b/providers/huggingface/models/tencent/Hy4-preview.toml new file mode 100644 index 00000000000..5a2ca915a5b --- /dev/null +++ b/providers/huggingface/models/tencent/Hy4-preview.toml @@ -0,0 +1,13 @@ +# Effort: reasoning_effort = none|high; high is the default. +# https://huggingface.co/tencent/Hy4-preview#quickstart +base_model = "tencent/hy4-preview" +description = "Tencent Hy reasoning model for coding, instruction following, and agent tasks" +structured_output = true +reasoning_options = [{ type = "effort", values = ["none", "high"] }] + +[cost] +input = 0.834 +output = 2.501 + +[limit] +context = 1_000_000 diff --git a/providers/hyper/models/deepseek-v4.1-flash.toml b/providers/hyper/models/deepseek-v4.1-flash.toml index c703c4fcf0c..7fd6fdfb3b1 100644 --- a/providers/hyper/models/deepseek-v4.1-flash.toml +++ b/providers/hyper/models/deepseek-v4.1-flash.toml @@ -2,7 +2,7 @@ base_model = "deepseek/deepseek-v4.1-flash" [[reasoning_options]] type = "effort" -values = ["low", "high", "max"] +values = ["low", "high", "xhigh"] [cost] input = 0.3 @@ -10,4 +10,5 @@ output = 1.2 cache_read = 0.03 [limit] +context = 1_048_576 output = 32_768 diff --git a/providers/kenari/models/deepseek-v4-1-flash.toml b/providers/kenari/models/deepseek-v4-1-flash.toml new file mode 100644 index 00000000000..2a775a81e1f --- /dev/null +++ b/providers/kenari/models/deepseek-v4-1-flash.toml @@ -0,0 +1,11 @@ +# Source: https://kenari.id/v1/models (id deepseek-v4-1-flash, reasoning low/high/max) +# Kenari convention: subscription pricing, cost 0/0 like other Kenari entries. +base_model = "deepseek/deepseek-v4.1-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 diff --git a/providers/kilo/models/anthropic/claude-opus-4.toml b/providers/kilo/models/anthropic/claude-opus-4.toml deleted file mode 100644 index 6e622f068ab..00000000000 --- a/providers/kilo/models/anthropic/claude-opus-4.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Anthropic: Claude Opus 4 ($$$$)" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -family = "claude-opus" -release_date = "2025-05-22" -last_updated = "2025-05-22" -attachment = true -reasoning = true -temperature = true -tool_call = true -structured_output = false -open_weights = false - -[[reasoning_options]] -type = "effort" -values = ["none", "high"] - -[cost] -input = 15 -output = 75 -cache_read = 1.5 -cache_write = 18.75 - -[limit] -context = 200_000 -output = 32_000 - -[modalities] -input = ["image", "text", "pdf"] -output = ["text"] diff --git a/providers/kilo/models/anthropic/claude-opus-5.5.toml b/providers/kilo/models/anthropic/claude-opus-5.5.toml new file mode 100644 index 00000000000..53b1a0d9fb2 --- /dev/null +++ b/providers/kilo/models/anthropic/claude-opus-5.5.toml @@ -0,0 +1,14 @@ +base_model = "anthropic/claude-opus-5-5" +description = "Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code..." +temperature = true +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/kilo/models/baidu/ernie-4.5-vl-424b-a47b.toml b/providers/kilo/models/baidu/ernie-4.5-vl-424b-a47b.toml index a962a373363..ef65a04722c 100644 --- a/providers/kilo/models/baidu/ernie-4.5-vl-424b-a47b.toml +++ b/providers/kilo/models/baidu/ernie-4.5-vl-424b-a47b.toml @@ -1,4 +1,4 @@ -name = "Baidu: ERNIE 4.5 VL 424B A47B " +name = "Baidu: ERNIE 4.5 VL 424B A47B (retires Oct 8)" description = "Multimodal reasoning model for visual analysis, planning, and tool use" family = "ernie" release_date = "2025-06-30" diff --git a/providers/kilo/models/cohere/command-a-plus.toml b/providers/kilo/models/cohere/command-a-plus.toml new file mode 100644 index 00000000000..dcee2ea1432 --- /dev/null +++ b/providers/kilo/models/cohere/command-a-plus.toml @@ -0,0 +1,28 @@ +name = "Cohere: Command A+" +description = "Command A+ is Cohere's flagship model for enterprise agentic workflows. It accepts text and image inputs with a 192K context window, supports native tool calling with strict tool schemas, structured..." +family = "command-a" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.3 +output = 1.5 +cache_read = 0.15 + +[limit] +context = 192_000 +output = 64_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/kilo/models/deepseek/deepseek-r1-distill-llama-70b.toml b/providers/kilo/models/deepseek/deepseek-r1-distill-llama-70b.toml index 3635f13d091..18d5e9a099b 100644 --- a/providers/kilo/models/deepseek/deepseek-r1-distill-llama-70b.toml +++ b/providers/kilo/models/deepseek/deepseek-r1-distill-llama-70b.toml @@ -1,4 +1,4 @@ -name = "DeepSeek: R1 Distill Llama 70B" +name = "DeepSeek: R1 Distill Llama 70B (retires Sep 28)" description = "DeepSeek reasoning model for multi-step analysis, math, coding, and tools" family = "deepseek" release_date = "2025-01-23" diff --git a/providers/kilo/models/deepseek/deepseek-v3.1-terminus.toml b/providers/kilo/models/deepseek/deepseek-v3.1-terminus.toml index 79663dcfa0f..0715bd02e00 100644 --- a/providers/kilo/models/deepseek/deepseek-v3.1-terminus.toml +++ b/providers/kilo/models/deepseek/deepseek-v3.1-terminus.toml @@ -1,4 +1,4 @@ -name = "DeepSeek: DeepSeek V3.1 Terminus" +name = "DeepSeek: DeepSeek V3.1 Terminus (retires Sep 28)" description = "DeepSeek chat model for instruction following, coding, and analysis" family = "deepseek" release_date = "2025-09-22" diff --git a/providers/kilo/models/deepseek/deepseek-v3.2-exp.toml b/providers/kilo/models/deepseek/deepseek-v3.2-exp.toml index 7e9c21db938..0ac45019600 100644 --- a/providers/kilo/models/deepseek/deepseek-v3.2-exp.toml +++ b/providers/kilo/models/deepseek/deepseek-v3.2-exp.toml @@ -1,4 +1,4 @@ -name = "DeepSeek: DeepSeek V3.2 Exp" +name = "DeepSeek: DeepSeek V3.2 Exp (retires Sep 28)" description = "DeepSeek chat model for instruction following, coding, and analysis" family = "deepseek" release_date = "2025-09-29" diff --git a/providers/kilo/models/deepseek/deepseek-v4-flash-0731:free.toml b/providers/kilo/models/deepseek/deepseek-v4-flash-0731:free.toml deleted file mode 100644 index 73a49138fbe..00000000000 --- a/providers/kilo/models/deepseek/deepseek-v4-flash-0731:free.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "deepseek/deepseek-v4-flash-0731" -name = "DeepSeek: DeepSeek V4 Flash 0731 (free)" -description = "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows...." - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "high", "max"] - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 -output = 393_216 diff --git a/providers/kilo/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/kilo/models/deepseek/deepseek-v4-flash-vision-exp.toml index 41e3e4d74b4..a280435f49f 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -12,4 +12,4 @@ cache_read = 0.028 [limit] context = 1_048_576 -output = 262_144 +output = 943_718 diff --git a/providers/kilo/models/deepseek/deepseek-v4-pro.toml b/providers/kilo/models/deepseek/deepseek-v4-pro.toml index e58e24281d4..88500bd7dd8 100644 --- a/providers/kilo/models/deepseek/deepseek-v4-pro.toml +++ b/providers/kilo/models/deepseek/deepseek-v4-pro.toml @@ -11,5 +11,4 @@ output = 3.2 cache_read = 0.135 [limit] -context = 1_048_576 -output = 393_216 +context = 1_024_000 diff --git a/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml b/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml index 2d43303aad0..d755c215bcf 100644 --- a/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/kilo/models/deepseek/deepseek-v4.1-flash.toml @@ -12,3 +12,4 @@ cache_read = 0.006 [limit] context = 1_048_576 +output = 943_718 diff --git a/providers/kilo/models/kwaipilot/kat-coder-pro-v2.toml b/providers/kilo/models/kwaipilot/kat-coder-pro-v2.toml deleted file mode 100644 index 8e2499749ae..00000000000 --- a/providers/kilo/models/kwaipilot/kat-coder-pro-v2.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Kwaipilot: KAT-Coder-Pro V2" -description = "Coding model for repository understanding, refactors, and agentic engineering tasks" -family = "kat-coder" -release_date = "2026-03-27" -last_updated = "2026-03-27" -attachment = false -reasoning = false -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.3 -output = 1.2 -cache_read = 0.06 - -[limit] -context = 262_144 -output = 144_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/kilo/models/meta/muse-glimmer-30b.toml b/providers/kilo/models/meta/muse-glimmer-30b.toml index 2c352c0d845..aa01201020a 100644 --- a/providers/kilo/models/meta/muse-glimmer-30b.toml +++ b/providers/kilo/models/meta/muse-glimmer-30b.toml @@ -11,7 +11,7 @@ output = 1.1 cache_read = 0.04 [limit] -output = 117_964 +output = 16_384 [modalities] input = ["text", "image", "pdf"] diff --git a/providers/kilo/models/mistralai/mistral-small-3.1-24b-instruct.toml b/providers/kilo/models/mistralai/mistral-small-3.1-24b-instruct.toml index e5a92d90a1c..cddc0eb9495 100644 --- a/providers/kilo/models/mistralai/mistral-small-3.1-24b-instruct.toml +++ b/providers/kilo/models/mistralai/mistral-small-3.1-24b-instruct.toml @@ -6,7 +6,7 @@ last_updated = "2025-03-17" attachment = true reasoning = false temperature = true -tool_call = false +tool_call = true structured_output = false open_weights = false diff --git a/providers/kilo/models/moonshotai/kimi-k3.toml b/providers/kilo/models/moonshotai/kimi-k3.toml index 3af9cc14392..1ebab0a52b7 100644 --- a/providers/kilo/models/moonshotai/kimi-k3.toml +++ b/providers/kilo/models/moonshotai/kimi-k3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 3 +output = 15 +cache_read = 0.3 [limit] output = 943_718 diff --git a/providers/kilo/models/nex-agi/nex-n2.5-mini.toml b/providers/kilo/models/nex-agi/nex-n2.5-mini.toml new file mode 100644 index 00000000000..04ca31fff1a --- /dev/null +++ b/providers/kilo/models/nex-agi/nex-n2.5-mini.toml @@ -0,0 +1,28 @@ +name = "Nex AGI: Nex-N2.5-Mini" +description = "Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file..." +family = "agi" +release_date = "2026-09-08" +last_updated = "2026-09-08" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "medium", "high"] + +[cost] +input = 0.025 +output = 0.1 +cache_read = 0.0025 + +[limit] +context = 262_144 +output = 235_929 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/kilo/models/nex-agi/nex-n2.5-pro.toml b/providers/kilo/models/nex-agi/nex-n2.5-pro.toml new file mode 100644 index 00000000000..e565d70e36f --- /dev/null +++ b/providers/kilo/models/nex-agi/nex-n2.5-pro.toml @@ -0,0 +1,28 @@ +name = "Nex AGI: Nex-N2.5-Pro" +description = "Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file..." +family = "agi" +release_date = "2026-09-08" +last_updated = "2026-09-08" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "medium", "high"] + +[cost] +input = 0.075 +output = 0.25 +cache_read = 0.015 + +[limit] +context = 262_144 +output = 235_929 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/kilo/models/nvidia/nemotron-3-ultra-550b-a55b.toml b/providers/kilo/models/nvidia/nemotron-3-ultra-550b-a55b.toml index 212e280b8a9..25c72a0b0fb 100644 --- a/providers/kilo/models/nvidia/nemotron-3-ultra-550b-a55b.toml +++ b/providers/kilo/models/nvidia/nemotron-3-ultra-550b-a55b.toml @@ -12,5 +12,5 @@ output = 2.2 cache_read = 0.1 [limit] -context = 256_000 -output = 32_768 +context = 202_800 +output = 182_520 diff --git a/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml b/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml index 78104d7319f..5244dc1ac02 100644 --- a/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml +++ b/providers/kilo/models/nvidia/nemotron-3.5-lightning.toml @@ -10,4 +10,4 @@ input = 0.065 output = 0.18 [limit] -output = 131_072 +output = 235_929 diff --git a/providers/kilo/models/openai/gpt-6-luna-pro.toml b/providers/kilo/models/openai/gpt-6-luna-pro.toml new file mode 100644 index 00000000000..3e57329bcf1 --- /dev/null +++ b/providers/kilo/models/openai/gpt-6-luna-pro.toml @@ -0,0 +1,29 @@ +name = "OpenAI: GPT-6 Luna Pro" +description = "GPT-6 Luna Pro is the same underlying model as [GPT-6 Luna](https://openrouter.ai/openai/gpt-6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode" +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/kilo/models/openai/gpt-6-luna.toml b/providers/kilo/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..9e9940bc126 --- /dev/null +++ b/providers/kilo/models/openai/gpt-6-luna.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-6-luna" +description = "GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic..." + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 diff --git a/providers/kilo/models/openai/gpt-6-sol-pro.toml b/providers/kilo/models/openai/gpt-6-sol-pro.toml new file mode 100644 index 00000000000..f1d263d64fa --- /dev/null +++ b/providers/kilo/models/openai/gpt-6-sol-pro.toml @@ -0,0 +1,29 @@ +name = "OpenAI: GPT-6 Sol Pro" +description = "GPT-6 Sol Pro is the same underlying model as [GPT-6 Sol](https://openrouter.ai/openai/gpt-6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode" +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/kilo/models/openai/gpt-6-sol.toml b/providers/kilo/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..5960d039080 --- /dev/null +++ b/providers/kilo/models/openai/gpt-6-sol.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-6-sol" +description = "GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional..." + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 diff --git a/providers/kilo/models/openai/gpt-oss-20b.toml b/providers/kilo/models/openai/gpt-oss-20b.toml index 895f7fb4602..98aeb108997 100644 --- a/providers/kilo/models/openai/gpt-oss-20b.toml +++ b/providers/kilo/models/openai/gpt-oss-20b.toml @@ -6,8 +6,5 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.02 -output = 0.1 - -[limit] -output = 117_964 +input = 0.018 +output = 0.09 diff --git a/providers/kilo/models/prism-ml/ternary-bonsai-2-27b.toml b/providers/kilo/models/prism-ml/ternary-bonsai-2-27b.toml new file mode 100644 index 00000000000..a1eb487d1bb --- /dev/null +++ b/providers/kilo/models/prism-ml/ternary-bonsai-2-27b.toml @@ -0,0 +1,26 @@ +name = "PrismML: Ternary Bonsai 2 27B" +description = "Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window. Ternary compression shrinks..." +release_date = "2026-09-18" +last_updated = "2026-09-18" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "medium", "xhigh"] + +[cost] +input = 0.075 +output = 0.5 + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/kilo/models/qwen/qwen3-next-80b-a3b-thinking.toml b/providers/kilo/models/qwen/qwen3-next-80b-a3b-thinking.toml index 7c76fb082af..5d67b671b7a 100644 --- a/providers/kilo/models/qwen/qwen3-next-80b-a3b-thinking.toml +++ b/providers/kilo/models/qwen/qwen3-next-80b-a3b-thinking.toml @@ -9,3 +9,7 @@ values = ["high"] [cost] input = 0.15 output = 1.2 + +[limit] +context = 262_144 +output = 235_929 diff --git a/providers/kilo/models/qwen/qwen3.5-35b-a3b.toml b/providers/kilo/models/qwen/qwen3.5-35b-a3b.toml index 96076003f63..bdebd95519c 100644 --- a/providers/kilo/models/qwen/qwen3.5-35b-a3b.toml +++ b/providers/kilo/models/qwen/qwen3.5-35b-a3b.toml @@ -9,5 +9,9 @@ values = ["none", "high"] input = 0.1625 output = 1.3 +[limit] +context = 256_000 +output = 16_384 + [modalities] input = ["text", "image", "video"] diff --git a/providers/kilo/models/qwen/qwen3.5-9b.toml b/providers/kilo/models/qwen/qwen3.5-9b.toml index bc4a13e3e11..7c1d5856e31 100644 --- a/providers/kilo/models/qwen/qwen3.5-9b.toml +++ b/providers/kilo/models/qwen/qwen3.5-9b.toml @@ -10,4 +10,5 @@ input = 0.1 output = 0.15 [limit] -output = 235_929 +context = 256_000 +output = 32_768 diff --git a/providers/kilo/models/qwen/qwen3.6-27b.toml b/providers/kilo/models/qwen/qwen3.6-27b.toml index 61085517bb0..6ca4eb2df17 100644 --- a/providers/kilo/models/qwen/qwen3.6-27b.toml +++ b/providers/kilo/models/qwen/qwen3.6-27b.toml @@ -9,5 +9,8 @@ values = ["none", "high"] input = 0.45 output = 2.7 +[limit] +output = 262_140 + [modalities] input = ["text", "image", "video"] diff --git a/providers/kilo/models/qwen/qwen3.6-35b-a3b.toml b/providers/kilo/models/qwen/qwen3.6-35b-a3b.toml index 1385363812a..21f9408b4a8 100644 --- a/providers/kilo/models/qwen/qwen3.6-35b-a3b.toml +++ b/providers/kilo/models/qwen/qwen3.6-35b-a3b.toml @@ -6,8 +6,8 @@ type = "effort" values = ["none", "high"] [cost] -input = 0.1 -output = 0.9 +input = 0.15 +output = 1 cache_read = 0.05 [limit] diff --git a/providers/kilo/models/qwen/qwen3.8-27b.toml b/providers/kilo/models/qwen/qwen3.8-27b.toml index 9b18fb98b33..820f69fb194 100644 --- a/providers/kilo/models/qwen/qwen3.8-27b.toml +++ b/providers/kilo/models/qwen/qwen3.8-27b.toml @@ -12,4 +12,5 @@ cache_read = 0.085 cache_write = 0.53125 [limit] +context = 1_000_000 output = 131_072 diff --git a/providers/kilo/models/qwen/qwen3.8-omni-flash.toml b/providers/kilo/models/qwen/qwen3.8-omni-flash.toml new file mode 100644 index 00000000000..574800249bc --- /dev/null +++ b/providers/kilo/models/qwen/qwen3.8-omni-flash.toml @@ -0,0 +1,13 @@ +base_model = "alibaba/qwen3.8-omni-flash" +description = "Qwen3.8 Omni Flash is an omni-modal reasoning model from Alibaba, the first Qwen model built around agentic capabilities with native audio-video understanding. It is suited for audio-video analysis and summarization,..." +temperature = true +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.15 +output = 0.47 +cache_read = 0.016 diff --git a/providers/kilo/models/tencent/hy3.toml b/providers/kilo/models/tencent/hy3.toml index 8ea8b76bcff..c3469c1e9dc 100644 --- a/providers/kilo/models/tencent/hy3.toml +++ b/providers/kilo/models/tencent/hy3.toml @@ -7,9 +7,9 @@ type = "effort" values = ["none", "low", "high"] [cost] -input = 0.14 -output = 0.58 -cache_read = 0.035 +input = 0.13 +output = 0.53 +cache_read = 0.033 [limit] context = 262_144 diff --git a/providers/kilo/models/upstage/solar-pro-3.toml b/providers/kilo/models/upstage/solar-pro-3.toml index 54205b2bb45..ad7adb31fe8 100644 --- a/providers/kilo/models/upstage/solar-pro-3.toml +++ b/providers/kilo/models/upstage/solar-pro-3.toml @@ -12,7 +12,7 @@ open_weights = false [[reasoning_options]] type = "effort" -values = ["none", "high"] +values = ["none", "minimal", "low", "medium", "high"] [cost] input = 0.15 diff --git a/providers/kilo/models/upstage/solar-pro4.toml b/providers/kilo/models/upstage/solar-pro4.toml index 4b483744118..d3486cf5460 100644 --- a/providers/kilo/models/upstage/solar-pro4.toml +++ b/providers/kilo/models/upstage/solar-pro4.toml @@ -12,7 +12,7 @@ open_weights = false [[reasoning_options]] type = "effort" -values = ["none", "high"] +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 0.3 diff --git a/providers/kilo/models/x-ai/grok-4.7.toml b/providers/kilo/models/x-ai/grok-4.7.toml new file mode 100644 index 00000000000..7e904bb941d --- /dev/null +++ b/providers/kilo/models/x-ai/grok-4.7.toml @@ -0,0 +1,14 @@ +base_model = "xai/grok-4.7" +description = "Grok 4.7 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM." + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 1.6 +output = 4.8 +cache_read = 0.4 + +[limit] +output = 450_000 diff --git a/providers/kilo/models/xiaomi/mimo-v2.6-flash.toml b/providers/kilo/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..0a1c6e4fe63 --- /dev/null +++ b/providers/kilo/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,12 @@ +base_model = "xiaomi/mimo-v2.6-flash" +description = "MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for..." +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 diff --git a/providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..ad9bc52de83 --- /dev/null +++ b/providers/kilo/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,12 @@ +base_model = "xiaomi/mimo-v2.6-pro-ultraspeed" +description = "MiMo-V2.6-Pro-UltraSpeed is the fast speed edition of Xiaomi's flagship foundation model, MiMo-V2.6-Pro. Built from the same 1T MiMo-V2.6-Pro checkpoint, it matches the original model in quality while delivering roughly 10x..." +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 4.35 +output = 8.7 +cache_read = 0.036 diff --git a/providers/kilo/models/xiaomi/mimo-v2.6-pro.toml b/providers/kilo/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..1285f109afd --- /dev/null +++ b/providers/kilo/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,12 @@ +base_model = "xiaomi/mimo-v2.6-pro" +description = "MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding..." +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 diff --git a/providers/kilo/models/z-ai/glm-5.2.toml b/providers/kilo/models/z-ai/glm-5.2.toml index f8addeabfc8..bb89edf75cf 100644 --- a/providers/kilo/models/z-ai/glm-5.2.toml +++ b/providers/kilo/models/z-ai/glm-5.2.toml @@ -12,4 +12,3 @@ cache_read = 0.26 [limit] context = 1_048_576 -output = 163_840 diff --git a/providers/kilo/models/z-ai/glm-5.3-flash.toml b/providers/kilo/models/z-ai/glm-5.3-flash.toml index 0104ea56ade..c464db465b8 100644 --- a/providers/kilo/models/z-ai/glm-5.3-flash.toml +++ b/providers/kilo/models/z-ai/glm-5.3-flash.toml @@ -12,6 +12,7 @@ cache_read = 0.03 [limit] context = 1_048_576 +output = 943_718 [modalities] input = ["text", "image", "video"] diff --git a/providers/kilo/models/z-ai/glm-5.3-flashx.toml b/providers/kilo/models/z-ai/glm-5.3-flashx.toml new file mode 100644 index 00000000000..a730a63619c --- /dev/null +++ b/providers/kilo/models/z-ai/glm-5.3-flashx.toml @@ -0,0 +1,28 @@ +name = "Z.ai: GLM 5.3 FlashX" +description = "GLM-5.3-FlashX is the high-speed variant of Z.ai's GLM-5.3-Flash, a native multimodal model delivering inference speeds of up to 200 tokens/s. Built on the same hybrid sparse and linear attention architecture..." +family = "glm" +release_date = "2026-09-18" +last_updated = "2026-09-18" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.37 +output = 1.25 +cache_read = 0.075 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/kilo/models/z-ai/glm-5.3.toml b/providers/kilo/models/z-ai/glm-5.3.toml index 70fec865ada..e9d5bdad891 100644 --- a/providers/kilo/models/z-ai/glm-5.3.toml +++ b/providers/kilo/models/z-ai/glm-5.3.toml @@ -11,5 +11,4 @@ output = 4.4 cache_read = 0.26 [limit] -context = 1_048_575 -output = 943_717 +context = 1_048_576 diff --git a/providers/kilo/models/~anthropic/claude-opus-latest.toml b/providers/kilo/models/~anthropic/claude-opus-latest.toml index 7eae0b14770..709f3bb1f1b 100644 --- a/providers/kilo/models/~anthropic/claude-opus-latest.toml +++ b/providers/kilo/models/~anthropic/claude-opus-latest.toml @@ -12,13 +12,13 @@ open_weights = false [[reasoning_options]] type = "effort" -values = ["none", "low", "medium", "high", "xhigh", "max"] +values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 5 -output = 25 -cache_read = 0.5 -cache_write = 6.25 +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 [limit] context = 1_000_000 diff --git a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml index 1a05f7dcedb..6b3302e301e 100644 --- a/providers/kilo/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.1485 -output = 0.594 -cache_read = 0.004455 +input = 0.039 +output = 0.6 +cache_read = 0.038 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml index b14064bf897..7666171face 100644 --- a/providers/kilo/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-pro-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.4 +output = 4.3 +cache_read = 0.033 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml index 396a029912a..0412502ff5d 100644 --- a/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/kilo/models/~deepseek/deepseek-v4-flash-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.0558 -output = 0.1767 -cache_read = 0.0088 +input = 0.03 +output = 0.8 +cache_read = 0.008 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~moonshotai/kimi-latest.toml b/providers/kilo/models/~moonshotai/kimi-latest.toml index 9bbf3c7be77..163a53ae425 100644 --- a/providers/kilo/models/~moonshotai/kimi-latest.toml +++ b/providers/kilo/models/~moonshotai/kimi-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 1.49 +output = 14.5 +cache_read = 0.21 [limit] context = 1_048_576 diff --git a/providers/kilo/models/~openai/gpt-luna-latest.toml b/providers/kilo/models/~openai/gpt-luna-latest.toml index 6f3028a0af2..d593f0aca35 100644 --- a/providers/kilo/models/~openai/gpt-luna-latest.toml +++ b/providers/kilo/models/~openai/gpt-luna-latest.toml @@ -15,10 +15,10 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 0.2 -output = 1.2 -cache_read = 0.02 -cache_write = 0.25 +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 [limit] context = 1_050_000 diff --git a/providers/kilo/models/~x-ai/grok-latest.toml b/providers/kilo/models/~x-ai/grok-latest.toml index 5f182ec62c7..85f00745faa 100644 --- a/providers/kilo/models/~x-ai/grok-latest.toml +++ b/providers/kilo/models/~x-ai/grok-latest.toml @@ -15,9 +15,9 @@ type = "effort" values = ["low", "medium", "high", "xhigh"] [cost] -input = 2 -output = 6 -cache_read = 0.5 +input = 1.6 +output = 4.8 +cache_read = 0.4 [limit] context = 500_000 diff --git a/providers/kilo/models/~z-ai/glm-latest.toml b/providers/kilo/models/~z-ai/glm-latest.toml index 38aa9fed430..2ce00d3e342 100644 --- a/providers/kilo/models/~z-ai/glm-latest.toml +++ b/providers/kilo/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8775 -output = 2.97 -cache_read = 0.1755 +input = 0.5625 +output = 2.5 +cache_read = 0.125 [limit] -context = 262_144 -output = 235_929 +context = 1_048_576 +output = 131_072 [modalities] input = ["text"] diff --git a/providers/kimi-for-coding/logo.svg b/providers/kimi-code-plan-cn/logo.svg similarity index 100% rename from providers/kimi-for-coding/logo.svg rename to providers/kimi-code-plan-cn/logo.svg diff --git a/providers/kimi-for-coding/models/k3-256k.toml b/providers/kimi-code-plan-cn/models/k3-256k.toml similarity index 93% rename from providers/kimi-for-coding/models/k3-256k.toml rename to providers/kimi-code-plan-cn/models/k3-256k.toml index 2e7cd5d2d1f..3d9c4ba810e 100644 --- a/providers/kimi-for-coding/models/k3-256k.toml +++ b/providers/kimi-code-plan-cn/models/k3-256k.toml @@ -22,3 +22,6 @@ context = 262_144 [modalities] input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-for-coding/models/k3.toml b/providers/kimi-code-plan-cn/models/k3.toml similarity index 91% rename from providers/kimi-for-coding/models/k3.toml rename to providers/kimi-code-plan-cn/models/k3.toml index dad0dadecd8..d0dbb42ade9 100644 --- a/providers/kimi-for-coding/models/k3.toml +++ b/providers/kimi-code-plan-cn/models/k3.toml @@ -16,3 +16,6 @@ input = 0 output = 0 cache_read = 0 cache_write = 0 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-for-coding/models/kimi-for-coding-highspeed.toml b/providers/kimi-code-plan-cn/models/kimi-for-coding-highspeed.toml similarity index 81% rename from providers/kimi-for-coding/models/kimi-for-coding-highspeed.toml rename to providers/kimi-code-plan-cn/models/kimi-for-coding-highspeed.toml index 32d5102804a..407bcf710f8 100644 --- a/providers/kimi-for-coding/models/kimi-for-coding-highspeed.toml +++ b/providers/kimi-code-plan-cn/models/kimi-for-coding-highspeed.toml @@ -10,3 +10,6 @@ cache_write = 0 [limit] output = 32_768 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-for-coding/models/kimi-for-coding.toml b/providers/kimi-code-plan-cn/models/kimi-for-coding.toml similarity index 73% rename from providers/kimi-for-coding/models/kimi-for-coding.toml rename to providers/kimi-code-plan-cn/models/kimi-for-coding.toml index 2dc81d0d56d..570fea3fd8a 100644 --- a/providers/kimi-for-coding/models/kimi-for-coding.toml +++ b/providers/kimi-code-plan-cn/models/kimi-for-coding.toml @@ -1,4 +1,6 @@ # https://www.kimi.com/code/docs/en/third-party-tools/opencode.html +# OpenAI-compatible chat completions on api.kimi.com (verified 2026-09-18 with +# kimi-code OAuth credentials; reasoning returned via reasoning_content). # Toggle: thinking.type = "enabled" | "disabled" | "adaptive" # Effort: output_config.effort = "low" | "high" | "max" (default: "max") # Retain the existing output limit; the K2.8 announcement only specifies context. @@ -20,3 +22,6 @@ cache_write = 0 [limit] output = 32_768 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-code-plan-cn/provider.toml b/providers/kimi-code-plan-cn/provider.toml new file mode 100644 index 00000000000..ab7b8962d83 --- /dev/null +++ b/providers/kimi-code-plan-cn/provider.toml @@ -0,0 +1,10 @@ +# NOTE: api.kimi.com/coding serves BOTH Anthropic (/v1/messages) and +# OpenAI-compatible (/v1/chat/completions) protocols. All four models +# verified on /chat/completions 2026-09-18 with kimi-code OAuth credentials +# (reasoning returned via reasoning_content). The global deployment +# (api.kimi.ai) is the separate kimi-code-plan-global provider. +name = "Kimi For Coding (kimi.com)" +env = ["KIMI_API_KEY"] +npm = "@ai-sdk/openai-compatible" +doc = "https://www.kimi.com/code/docs/en/kimi-code/models.html" +api = "https://api.kimi.com/coding/v1" diff --git a/providers/kimi-code-plan-global/logo.svg b/providers/kimi-code-plan-global/logo.svg new file mode 100644 index 00000000000..d41029890d2 --- /dev/null +++ b/providers/kimi-code-plan-global/logo.svg @@ -0,0 +1,3 @@ + + + diff --git a/providers/kimi-code-plan-global/models/k3-256k.toml b/providers/kimi-code-plan-global/models/k3-256k.toml new file mode 100644 index 00000000000..36af21989bd --- /dev/null +++ b/providers/kimi-code-plan-global/models/k3-256k.toml @@ -0,0 +1,29 @@ +# Verified 2026-09-18 against https://api.kimi.ai/coding/v1/chat/completions +# with kimi-code OAuth credentials; reasoning returned via reasoning_content. +# k3-256k is the 256K-context variant of Kimi K3 on the Kimi For Coding +# endpoint. Unlike full K3 it accepts no video input (image only). +# reasoning_options per the Kimi Code docs: +# reasoning_effort = "low" | "high" | "max" (default "high") +# Cost is zeroed per this provider's subscription convention. +base_model = "moonshotai/kimi-k3" +name = "Kimi K3-256K" +description = "256K-context version of Kimi K3, reducing token consumption for shorter coding sessions" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 + +[limit] +context = 262_144 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-code-plan-global/models/k3.toml b/providers/kimi-code-plan-global/models/k3.toml new file mode 100644 index 00000000000..e2cc98df82b --- /dev/null +++ b/providers/kimi-code-plan-global/models/k3.toml @@ -0,0 +1,23 @@ +# Verified 2026-09-18 against https://api.kimi.ai/coding/v1/chat/completions +# with kimi-code OAuth credentials; reasoning returned via reasoning_content. +# reasoning_options mirror the Moonshot AI platform API surface: +# thinking.type = "enabled" | "disabled" | "adaptive" +# output_config.effort = "low" | "high" | "max" +# Cost is zeroed per this provider's subscription convention. +base_model = "moonshotai/kimi-k3" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-code-plan-global/models/kimi-for-coding-highspeed.toml b/providers/kimi-code-plan-global/models/kimi-for-coding-highspeed.toml new file mode 100644 index 00000000000..ff4e80c9104 --- /dev/null +++ b/providers/kimi-code-plan-global/models/kimi-for-coding-highspeed.toml @@ -0,0 +1,17 @@ +# Verified 2026-09-18 against https://api.kimi.ai/coding/v1/chat/completions +# with kimi-code OAuth credentials; reasoning returned via reasoning_content. +base_model = "moonshotai/kimi-k2.7-code-highspeed" +name = "Kimi For Coding HighSpeed" +reasoning_options = [] + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 + +[limit] +output = 32_768 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-code-plan-global/models/kimi-for-coding.toml b/providers/kimi-code-plan-global/models/kimi-for-coding.toml new file mode 100644 index 00000000000..766cc5bfb99 --- /dev/null +++ b/providers/kimi-code-plan-global/models/kimi-for-coding.toml @@ -0,0 +1,25 @@ +# Verified 2026-09-18 against https://api.kimi.ai/coding/v1/chat/completions +# with kimi-code OAuth credentials; reasoning returned via reasoning_content. +# Toggle: thinking.type = "enabled" | "disabled" | "adaptive" +# Effort: output_config.effort = "low" | "high" | "max" (default: "max") +base_model = "moonshotai/kimi-k2.8-preview" +name = "kimi-for-coding" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 + +[limit] +output = 32_768 + +[interleaved] +field = "reasoning_content" diff --git a/providers/kimi-code-plan-global/provider.toml b/providers/kimi-code-plan-global/provider.toml new file mode 100644 index 00000000000..cf4da1d06d8 --- /dev/null +++ b/providers/kimi-code-plan-global/provider.toml @@ -0,0 +1,9 @@ +# Global deployment of Kimi For Coding on api.kimi.ai (separate host and OAuth +# issuer auth.kimi.ai). OpenAI-compatible chat completions; catalog verified +# via GET /models on 2026-09-18 (kimi-for-coding, kimi-for-coding-highspeed, +# k3, k3-256k) with kimi-code OAuth credentials. +name = "Kimi For Coding (kimi.ai)" +env = ["KIMI_API_KEY"] +npm = "@ai-sdk/openai-compatible" +doc = "https://www.kimi.ai/code/docs/en/kimi-code/models.html" +api = "https://api.kimi.ai/coding/v1" diff --git a/providers/kimi-for-coding/provider.toml b/providers/kimi-for-coding/provider.toml deleted file mode 100644 index f8d71f33de3..00000000000 --- a/providers/kimi-for-coding/provider.toml +++ /dev/null @@ -1,10 +0,0 @@ -# NOTE: api.kimi.com/coding serves BOTH Anthropic (/v1/messages) and -# OpenAI-compatible (/v1/chat/completions) protocols (verified 2026-07). -# Registered as @ai-sdk/anthropic: the Messages surface is the officially -# documented one (see doc) and round-trips reasoning natively via -# thinking blocks. -name = "Kimi For Coding" -env = ["KIMI_API_KEY"] -npm = "@ai-sdk/anthropic" -doc = "https://www.kimi.com/code/docs/en/kimi-code/models.html" -api = "https://api.kimi.com/coding/v1" diff --git a/providers/llmgateway-providers/models/anthropic/claude-opus-5-5.toml b/providers/llmgateway-providers/models/anthropic/claude-opus-5-5.toml new file mode 100644 index 00000000000..090b539ee3e --- /dev/null +++ b/providers/llmgateway-providers/models/anthropic/claude-opus-5-5.toml @@ -0,0 +1,13 @@ +base_model = "anthropic/claude-opus-5-5" +name = "Claude Opus 5.5 (Anthropic)" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/llmgateway-providers/models/baidu/deepseek-v4-flash.toml b/providers/llmgateway-providers/models/baidu/deepseek-v4-flash.toml index 95bedd8498e..40383b0961a 100644 --- a/providers/llmgateway-providers/models/baidu/deepseek-v4-flash.toml +++ b/providers/llmgateway-providers/models/baidu/deepseek-v4-flash.toml @@ -12,7 +12,7 @@ values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 0.44 output = 1.32 -cache_read = 0.044 +cache_read = 0.014 [limit] context = 1_048_576 diff --git a/providers/llmgateway-providers/models/baidu/deepseek-v4-pro.toml b/providers/llmgateway-providers/models/baidu/deepseek-v4-pro.toml index bab1be34eeb..e9463e0fe2a 100644 --- a/providers/llmgateway-providers/models/baidu/deepseek-v4-pro.toml +++ b/providers/llmgateway-providers/models/baidu/deepseek-v4-pro.toml @@ -12,7 +12,7 @@ values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 1.32 output = 3.96 -cache_read = 0.132 +cache_read = 0.042 [limit] context = 1_048_576 diff --git a/providers/llmgateway-providers/models/baidu/deepseek-v4.1-flash.toml b/providers/llmgateway-providers/models/baidu/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..a22c803e3f5 --- /dev/null +++ b/providers/llmgateway-providers/models/baidu/deepseek-v4.1-flash.toml @@ -0,0 +1,15 @@ +base_model = "deepseek/deepseek-v4.1-flash" +name = "DeepSeek V4.1 Flash (Baidu)" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.006 + +[limit] +context = 1_048_576 +output = 393_216 diff --git a/providers/llmgateway-providers/models/deepinfra/glm-5.3.toml b/providers/llmgateway-providers/models/deepinfra/glm-5.3.toml new file mode 100644 index 00000000000..49f3f113f39 --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/glm-5.3.toml @@ -0,0 +1,14 @@ +base_model = "zhipuai/glm-5.3" +name = "GLM-5.3 (DeepInfra)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high", "max"] + +[cost] +input = 1.2 +output = 4 +cache_read = 0.2 + +[limit] +context = 1_048_576 diff --git a/providers/llmgateway-providers/models/deepinfra/inkling-small.toml b/providers/llmgateway-providers/models/deepinfra/inkling-small.toml new file mode 100644 index 00000000000..0109639620f --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/inkling-small.toml @@ -0,0 +1,16 @@ +base_model = "thinkingmachines/inkling-small" +name = "Inkling Small (DeepInfra)" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.45 +output = 1.2 +cache_read = 0.1 + +[limit] +context = 524_288 +output = 262_144 diff --git a/providers/llmgateway-providers/models/deepinfra/inkling.toml b/providers/llmgateway-providers/models/deepinfra/inkling.toml new file mode 100644 index 00000000000..41e528e739e --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/inkling.toml @@ -0,0 +1,17 @@ +base_model = "thinkingmachines/inkling" +name = "Inkling (DeepInfra)" +tool_call = false +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.95 +output = 4.05 +cache_read = 0.16 + +[limit] +context = 524_288 +output = 262_144 diff --git a/providers/llmgateway-providers/models/deepinfra/muse-glimmer-30b.toml b/providers/llmgateway-providers/models/deepinfra/muse-glimmer-30b.toml new file mode 100644 index 00000000000..0c26147df21 --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/muse-glimmer-30b.toml @@ -0,0 +1,14 @@ +base_model = "meta/muse-glimmer-30b" +name = "Muse Glimmer 30B (DeepInfra)" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.04 + +[limit] +output = 16_384 diff --git a/providers/llmgateway-providers/models/deepinfra/nemotron-3.5-lightning.toml b/providers/llmgateway-providers/models/deepinfra/nemotron-3.5-lightning.toml new file mode 100644 index 00000000000..aa669cc8be8 --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/nemotron-3.5-lightning.toml @@ -0,0 +1,13 @@ +base_model = "nvidia/nemotron-3.5-lightning" +name = "Nemotron 3.5 Lightning (DeepInfra)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.08 +output = 0.2 + +[limit] +output = 131_072 diff --git a/providers/llmgateway-providers/models/deepinfra/qwen3.8-2.4t-a95b.toml b/providers/llmgateway-providers/models/deepinfra/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..09c8ba4899d --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,11 @@ +base_model = "alibaba/qwen3.8-2.4t-a95b" +name = "Qwen3.8 2.4T A95B (DeepInfra)" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 6 +cache_read = 0.2 diff --git a/providers/llmgateway-providers/models/deepinfra/qwen3.8-27b.toml b/providers/llmgateway-providers/models/deepinfra/qwen3.8-27b.toml new file mode 100644 index 00000000000..3b37e4d93ab --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/qwen3.8-27b.toml @@ -0,0 +1,14 @@ +base_model = "alibaba/qwen3.8-27b" +name = "Qwen3.8 27B (DeepInfra)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.2 +output = 2.5 +cache_read = 0.05 + +[limit] +output = 235_929 diff --git a/providers/llmgateway-providers/models/deepinfra/step-3.7-flash.toml b/providers/llmgateway-providers/models/deepinfra/step-3.7-flash.toml new file mode 100644 index 00000000000..d9f47a5da91 --- /dev/null +++ b/providers/llmgateway-providers/models/deepinfra/step-3.7-flash.toml @@ -0,0 +1,17 @@ +base_model = "stepfun/step-3.7-flash" +base_model_omit = ["limit.input"] +name = "Step 3.7 Flash (DeepInfra)" +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.2 +output = 1.15 +cache_read = 0.04 + +[limit] +context = 262_144 +output = 32_768 diff --git a/providers/llmgateway-providers/models/gonka24/glm-5.3-flash.toml b/providers/llmgateway-providers/models/gonka24/glm-5.3-flash.toml new file mode 100644 index 00000000000..921d7dc252d --- /dev/null +++ b/providers/llmgateway-providers/models/gonka24/glm-5.3-flash.toml @@ -0,0 +1,20 @@ +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3 Flash (Gonka24)" +attachment = false +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.07 +output = 0.19 +cache_read = 0.015 + +[limit] +context = 200_000 +output = 16_384 + +[modalities] +input = ["text"] diff --git a/providers/llmgateway-providers/models/novita/qwen3.8-2.4t-a95b.toml b/providers/llmgateway-providers/models/novita/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..6cac460cd05 --- /dev/null +++ b/providers/llmgateway-providers/models/novita/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,14 @@ +base_model = "alibaba/qwen3.8-2.4t-a95b" +name = "Qwen3.8 2.4T A95B (NovitaAI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 6 +cache_read = 0.25 + +[limit] +context = 1_000_000 diff --git a/providers/llmgateway-providers/models/novita/step-3.7-flash.toml b/providers/llmgateway-providers/models/novita/step-3.7-flash.toml new file mode 100644 index 00000000000..ca5843d5fb8 --- /dev/null +++ b/providers/llmgateway-providers/models/novita/step-3.7-flash.toml @@ -0,0 +1,16 @@ +base_model = "stepfun/step-3.7-flash" +base_model_omit = ["limit.input"] +name = "Step 3.7 Flash (NovitaAI)" +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.2 +output = 1.15 +cache_read = 0.04 + +[limit] +context = 262_144 diff --git a/providers/llmgateway-providers/models/openai/gpt-6-luna.toml b/providers/llmgateway-providers/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..6383b613669 --- /dev/null +++ b/providers/llmgateway-providers/models/openai/gpt-6-luna.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-6-luna" +name = "GPT-6 Luna (OpenAI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 diff --git a/providers/llmgateway-providers/models/openai/gpt-6-sol.toml b/providers/llmgateway-providers/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..88bcf456342 --- /dev/null +++ b/providers/llmgateway-providers/models/openai/gpt-6-sol.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-6-sol" +name = "GPT-6 Sol (OpenAI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 diff --git a/providers/llmgateway-providers/models/scx-ai-gp/deepseek-v4.1-flash.toml b/providers/llmgateway-providers/models/scx-ai-gp/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..9ea85540b29 --- /dev/null +++ b/providers/llmgateway-providers/models/scx-ai-gp/deepseek-v4.1-flash.toml @@ -0,0 +1,14 @@ +base_model = "deepseek/deepseek-v4.1-flash" +name = "DeepSeek V4.1 Flash (SCX.ai)" +attachment = false +reasoning = false +tool_call = false +structured_output = false + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.01 + +[modalities] +input = ["text"] diff --git a/providers/llmgateway-providers/models/together-ai/glm-5.3-flash.toml b/providers/llmgateway-providers/models/together-ai/glm-5.3-flash.toml new file mode 100644 index 00000000000..af0987b5584 --- /dev/null +++ b/providers/llmgateway-providers/models/together-ai/glm-5.3-flash.toml @@ -0,0 +1,15 @@ +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3 Flash (Together AI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.03 + +[limit] +context = 1_048_576 +output = 943_717 diff --git a/providers/llmgateway-providers/models/vichar-ai/glm-5.3.toml b/providers/llmgateway-providers/models/together-ai/glm-5.3.toml similarity index 51% rename from providers/llmgateway-providers/models/vichar-ai/glm-5.3.toml rename to providers/llmgateway-providers/models/together-ai/glm-5.3.toml index 561369dc9e2..67971580a73 100644 --- a/providers/llmgateway-providers/models/vichar-ai/glm-5.3.toml +++ b/providers/llmgateway-providers/models/together-ai/glm-5.3.toml @@ -1,10 +1,9 @@ base_model = "zhipuai/glm-5.3" -name = "GLM-5.3 (vichar-ai)" -structured_output = false +name = "GLM-5.3 (Together AI)" [[reasoning_options]] type = "effort" -values = ["low", "high", "max"] +values = ["none", "low", "high", "max"] [cost] input = 1.4 @@ -12,5 +11,5 @@ output = 4.4 cache_read = 0.26 [limit] -context = 1_048_000 -output = 128_000 +context = 1_048_576 +output = 943_717 diff --git a/providers/llmgateway-providers/models/together-ai/inkling.toml b/providers/llmgateway-providers/models/together-ai/inkling.toml new file mode 100644 index 00000000000..1d6f4d10c7c --- /dev/null +++ b/providers/llmgateway-providers/models/together-ai/inkling.toml @@ -0,0 +1,16 @@ +base_model = "thinkingmachines/inkling" +name = "Inkling (Together AI)" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 1 +output = 4.05 +cache_read = 0.17 + +[limit] +context = 524_288 +output = 471_859 diff --git a/providers/llmgateway-providers/models/together-ai/muse-glimmer-30b.toml b/providers/llmgateway-providers/models/together-ai/muse-glimmer-30b.toml new file mode 100644 index 00000000000..99056ca7d36 --- /dev/null +++ b/providers/llmgateway-providers/models/together-ai/muse-glimmer-30b.toml @@ -0,0 +1,14 @@ +base_model = "meta/muse-glimmer-30b" +name = "Muse Glimmer 30B (Together AI)" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.35 +output = 1.5 +cache_read = 0.04 + +[limit] +output = 117_964 diff --git a/providers/llmgateway-providers/models/together-ai/qwen3.8-2.4t-a95b.toml b/providers/llmgateway-providers/models/together-ai/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..cd2eecc8a54 --- /dev/null +++ b/providers/llmgateway-providers/models/together-ai/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,15 @@ +base_model = "alibaba/qwen3.8-2.4t-a95b" +name = "Qwen3.8 2.4T A95B (Together AI)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.25 + +[limit] +context = 1_010_000 +output = 909_000 diff --git a/providers/llmgateway-providers/models/vichar-ai/glm-5.3-flash.toml b/providers/llmgateway-providers/models/vichar-ai/glm-5.3-flash.toml deleted file mode 100644 index a8d0eb4cba1..00000000000 --- a/providers/llmgateway-providers/models/vichar-ai/glm-5.3-flash.toml +++ /dev/null @@ -1,20 +0,0 @@ -base_model = "zhipuai/glm-5.3-flash" -name = "GLM-5.3 Flash (vichar-ai)" -attachment = false -structured_output = false - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0.15 -output = 0.5 -cache_read = 0.03 - -[limit] -context = 1_048_000 -output = 128_000 - -[modalities] -input = ["text"] diff --git a/providers/llmgateway-providers/models/xai/grok-4-7.toml b/providers/llmgateway-providers/models/xai/grok-4-7.toml new file mode 100644 index 00000000000..03bd9afa268 --- /dev/null +++ b/providers/llmgateway-providers/models/xai/grok-4-7.toml @@ -0,0 +1,28 @@ +name = "Grok 4.7 (xAI)" +description = "Grok model for agentic tool use, reasoning, coding, and live assistance" +family = "grok" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[limit] +context = 500_000 +output = 500_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-flash.toml b/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..718a69cf405 --- /dev/null +++ b/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,15 @@ +base_model = "xiaomi/mimo-v2.6-flash" +name = "MiMo V2.6 Flash (Xiaomi)" +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[limit] +context = 1_000_000 diff --git a/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-pro.toml b/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..3fe108bfc5a --- /dev/null +++ b/providers/llmgateway-providers/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,15 @@ +base_model = "xiaomi/mimo-v2.6-pro" +name = "MiMo V2.6 Pro (Xiaomi)" +structured_output = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 + +[limit] +context = 1_000_000 diff --git a/providers/llmgateway/models/claude-opus-5-5.toml b/providers/llmgateway/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..85ea7179304 --- /dev/null +++ b/providers/llmgateway/models/claude-opus-5-5.toml @@ -0,0 +1,11 @@ +base_model = "anthropic/claude-opus-5-5" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/llmgateway/models/glm-5.3-flash.toml b/providers/llmgateway/models/glm-5.3-flash.toml index d46987e8acc..5aecfa64069 100644 --- a/providers/llmgateway/models/glm-5.3-flash.toml +++ b/providers/llmgateway/models/glm-5.3-flash.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] -input = 0.088 -output = 0.25 -cache_read = 0.025 +input = 0.07 +output = 0.19 +cache_read = 0.015 [limit] context = 1_048_576 diff --git a/providers/llmgateway/models/glm-5.3.toml b/providers/llmgateway/models/glm-5.3.toml index 66cc365baaa..6be69ce3acf 100644 --- a/providers/llmgateway/models/glm-5.3.toml +++ b/providers/llmgateway/models/glm-5.3.toml @@ -2,7 +2,7 @@ base_model = "zhipuai/glm-5.3" [[reasoning_options]] type = "effort" -values = ["low", "high", "max"] +values = ["none", "low", "high", "max"] [cost] input = 1.2 diff --git a/providers/llmgateway/models/gpt-6-luna.toml b/providers/llmgateway/models/gpt-6-luna.toml new file mode 100644 index 00000000000..f058ea629ce --- /dev/null +++ b/providers/llmgateway/models/gpt-6-luna.toml @@ -0,0 +1,11 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 diff --git a/providers/llmgateway/models/gpt-6-sol.toml b/providers/llmgateway/models/gpt-6-sol.toml new file mode 100644 index 00000000000..e4694368a0c --- /dev/null +++ b/providers/llmgateway/models/gpt-6-sol.toml @@ -0,0 +1,11 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 diff --git a/providers/llmgateway/models/grok-4-7.toml b/providers/llmgateway/models/grok-4-7.toml new file mode 100644 index 00000000000..4e89d64b78f --- /dev/null +++ b/providers/llmgateway/models/grok-4-7.toml @@ -0,0 +1,28 @@ +name = "Grok 4.7" +description = "Grok model for agentic tool use, reasoning, coding, and live assistance" +family = "grok" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[limit] +context = 500_000 +output = 500_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/llmgateway/models/inkling-small.toml b/providers/llmgateway/models/inkling-small.toml new file mode 100644 index 00000000000..954d4c25f03 --- /dev/null +++ b/providers/llmgateway/models/inkling-small.toml @@ -0,0 +1,13 @@ +base_model = "thinkingmachines/inkling-small" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.45 +output = 1.2 +cache_read = 0.1 + +[limit] +context = 524_288 diff --git a/providers/llmgateway/models/inkling.toml b/providers/llmgateway/models/inkling.toml new file mode 100644 index 00000000000..1d07005aa33 --- /dev/null +++ b/providers/llmgateway/models/inkling.toml @@ -0,0 +1,13 @@ +base_model = "thinkingmachines/inkling" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.95 +output = 4.05 +cache_read = 0.16 + +[limit] +context = 524_288 diff --git a/providers/llmgateway/models/mimo-v2.6-flash.toml b/providers/llmgateway/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..a71af774946 --- /dev/null +++ b/providers/llmgateway/models/mimo-v2.6-flash.toml @@ -0,0 +1,13 @@ +base_model = "xiaomi/mimo-v2.6-flash" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[limit] +context = 1_000_000 diff --git a/providers/llmgateway/models/mimo-v2.6-pro.toml b/providers/llmgateway/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..36f93eaabd5 --- /dev/null +++ b/providers/llmgateway/models/mimo-v2.6-pro.toml @@ -0,0 +1,13 @@ +base_model = "xiaomi/mimo-v2.6-pro" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 + +[limit] +context = 1_000_000 diff --git a/providers/llmgateway/models/muse-glimmer-30b.toml b/providers/llmgateway/models/muse-glimmer-30b.toml new file mode 100644 index 00000000000..9d06a10be3b --- /dev/null +++ b/providers/llmgateway/models/muse-glimmer-30b.toml @@ -0,0 +1,10 @@ +base_model = "meta/muse-glimmer-30b" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.04 diff --git a/providers/llmgateway/models/nemotron-3.5-lightning.toml b/providers/llmgateway/models/nemotron-3.5-lightning.toml new file mode 100644 index 00000000000..6e7dba809e3 --- /dev/null +++ b/providers/llmgateway/models/nemotron-3.5-lightning.toml @@ -0,0 +1,9 @@ +base_model = "nvidia/nemotron-3.5-lightning" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.08 +output = 0.2 diff --git a/providers/llmgateway/models/qwen3.8-2.4t-a95b.toml b/providers/llmgateway/models/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..02fe65d078f --- /dev/null +++ b/providers/llmgateway/models/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,13 @@ +base_model = "alibaba/qwen3.8-2.4t-a95b" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 6 +cache_read = 0.25 + +[limit] +context = 1_010_000 diff --git a/providers/llmgateway/models/step-3.7-flash.toml b/providers/llmgateway/models/step-3.7-flash.toml new file mode 100644 index 00000000000..ad0acb372ea --- /dev/null +++ b/providers/llmgateway/models/step-3.7-flash.toml @@ -0,0 +1,14 @@ +base_model = "stepfun/step-3.7-flash" +base_model_omit = ["limit.input"] + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.2 +output = 1.15 +cache_read = 0.04 + +[limit] +context = 262_144 diff --git a/providers/llmtech/models/unsloth/Qwen3.8-27B-NVFP4.toml b/providers/llmtech/models/nvidia/Qwen3.8-27B-NVFP4.toml similarity index 100% rename from providers/llmtech/models/unsloth/Qwen3.8-27B-NVFP4.toml rename to providers/llmtech/models/nvidia/Qwen3.8-27B-NVFP4.toml diff --git a/providers/merge-gateway/models/anthropic/claude-opus-5-5.toml b/providers/merge-gateway/models/anthropic/claude-opus-5-5.toml new file mode 100644 index 00000000000..5f5b832e6f8 --- /dev/null +++ b/providers/merge-gateway/models/anthropic/claude-opus-5-5.toml @@ -0,0 +1,9 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true +reasoning_options = [] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/merge-gateway/models/openai/gpt-6-astra.toml b/providers/merge-gateway/models/openai/gpt-6-astra.toml index 223b48abe9e..da7cc6ca051 100644 --- a/providers/merge-gateway/models/openai/gpt-6-astra.toml +++ b/providers/merge-gateway/models/openai/gpt-6-astra.toml @@ -8,6 +8,7 @@ values = ["low", "medium", "high", "xhigh", "max"] input = 10 output = 50 cache_read = 1 +cache_write = 12.5 [modalities] input = ["text", "image"] diff --git a/providers/merge-gateway/models/openai/gpt-6-luna.toml b/providers/merge-gateway/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..7b7f12548e8 --- /dev/null +++ b/providers/merge-gateway/models/openai/gpt-6-luna.toml @@ -0,0 +1,17 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[modalities] +input = ["text", "image"] diff --git a/providers/merge-gateway/models/openai/gpt-6-sol.toml b/providers/merge-gateway/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..e0b3cd3f91b --- /dev/null +++ b/providers/merge-gateway/models/openai/gpt-6-sol.toml @@ -0,0 +1,17 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[modalities] +input = ["text", "image"] diff --git a/providers/merge-gateway/models/qwen/qwen3.5-27b.toml b/providers/merge-gateway/models/qwen/qwen3.5-27b.toml index f0885b3822b..d7d8d42decc 100644 --- a/providers/merge-gateway/models/qwen/qwen3.5-27b.toml +++ b/providers/merge-gateway/models/qwen/qwen3.5-27b.toml @@ -1,5 +1,6 @@ # Source: https://api-gateway.merge.dev/v1/models?model=qwen%2Fqwen3.5-27b (accessed 2026-07-21) base_model = "alibaba/qwen3.5-27b" +attachment = false [[reasoning_options]] type = "toggle" @@ -10,8 +11,8 @@ output = 0.688 cache_read = 0.0172 [limit] -context = 256_000 -output = 64_000 +context = 131_072 +output = 32_768 [modalities] -input = ["text", "image"] +input = ["text"] diff --git a/providers/merge-gateway/models/qwen/qwen3.5-397b-a17b.toml b/providers/merge-gateway/models/qwen/qwen3.5-397b-a17b.toml index 0264c6bc733..84592d1b8dd 100644 --- a/providers/merge-gateway/models/qwen/qwen3.5-397b-a17b.toml +++ b/providers/merge-gateway/models/qwen/qwen3.5-397b-a17b.toml @@ -2,6 +2,7 @@ # Selected route reasoning.controls = ["thinking"], disable_supported = true. base_model = "alibaba/qwen3.5-397b-a17b" name = "Qwen3.5 397B A17B" +attachment = false [[reasoning_options]] type = "toggle" @@ -12,8 +13,8 @@ output = 1.032 cache_read = 0.0344 [limit] -context = 256_000 -output = 64_000 +context = 131_072 +output = 32_768 [modalities] -input = ["text", "image"] +input = ["text"] diff --git a/providers/merge-gateway/models/xai/grok-4.7.toml b/providers/merge-gateway/models/xai/grok-4.7.toml new file mode 100644 index 00000000000..19465981ea7 --- /dev/null +++ b/providers/merge-gateway/models/xai/grok-4.7.toml @@ -0,0 +1,13 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[modalities] +input = ["text", "image"] diff --git a/providers/minimax-cn-coding-plan/provider.toml b/providers/minimax-cn-coding-plan/provider.toml index b72447a7d59..28bd4e0b23d 100644 --- a/providers/minimax-cn-coding-plan/provider.toml +++ b/providers/minimax-cn-coding-plan/provider.toml @@ -1,5 +1,10 @@ -name = "MiniMax Token Plan (minimaxi.com)" +# China API host migrated from api.minimaxi.com → api.minimax.cn +# Sources: +# - https://platform.minimaxi.com/docs/guides/quickstart-preparation +# - https://platform.minimaxi.com/docs/token-plan/other-tools +# Anthropic-compatible base: https://api.minimax.cn/anthropic +name = "MiniMax Token Plan (minimax.cn)" env = ["MINIMAX_API_KEY"] npm = "@ai-sdk/anthropic" doc = "https://platform.minimaxi.com/docs/token-plan/intro" -api = "https://api.minimaxi.com/anthropic/v1" +api = "https://api.minimax.cn/anthropic/v1" diff --git a/providers/minimax-cn/provider.toml b/providers/minimax-cn/provider.toml index a72226e7345..2a75ce8de3a 100644 --- a/providers/minimax-cn/provider.toml +++ b/providers/minimax-cn/provider.toml @@ -1,5 +1,10 @@ -name = "MiniMax (minimaxi.com)" +# China API host migrated from api.minimaxi.com → api.minimax.cn +# Sources: +# - https://platform.minimaxi.com/docs/guides/quickstart-preparation +# - https://platform.minimaxi.com/docs/api-reference/text-anthropic-api +# Anthropic-compatible base: https://api.minimax.cn/anthropic +name = "MiniMax (minimax.cn)" env = ["MINIMAX_API_KEY"] npm = "@ai-sdk/anthropic" doc = "https://platform.minimaxi.com/docs/guides/quickstart" -api = "https://api.minimaxi.com/anthropic/v1" +api = "https://api.minimax.cn/anthropic/v1" diff --git a/providers/nano-gpt/models/TEE/deepseek-v4-flash.toml b/providers/nano-gpt/models/TEE/deepseek-v4-flash.toml deleted file mode 100644 index e9d130676e5..00000000000 --- a/providers/nano-gpt/models/TEE/deepseek-v4-flash.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "deepseek/deepseek-v4-flash" -name = "DeepSeek V4 Flash TEE" -reasoning_options = [] - -[cost] -input = 0.2 -output = 0.4 -cache_read = 0.04 - -[limit] -context = 1_048_576 -input = 1_048_576 -output = 393_216 diff --git a/providers/nano-gpt/models/abliteration-ai/abliterated-model-large-v2.toml b/providers/nano-gpt/models/abliteration-ai/abliterated-model-large-v2.toml index 2228e8805bd..1a9af392dac 100644 --- a/providers/nano-gpt/models/abliteration-ai/abliterated-model-large-v2.toml +++ b/providers/nano-gpt/models/abliteration-ai/abliterated-model-large-v2.toml @@ -13,9 +13,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 5 +input = 3 output = 5 -cache_read = 0.5 +cache_read = 0.3 [limit] context = 1_000_000 diff --git a/providers/nano-gpt/models/abliteration-ai/abliterated-model-large.toml b/providers/nano-gpt/models/abliteration-ai/abliterated-model-large.toml index 71a9a723881..58ef18e1471 100644 --- a/providers/nano-gpt/models/abliteration-ai/abliterated-model-large.toml +++ b/providers/nano-gpt/models/abliteration-ai/abliterated-model-large.toml @@ -13,9 +13,9 @@ type = "effort" values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] -input = 5 +input = 3 output = 5 -cache_read = 0.5 +cache_read = 0.3 [limit] context = 1_000_000 diff --git a/providers/nano-gpt/models/abliteration-ai/abliterated-model.toml b/providers/nano-gpt/models/abliteration-ai/abliterated-model.toml index ac86f6f5738..46dfe7fcc9f 100644 --- a/providers/nano-gpt/models/abliteration-ai/abliterated-model.toml +++ b/providers/nano-gpt/models/abliteration-ai/abliterated-model.toml @@ -13,9 +13,9 @@ type = "effort" values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] -input = 3 +input = 1 output = 3 -cache_read = 0.3 +cache_read = 0.1 [limit] context = 262_144 diff --git a/providers/opper/models/anthropic/claude-sonnet-5.toml b/providers/nano-gpt/models/anthropic/claude-opus-5.5.toml similarity index 53% rename from providers/opper/models/anthropic/claude-sonnet-5.toml rename to providers/nano-gpt/models/anthropic/claude-opus-5.5.toml index c49fbcbdf8a..5948dbab4f8 100644 --- a/providers/opper/models/anthropic/claude-sonnet-5.toml +++ b/providers/nano-gpt/models/anthropic/claude-opus-5.5.toml @@ -1,4 +1,4 @@ -base_model = "anthropic/claude-sonnet-5" +base_model = "anthropic/claude-opus-5-5" structured_output = true [[reasoning_options]] @@ -6,10 +6,10 @@ type = "effort" values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 2 -output = 10 +input = 4 +output = 20 cache_read = 0.2 -cache_write = 2.5 +cache_write = 5 -[interleaved] -field = "reasoning_content" +[limit] +input = 1_000_000 diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-latest.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-latest.toml index abb7e798efd..fb7f9db2975 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-latest.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-latest.toml @@ -19,9 +19,9 @@ output = 0.16 cache_read = 0.013 [limit] -context = 1_048_576 -input = 1_048_576 -output = 384_000 +context = 1_000_000 +input = 1_000_000 +output = 131_072 [modalities] input = ["text"] diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml index b4cce19afb3..a62995b0beb 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.22 -output = 0.66 -cache_read = 0.007 +input = 0.44 +output = 1.32 +cache_read = 0.014 [limit] context = 1_048_576 diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash.toml b/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash.toml index 8e61a5cbf70..c7745ff077e 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash.toml @@ -5,9 +5,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.1 -output = 0.4 -cache_read = 0.003 +input = 0.13 +output = 0.52 +cache_read = 0.006 [limit] input = 1_000_000 diff --git a/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash:thinking.toml b/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash:thinking.toml index 90c16dc648c..c26ac296879 100644 --- a/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash:thinking.toml +++ b/providers/nano-gpt/models/deepseek/deepseek-v4.1-flash:thinking.toml @@ -6,9 +6,9 @@ type = "effort" values = ["none", "low", "high", "max"] [cost] -input = 0.1 -output = 0.4 -cache_read = 0.003 +input = 0.13 +output = 0.52 +cache_read = 0.006 [limit] input = 1_000_000 diff --git a/providers/nano-gpt/models/gemma-4-12b-it.toml b/providers/nano-gpt/models/gemma-4-12b-it.toml index 6959fb985f1..3d3dd5e5581 100644 --- a/providers/nano-gpt/models/gemma-4-12b-it.toml +++ b/providers/nano-gpt/models/gemma-4-12b-it.toml @@ -1,13 +1,6 @@ +base_model = "google/gemma-4-12b-it" name = "Gemma 4 12B Instruct" -description = "Google's Gemma 4 12B Instruct is an open-weight multimodal model for text, image, audio, and video understanding, with tool calling and structured output support." -family = "gemma" -release_date = "2026-08-01" -last_updated = "2026-08-01" -attachment = true reasoning = false -tool_call = true -structured_output = true -open_weights = true [cost] input = 0.05 @@ -17,8 +10,6 @@ cache_read = 0.025 [limit] context = 131_072 input = 131_072 -output = 32_768 [modalities] input = ["text", "image", "video", "audio"] -output = ["text"] diff --git a/providers/nano-gpt/models/google/diffusiongemma.toml b/providers/nano-gpt/models/google/diffusiongemma.toml new file mode 100644 index 00000000000..244dd04defb --- /dev/null +++ b/providers/nano-gpt/models/google/diffusiongemma.toml @@ -0,0 +1,27 @@ +name = "DiffusionGemma" +description = "DiffusionGemma is a high-speed diffusion-based version of Gemma 4 26B A4B. It supports optional reasoning and a 262,144-token context window." +release_date = "2026-09-19" +last_updated = "2026-09-19" +attachment = false +reasoning = true +tool_call = false +structured_output = false +open_weights = true + +[[reasoning_options]] +type = "effort" +values = ["none", "xhigh"] + +[cost] +input = 0.05 +output = 0.15 +cache_read = 0.025 + +[limit] +context = 262_144 +input = 262_144 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/nano-gpt/models/google/gemma-4-26b-a4b-it-cybersecurity.toml b/providers/nano-gpt/models/google/gemma-4-26b-a4b-it-cybersecurity.toml new file mode 100644 index 00000000000..502d73081ea --- /dev/null +++ b/providers/nano-gpt/models/google/gemma-4-26b-a4b-it-cybersecurity.toml @@ -0,0 +1,25 @@ +name = "Gemma 4 26B A4B Cybersecurity" +description = "Gemma 4 26B A4B Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports optional reasoning, image understanding, tool calling, and a 262,144-token context window." +family = "gemma" +release_date = "2026-09-19" +last_updated = "2026-09-19" +attachment = true +reasoning = true +tool_call = true +structured_output = false +open_weights = true +reasoning_options = [] + +[cost] +input = 0.1056 +output = 0.3344 +cache_read = 0.0528 + +[limit] +context = 262_144 +input = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/nano-gpt/models/slowburn/gemma4-31b-splituntied.toml b/providers/nano-gpt/models/google/gemma4-31b-splituntied.toml similarity index 65% rename from providers/nano-gpt/models/slowburn/gemma4-31b-splituntied.toml rename to providers/nano-gpt/models/google/gemma4-31b-splituntied.toml index 60615a27959..2a2440d5a87 100644 --- a/providers/nano-gpt/models/slowburn/gemma4-31b-splituntied.toml +++ b/providers/nano-gpt/models/google/gemma4-31b-splituntied.toml @@ -1,5 +1,5 @@ name = "Gemma 4 31B Split-Untied" -description = "Slowburn's Split-Untied is a text-only Gemma 4 31B community finetune with an untied BF16 output head, built for creative writing, roleplay, expressive dialogue, and tool use." +description = "Blazed-Forge's Split-Untied is a text-only Gemma 4 31B community finetune with an untied BF16 output head, built for creative writing, roleplay, expressive dialogue, and tool use." family = "gemma" release_date = "2026-09-17" last_updated = "2026-09-17" diff --git a/providers/nano-gpt/models/mistralai/mistral-large-3-675b-instruct-2512.toml b/providers/nano-gpt/models/mistralai/mistral-large-3-675b-instruct-2512.toml deleted file mode 100644 index 0a0650a3987..00000000000 --- a/providers/nano-gpt/models/mistralai/mistral-large-3-675b-instruct-2512.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Mistral Large 3 675B" -description = "Flagship Mistral model for advanced reasoning, coding, and multilingual work" -family = "mistral-large" -release_date = "2025-12-25" -last_updated = "2025-12-02" -attachment = true -reasoning = false -tool_call = false -structured_output = false -open_weights = true - -[cost] -input = 1 -output = 3 -cache_read = 0.5 - -[limit] -context = 262_144 -input = 262_144 -output = 256_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/nano-gpt/models/nano/lumen-stealth.toml b/providers/nano-gpt/models/nano/lumen-stealth.toml new file mode 100644 index 00000000000..9b27918c763 --- /dev/null +++ b/providers/nano-gpt/models/nano/lumen-stealth.toml @@ -0,0 +1,23 @@ +name = "Lumen Stealth" +description = "Experimental multimodal model focused on reasoning, creative writing, roleplay, and agentic workflows. Available temporarily for evaluation ahead of public release. During this evaluation, prompts and responses are logged and may be reviewed to evaluate and improve the model. Do not send sensitive or confidential information." +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +tool_call = false +structured_output = false +open_weights = false +reasoning_options = [] + +[cost] +input = 0.05 +output = 0 + +[limit] +context = 200_000 +input = 200_000 +output = 100_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/nano-gpt/models/nvidia/nemotron-3.5-content-safety.toml b/providers/nano-gpt/models/nvidia/nemotron-3.5-content-safety.toml new file mode 100644 index 00000000000..8564f16dea6 --- /dev/null +++ b/providers/nano-gpt/models/nvidia/nemotron-3.5-content-safety.toml @@ -0,0 +1,17 @@ +base_model = "nvidia/nemotron-3.5-content-safety" +attachment = false +structured_output = false +reasoning_options = [] + +[cost] +input = 0.05 +output = 0.15 +cache_read = 0.025 + +[limit] +context = 131_072 +input = 131_072 +output = 32_768 + +[modalities] +input = ["text"] diff --git a/providers/nano-gpt/models/openai/gpt-6-luna-pro.toml b/providers/nano-gpt/models/openai/gpt-6-luna-pro.toml new file mode 100644 index 00000000000..cc8bc7d726f --- /dev/null +++ b/providers/nano-gpt/models/openai/gpt-6-luna-pro.toml @@ -0,0 +1,29 @@ +name = "GPT 6 Luna Pro" +description = "GPT-6 Luna Pro uses the same underlying model as GPT-6 Luna with Pro reasoning mode enabled for higher-quality responses on complex tasks." +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.05 +output = 0.25 +cache_read = 0.005 +cache_write = 0.0625 + +[limit] +context = 1_050_000 +input = 1_050_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/nano-gpt/models/openai/gpt-6-luna.toml b/providers/nano-gpt/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..23f8b252eec --- /dev/null +++ b/providers/nano-gpt/models/openai/gpt-6-luna.toml @@ -0,0 +1,15 @@ +base_model = "openai/gpt-6-luna" +name = "GPT 6 Luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.05 +output = 0.25 +cache_read = 0.005 +cache_write = 0.0625 + +[limit] +input = 1_050_000 diff --git a/providers/kilo/models/openai/gpt-5.6-sol-discounted.toml b/providers/nano-gpt/models/openai/gpt-6-sol-pro.toml similarity index 52% rename from providers/kilo/models/openai/gpt-5.6-sol-discounted.toml rename to providers/nano-gpt/models/openai/gpt-6-sol-pro.toml index 6008af017c3..1720ef7ae10 100644 --- a/providers/kilo/models/openai/gpt-5.6-sol-discounted.toml +++ b/providers/nano-gpt/models/openai/gpt-6-sol-pro.toml @@ -1,13 +1,12 @@ -name = "OpenAI: GPT-5.6 Sol (50% off)" -description = "GPT-5.6 Sol served by OpenAI through Vercel AI Gateway at 50% lower cost than other available inference providers. This promotion runs through September 18, 2026." +name = "GPT 6 Sol Pro" +description = "GPT-6 Sol Pro uses the same underlying model as GPT-6 Sol with Pro reasoning mode enabled for higher-quality responses on complex tasks." family = "gpt" -release_date = "2025-08-26" -last_updated = "2025-08-26" +release_date = "2026-09-22" +last_updated = "2026-09-22" attachment = true reasoning = true -temperature = true tool_call = true -structured_output = false +structured_output = true open_weights = false [[reasoning_options]] @@ -17,12 +16,12 @@ values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] input = 2 output = 10 -reasoning = 0 cache_read = 0.2 cache_write = 2.5 [limit] context = 1_050_000 +input = 1_050_000 output = 128_000 [modalities] diff --git a/providers/nano-gpt/models/openai/gpt-6-sol.toml b/providers/nano-gpt/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..ee3ffe25af3 --- /dev/null +++ b/providers/nano-gpt/models/openai/gpt-6-sol.toml @@ -0,0 +1,15 @@ +base_model = "openai/gpt-6-sol" +name = "GPT 6 Sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[limit] +input = 1_050_000 diff --git a/providers/nano-gpt/models/prism-ml/ternary-bonsai-2-27b.toml b/providers/nano-gpt/models/prism-ml/ternary-bonsai-2-27b.toml new file mode 100644 index 00000000000..32eb4c91ae5 --- /dev/null +++ b/providers/nano-gpt/models/prism-ml/ternary-bonsai-2-27b.toml @@ -0,0 +1,27 @@ +name = "Ternary Bonsai 2 27B" +description = "Ternary Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window." +release_date = "2026-09-18" +last_updated = "2026-09-18" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = true + +[[reasoning_options]] +type = "effort" +values = ["medium", "xhigh"] + +[cost] +input = 0.075 +output = 0.5 +cache_read = 0.0375 + +[limit] +context = 262_144 +input = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/nano-gpt/models/qwen/qwen3-14b.toml b/providers/nano-gpt/models/qwen/qwen3-14b.toml index 21b6ecf5bc8..0ec7e2615e0 100644 --- a/providers/nano-gpt/models/qwen/qwen3-14b.toml +++ b/providers/nano-gpt/models/qwen/qwen3-14b.toml @@ -6,7 +6,7 @@ release_date = "2024-01-01" last_updated = "2024-01-01" attachment = false reasoning = false -tool_call = false +tool_call = true structured_output = false open_weights = true diff --git a/providers/nano-gpt/models/qwen/qwen3-8b.toml b/providers/nano-gpt/models/qwen/qwen3-8b.toml index 5dd994f1ea3..96cac74c86b 100644 --- a/providers/nano-gpt/models/qwen/qwen3-8b.toml +++ b/providers/nano-gpt/models/qwen/qwen3-8b.toml @@ -5,7 +5,7 @@ release_date = "2024-01-01" last_updated = "2024-01-01" attachment = false reasoning = false -tool_call = false +tool_call = true structured_output = false open_weights = true diff --git a/providers/nano-gpt/models/qwen/qwen3-vl-235b-a22b-instruct.toml b/providers/nano-gpt/models/qwen/qwen3-vl-235b-a22b-instruct.toml index a648f070f2e..8b97c180ac4 100644 --- a/providers/nano-gpt/models/qwen/qwen3-vl-235b-a22b-instruct.toml +++ b/providers/nano-gpt/models/qwen/qwen3-vl-235b-a22b-instruct.toml @@ -4,7 +4,7 @@ structured_output = false [cost] input = 0.3 -output = 1.2 +output = 1.9 cache_read = 0.15 [limit] diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-cybersecurity.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-cybersecurity.toml new file mode 100644 index 00000000000..fcae44481c3 --- /dev/null +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-cybersecurity.toml @@ -0,0 +1,25 @@ +name = "Qwen 3.8 27B Cybersecurity" +description = "Qwen 3.8 27B Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports optional reasoning, image understanding, tool calling, and a 262,144-token context window." +family = "qwen" +release_date = "2026-09-19" +last_updated = "2026-09-19" +attachment = true +reasoning = true +tool_call = true +structured_output = false +open_weights = true +reasoning_options = [] + +[cost] +input = 0.1 +output = 0.6 +cache_read = 0.05 + +[limit] +context = 262_144 +input = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-hemmingway.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-hemmingway.toml new file mode 100644 index 00000000000..e8e4e1ce33f --- /dev/null +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-hemmingway.toml @@ -0,0 +1,25 @@ +name = "Qwen 3.8 27B Hemingway" +description = "Qwen 3.8 27B Hemingway is an open-weight NVFP4 multimodal creative finetune for long-form prose, character dialogue, storytelling, and roleplay." +family = "qwen" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +tool_call = true +structured_output = false +open_weights = true +reasoning_options = [] + +[cost] +input = 0.25 +output = 1.5 +cache_read = 0.125 + +[limit] +context = 262_144 +input = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml index 516d35fc982..3dc414edb50 100644 --- a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored.toml @@ -11,8 +11,8 @@ open_weights = true reasoning_options = [] [cost] -input = 0.25 -output = 1.5 +input = 0.15 +output = 1.2 cache_read = 0.125 [limit] diff --git a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml index 169511f2187..d4ee170d540 100644 --- a/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml +++ b/providers/nano-gpt/models/qwen/qwen3.8-27b-uncensored:thinking.toml @@ -11,8 +11,8 @@ open_weights = true reasoning_options = [] [cost] -input = 0.25 -output = 1.5 +input = 0.15 +output = 1.2 cache_read = 0.125 [limit] diff --git a/providers/nano-gpt/models/qwen3-vl-235b-a22b-instruct-original.toml b/providers/nano-gpt/models/qwen3-vl-235b-a22b-instruct-original.toml index 2f39269ea3f..851b51ce84f 100644 --- a/providers/nano-gpt/models/qwen3-vl-235b-a22b-instruct-original.toml +++ b/providers/nano-gpt/models/qwen3-vl-235b-a22b-instruct-original.toml @@ -9,9 +9,9 @@ structured_output = false open_weights = false [cost] -input = 0.5 -output = 1.2 -cache_read = 0.25 +input = 0.3 +output = 1.9 +cache_read = 0.15 [limit] context = 32_768 diff --git a/providers/nano-gpt/models/stepfun/step-5-preview.toml b/providers/nano-gpt/models/stepfun/step-5-preview.toml new file mode 100644 index 00000000000..20ebdd2b522 --- /dev/null +++ b/providers/nano-gpt/models/stepfun/step-5-preview.toml @@ -0,0 +1,10 @@ +base_model = "stepfun/step-5-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1 +output = 2.7 +cache_read = 0.05 diff --git a/providers/nano-gpt/models/typesafe/jev-latest.toml b/providers/nano-gpt/models/typesafe/jev-latest.toml new file mode 100644 index 00000000000..49c0acf91f0 --- /dev/null +++ b/providers/nano-gpt/models/typesafe/jev-latest.toml @@ -0,0 +1,12 @@ +base_model = "typesafe/jev-latest" +name = "Jev Latest" +structured_output = false + +[cost] +input = 0.042 +output = 0 +cache_read = 0.021 + +[limit] +context = 32_000 +input = 32_000 diff --git a/providers/nano-gpt/models/x-ai/grok-4.7.toml b/providers/nano-gpt/models/x-ai/grok-4.7.toml new file mode 100644 index 00000000000..81b4067b343 --- /dev/null +++ b/providers/nano-gpt/models/x-ai/grok-4.7.toml @@ -0,0 +1,16 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 1.6 +output = 4.8 +cache_read = 0.4 + +[limit] +input = 500_000 + +[modalities] +input = ["text", "image"] diff --git a/providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml b/providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..79ac140f7b6 --- /dev/null +++ b/providers/nano-gpt/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,16 @@ +base_model = "xiaomi/mimo-v2.6-flash" +name = "MiMo V2.6 Flash" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 +cache_write = 0 + +[limit] +input = 1_048_576 diff --git a/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..8950d4913de --- /dev/null +++ b/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,16 @@ +base_model = "xiaomi/mimo-v2.6-pro-ultraspeed" +name = "MiMo V2.6 Pro UltraSpeed" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 4.35 +output = 8.7 +cache_read = 0.036 +cache_write = 0 + +[limit] +input = 1_048_576 diff --git a/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml b/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..a99c234165d --- /dev/null +++ b/providers/nano-gpt/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,16 @@ +base_model = "xiaomi/mimo-v2.6-pro" +name = "MiMo V2.6 Pro" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 +cache_write = 0 + +[limit] +input = 1_048_576 diff --git a/providers/nano-gpt/models/z-ai/glm-4.7-flash-original.toml b/providers/nano-gpt/models/z-ai/glm-4.7-flash-original.toml deleted file mode 100644 index 3491643442e..00000000000 --- a/providers/nano-gpt/models/z-ai/glm-4.7-flash-original.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "GLM 4.7 Flash Original" -description = "GLM-4.7-Flash is a lightweight 30B model optimized for coding and agentic tasks. Balances high performance with efficiency, perfect for local deployment." -family = "glm-flash" -release_date = "2026-01-19" -last_updated = "2026-01-19" -attachment = false -reasoning = true -tool_call = true -structured_output = true -open_weights = true -reasoning_options = [] - -[cost] -input = 0.07 -output = 0.4 -cache_read = 0.035 - -[limit] -context = 200_000 -input = 200_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/nano-gpt/models/z-ai/glm-4.7-flash-original:thinking.toml b/providers/nano-gpt/models/z-ai/glm-4.7-flash-original:thinking.toml deleted file mode 100644 index 65480647fea..00000000000 --- a/providers/nano-gpt/models/z-ai/glm-4.7-flash-original:thinking.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "GLM 4.7 Flash Original Thinking" -description = "GLM-4.7-Flash with extended thinking capabilities for complex reasoning. Lightweight 30B model optimized for coding and agentic tasks." -family = "glm-flash" -release_date = "2026-01-19" -last_updated = "2026-01-19" -attachment = false -reasoning = true -tool_call = true -structured_output = false -open_weights = true -reasoning_options = [] - -[cost] -input = 0.07 -output = 0.4 -cache_read = 0.035 - -[limit] -context = 200_000 -input = 200_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/nano-gpt/models/z-ai/glm-5.3-flash-cybersecurity.toml b/providers/nano-gpt/models/z-ai/glm-5.3-flash-cybersecurity.toml new file mode 100644 index 00000000000..b55aaf3a6a5 --- /dev/null +++ b/providers/nano-gpt/models/z-ai/glm-5.3-flash-cybersecurity.toml @@ -0,0 +1,28 @@ +name = "GLM 5.3 Flash Cybersecurity" +description = "GLM 5.3 Flash Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports always-on reasoning, image understanding, tool calling, and a 1,048,576-token context window." +family = "glm" +release_date = "2026-09-19" +last_updated = "2026-09-19" +attachment = true +reasoning = true +tool_call = true +structured_output = false +open_weights = true + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.075 + +[limit] +context = 1_048_576 +input = 1_048_576 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/nebius/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml b/providers/nebius/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml new file mode 100644 index 00000000000..b439ca31a66 --- /dev/null +++ b/providers/nebius/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml @@ -0,0 +1,28 @@ +# Sources: +# - https://tokenfactory.nebius.com/ (model catalog) +# - https://tokenfactory.nebius.com/api/public/models_info (pricing, context, max length) +# Accessed 2026-09-18. +# Effort: reasoning_effort = none|low|high|max (none = thinking off). Same gateway +# behavior as the other DeepSeek V4 entries on this host (DeepSeek-V4-Flash-0731, +# DeepSeek-V4-Pro): the generic schema accepts none/minimal/low/medium/high/xhigh/max, +# but the mid tiers collapse (minimal=low=medium, high=xhigh) and none disables +# thinking, leaving none/low/high/max as the distinct levels. +# Served window: models_info max_model_len = 979_000 for this snapshot with no separate +# output cap, so context and output both take the full served window. +# No discounted prompt-cache tier is published in the public catalog: cached and fresh +# input are billed identically, so cache_read = input (same rationale as +# DeepSeek-V4-Flash-0731 / Kimi-K3 / GLM-5.3-Flash). +base_model = "deepseek/deepseek-v4-pro-0813" +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1.32 +output = 3.96 +cache_read = 1.32 + +[limit] +context = 979_000 +output = 979_000 diff --git a/providers/nebius/models/deepseek-ai/DeepSeek-V4.1-Flash.toml b/providers/nebius/models/deepseek-ai/DeepSeek-V4.1-Flash.toml new file mode 100644 index 00000000000..cc631b9cfee --- /dev/null +++ b/providers/nebius/models/deepseek-ai/DeepSeek-V4.1-Flash.toml @@ -0,0 +1,30 @@ +# Sources: +# - https://tokenfactory.nebius.com/ (model catalog) +# - https://tokenfactory.nebius.com/api/public/models_info (pricing, context, max length) +# Accessed 2026-09-18. +# Effort: reasoning_effort = none|low|high|max (none = thinking off). Same gateway +# behavior as the other DeepSeek V4 entries on this host (DeepSeek-V4-Flash-0731, +# DeepSeek-V4-Pro): the generic schema accepts none/minimal/low/medium/high/xhigh/max, +# but the mid tiers collapse (minimal=low=medium, high=xhigh) and none disables +# thinking, leaving none/low/high/max as the distinct levels. +# Server reports model type image2text, so attachment/modalities are inherited from the +# lab entry. +# Served window: models_info max_model_len = 1_048_000 for this model with no separate +# output cap, so context and output both take the full served window. +# No discounted prompt-cache tier is published in the public catalog: cached and fresh +# input are billed identically, so cache_read = input (same rationale as +# DeepSeek-V4-Flash-0731 / Kimi-K3 / GLM-5.3-Flash). +base_model = "deepseek/deepseek-v4.1-flash" +reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.3 + +[limit] +context = 1_048_000 +output = 1_048_000 diff --git a/providers/nebius/models/zai-org/GLM-5.3.toml b/providers/nebius/models/zai-org/GLM-5.3.toml new file mode 100644 index 00000000000..be8b6208f43 --- /dev/null +++ b/providers/nebius/models/zai-org/GLM-5.3.toml @@ -0,0 +1,28 @@ +# Sources: +# - https://tokenfactory.nebius.com/ (model catalog) +# - https://tokenfactory.nebius.com/api/public/models_info (pricing, context, max length) +# - https://docs.bigmodel.cn/cn/guide/models/text/glm-5.3 (effort set; thinking cannot +# be disabled, so there is no toggle — effort per lab: low|high|max, default max) +# Accessed 2026-09-18. +# Served window: models_info max_model_len = 1_024_000 with no separate output cap, so +# context and output both take the full served window. +# No discounted prompt-cache tier is published in the public catalog: cached and fresh +# input are billed identically, so cache_read = input (same rationale as +# GLM-5.3-Flash / DeepSeek-V4-Flash-0731 / Kimi-K3). +base_model = "zhipuai/glm-5.3" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1.4 +output = 4.4 +cache_read = 1.4 + +[limit] +context = 1_024_000 +output = 1_024_000 diff --git a/providers/nebul/models/moonshotai/Kimi-K3.toml b/providers/nebul/models/moonshotai/Kimi-K3.toml index 2c90aaf6ce6..85a66cacacc 100644 --- a/providers/nebul/models/moonshotai/Kimi-K3.toml +++ b/providers/nebul/models/moonshotai/Kimi-K3.toml @@ -1,17 +1,16 @@ -# Reasoning control per Nebul's docs: GET /v1/model/info `reasoning_efforts` is the -# per-model source of truth for "the reasoning_effort values each model meaningfully -# accepts" (https://docs.nebul.io/docs/inference-api/models/model-catalog). It -# advertises none for this model — the live response also marks this served ID with -# supports_reasoning = false — and the Chat Completions API documents no on/off -# toggle (https://docs.nebul.io/docs/inference-api/models/chat-completions), so this -# host exposes no caller control here. -# Trace field: reasoning_content per https://docs.nebul.io/docs/inference-api/advanced-topics/reasoning +# Reasoning on this host: GET /v1/model/info marks this served ID with +# supports_reasoning = false and exposes no reasoning_efforts for it +# (https://api.inference.nebul.io/v1/model/info, checked 2026-09-22). Per +# Nebul's docs that catalog is the source of truth for reasoning-capable +# models, and the reasoning guide does not list Kimi among the +# trace-returning families +# (https://docs.nebul.io/docs/inference-api/advanced-topics/reasoning). +# Nebul serves this ID with thinking disabled — no reasoning controls and no +# reasoning_content traces — so override the inherited lab default with +# reasoning = false and declare no reasoning_options / interleaved. base_model = "moonshotai/kimi-k3" -reasoning_options = [] - -[interleaved] -field = "reasoning_content" +reasoning = false [cost] input = 4.73 diff --git a/providers/neuralwatt/models/deepseek-v4-flash-flex.toml b/providers/neuralwatt/models/deepseek-v4-flash-flex.toml index 090d4d2f3f7..fae8619652d 100644 --- a/providers/neuralwatt/models/deepseek-v4-flash-flex.toml +++ b/providers/neuralwatt/models/deepseek-v4-flash-flex.toml @@ -1,5 +1,6 @@ # Flex-tier variant of deepseek-v4-flash: same model, context window -# (1,048,560), and output cap (65,536) as the standard tier; requested via the +# (1,048,560), and output cap (393,216, raised from 65,536 per GET /v1/models +# checked 2026-09-18) as the standard tier; requested via the # `-flex` model ID or service_tier="flex" and requires stream=true # (non-streaming requests fall through to the standard tier). # Cost is the flex rate: 0.65x of standard pricing (35% off) per the flex-tier @@ -26,4 +27,4 @@ cache_read = 0.0182 [limit] context = 1_048_560 -output = 65_536 +output = 393_216 diff --git a/providers/neuralwatt/models/deepseek-v4-flash-speed.toml b/providers/neuralwatt/models/deepseek-v4-flash-speed.toml new file mode 100644 index 00000000000..adf0fbd952c --- /dev/null +++ b/providers/neuralwatt/models/deepseek-v4-flash-speed.toml @@ -0,0 +1,26 @@ +# Speed variant serving the DeepSeek V4 Flash 0731 checkpoint: same pricing +# as deepseek-v4-flash ($0.14/$0.28/$0.028) with a 393,216-token output cap, +# per the public GET /v1/models metadata (checked 2026-09-18). Text-only +# input on this host. Reasoning matches the standard tier: reasoning_effort +# = none|high|max (default none, reasoning off by default; aliases +# low/minimal/medium -> high, xhigh -> max). Off is effort=none, so no +# separate toggle. thinking_token_budget is rejected with a 400 on this +# generation, so it is not declared. +# https://portal.neuralwatt.com/docs/api/chat-completions +# https://portal.neuralwatt.com/docs/api/models +base_model = "deepseek/deepseek-v4-flash-0731" +name = "DeepSeek V4 Flash (Speed)" +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high", "max"] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.028 + +[limit] +context = 1_048_560 +output = 393_216 diff --git a/providers/neuralwatt/models/deepseek-v4-flash.toml b/providers/neuralwatt/models/deepseek-v4-flash.toml index 5d30b1cbdca..4a32afaee6c 100644 --- a/providers/neuralwatt/models/deepseek-v4-flash.toml +++ b/providers/neuralwatt/models/deepseek-v4-flash.toml @@ -8,6 +8,8 @@ # is rejected with a 400 on this model, so it is not declared. # https://portal.neuralwatt.com/docs/api/models # https://portal.neuralwatt.com/pricing +# Output cap raised to 393,216 tokens per GET /v1/models (checked +# 2026-09-18; was 65,536). # https://portal.neuralwatt.com/docs/api/chat-completions base_model = "deepseek/deepseek-v4-flash" interleaved = true @@ -23,4 +25,4 @@ cache_read = 0.028 [limit] context = 1_048_560 -output = 65_536 +output = 393_216 diff --git a/providers/neuralwatt/models/deepseek-v4-pro.toml b/providers/neuralwatt/models/deepseek-v4-pro.toml index 7869e9bb888..7b3f48e037a 100644 --- a/providers/neuralwatt/models/deepseek-v4-pro.toml +++ b/providers/neuralwatt/models/deepseek-v4-pro.toml @@ -1,8 +1,6 @@ -# Preview rollout: deepseek-v4-pro is in private preview on Neuralwatt — -# visible to paid users after requesting access per model card, per the -# Preview Models guide. Pricing, output cap, and cache-read from the -# authenticated GET /v1/models metadata (checked 2026-08-27); V4-Pro is not -# yet in the public catalog or token-pricing table. +# Deprecated: Neuralwatt retired deepseek-v4-pro from GET /v1/models +# (checked 2026-09-18); it never left private preview. Retained for pricing +# history. # Reasoning per the chat-completions per-model table (docs verified against # the live API on 2026-08-26): reasoning_effort accepts none|low|high|max as # distinct levels with low as the default (reasoning on by default); aliases @@ -10,9 +8,8 @@ # accepted on this model (unlike deepseek-v4-flash, which rejects it); no # bounds or disable sentinel documented. # https://portal.neuralwatt.com/docs/api/chat-completions -# https://portal.neuralwatt.com/docs/guides/preview-models base_model = "deepseek/deepseek-v4-pro" -status = "beta" +status = "deprecated" interleaved = true # Effort: reasoning_effort = none|low|high|max (off is effort=none; aliases diff --git a/providers/neuralwatt/models/deepseek-v4.1-flash-flex.toml b/providers/neuralwatt/models/deepseek-v4.1-flash-flex.toml new file mode 100644 index 00000000000..a3dcc0edcd1 --- /dev/null +++ b/providers/neuralwatt/models/deepseek-v4.1-flash-flex.toml @@ -0,0 +1,26 @@ +# Flex-tier variant of deepseek-v4.1-flash: same model, context (1,048,560), +# vision, and effort ladder as the standard tier; requested via the `-flex` +# model ID or service_tier="flex" and requires stream=true (non-streaming +# requests fall through to the standard tier). Cost is the flex rate: 0.65x +# of standard pricing (35% off) per the flex-tier guide, applied to the +# standard rates advertised by GET /v1/models ($0.15/$0.6/$0.015, checked +# 2026-09-18). thinking_token_budget is rejected with a 400 on the V4-Flash +# generation and is not documented for V4.1-Flash, so it is not declared. +# https://portal.neuralwatt.com/docs/guides/flex-tier +# https://portal.neuralwatt.com/docs/api/chat-completions +base_model = "deepseek/deepseek-v4.1-flash" +name = "DeepSeek V4.1 Flash Flex" +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high", "xhigh", "max"] + +[cost] +input = 0.0975 +output = 0.39 +cache_read = 0.00975 + +[limit] +context = 1_048_560 +output = 393_216 diff --git a/providers/neuralwatt/models/deepseek-v4.1-flash.toml b/providers/neuralwatt/models/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..240f6508802 --- /dev/null +++ b/providers/neuralwatt/models/deepseek-v4.1-flash.toml @@ -0,0 +1,25 @@ +# Pricing, limits, vision, and effort ladder from the public GET /v1/models +# metadata (checked 2026-09-18): $0.15/$0.6/$0.015, 1,048,560-token context, +# 393,216-token output cap, 20-image cap. Effort: reasoning_effort = +# none|low|high|xhigh|max (default high, reasoning on by default; aliases +# medium -> high, minimal -> low). Off is effort=none, so no separate toggle. +# thinking_token_budget is rejected with a 400 on the V4-Flash generation +# (checked 2026-08-26) and is not documented for V4.1-Flash, so it is not +# declared. +# https://portal.neuralwatt.com/docs/api/chat-completions +# https://portal.neuralwatt.com/docs/api/models +base_model = "deepseek/deepseek-v4.1-flash" +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high", "xhigh", "max"] + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.015 + +[limit] +context = 1_048_560 +output = 393_216 diff --git a/providers/neuralwatt/models/glm-5.2-fast.toml b/providers/neuralwatt/models/glm-5.2-fast.toml index a61cf125702..241dd01adcd 100644 --- a/providers/neuralwatt/models/glm-5.2-fast.toml +++ b/providers/neuralwatt/models/glm-5.2-fast.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Fast" @@ -5,6 +7,7 @@ description = "Efficient GLM model for fast reasoning, coding, and agent workflo release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2-flex.toml b/providers/neuralwatt/models/glm-5.2-flex.toml index e42b7fb1cc5..b8fce8e6ba2 100644 --- a/providers/neuralwatt/models/glm-5.2-flex.toml +++ b/providers/neuralwatt/models/glm-5.2-flex.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Flex" @@ -5,6 +7,7 @@ description = "Flagship GLM model for hybrid reasoning, coding, and agentic engi release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2-short-fast-flex.toml b/providers/neuralwatt/models/glm-5.2-short-fast-flex.toml index eea25d5f58e..f463c21d078 100644 --- a/providers/neuralwatt/models/glm-5.2-short-fast-flex.toml +++ b/providers/neuralwatt/models/glm-5.2-short-fast-flex.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Short Fast Flex" @@ -5,6 +7,7 @@ description = "Efficient GLM model for fast reasoning, coding, and agent workflo release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2-short-fast.toml b/providers/neuralwatt/models/glm-5.2-short-fast.toml index fa48e039e59..20da2db40c4 100644 --- a/providers/neuralwatt/models/glm-5.2-short-fast.toml +++ b/providers/neuralwatt/models/glm-5.2-short-fast.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Short Fast" @@ -5,6 +7,7 @@ description = "Efficient GLM model for fast reasoning, coding, and agent workflo release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2-short-flex.toml b/providers/neuralwatt/models/glm-5.2-short-flex.toml index 604df60b11d..0b1cc1ed110 100644 --- a/providers/neuralwatt/models/glm-5.2-short-flex.toml +++ b/providers/neuralwatt/models/glm-5.2-short-flex.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Short Flex" @@ -5,6 +7,7 @@ description = "Flagship GLM model for hybrid reasoning, coding, and agentic engi release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2-short.toml b/providers/neuralwatt/models/glm-5.2-short.toml index 26127431e9c..a324bae38c5 100644 --- a/providers/neuralwatt/models/glm-5.2-short.toml +++ b/providers/neuralwatt/models/glm-5.2-short.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2 Short" @@ -5,6 +7,7 @@ description = "Flagship GLM model for hybrid reasoning, coding, and agentic engi release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.2.toml b/providers/neuralwatt/models/glm-5.2.toml index 0b7b1191535..b83cbca3c04 100644 --- a/providers/neuralwatt/models/glm-5.2.toml +++ b/providers/neuralwatt/models/glm-5.2.toml @@ -1,3 +1,5 @@ +# Deprecated: Neuralwatt retired the GLM-5.2 family from GET /v1/models +# (checked 2026-09-18); retained for pricing history. # Budget: thinking_token_budget (integer reasoning tokens) base_model = "zhipuai/glm-5.2" name = "GLM 5.2" @@ -5,6 +7,7 @@ description = "Flagship GLM model for hybrid reasoning, coding, and agentic engi release_date = "2026-06-17" last_updated = "2026-06-17" structured_output = false +status = "deprecated" interleaved = true diff --git a/providers/neuralwatt/models/glm-5.3-flash-flex.toml b/providers/neuralwatt/models/glm-5.3-flash-flex.toml new file mode 100644 index 00000000000..502beae0882 --- /dev/null +++ b/providers/neuralwatt/models/glm-5.3-flash-flex.toml @@ -0,0 +1,34 @@ +# Flex-tier variant of glm-5.3-flash: same model, context (1,048,560), +# vision, and effort ladder as the standard tier; requested via the `-flex` +# model ID or service_tier="flex" and requires stream=true (non-streaming +# requests fall through to the standard tier). Cost is the flex rate: 0.65x +# of standard pricing (35% off) per the flex-tier guide, applied to the +# standard rates advertised by GET /v1/models ($0.15/$0.5/$0.03, checked +# 2026-09-18). JSON mode is off on this host, unlike the lab default. +# https://portal.neuralwatt.com/docs/guides/flex-tier +# https://portal.neuralwatt.com/docs/api/chat-completions +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3 Flash Flex" +structured_output = false + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.0975 +output = 0.325 +cache_read = 0.0195 + +[limit] +context = 1_048_560 +output = 1_048_560 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/neuralwatt/models/glm-5.3-flash.toml b/providers/neuralwatt/models/glm-5.3-flash.toml new file mode 100644 index 00000000000..8f87fce7fcc --- /dev/null +++ b/providers/neuralwatt/models/glm-5.3-flash.toml @@ -0,0 +1,35 @@ +# Exited preview 2026-09-11. Pricing, vision, and effort ladder from the +# public GET /v1/models metadata (checked 2026-09-18): $0.15/$0.5/$0.03, +# 1,048,560-token context, 20-image cap. Reasoning cannot be disabled +# (reasoning.mandatory = true), so there is no none level and no toggle; +# low, high and max are distinct efforts, with minimal/medium/xhigh accepted +# as aliases onto them. Budget: thinking_token_budget (integer reasoning +# tokens). Neuralwatt serves text+image input only (lab metadata also lists +# video/pdf). JSON mode is off on this host, unlike the lab default. +# https://portal.neuralwatt.com/docs/api/chat-completions +# https://portal.neuralwatt.com/models/glm-5.3-flash +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3 Flash" +structured_output = false + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.03 + +[limit] +context = 1_048_560 +output = 1_048_560 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/neuralwatt/models/glm-5.3-flex.toml b/providers/neuralwatt/models/glm-5.3-flex.toml new file mode 100644 index 00000000000..fa4a8aaef2c --- /dev/null +++ b/providers/neuralwatt/models/glm-5.3-flex.toml @@ -0,0 +1,30 @@ +# Flex-tier variant of glm-5.3: same model, context (1,048,560), and effort +# ladder as the standard tier; requested via the `-flex` model ID or +# service_tier="flex" and requires stream=true (non-streaming requests fall +# through to the standard tier). Cost is the flex rate: 0.65x of standard +# pricing (35% off) per the flex-tier guide, applied to the standard rates +# advertised by GET /v1/models ($1.45/$4.5/$0.145, checked 2026-09-18). +# Budget: thinking_token_budget (integer reasoning tokens) +# https://portal.neuralwatt.com/docs/guides/flex-tier +# https://portal.neuralwatt.com/docs/api/chat-completions +base_model = "zhipuai/glm-5.3" +name = "GLM 5.3 Flex" +structured_output = false + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.9425 +output = 2.925 +cache_read = 0.09425 + +[limit] +context = 1_048_560 +output = 1_048_560 diff --git a/providers/neuralwatt/models/glm-5.3.toml b/providers/neuralwatt/models/glm-5.3.toml index 700f54f026e..5e77d0a249d 100644 --- a/providers/neuralwatt/models/glm-5.3.toml +++ b/providers/neuralwatt/models/glm-5.3.toml @@ -1,17 +1,13 @@ -# Preview rollout: glm-5.3 is a gated preview on Neuralwatt, visible to -# accounts granted access per the Preview Models guide. Its list price is -# Neuralwatt's GLM 5.2 parity default and is marked for review at launch, -# per the model description in GET /v1/models (checked 2026-08-30). +# Exited preview 2026-09-10; the preview-parity list price ($1.45/$4.5/$0.145) +# was confirmed as the launch price per GET /v1/models (checked 2026-09-18). # Reasoning cannot be disabled (reasoning.mandatory = true), so there is no # none level and no toggle. The API supports low, high and max as distinct # efforts; minimal, medium and xhigh are accepted as aliases onto them. # Budget: thinking_token_budget (integer reasoning tokens) # https://portal.neuralwatt.com/docs/api/chat-completions -# https://portal.neuralwatt.com/docs/guides/preview-models base_model = "zhipuai/glm-5.3" name = "GLM 5.3" structured_output = false -status = "beta" interleaved = true diff --git a/providers/neuralwatt/models/qwen-3.8-27b-flex.toml b/providers/neuralwatt/models/qwen-3.8-27b-flex.toml new file mode 100644 index 00000000000..7021462ac7f --- /dev/null +++ b/providers/neuralwatt/models/qwen-3.8-27b-flex.toml @@ -0,0 +1,33 @@ +# Flex-tier variant of qwen-3.8-27b: same model, context (262,128), vision, +# and effort ladder as the standard tier; requested via the `-flex` model ID +# or service_tier="flex" and requires stream=true (non-streaming requests +# fall through to the standard tier). Cost is the flex rate: 0.65x of +# standard pricing (35% off) per the flex-tier guide, applied to the standard +# rates advertised by GET /v1/models ($0.45/$3.2/$0.25, checked 2026-09-18). +# Budget: thinking_token_budget (integer reasoning tokens) +# Neuralwatt serves text+image input only (lab metadata also lists video). +# https://portal.neuralwatt.com/docs/guides/flex-tier +# https://portal.neuralwatt.com/docs/api/chat-completions +base_model = "alibaba/qwen3.8-27b" +name = "Qwen3.8 27B Flex" +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.2925 +output = 2.08 +cache_read = 0.1625 + +[limit] +context = 262_128 +output = 131_072 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/neuralwatt/models/qwen-3.8-27b.toml b/providers/neuralwatt/models/qwen-3.8-27b.toml index baa1ccb1e23..e2749dc1438 100644 --- a/providers/neuralwatt/models/qwen-3.8-27b.toml +++ b/providers/neuralwatt/models/qwen-3.8-27b.toml @@ -1,6 +1,5 @@ -# Preview rollout: qwen-3.8-27b is in preview on Neuralwatt — early-access -# program, granted via the portal enroll page; absent from the public -# /v1/models catalog until access is granted. +# Left preview and joined the public GET /v1/models catalog; the output cap +# was raised from 65,536 to 131,072 tokens (checked 2026-09-18). # Effort: reasoning_effort = none|low|medium|xhigh (default xhigh, reasoning # on by default); aliases max/xhigh/high -> xhigh and minimal -> low, per the # chat-completions per-model table (docs verified against the live API on @@ -9,12 +8,10 @@ # model. # Neuralwatt serves text+image input only (lab metadata also lists video). # Pricing, context, output cap, and effort ladder from the model page and the -# authenticated GET /v1/models metadata (checked 2026-08-27): +# GET /v1/models metadata (checked 2026-09-18): # https://portal.neuralwatt.com/models/qwen-3.8-27b # https://portal.neuralwatt.com/docs/api/chat-completions -# https://portal.neuralwatt.com/docs/guides/preview-models base_model = "alibaba/qwen3.8-27b" -status = "beta" interleaved = true [[reasoning_options]] @@ -31,7 +28,7 @@ cache_read = 0.25 [limit] context = 262_128 -output = 65_536 +output = 131_072 [modalities] input = ["text", "image"] diff --git a/providers/neuralwatt/models/qwen3.6-35b-flex.toml b/providers/neuralwatt/models/qwen3.6-35b-flex.toml new file mode 100644 index 00000000000..3f8bd238eca --- /dev/null +++ b/providers/neuralwatt/models/qwen3.6-35b-flex.toml @@ -0,0 +1,32 @@ +# Flex-tier variant of qwen3.6-35b: same model, context (131,056), vision, +# and effort ladder as the standard tier; requested via the `-flex` model ID +# or service_tier="flex" and requires stream=true (non-streaming requests +# fall through to the standard tier). Cost is the flex rate: 0.65x of +# standard pricing (35% off) per the flex-tier guide, applied to the standard +# rates advertised by GET /v1/models ($0.29/$1.15/$0.029, checked 2026-09-18). +# Budget: thinking_token_budget (integer reasoning tokens) +# https://portal.neuralwatt.com/docs/guides/flex-tier +# https://portal.neuralwatt.com/docs/api/chat-completions +base_model = "alibaba/qwen3.6-35b-a3b" +name = "Qwen3.6 35B Flex" + +interleaved = true + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1885 +output = 0.7475 +cache_read = 0.01885 + +[limit] +context = 131_056 +output = 131_056 + +[modalities] +input = ["text", "image"] diff --git a/providers/ofox/models/anthropic/claude-opus-5.5.toml b/providers/ofox/models/anthropic/claude-opus-5.5.toml new file mode 100644 index 00000000000..23287398bb0 --- /dev/null +++ b/providers/ofox/models/anthropic/claude-opus-5.5.toml @@ -0,0 +1,16 @@ +# Sources: +# https://ofox.ai/models/anthropic/claude-opus-5.5 +# https://platform.claude.com/docs/en/models/opus-5-5/overview +# https://www.anthropic.com/claude-opus-5-5 +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[provider] +npm = "@ai-sdk/anthropic" +api = "https://api.ofox.ai/anthropic/v1" diff --git a/providers/ofox/models/openai/gpt-6-luna.toml b/providers/ofox/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..990db0bda79 --- /dev/null +++ b/providers/ofox/models/openai/gpt-6-luna.toml @@ -0,0 +1,20 @@ +# Sources: +# - Ofox model page: https://ofox.ai/models/openai/gpt-6-luna +# - Ofox catalog API: https://api.ofox.ai/v2/models/catalog?include=provider_price&search=gpt-6-luna +# - OpenAI model docs (pricing/reasoning): https://developers.openai.com/api/docs/models/gpt-6-luna +# Ofox provider_price override is flat $0.08/$0.40 (+ cache) with no context tier in the catalog, so no [[cost.tiers]]. +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.08 +output = 0.4 +cache_read = 0.008 +cache_write = 0.1 + +[provider] +npm = "@ai-sdk/openai" +api = "https://api.ofox.ai/v1" diff --git a/providers/ofox/models/openai/gpt-6-sol.toml b/providers/ofox/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..c686a7a19a1 --- /dev/null +++ b/providers/ofox/models/openai/gpt-6-sol.toml @@ -0,0 +1,20 @@ +# Sources: +# - Ofox model page: https://ofox.ai/models/openai/gpt-6-sol +# - Ofox catalog API: https://api.ofox.ai/v2/models/catalog?include=provider_price&search=gpt-6-sol +# - OpenAI model docs (reasoning effort + list pricing): https://developers.openai.com/api/docs/models/gpt-6-sol +# Ofox provider_price is the -20% promo override ($1.6/$8 + cache vs $2/$10 list); flat with no >=272k context tier in the catalog, so no [[cost.tiers]]. +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 1.6 +output = 8 +cache_read = 0.16 +cache_write = 2 + +[provider] +npm = "@ai-sdk/openai" +api = "https://api.ofox.ai/v1" diff --git a/providers/ofox/models/x-ai/grok-4.7.toml b/providers/ofox/models/x-ai/grok-4.7.toml new file mode 100644 index 00000000000..b829098d460 --- /dev/null +++ b/providers/ofox/models/x-ai/grok-4.7.toml @@ -0,0 +1,16 @@ +# Ofox: x-ai/grok-4.7 → xAI Grok 4.7 +# Sources: +# - https://api.ofox.ai/v1/models/x-ai/grok-4.7 (prompt $2/M, completion $6/M, cache_read $0.5/M; context 500k; max_completion 65536; supported_parameters includes reasoning) +# - https://ofox.ai/models/x-ai/grok-4.7 (same pricing; max output 66K; reasoning + tools + prompt caching) +# - https://docs.x.ai/developers/models/grok-4.7 (reasoning_effort low|medium|high|xhigh; cannot disable) +# reasoning_options match first-party xAI + relay peers (no none) +base_model = "xai/grok-4.7" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[limit] +output = 65_536 diff --git a/providers/openai/models/gpt-6-luna.toml b/providers/openai/models/gpt-6-luna.toml new file mode 100644 index 00000000000..3a132f7fabb --- /dev/null +++ b/providers/openai/models/gpt-6-luna.toml @@ -0,0 +1,24 @@ +# Pricing: https://developers.openai.com/api/docs/models/gpt-6-luna +# Reasoning and modes: https://developers.openai.com/api/docs/guides/latest-model?model=gpt-6-luna +base_model = "openai/gpt-6-luna" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.10 +output = 0.50 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.20 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[experimental.modes.fast] +cost = { input = 0.20, output = 1.00, cache_read = 0.02, cache_write = 0.25 } +provider = { body = { service_tier = "priority" } } + +[experimental.modes.pro] +provider = { body = { reasoning = { mode = "pro" } } } diff --git a/providers/openai/models/gpt-6-sol.toml b/providers/openai/models/gpt-6-sol.toml new file mode 100644 index 00000000000..e141e9653ad --- /dev/null +++ b/providers/openai/models/gpt-6-sol.toml @@ -0,0 +1,24 @@ +# Pricing: https://developers.openai.com/api/docs/models/gpt-6-sol +# Reasoning and modes: https://developers.openai.com/api/docs/guides/latest-model?model=gpt-6-sol +base_model = "openai/gpt-6-sol" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 2.00 +output = 10.00 +cache_read = 0.20 +cache_write = 2.50 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4.00 +output = 15.00 +cache_read = 0.40 +cache_write = 5.00 + +[experimental.modes.fast] +cost = { input = 4.00, output = 20.00, cache_read = 0.40, cache_write = 5.00 } +provider = { body = { service_tier = "priority" } } + +[experimental.modes.pro] +provider = { body = { reasoning = { mode = "pro" } } } diff --git a/providers/opencode-go/models/grok-4.7.toml b/providers/opencode-go/models/grok-4.7.toml new file mode 100644 index 00000000000..edde035f699 --- /dev/null +++ b/providers/opencode-go/models/grok-4.7.toml @@ -0,0 +1,19 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 + +[provider] +npm = "@ai-sdk/openai" diff --git a/providers/opencode-go/models/mimo-v2.6-flash.toml b/providers/opencode-go/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..06b0dcf1faa --- /dev/null +++ b/providers/opencode-go/models/mimo-v2.6-flash.toml @@ -0,0 +1,10 @@ +base_model = "xiaomi/mimo-v2.6-flash" +reasoning_options = [] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 diff --git a/providers/opencode-go/models/mimo-v2.6-pro.toml b/providers/opencode-go/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..7513e538ec8 --- /dev/null +++ b/providers/opencode-go/models/mimo-v2.6-pro.toml @@ -0,0 +1,10 @@ +base_model = "xiaomi/mimo-v2.6-pro" +reasoning_options = [] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.003625 diff --git a/providers/opencode-go/models/minimax-m2.5.toml b/providers/opencode-go/models/minimax-m2.5.toml index 81fcd79363a..f6973ebf170 100644 --- a/providers/opencode-go/models/minimax-m2.5.toml +++ b/providers/opencode-go/models/minimax-m2.5.toml @@ -15,7 +15,8 @@ status = "deprecated" [cost] input = 0.3 output = 1.2 -cache_read = 0.03 +cache_read = 0.06 +cache_write = 0.375 [limit] context = 204_800 diff --git a/providers/opencode-go/models/minimax-m2.7.toml b/providers/opencode-go/models/minimax-m2.7.toml index 311a9c4af6f..33faddac7d9 100644 --- a/providers/opencode-go/models/minimax-m2.7.toml +++ b/providers/opencode-go/models/minimax-m2.7.toml @@ -15,6 +15,7 @@ open_weights = true input = 0.3 output = 1.2 cache_read = 0.06 +cache_write = 0.375 [limit] context = 204_800 diff --git a/providers/opencode-go/provider.toml b/providers/opencode-go/provider.toml index 833a7cb0002..7ddf38e1f5d 100644 --- a/providers/opencode-go/provider.toml +++ b/providers/opencode-go/provider.toml @@ -8,4 +8,4 @@ npm = "@ai-sdk/openai-compatible" # by the public Zen page. # https://opencode.ai/docs/zen#endpoints api = "https://opencode.ai/zen/go/v1" -doc = "https://opencode.ai/docs/zen" +doc = "https://opencode.ai/docs/go" diff --git a/providers/opencode/models/claude-opus-5-5.toml b/providers/opencode/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..e39cc456019 --- /dev/null +++ b/providers/opencode/models/claude-opus-5-5.toml @@ -0,0 +1,12 @@ +# Pricing: https://platform.claude.com/docs/en/models/opus-5-5/overview +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 + +[provider] +npm = "@ai-sdk/anthropic" diff --git a/providers/opper/models/moonshot/kimi-k3.toml b/providers/opencode/models/deepseek-v4.1-flash.toml similarity index 51% rename from providers/opper/models/moonshot/kimi-k3.toml rename to providers/opencode/models/deepseek-v4.1-flash.toml index 0862dd21e4c..20a4928d010 100644 --- a/providers/opper/models/moonshot/kimi-k3.toml +++ b/providers/opencode/models/deepseek-v4.1-flash.toml @@ -1,13 +1,14 @@ -base_model = "moonshotai/kimi-k3" +base_model = "deepseek/deepseek-v4.1-flash" +name = "DeepSeek V4.1 Flash" [[reasoning_options]] type = "effort" values = ["low", "high", "max"] -[cost] -input = 3 -output = 15 -cache_read = 0.3 - [interleaved] field = "reasoning_content" + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.006 diff --git a/providers/opencode/models/gpt-6-luna.toml b/providers/opencode/models/gpt-6-luna.toml new file mode 100644 index 00000000000..1263a99bdc4 --- /dev/null +++ b/providers/opencode/models/gpt-6-luna.toml @@ -0,0 +1,19 @@ +# Pricing and controls: https://developers.openai.com/api/docs/models/gpt-6-luna +base_model = "openai/gpt-6-luna" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[provider] +npm = "@ai-sdk/openai" diff --git a/providers/opencode/models/gpt-6-sol.toml b/providers/opencode/models/gpt-6-sol.toml new file mode 100644 index 00000000000..eefd5c4f407 --- /dev/null +++ b/providers/opencode/models/gpt-6-sol.toml @@ -0,0 +1,19 @@ +# Pricing and controls: https://developers.openai.com/api/docs/models/gpt-6-sol +base_model = "openai/gpt-6-sol" +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 + +[provider] +npm = "@ai-sdk/openai" diff --git a/providers/opencode/models/grok-4.7.toml b/providers/opencode/models/grok-4.7.toml new file mode 100644 index 00000000000..e121d9b0cd5 --- /dev/null +++ b/providers/opencode/models/grok-4.7.toml @@ -0,0 +1,21 @@ +# Pricing: 30% off OpenCode Go Grok 4.7 rates, including long-context tiers. +base_model = "xai/grok-4.7" +name = "Grok 4.7 (30% Off)" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 1.4 +output = 4.2 +cache_read = 0.35 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.8 +output = 8.4 +cache_read = 0.7 + +[provider] +npm = "@ai-sdk/openai" diff --git a/providers/opencode/models/jev-1.13-free.toml b/providers/opencode/models/jev-1.13-free.toml new file mode 100644 index 00000000000..3c2a23a4f39 --- /dev/null +++ b/providers/opencode/models/jev-1.13-free.toml @@ -0,0 +1,7 @@ +# Availability: https://opencode.ai/zen/v1/models +base_model = "typesafe/jev-latest" +name = "Jev 1.13 Free" + +[cost] +input = 0 +output = 0 diff --git a/providers/opencode/models/jev-1.13.toml b/providers/opencode/models/jev-1.13.toml new file mode 100644 index 00000000000..c31caeddd67 --- /dev/null +++ b/providers/opencode/models/jev-1.13.toml @@ -0,0 +1,8 @@ +# Availability: https://opencode.ai/zen/v1/models +# Pricing: https://docs.typesafe.ai/models.md +base_model = "typesafe/jev-latest" +name = "Jev 1.13" + +[cost] +input = 0.042 +output = 0 diff --git a/providers/opencode/models/jev-latest.toml b/providers/opencode/models/jev-latest.toml deleted file mode 100644 index 409cf1a8931..00000000000 --- a/providers/opencode/models/jev-latest.toml +++ /dev/null @@ -1,6 +0,0 @@ -# Pricing source: https://typesafe.ai/blog/introducing-system-one-models-and-jev -base_model = "typesafe/jev-latest" - -[cost] -input = 0.042 -output = 0 diff --git a/providers/opencode/models/mimo-v2.5-free.toml b/providers/opencode/models/mimo-v2.5-free.toml index 3a1d9a269a6..d799e874914 100644 --- a/providers/opencode/models/mimo-v2.5-free.toml +++ b/providers/opencode/models/mimo-v2.5-free.toml @@ -10,6 +10,7 @@ temperature = true tool_call = true knowledge = "2024-12" open_weights = true +status = "deprecated" [interleaved] field = "reasoning_content" diff --git a/providers/opencode/models/mimo-v2.6-flash-free.toml b/providers/opencode/models/mimo-v2.6-flash-free.toml new file mode 100644 index 00000000000..753e0075aa2 --- /dev/null +++ b/providers/opencode/models/mimo-v2.6-flash-free.toml @@ -0,0 +1,15 @@ +base_model = "xiaomi/mimo-v2.6-flash" +name = "MiMo-V2.6-Flash Free" +reasoning_options = [] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 + +[limit] +context = 200_000 +output = 32_000 diff --git a/providers/opencode/models/muse-spark-1.3.toml b/providers/opencode/models/muse-spark-1.3.toml index cfaa3f534d3..80f0b886543 100644 --- a/providers/opencode/models/muse-spark-1.3.toml +++ b/providers/opencode/models/muse-spark-1.3.toml @@ -1,5 +1,5 @@ base_model = "meta/muse-spark-1.3" -reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high", "xhigh", "max"] }] +reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high", "xhigh"] }] [cost] input = 1.25 diff --git a/providers/opencode/models/qwen3.8-flash.toml b/providers/opencode/models/qwen3.8-flash.toml new file mode 100644 index 00000000000..1cd079b5cb3 --- /dev/null +++ b/providers/opencode/models/qwen3.8-flash.toml @@ -0,0 +1,25 @@ +# https://opencode.ai/docs/go/#endpoints +# https://help.aliyun.com/en/model-studio/anthropic-api-messages +# Toggle (Messages): thinking.type = enabled|disabled +# Effort (Messages): output_config.effort = low|medium|xhigh +base_model = "alibaba/qwen3.8-flash" +temperature = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.15 +output = 0.47 +cache_read = 0.016 +cache_write = 0.2 + +[provider] +npm = "@ai-sdk/anthropic" \ No newline at end of file diff --git a/providers/openrouter/models/anthropic/claude-opus-4.toml b/providers/openrouter/models/anthropic/claude-opus-4.toml deleted file mode 100644 index 094f7f59203..00000000000 --- a/providers/openrouter/models/anthropic/claude-opus-4.toml +++ /dev/null @@ -1,31 +0,0 @@ -# Toggle: reasoning.enabled = true|false -# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens -name = "Claude Opus 4" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -family = "claude-opus" -release_date = "2025-05-22" -last_updated = "2025-05-22" -attachment = true -reasoning = true -temperature = true -tool_call = true -structured_output = false -knowledge = "2025-01-31" -open_weights = false - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 15 -output = 75 -cache_read = 1.5 -cache_write = 18.75 - -[limit] -context = 200_000 -output = 32_000 - -[modalities] -input = ["image", "text", "pdf"] -output = ["text"] diff --git a/providers/openrouter/models/anthropic/claude-opus-5.5.toml b/providers/openrouter/models/anthropic/claude-opus-5.5.toml new file mode 100644 index 00000000000..70709244ff6 --- /dev/null +++ b/providers/openrouter/models/anthropic/claude-opus-5.5.toml @@ -0,0 +1,14 @@ +base_model = "anthropic/claude-opus-5-5" +description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" +temperature = true +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/openrouter/models/anthropic/claude-sonnet-4.toml b/providers/openrouter/models/anthropic/claude-sonnet-4.toml index 7baa5f4f635..b7b54a299a9 100644 --- a/providers/openrouter/models/anthropic/claude-sonnet-4.toml +++ b/providers/openrouter/models/anthropic/claude-sonnet-4.toml @@ -30,7 +30,7 @@ cache_read = 0.6 cache_write = 7.5 [limit] -context = 1_000_000 +context = 200_000 output = 64_000 [modalities] diff --git a/providers/openrouter/models/cohere/command-a-plus.toml b/providers/openrouter/models/cohere/command-a-plus.toml new file mode 100644 index 00000000000..a3ee649020e --- /dev/null +++ b/providers/openrouter/models/cohere/command-a-plus.toml @@ -0,0 +1,29 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +name = "Command A+" +description = "Cohere command model for multilingual enterprise agents, tools, and chat" +family = "command-a" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.3 +output = 1.5 +cache_read = 0.15 + +[limit] +context = 192_000 +output = 64_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml index df31c0bbb32..3de935d9215 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.06 -output = 0.12 -cache_read = 0.012 +input = 0.04 +output = 0.64 +cache_read = 0.016 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731:free.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-0731:free.toml deleted file mode 100644 index 1f431a902ab..00000000000 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-0731:free.toml +++ /dev/null @@ -1,20 +0,0 @@ -# Toggle: reasoning.enabled = true|false -# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens -base_model = "deepseek/deepseek-v4-flash-0731" -name = "DeepSeek V4 Flash 0731 (free)" -description = "Fast DeepSeek model for efficient chat, coding help, and agent loops" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 -output = 393_216 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-flash-vision-exp.toml b/providers/openrouter/models/deepseek/deepseek-v4-flash-vision-exp.toml index 0417e3027d5..2186c60b6fb 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-flash-vision-exp.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-flash-vision-exp.toml @@ -11,10 +11,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.2156 -output = 0.6468 -cache_read = 0.00686 +input = 0.22 +output = 0.66 +cache_read = 0.007 [limit] context = 1_048_576 -output = 262_144 +output = 943_718 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml index 7c5828b16e2..b8d53b0809c 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro-0813.toml @@ -10,9 +10,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 1.32 +output = 3.96 +cache_read = 0.044 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml index 95460b6f5f3..e7683af4617 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4-pro.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4-pro.toml @@ -13,10 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 1.6 -output = 3.2 -cache_read = 0.135 +input = 0.95526 +output = 1.91052 +cache_read = 0.079605 [limit] context = 1_048_576 -output = 393_216 diff --git a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml index 16f058ed841..ffb1c4c680d 100644 --- a/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/openrouter/models/deepseek/deepseek-v4.1-flash.toml @@ -11,9 +11,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.15 +input = 0.039 output = 0.6 -cache_read = 0.003 +cache_read = 0.038 [limit] context = 1_048_576 +output = 943_718 diff --git a/providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml b/providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml deleted file mode 100644 index 80b0f00e52d..00000000000 --- a/providers/openrouter/models/kwaipilot/kat-coder-pro-v2.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "KAT-Coder-Pro V2" -description = "Coding model for repository understanding, refactors, and agentic engineering tasks" -family = "kat-coder" -release_date = "2026-03-27" -last_updated = "2026-03-27" -attachment = false -reasoning = false -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.3 -output = 1.2 -cache_read = 0.06 - -[limit] -context = 262_144 -output = 144_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/openrouter/models/meta/muse-glimmer-30b.toml b/providers/openrouter/models/meta/muse-glimmer-30b.toml index 12578d9072c..d4077f81891 100644 --- a/providers/openrouter/models/meta/muse-glimmer-30b.toml +++ b/providers/openrouter/models/meta/muse-glimmer-30b.toml @@ -11,8 +11,8 @@ values = ["low", "medium", "high", "xhigh"] [cost] input = 0.3 -output = 1.1 +output = 1.2 cache_read = 0.04 [limit] -output = 117_964 +output = 16_384 diff --git a/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml b/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml index 45893fba68d..bc32aea52eb 100644 --- a/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml +++ b/providers/openrouter/models/mistralai/mistral-small-3.1-24b-instruct.toml @@ -6,7 +6,7 @@ last_updated = "2025-03-17" attachment = true reasoning = false temperature = true -tool_call = false +tool_call = true structured_output = false knowledge = "2023-10-31" open_weights = true diff --git a/providers/openrouter/models/moonshotai/kimi-k2.7-code.toml b/providers/openrouter/models/moonshotai/kimi-k2.7-code.toml index 307383582d9..3fb61dea6a2 100644 --- a/providers/openrouter/models/moonshotai/kimi-k2.7-code.toml +++ b/providers/openrouter/models/moonshotai/kimi-k2.7-code.toml @@ -4,7 +4,7 @@ reasoning_options = [] [cost] input = 0.7062 -output = 3.21 +output = 3.3 cache_read = 0.18 [limit] diff --git a/providers/openrouter/models/moonshotai/kimi-k3.toml b/providers/openrouter/models/moonshotai/kimi-k3.toml index 618c6fb2bbd..c11eb6c3dae 100644 --- a/providers/openrouter/models/moonshotai/kimi-k3.toml +++ b/providers/openrouter/models/moonshotai/kimi-k3.toml @@ -12,9 +12,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 3 +output = 15 +cache_read = 0.3 [limit] output = 943_718 diff --git a/providers/openrouter/models/nex-agi/nex-n2.5-mini.toml b/providers/openrouter/models/nex-agi/nex-n2.5-mini.toml new file mode 100644 index 00000000000..539fc6a6686 --- /dev/null +++ b/providers/openrouter/models/nex-agi/nex-n2.5-mini.toml @@ -0,0 +1,28 @@ +name = "Nex-N2.5-Mini" +description = "Multimodal reasoning model for visual analysis, planning, and tool use" +family = "agi" +release_date = "2026-09-08" +last_updated = "2026-09-08" +attachment = true +reasoning = true +temperature = true +tool_call = false +structured_output = true +open_weights = true + +[[reasoning_options]] +type = "effort" +values = ["none", "medium", "high"] + +[cost] +input = 0.025 +output = 0.1 +cache_read = 0.0025 + +[limit] +context = 262_144 +output = 235_929 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/nex-agi/nex-n2.5-pro.toml b/providers/openrouter/models/nex-agi/nex-n2.5-pro.toml new file mode 100644 index 00000000000..0cb11065de4 --- /dev/null +++ b/providers/openrouter/models/nex-agi/nex-n2.5-pro.toml @@ -0,0 +1,28 @@ +name = "Nex-N2.5-Pro" +description = "Multimodal reasoning model for visual analysis, planning, and tool use" +family = "agi" +release_date = "2026-09-08" +last_updated = "2026-09-08" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[[reasoning_options]] +type = "effort" +values = ["none", "medium", "high"] + +[cost] +input = 0.075 +output = 0.25 +cache_read = 0.015 + +[limit] +context = 262_144 +output = 235_929 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml b/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml index 297dfb56987..bf732029012 100644 --- a/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml +++ b/providers/openrouter/models/nvidia/nemotron-3-nano-30b-a3b.toml @@ -7,8 +7,9 @@ structured_output = true type = "toggle" [cost] -input = 0.06 -output = 0.24 +input = 0.05 +output = 0.2 +cache_read = 0.03 [limit] output = 235_929 diff --git a/providers/openrouter/models/nvidia/nemotron-3-ultra-550b-a55b.toml b/providers/openrouter/models/nvidia/nemotron-3-ultra-550b-a55b.toml index f471e71439f..5dcd117448e 100644 --- a/providers/openrouter/models/nvidia/nemotron-3-ultra-550b-a55b.toml +++ b/providers/openrouter/models/nvidia/nemotron-3-ultra-550b-a55b.toml @@ -14,10 +14,10 @@ values = ["medium", "high"] type = "budget_tokens" [cost] -input = 0.625 -output = 3.125 -cache_read = 0.1875 +input = 0.6 +output = 2.4 +cache_read = 0.12 [limit] context = 262_144 -output = 32_768 +output = 182_520 diff --git a/providers/openrouter/models/nvidia/nemotron-3.5-lightning.toml b/providers/openrouter/models/nvidia/nemotron-3.5-lightning.toml index 3b251679339..842e5779095 100644 --- a/providers/openrouter/models/nvidia/nemotron-3.5-lightning.toml +++ b/providers/openrouter/models/nvidia/nemotron-3.5-lightning.toml @@ -7,9 +7,9 @@ description = "Nemotron model for efficient reasoning, coding, and specialized A type = "toggle" [cost] -input = 0.08 +input = 0.07 output = 0.2 cache_read = 0.04 [limit] -output = 131_072 +output = 235_929 diff --git a/providers/openrouter/models/openai/gpt-6-luna-pro.toml b/providers/openrouter/models/openai/gpt-6-luna-pro.toml new file mode 100644 index 00000000000..f3f1dd66bc1 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-6-luna-pro.toml @@ -0,0 +1,36 @@ +name = "GPT-6 Luna Pro" +description = "Frontier GPT model for professional reasoning, coding, and multimodal work" +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-6-luna.toml b/providers/openrouter/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..be8a0fb9e1b --- /dev/null +++ b/providers/openrouter/models/openai/gpt-6-luna.toml @@ -0,0 +1,19 @@ +base_model = "openai/gpt-6-luna" +description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 diff --git a/providers/openrouter/models/openai/gpt-6-sol-pro.toml b/providers/openrouter/models/openai/gpt-6-sol-pro.toml new file mode 100644 index 00000000000..a34a53e3bb1 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-6-sol-pro.toml @@ -0,0 +1,36 @@ +name = "GPT-6 Sol Pro" +description = "Frontier GPT model for professional reasoning, coding, and multimodal work" +family = "gpt" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = false +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 + +[limit] +context = 1_050_000 +output = 128_000 + +[modalities] +input = ["pdf", "image", "text"] +output = ["text"] diff --git a/providers/openrouter/models/openai/gpt-6-sol.toml b/providers/openrouter/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..83f598548a3 --- /dev/null +++ b/providers/openrouter/models/openai/gpt-6-sol.toml @@ -0,0 +1,19 @@ +base_model = "openai/gpt-6-sol" +description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 diff --git a/providers/openrouter/models/openai/gpt-oss-20b.toml b/providers/openrouter/models/openai/gpt-oss-20b.toml index b65dd3718e0..98aeb108997 100644 --- a/providers/openrouter/models/openai/gpt-oss-20b.toml +++ b/providers/openrouter/models/openai/gpt-oss-20b.toml @@ -6,9 +6,5 @@ type = "effort" values = ["low", "medium", "high"] [cost] -input = 0.03 -output = 0.13 -cache_read = 0.03 - -[limit] -output = 117_964 +input = 0.018 +output = 0.09 diff --git a/providers/openrouter/models/prism-ml/ternary-bonsai-2-27b.toml b/providers/openrouter/models/prism-ml/ternary-bonsai-2-27b.toml new file mode 100644 index 00000000000..316e8fde320 --- /dev/null +++ b/providers/openrouter/models/prism-ml/ternary-bonsai-2-27b.toml @@ -0,0 +1,31 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +name = "Ternary Bonsai 2 27B" +description = "Multimodal reasoning model for visual analysis, planning, and tool use" +release_date = "2026-09-18" +last_updated = "2026-09-18" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["medium", "xhigh"] + +[cost] +input = 0.075 +output = 0.5 + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml b/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml index 14e222d894d..35f13bee686 100644 --- a/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml +++ b/providers/openrouter/models/qwen/qwen3-next-80b-a3b-thinking.toml @@ -8,3 +8,4 @@ output = 1.2 [limit] context = 262_144 +output = 235_929 diff --git a/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml b/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml index 34aa7a00097..8257248c271 100644 --- a/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml +++ b/providers/openrouter/models/qwen/qwen3.5-35b-a3b.toml @@ -6,8 +6,12 @@ base_model = "alibaba/qwen3.5-35b-a3b" type = "toggle" [cost] -input = 0.1625 -output = 1.3 +input = 0.3125 +output = 1.25 +cache_read = 0.15625 + +[limit] +output = 16_384 [modalities] input = ["text", "image", "video"] diff --git a/providers/openrouter/models/qwen/qwen3.5-9b.toml b/providers/openrouter/models/qwen/qwen3.5-9b.toml index 34660f4a8ed..8f34db0bc11 100644 --- a/providers/openrouter/models/qwen/qwen3.5-9b.toml +++ b/providers/openrouter/models/qwen/qwen3.5-9b.toml @@ -1,3 +1,5 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens base_model = "alibaba/qwen3.5-9b" [[reasoning_options]] @@ -8,4 +10,4 @@ input = 0.1 output = 0.15 [limit] -output = 235_929 +output = 32_768 diff --git a/providers/openrouter/models/qwen/qwen3.6-27b.toml b/providers/openrouter/models/qwen/qwen3.6-27b.toml index 15ca4007021..c188473afdb 100644 --- a/providers/openrouter/models/qwen/qwen3.6-27b.toml +++ b/providers/openrouter/models/qwen/qwen3.6-27b.toml @@ -6,9 +6,12 @@ base_model = "alibaba/qwen3.6-27b" type = "toggle" [cost] -input = 0.3 -output = 2 -cache_read = 0.03 +input = 0.32 +output = 2.7 +cache_read = 0.15 + +[limit] +output = 262_140 [modalities] input = ["text", "image", "video"] diff --git a/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml b/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml index ac95b282785..a223e74df58 100644 --- a/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml +++ b/providers/openrouter/models/qwen/qwen3.6-35b-a3b.toml @@ -6,8 +6,8 @@ base_model = "alibaba/qwen3.6-35b-a3b" type = "toggle" [cost] -input = 0.1 -output = 0.9 +input = 0.15 +output = 1 cache_read = 0.05 [limit] diff --git a/providers/openrouter/models/qwen/qwen3.8-27b.toml b/providers/openrouter/models/qwen/qwen3.8-27b.toml index 488923c2abf..d1ddcffa2ed 100644 --- a/providers/openrouter/models/qwen/qwen3.8-27b.toml +++ b/providers/openrouter/models/qwen/qwen3.8-27b.toml @@ -11,9 +11,9 @@ type = "effort" values = ["low", "medium", "xhigh"] [cost] -input = 0.214 -output = 2.55 -cache_read = 0.15 +input = 0.42 +output = 3 +cache_read = 0.085 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/qwen/qwen3.8-omni-flash.toml b/providers/openrouter/models/qwen/qwen3.8-omni-flash.toml new file mode 100644 index 00000000000..d5769de2059 --- /dev/null +++ b/providers/openrouter/models/qwen/qwen3.8-omni-flash.toml @@ -0,0 +1,16 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +base_model = "alibaba/qwen3.8-omni-flash" +temperature = true +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.15 +output = 0.47 +cache_read = 0.016 diff --git a/providers/openrouter/models/upstage/solar-pro-3.toml b/providers/openrouter/models/upstage/solar-pro-3.toml index 146505f4863..c9eda434a24 100644 --- a/providers/openrouter/models/upstage/solar-pro-3.toml +++ b/providers/openrouter/models/upstage/solar-pro-3.toml @@ -13,7 +13,8 @@ structured_output = true open_weights = false [[reasoning_options]] -type = "toggle" +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] [cost] input = 0.15 diff --git a/providers/openrouter/models/upstage/solar-pro4.toml b/providers/openrouter/models/upstage/solar-pro4.toml index adc9761fb07..7e3d19352a7 100644 --- a/providers/openrouter/models/upstage/solar-pro4.toml +++ b/providers/openrouter/models/upstage/solar-pro4.toml @@ -13,7 +13,8 @@ structured_output = true open_weights = false [[reasoning_options]] -type = "toggle" +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 0.09 diff --git a/providers/openrouter/models/x-ai/grok-4.7.toml b/providers/openrouter/models/x-ai/grok-4.7.toml new file mode 100644 index 00000000000..3ffa2ff7ae3 --- /dev/null +++ b/providers/openrouter/models/x-ai/grok-4.7.toml @@ -0,0 +1,20 @@ +base_model = "xai/grok-4.7" +description = "Grok model for agentic tool use, reasoning, coding, and live assistance" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 1.6 +output = 4.8 +cache_read = 0.4 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 3.2 +output = 9.6 +cache_read = 0.8 + +[limit] +output = 450_000 diff --git a/providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml b/providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..e1bb054f013 --- /dev/null +++ b/providers/openrouter/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,13 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +base_model = "xiaomi/mimo-v2.6-flash" +description = "MiMo flash model for fast multimodal assistance and agent workflows" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 diff --git a/providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..e7a245618f3 --- /dev/null +++ b/providers/openrouter/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,13 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +base_model = "xiaomi/mimo-v2.6-pro-ultraspeed" +description = "MiMo pro model for strong multimodal reasoning and agent execution" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 4.35 +output = 8.7 +cache_read = 0.036 diff --git a/providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml b/providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..7f938318372 --- /dev/null +++ b/providers/openrouter/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,13 @@ +# Toggle: reasoning.enabled = true|false +# https://openrouter.ai/docs/guides/best-practices/reasoning-tokens +base_model = "xiaomi/mimo-v2.6-pro" +description = "MiMo pro model for strong multimodal reasoning and agent execution" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 diff --git a/providers/openrouter/models/z-ai/glm-5.2.toml b/providers/openrouter/models/z-ai/glm-5.2.toml index 426a4d8f486..e1246ed1b0a 100644 --- a/providers/openrouter/models/z-ai/glm-5.2.toml +++ b/providers/openrouter/models/z-ai/glm-5.2.toml @@ -13,10 +13,9 @@ type = "effort" values = ["high", "xhigh"] [cost] -input = 0.5625 -output = 1.8 -cache_read = 0.105 +input = 0.6496 +output = 2.0416 +cache_read = 0.12064 [limit] context = 1_048_576 -output = 163_840 diff --git a/providers/openrouter/models/z-ai/glm-5.3-flash.toml b/providers/openrouter/models/z-ai/glm-5.3-flash.toml index f82829f96ea..188a84f8fb5 100644 --- a/providers/openrouter/models/z-ai/glm-5.3-flash.toml +++ b/providers/openrouter/models/z-ai/glm-5.3-flash.toml @@ -6,12 +6,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.09 -output = 0.3 -cache_read = 0.018 +input = 0.15 +output = 0.5 +cache_read = 0.05 [limit] context = 1_310_720 +output = 943_718 [modalities] input = ["text", "image", "video"] diff --git a/providers/openrouter/models/z-ai/glm-5.3-flashx.toml b/providers/openrouter/models/z-ai/glm-5.3-flashx.toml new file mode 100644 index 00000000000..ea5b2ee8747 --- /dev/null +++ b/providers/openrouter/models/z-ai/glm-5.3-flashx.toml @@ -0,0 +1,28 @@ +name = "GLM 5.3 FlashX" +description = "GLM vision model for visual reasoning, documents, and multimodal agents" +family = "glm" +release_date = "2026-09-18" +last_updated = "2026-09-18" +attachment = true +reasoning = true +temperature = true +tool_call = true +structured_output = false +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.37 +output = 1.25 +cache_read = 0.075 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/openrouter/models/z-ai/glm-5.3.toml b/providers/openrouter/models/z-ai/glm-5.3.toml index d077461d393..8b9af97ef26 100644 --- a/providers/openrouter/models/z-ai/glm-5.3.toml +++ b/providers/openrouter/models/z-ai/glm-5.3.toml @@ -6,10 +6,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 1.4 -output = 4.4 -cache_read = 0.26 +input = 0.84 +output = 2.64 +cache_read = 0.156 [limit] context = 1_310_720 -output = 943_717 diff --git a/providers/openrouter/models/~anthropic/claude-opus-latest.toml b/providers/openrouter/models/~anthropic/claude-opus-latest.toml index f4dce5e6b4e..8057256dd8c 100644 --- a/providers/openrouter/models/~anthropic/claude-opus-latest.toml +++ b/providers/openrouter/models/~anthropic/claude-opus-latest.toml @@ -12,18 +12,15 @@ tool_call = true structured_output = true open_weights = false -[[reasoning_options]] -type = "toggle" - [[reasoning_options]] type = "effort" values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 5 -output = 25 -cache_read = 0.5 -cache_write = 6.25 +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 [limit] context = 1_000_000 diff --git a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml index 07784758cd2..c649cf1995c 100644 --- a/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.1485 -output = 0.594 -cache_read = 0.004455 +input = 0.039 +output = 0.6 +cache_read = 0.038 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml index ea09a0d12b6..3112dca3caa 100644 --- a/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-pro-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.66 -output = 1.98 -cache_read = 0.022 +input = 0.4 +output = 4.3 +cache_read = 0.033 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml index 7356cb84c22..6c267e8f600 100644 --- a/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml +++ b/providers/openrouter/models/~deepseek/deepseek-v4-flash-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.0558 -output = 0.1767 -cache_read = 0.0088 +input = 0.03 +output = 0.8 +cache_read = 0.008 [limit] context = 1_310_720 diff --git a/providers/openrouter/models/~moonshotai/kimi-latest.toml b/providers/openrouter/models/~moonshotai/kimi-latest.toml index d1d254d74de..4c30be833f9 100644 --- a/providers/openrouter/models/~moonshotai/kimi-latest.toml +++ b/providers/openrouter/models/~moonshotai/kimi-latest.toml @@ -20,9 +20,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 2.1 -output = 10.95 -cache_read = 0.23 +input = 1.49 +output = 14.5 +cache_read = 0.21 [limit] context = 1_048_576 diff --git a/providers/openrouter/models/~openai/gpt-luna-latest.toml b/providers/openrouter/models/~openai/gpt-luna-latest.toml index e349af31375..5324cf22735 100644 --- a/providers/openrouter/models/~openai/gpt-luna-latest.toml +++ b/providers/openrouter/models/~openai/gpt-luna-latest.toml @@ -16,17 +16,17 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 0.2 -output = 1.2 -cache_read = 0.02 -cache_write = 0.25 +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 [[cost.tiers]] tier = { type = "context", size = 272_000 } -input = 0.4 -output = 1.8 -cache_read = 0.04 -cache_write = 0.5 +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 [limit] context = 1_050_000 diff --git a/providers/openrouter/models/~x-ai/grok-latest.toml b/providers/openrouter/models/~x-ai/grok-latest.toml index ef7efee492e..2ce36baf40e 100644 --- a/providers/openrouter/models/~x-ai/grok-latest.toml +++ b/providers/openrouter/models/~x-ai/grok-latest.toml @@ -15,15 +15,15 @@ type = "effort" values = ["low", "medium", "high", "xhigh"] [cost] -input = 2 -output = 6 -cache_read = 0.5 +input = 1.6 +output = 4.8 +cache_read = 0.4 [[cost.tiers]] tier = { type = "context", size = 200_000 } -input = 4 -output = 12 -cache_read = 1 +input = 3.2 +output = 9.6 +cache_read = 0.8 [limit] context = 500_000 diff --git a/providers/openrouter/models/~z-ai/glm-latest.toml b/providers/openrouter/models/~z-ai/glm-latest.toml index 3e3ce108967..9bbf416c3f6 100644 --- a/providers/openrouter/models/~z-ai/glm-latest.toml +++ b/providers/openrouter/models/~z-ai/glm-latest.toml @@ -15,13 +15,13 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.8775 -output = 2.97 -cache_read = 0.1755 +input = 0.5625 +output = 2.5 +cache_read = 0.125 [limit] context = 1_310_720 -output = 235_929 +output = 131_072 [modalities] input = ["text"] diff --git a/providers/opper/models/anthropic/claude-haiku-4-5.toml b/providers/opper/models/anthropic/claude-haiku-4-5.toml deleted file mode 100644 index 63e9f09896d..00000000000 --- a/providers/opper/models/anthropic/claude-haiku-4-5.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "anthropic/claude-haiku-4-5" -structured_output = true - -reasoning_options = [] - -[cost] -input = 1 -output = 5 -cache_read = 0.1 -cache_write = 1.25 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-opus-4-7.toml b/providers/opper/models/anthropic/claude-opus-4-7.toml deleted file mode 100644 index d4f86815775..00000000000 --- a/providers/opper/models/anthropic/claude-opus-4-7.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "anthropic/claude-opus-4-7" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - -[cost] -input = 5 -output = 25 -cache_read = 0.5 -cache_write = 6.25 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-opus-5.toml b/providers/opper/models/anthropic/claude-opus-5.toml deleted file mode 100644 index 634c2f987e6..00000000000 --- a/providers/opper/models/anthropic/claude-opus-5.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "anthropic/claude-opus-5" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - -[cost] -input = 5 -output = 25 -cache_read = 0.5 -cache_write = 6.25 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-sonnet-4-5.toml b/providers/opper/models/anthropic/claude-sonnet-4-5.toml deleted file mode 100644 index fdc69c21166..00000000000 --- a/providers/opper/models/anthropic/claude-sonnet-4-5.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "anthropic/claude-sonnet-4-5" -structured_output = true - -reasoning_options = [] - -[cost] -input = 3 -output = 15 -cache_read = 0.3 -cache_write = 3.75 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-sonnet-4-6.toml b/providers/opper/models/anthropic/claude-sonnet-4-6.toml deleted file mode 100644 index aeed8195b8f..00000000000 --- a/providers/opper/models/anthropic/claude-sonnet-4-6.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "anthropic/claude-sonnet-4-6" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "max"] - -[cost] -input = 3 -output = 15 -cache_read = 0.3 -cache_write = 3.75 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/claude-fable-5-1.toml b/providers/opper/models/claude-fable-5-1.toml new file mode 100644 index 00000000000..a17d2fb8c29 --- /dev/null +++ b/providers/opper/models/claude-fable-5-1.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-fable-5-1" +structured_output = true + +reasoning_options = [] + +[cost] +input = 10 +output = 50 +cache_read = 0.25 +cache_write = 12.5 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-fable-5.toml b/providers/opper/models/claude-fable-5.toml similarity index 50% rename from providers/opper/models/anthropic/claude-fable-5.toml rename to providers/opper/models/claude-fable-5.toml index 78ec64e8c7f..2bd15839fc0 100644 --- a/providers/opper/models/anthropic/claude-fable-5.toml +++ b/providers/opper/models/claude-fable-5.toml @@ -1,9 +1,9 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. base_model = "anthropic/claude-fable-5" structured_output = true -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] +reasoning_options = [] [cost] input = 10 diff --git a/providers/opper/models/claude-haiku-4-5.toml b/providers/opper/models/claude-haiku-4-5.toml new file mode 100644 index 00000000000..79bd6828653 --- /dev/null +++ b/providers/opper/models/claude-haiku-4-5.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-haiku-4-5" +structured_output = true + +reasoning_options = [] + +[cost] +input = 1.1 +output = 5.5 +cache_read = 0.11 +cache_write = 1.375 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-opus-4-5.toml b/providers/opper/models/claude-opus-4-5.toml similarity index 50% rename from providers/opper/models/anthropic/claude-opus-4-5.toml rename to providers/opper/models/claude-opus-4-5.toml index 45ac31901cf..fe7f973c895 100644 --- a/providers/opper/models/anthropic/claude-opus-4-5.toml +++ b/providers/opper/models/claude-opus-4-5.toml @@ -1,9 +1,9 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. base_model = "anthropic/claude-opus-4-5" structured_output = true -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] +reasoning_options = [] [cost] input = 5 diff --git a/providers/opper/models/anthropic/claude-opus-4-6.toml b/providers/opper/models/claude-opus-4-6.toml similarity index 50% rename from providers/opper/models/anthropic/claude-opus-4-6.toml rename to providers/opper/models/claude-opus-4-6.toml index dd56adf14b9..548c9efc22f 100644 --- a/providers/opper/models/anthropic/claude-opus-4-6.toml +++ b/providers/opper/models/claude-opus-4-6.toml @@ -1,9 +1,9 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. base_model = "anthropic/claude-opus-4-6" structured_output = true -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "max"] +reasoning_options = [] [cost] input = 5 diff --git a/providers/opper/models/claude-opus-4-7.toml b/providers/opper/models/claude-opus-4-7.toml new file mode 100644 index 00000000000..8b222c7b3c4 --- /dev/null +++ b/providers/opper/models/claude-opus-4-7.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-opus-4-7" +structured_output = true + +reasoning_options = [] + +[cost] +input = 5.5 +output = 27.5 +cache_read = 0.55 +cache_write = 6.875 + +[limit] +output = 64000 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/anthropic/claude-opus-4-8.toml b/providers/opper/models/claude-opus-4-8.toml similarity index 50% rename from providers/opper/models/anthropic/claude-opus-4-8.toml rename to providers/opper/models/claude-opus-4-8.toml index 83e1e7059c1..0d7904267ce 100644 --- a/providers/opper/models/anthropic/claude-opus-4-8.toml +++ b/providers/opper/models/claude-opus-4-8.toml @@ -1,9 +1,9 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. base_model = "anthropic/claude-opus-4-8" structured_output = true -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] +reasoning_options = [] [cost] input = 5 diff --git a/providers/opper/models/claude-opus-5.toml b/providers/opper/models/claude-opus-5.toml new file mode 100644 index 00000000000..052e3ac7cd9 --- /dev/null +++ b/providers/opper/models/claude-opus-5.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-opus-5" +structured_output = true + +reasoning_options = [] + +[cost] +input = 5.5 +output = 27.5 +cache_read = 0.55 +cache_write = 6.875 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/claude-sonnet-4-5.toml b/providers/opper/models/claude-sonnet-4-5.toml new file mode 100644 index 00000000000..953685007ad --- /dev/null +++ b/providers/opper/models/claude-sonnet-4-5.toml @@ -0,0 +1,23 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "anthropic/claude-sonnet-4-5" +structured_output = true + +reasoning_options = [] + +[cost] +input = 3.3 +output = 16.5 +cache_read = 0.33 +cache_write = 4.125 + +[[cost.tiers]] +tier = { type = "context", size = 200000 } +input = 6.6 +output = 24.75 +cache_read = 0.66 +cache_write = 8.25 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/claude-sonnet-4-6.toml b/providers/opper/models/claude-sonnet-4-6.toml new file mode 100644 index 00000000000..8efa496556f --- /dev/null +++ b/providers/opper/models/claude-sonnet-4-6.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-sonnet-4-6" +structured_output = true + +reasoning_options = [] + +[cost] +input = 3.3 +output = 16.5 +cache_read = 0.33 +cache_write = 4.125 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/claude-sonnet-5.toml b/providers/opper/models/claude-sonnet-5.toml new file mode 100644 index 00000000000..3b91951e85f --- /dev/null +++ b/providers/opper/models/claude-sonnet-5.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "anthropic/claude-sonnet-5" +structured_output = true + +reasoning_options = [] + +[cost] +input = 2.2 +output = 11 +cache_read = 0.22 +cache_write = 2.75 + +[limit] +output = 64000 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/deepseek-v4-flash.toml b/providers/opper/models/deepseek-v4-flash.toml new file mode 100644 index 00000000000..b332afbbd8d --- /dev/null +++ b/providers/opper/models/deepseek-v4-flash.toml @@ -0,0 +1,12 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "deepseek/deepseek-v4-flash" + +reasoning_options = [] + +[cost] +input = 0.25 +output = 0.66 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/deepseek-v4-pro.toml b/providers/opper/models/deepseek-v4-pro.toml new file mode 100644 index 00000000000..fb6c5faa0c1 --- /dev/null +++ b/providers/opper/models/deepseek-v4-pro.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "deepseek/deepseek-v4-pro" + +reasoning_options = [] + +[cost] +input = 1.78812 +output = 3.57624 + +[limit] +output = 65536 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/devstral-2512.toml b/providers/opper/models/devstral-2512.toml new file mode 100644 index 00000000000..d3595c76c26 --- /dev/null +++ b/providers/opper/models/devstral-2512.toml @@ -0,0 +1,11 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +base_model = "mistral/devstral-2512" +structured_output = true + +[cost] +input = 0.4 +output = 2 + +[limit] +context = 256000 +output = 8192 diff --git a/providers/opper/models/gemini-3-flash-preview.toml b/providers/opper/models/gemini-3-flash-preview.toml new file mode 100644 index 00000000000..0e6941f334a --- /dev/null +++ b/providers/opper/models/gemini-3-flash-preview.toml @@ -0,0 +1,13 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemini-3-flash-preview" + +reasoning_options = [] + +[cost] +input = 0.5 +output = 3 +cache_read = 0.05 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini-3.1-pro-preview.toml b/providers/opper/models/gemini-3.1-pro-preview.toml new file mode 100644 index 00000000000..e25eda06eb1 --- /dev/null +++ b/providers/opper/models/gemini-3.1-pro-preview.toml @@ -0,0 +1,20 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "google/gemini-3.1-pro-preview" + +reasoning_options = [] + +[cost] +input = 2 +output = 12 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 200000 } +input = 4 +output = 18 +cache_read = 0.4 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini-3.5-flash-lite.toml b/providers/opper/models/gemini-3.5-flash-lite.toml new file mode 100644 index 00000000000..0f10955993d --- /dev/null +++ b/providers/opper/models/gemini-3.5-flash-lite.toml @@ -0,0 +1,13 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemini-3.5-flash-lite" + +reasoning_options = [] + +[cost] +input = 0.3 +output = 2.5 +cache_read = 0.03 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini-3.5-flash.toml b/providers/opper/models/gemini-3.5-flash.toml new file mode 100644 index 00000000000..b597560116a --- /dev/null +++ b/providers/opper/models/gemini-3.5-flash.toml @@ -0,0 +1,13 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemini-3.5-flash" + +reasoning_options = [] + +[cost] +input = 1.5 +output = 9 +cache_read = 0.15 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini-3.7-flash.toml b/providers/opper/models/gemini-3.7-flash.toml new file mode 100644 index 00000000000..514a0c8cd9f --- /dev/null +++ b/providers/opper/models/gemini-3.7-flash.toml @@ -0,0 +1,13 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemini-3.7-flash" + +reasoning_options = [] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini-3.8-flash.toml b/providers/opper/models/gemini-3.8-flash.toml new file mode 100644 index 00000000000..b7dfba7907c --- /dev/null +++ b/providers/opper/models/gemini-3.8-flash.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# All five members have params but omit params.temperature; Opper drops sampling controls +# for such cards before dispatch, so temperature is accepted but not respected. +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemini-3.8-flash" +temperature = false + +reasoning_options = [] + +[cost] +input = 0.825 +output = 4.125 +cache_read = 0.0825 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gemini/gemini-3-flash-preview.toml b/providers/opper/models/gemini/gemini-3-flash-preview.toml deleted file mode 100644 index b5b4efd521b..00000000000 --- a/providers/opper/models/gemini/gemini-3-flash-preview.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3-flash-preview" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 0.5 -output = 3 -cache_read = 0.05 diff --git a/providers/opper/models/gemini/gemini-3.1-pro-preview.toml b/providers/opper/models/gemini/gemini-3.1-pro-preview.toml deleted file mode 100644 index 8cb3844011e..00000000000 --- a/providers/opper/models/gemini/gemini-3.1-pro-preview.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemini-3.1-pro-preview" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 2 -output = 12 -cache_read = 0.2 - -[[cost.tiers]] -tier = { size = 200_000 } -input = 4.00 -output = 18.00 -cache_read = 0.40 diff --git a/providers/opper/models/gemini/gemini-3.5-flash-lite.toml b/providers/opper/models/gemini/gemini-3.5-flash-lite.toml deleted file mode 100644 index e5452932870..00000000000 --- a/providers/opper/models/gemini/gemini-3.5-flash-lite.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3.5-flash-lite" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 0.3 -output = 2.5 -cache_read = 0.03 diff --git a/providers/opper/models/gemini/gemini-3.5-flash.toml b/providers/opper/models/gemini/gemini-3.5-flash.toml deleted file mode 100644 index aa9842db8a7..00000000000 --- a/providers/opper/models/gemini/gemini-3.5-flash.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3.5-flash" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 1.5 -output = 9 -cache_read = 0.15 diff --git a/providers/opper/models/gemma-4-31b-it.toml b/providers/opper/models/gemma-4-31b-it.toml new file mode 100644 index 00000000000..8ee52039c78 --- /dev/null +++ b/providers/opper/models/gemma-4-31b-it.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "google/gemma-4-31b-it" + +reasoning_options = [] + +[cost] +input = 0.46488 +output = 2.44062 + +[limit] +context = 256000 +output = 8192 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/glm-5.2.toml b/providers/opper/models/glm-5.2.toml new file mode 100644 index 00000000000..bc1044a1917 --- /dev/null +++ b/providers/opper/models/glm-5.2.toml @@ -0,0 +1,12 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "zhipuai/glm-5.2" + +reasoning_options = [] + +[cost] +input = 1.62708 +output = 5.811 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/glm-5.3-flash.toml b/providers/opper/models/glm-5.3-flash.toml new file mode 100644 index 00000000000..c7db468e1b7 --- /dev/null +++ b/providers/opper/models/glm-5.3-flash.toml @@ -0,0 +1,21 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "zhipuai/glm-5.3-flash" +attachment = false +structured_output = false + +reasoning_options = [] + +[cost] +input = 0.2 +output = 0.5 +cache_read = 0.07 + +[limit] +output = 128000 + +[modalities] +input = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/glm-5.3.toml b/providers/opper/models/glm-5.3.toml new file mode 100644 index 00000000000..b6c8534060c --- /dev/null +++ b/providers/opper/models/glm-5.3.toml @@ -0,0 +1,14 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: members do not share a supported effort level. +base_model = "zhipuai/glm-5.3" +structured_output = false + +reasoning_options = [] + +[cost] +input = 1.75 +output = 4.6488 +cache_read = 0.44 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.3-chat-latest.toml b/providers/opper/models/gpt-5.3-chat-latest.toml new file mode 100644 index 00000000000..7568b754041 --- /dev/null +++ b/providers/opper/models/gpt-5.3-chat-latest.toml @@ -0,0 +1,9 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Live pool call on 2026-09-08 reports this upstream model has been deprecated. +base_model = "openai/gpt-5.3-chat-latest" +status = "deprecated" + +[cost] +input = 1.75 +output = 14 +cache_read = 0.175 diff --git a/providers/opper/models/gpt-5.3-codex.toml b/providers/opper/models/gpt-5.3-codex.toml new file mode 100644 index 00000000000..ea0de06a8c5 --- /dev/null +++ b/providers/opper/models/gpt-5.3-codex.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; OpenAI native levels on all pool members. +# Catalog context equals the lab input cap; retain the lab total context window. +base_model = "openai/gpt-5.3-codex" +temperature = false + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 1.75 +output = 14 +cache_read = 0.175 + +[modalities] +input = ["text", "pdf"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.4-mini.toml b/providers/opper/models/gpt-5.4-mini.toml new file mode 100644 index 00000000000..5daca15ba01 --- /dev/null +++ b/providers/opper/models/gpt-5.4-mini.toml @@ -0,0 +1,14 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Native OpenAI effort levels verified through this single-member pool on 2026-09-08. +# Effort: top-level reasoning_effort; levels shared by the pool members. +base_model = "openai/gpt-5.4-mini" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 0.75 +output = 4.5 +cache_read = 0.075 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.4-nano.toml b/providers/opper/models/gpt-5.4-nano.toml new file mode 100644 index 00000000000..dbbf8bf940a --- /dev/null +++ b/providers/opper/models/gpt-5.4-nano.toml @@ -0,0 +1,14 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; OpenAI native levels on all pool members. +base_model = "openai/gpt-5.4-nano" +temperature = false + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 0.2 +output = 1.25 +cache_read = 0.02 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.4-pro.toml b/providers/opper/models/gpt-5.4-pro.toml new file mode 100644 index 00000000000..03bb664a101 --- /dev/null +++ b/providers/opper/models/gpt-5.4-pro.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.4-pro" + +reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] + +[cost] +input = 30 +output = 180 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 60 +output = 270 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.4.toml b/providers/opper/models/gpt-5.4.toml new file mode 100644 index 00000000000..b83170741b4 --- /dev/null +++ b/providers/opper/models/gpt-5.4.toml @@ -0,0 +1,20 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.4" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 2.5 +output = 15 +cache_read = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 5 +output = 22.5 +cache_read = 0.5 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.5-pro.toml b/providers/opper/models/gpt-5.5-pro.toml new file mode 100644 index 00000000000..254bf5b66c9 --- /dev/null +++ b/providers/opper/models/gpt-5.5-pro.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Native OpenAI effort levels verified through this single-member pool on 2026-09-08. +# All pool members publish flat rates: no pricing.thresholds or input_surcharge_threshold_tokens. +# Opper billing uses this schedule; native context surcharges do not apply to these pools. +# Effort: top-level reasoning_effort; levels shared by the pool members. +base_model = "openai/gpt-5.5-pro" + +reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] + +[cost] +input = 30 +output = 180 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.5.toml b/providers/opper/models/gpt-5.5.toml new file mode 100644 index 00000000000..b05f12abb7f --- /dev/null +++ b/providers/opper/models/gpt-5.5.toml @@ -0,0 +1,20 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.5" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] + +[cost] +input = 5 +output = 30 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 10 +output = 45 +cache_read = 1 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.6-luna.toml b/providers/opper/models/gpt-5.6-luna.toml new file mode 100644 index 00000000000..254ff2a8555 --- /dev/null +++ b/providers/opper/models/gpt-5.6-luna.toml @@ -0,0 +1,25 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.6-luna" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.2 +output = 1.2 +cache_read = 0.02 +cache_write = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 0.4 +output = 1.8 +cache_read = 0.04 +cache_write = 0.5 + +[limit] +context = 1000000 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.6-sol.toml b/providers/opper/models/gpt-5.6-sol.toml new file mode 100644 index 00000000000..e7b4f94f5f8 --- /dev/null +++ b/providers/opper/models/gpt-5.6-sol.toml @@ -0,0 +1,25 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.6-sol" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 5 +output = 30 +cache_read = 0.5 +cache_write = 6.25 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 10 +output = 45 +cache_read = 1 +cache_write = 12.5 + +[limit] +context = 1000000 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-5.6-terra.toml b/providers/opper/models/gpt-5.6-terra.toml new file mode 100644 index 00000000000..209fe14cd29 --- /dev/null +++ b/providers/opper/models/gpt-5.6-terra.toml @@ -0,0 +1,25 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-5.6-terra" + +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 2 +output = 12 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 4 +output = 18 +cache_read = 0.4 +cache_write = 5 + +[limit] +context = 1000000 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-6-astra.toml b/providers/opper/models/gpt-6-astra.toml new file mode 100644 index 00000000000..62f6292c6cb --- /dev/null +++ b/providers/opper/models/gpt-6-astra.toml @@ -0,0 +1,25 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "openai/gpt-6-astra" + +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 + +[[cost.tiers]] +tier = { type = "context", size = 272000 } +input = 20 +output = 75 +cache_read = 2 +cache_write = 25 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-oss-120b.toml b/providers/opper/models/gpt-oss-120b.toml new file mode 100644 index 00000000000..1a576848f91 --- /dev/null +++ b/providers/opper/models/gpt-oss-120b.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "openai/gpt-oss-120b" + +reasoning_options = [] + +[cost] +input = 1.1622 +output = 4.88124 + +[limit] +context = 128000 +output = 8192 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/gpt-oss-20b.toml b/providers/opper/models/gpt-oss-20b.toml new file mode 100644 index 00000000000..ce3715afc7b --- /dev/null +++ b/providers/opper/models/gpt-oss-20b.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "openai/gpt-oss-20b" + +reasoning_options = [] + +[cost] +input = 0.11622 +output = 0.488124 + +[limit] +context = 128000 +output = 8192 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/grok-4.3.toml b/providers/opper/models/grok-4.3.toml new file mode 100644 index 00000000000..e40b4396000 --- /dev/null +++ b/providers/opper/models/grok-4.3.toml @@ -0,0 +1,18 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# All pool members publish flat rates: no pricing.thresholds or input_surcharge_threshold_tokens. +# Opper billing uses this schedule; native context surcharges do not apply to these pools. +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "xai/grok-4.3" + +reasoning_options = [] + +[cost] +input = 1.25 +output = 2.5 +cache_read = 0.2 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/grok-4.5.toml b/providers/opper/models/grok-4.5.toml new file mode 100644 index 00000000000..8cb4220459f --- /dev/null +++ b/providers/opper/models/grok-4.5.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# All pool members publish flat rates: no pricing.thresholds or input_surcharge_threshold_tokens. +# Opper billing uses this schedule; native context surcharges do not apply to these pools. +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "xai/grok-4.5" + +reasoning_options = [] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/grok-4.6.toml b/providers/opper/models/grok-4.6.toml new file mode 100644 index 00000000000..36c860ca4c1 --- /dev/null +++ b/providers/opper/models/grok-4.6.toml @@ -0,0 +1,20 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "xai/grok-4.6" + +reasoning_options = [] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200000 } +input = 4 +output = 12 +cache_read = 1 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/grok-build-0.1.toml b/providers/opper/models/grok-build-0.1.toml new file mode 100644 index 00000000000..13eef952286 --- /dev/null +++ b/providers/opper/models/grok-build-0.1.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# All pool members publish flat rates: no pricing.thresholds or input_surcharge_threshold_tokens. +# Opper billing uses this schedule; native context surcharges do not apply to these pools. +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "xai/grok-build-0.1" + +reasoning_options = [] + +[cost] +input = 1 +output = 2 +cache_read = 0.2 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/kimi-k3.toml b/providers/opper/models/kimi-k3.toml new file mode 100644 index 00000000000..b4519ce43f6 --- /dev/null +++ b/providers/opper/models/kimi-k3.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: members do not share a supported effort level. +base_model = "moonshotai/kimi-k3" +attachment = false + +reasoning_options = [] + +[cost] +input = 3 +output = 15 + +[modalities] +input = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/minimax-m2.7.toml b/providers/opper/models/minimax-m2.7.toml new file mode 100644 index 00000000000..ed41ef98618 --- /dev/null +++ b/providers/opper/models/minimax-m2.7.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "minimax/MiniMax-M2.7" +structured_output = true + +reasoning_options = [] + +[cost] +input = 0.69732 +output = 2.78928 + +[limit] +context = 196608 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/minimax-m3.toml b/providers/opper/models/minimax-m3.toml new file mode 100644 index 00000000000..fae074d323c --- /dev/null +++ b/providers/opper/models/minimax-m3.toml @@ -0,0 +1,28 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +# Context tiers are conservative pool price ceilings, including marginal-rate bands. +base_model = "minimax/MiniMax-M3" +structured_output = true + +reasoning_options = [] + +[cost] +input = 0.6 +output = 2.4 +cache_read = 0.12 + +[[cost.tiers]] +tier = { type = "context", size = 524288 } +input = 1.2 +output = 4.8 +cache_read = 0.24 + +[limit] +context = 1000000 +output = 131072 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/minimax/m3.toml b/providers/opper/models/minimax/m3.toml deleted file mode 100644 index d53af696874..00000000000 --- a/providers/opper/models/minimax/m3.toml +++ /dev/null @@ -1,24 +0,0 @@ -# Opper billed schedule, verified against GET /v3/compat/models and /v3/models (2026-08-20): -# base 0.6/2.4 (cache 0.12) up to 524_288, then 2x. This differs from MiniMax first-party -# list (0.30/1.20 base, upper band 0.60/2.40); the delta is flagged to Opper catalog -# maintainers and this entry will be updated if the billed schedule changes. -base_model = "minimax/MiniMax-M3" - -reasoning_options = [] - -[cost] -input = 0.6 -output = 2.4 -cache_read = 0.12 - -[[cost.tiers]] -tier = { size = 524_288 } -input = 1.2 -output = 4.8 -cache_read = 0.24 - -[limit] -context = 1_048_576 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/mistral-large-2512.toml b/providers/opper/models/mistral-large-2512.toml new file mode 100644 index 00000000000..1f67818b4a1 --- /dev/null +++ b/providers/opper/models/mistral-large-2512.toml @@ -0,0 +1,11 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +base_model = "mistral/mistral-large-2512" +structured_output = true + +[cost] +input = 0.5 +output = 1.5 + +[limit] +context = 256000 +output = 8192 diff --git a/providers/opper/models/mistral-small-2603.toml b/providers/opper/models/mistral-small-2603.toml new file mode 100644 index 00000000000..cff65202966 --- /dev/null +++ b/providers/opper/models/mistral-small-2603.toml @@ -0,0 +1,16 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "mistral/mistral-small-2603" +structured_output = false + +reasoning_options = [] + +[cost] +input = 0.5811 +output = 2.44062 + +[limit] +output = 8192 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/mistral/devstral-2512.toml b/providers/opper/models/mistral/devstral-2512.toml deleted file mode 100644 index 62661ddc309..00000000000 --- a/providers/opper/models/mistral/devstral-2512.toml +++ /dev/null @@ -1,5 +0,0 @@ -base_model = "mistral/devstral-2512" - -[cost] -input = 0.4 -output = 2 diff --git a/providers/opper/models/mistral/mistral-large-2512.toml b/providers/opper/models/mistral/mistral-large-2512.toml deleted file mode 100644 index 49df75a6342..00000000000 --- a/providers/opper/models/mistral/mistral-large-2512.toml +++ /dev/null @@ -1,5 +0,0 @@ -base_model = "mistral/mistral-large-2512" - -[cost] -input = 0.5 -output = 1.5 diff --git a/providers/opper/models/mistral/mistral-small-2603.toml b/providers/opper/models/mistral/mistral-small-2603.toml deleted file mode 100644 index c82dfc68357..00000000000 --- a/providers/opper/models/mistral/mistral-small-2603.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "mistral/mistral-small-2603" - -[[reasoning_options]] -type = "effort" -values = ["none", "high"] - -[cost] -input = 0.15 -output = 0.6 diff --git a/providers/opper/models/meta/muse-spark-1.2.toml b/providers/opper/models/muse-spark-1.2.toml similarity index 51% rename from providers/opper/models/meta/muse-spark-1.2.toml rename to providers/opper/models/muse-spark-1.2.toml index aea877da54a..200e6da0293 100644 --- a/providers/opper/models/meta/muse-spark-1.2.toml +++ b/providers/opper/models/muse-spark-1.2.toml @@ -1,3 +1,5 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. base_model = "meta/muse-spark-1.2" reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high", "xhigh"] }] @@ -6,3 +8,6 @@ reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "h input = 1.25 output = 4.25 cache_read = 0.15 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/muse-spark-1.3.toml b/providers/opper/models/muse-spark-1.3.toml new file mode 100644 index 00000000000..bba0703ecde --- /dev/null +++ b/providers/opper/models/muse-spark-1.3.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Standard-tier pool forwards native effort unchanged; max verified live on 2026-09-08. +# The catalog reasoning.supported list omits max, but the serving path accepts it. +# Effort: top-level reasoning_effort; levels shared by the pool members. +base_model = "meta/muse-spark-1.3" + +reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 1.25 +output = 4.25 +cache_read = 0.15 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/openai/gpt-5.3-chat-latest.toml b/providers/opper/models/openai/gpt-5.3-chat-latest.toml deleted file mode 100644 index 197284ef5f0..00000000000 --- a/providers/opper/models/openai/gpt-5.3-chat-latest.toml +++ /dev/null @@ -1,6 +0,0 @@ -base_model = "openai/gpt-5.3-chat-latest" - -[cost] -input = 1.75 -output = 14 -cache_read = 0.175 diff --git a/providers/opper/models/openai/gpt-5.3-codex.toml b/providers/opper/models/openai/gpt-5.3-codex.toml deleted file mode 100644 index e85f5642a44..00000000000 --- a/providers/opper/models/openai/gpt-5.3-codex.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-5.3-codex" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] - -[cost] -input = 1.75 -output = 14 -cache_read = 0.175 diff --git a/providers/opper/models/openai/gpt-5.4-mini.toml b/providers/opper/models/openai/gpt-5.4-mini.toml deleted file mode 100644 index a21229d609e..00000000000 --- a/providers/opper/models/openai/gpt-5.4-mini.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-5.4-mini" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] - -[cost] -input = 0.75 -output = 4.5 -cache_read = 0.075 diff --git a/providers/opper/models/openai/gpt-5.4-nano.toml b/providers/opper/models/openai/gpt-5.4-nano.toml deleted file mode 100644 index 0db45e05b9e..00000000000 --- a/providers/opper/models/openai/gpt-5.4-nano.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-5.4-nano" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] - -[cost] -input = 0.2 -output = 1.25 -cache_read = 0.02 diff --git a/providers/opper/models/openai/gpt-5.4-pro.toml b/providers/opper/models/openai/gpt-5.4-pro.toml deleted file mode 100644 index 6da8f6aa095..00000000000 --- a/providers/opper/models/openai/gpt-5.4-pro.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/gpt-5.4-pro" - -reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] - -[cost] -input = 30 -output = 180 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 60.00 -output = 270.00 diff --git a/providers/opper/models/openai/gpt-5.4.toml b/providers/opper/models/openai/gpt-5.4.toml deleted file mode 100644 index 7398a595b17..00000000000 --- a/providers/opper/models/openai/gpt-5.4.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "openai/gpt-5.4" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] - -[cost] -input = 2.5 -output = 15 -cache_read = 0.25 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 5.00 -output = 22.50 -cache_read = 0.50 diff --git a/providers/opper/models/openai/gpt-5.5-pro.toml b/providers/opper/models/openai/gpt-5.5-pro.toml deleted file mode 100644 index 77d5371cb8e..00000000000 --- a/providers/opper/models/openai/gpt-5.5-pro.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/gpt-5.5-pro" - -reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] - -[cost] -input = 30 -output = 180 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 60.00 -output = 270.00 diff --git a/providers/opper/models/openai/gpt-5.5.toml b/providers/opper/models/openai/gpt-5.5.toml deleted file mode 100644 index a58539b3e09..00000000000 --- a/providers/opper/models/openai/gpt-5.5.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "openai/gpt-5.5" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] - -[cost] -input = 5 -output = 30 -cache_read = 0.5 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 10.00 -output = 45.00 -cache_read = 1.00 diff --git a/providers/opper/models/openai/gpt-5.6-luna.toml b/providers/opper/models/openai/gpt-5.6-luna.toml deleted file mode 100644 index 07d577afa36..00000000000 --- a/providers/opper/models/openai/gpt-5.6-luna.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "openai/gpt-5.6-luna" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] - -[cost] -input = 0.2 -output = 1.2 -cache_read = 0.02 -cache_write = 0.25 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 0.40 -output = 1.80 -cache_read = 0.04 -cache_write = 0.50 diff --git a/providers/opper/models/openai/gpt-5.6-sol.toml b/providers/opper/models/openai/gpt-5.6-sol.toml deleted file mode 100644 index 31470933d23..00000000000 --- a/providers/opper/models/openai/gpt-5.6-sol.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "openai/gpt-5.6-sol" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] - -[cost] -input = 5 -output = 30 -cache_read = 0.5 -cache_write = 6.25 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 10.00 -output = 45.00 -cache_read = 1.00 -cache_write = 12.50 diff --git a/providers/opper/models/openai/gpt-5.6-terra.toml b/providers/opper/models/openai/gpt-5.6-terra.toml deleted file mode 100644 index 9af64197094..00000000000 --- a/providers/opper/models/openai/gpt-5.6-terra.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "openai/gpt-5.6-terra" - -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] - -[cost] -input = 2 -output = 12 -cache_read = 0.2 -cache_write = 2.5 - -[[cost.tiers]] -tier = { size = 272_000 } -input = 4.00 -output = 18.00 -cache_read = 0.40 -cache_write = 5.00 diff --git a/providers/opper/models/perplexity/sonar-pro.toml b/providers/opper/models/perplexity/sonar-pro.toml deleted file mode 100644 index f03972e3868..00000000000 --- a/providers/opper/models/perplexity/sonar-pro.toml +++ /dev/null @@ -1,5 +0,0 @@ -base_model = "perplexity/sonar-pro" - -[cost] -input = 3 -output = 15 diff --git a/providers/opper/models/perplexity/sonar-reasoning-pro.toml b/providers/opper/models/perplexity/sonar-reasoning-pro.toml deleted file mode 100644 index 6189d9d9ff8..00000000000 --- a/providers/opper/models/perplexity/sonar-reasoning-pro.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "perplexity/sonar-reasoning-pro" - -reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] - -[cost] -input = 2 -output = 8 diff --git a/providers/opper/models/perplexity/sonar.toml b/providers/opper/models/perplexity/sonar.toml deleted file mode 100644 index a6c921f515e..00000000000 --- a/providers/opper/models/perplexity/sonar.toml +++ /dev/null @@ -1,5 +0,0 @@ -base_model = "perplexity/sonar" - -[cost] -input = 1 -output = 1 diff --git a/providers/opper/models/qwen3-coder-next.toml b/providers/opper/models/qwen3-coder-next.toml new file mode 100644 index 00000000000..2816f8c2b7e --- /dev/null +++ b/providers/opper/models/qwen3-coder-next.toml @@ -0,0 +1,6 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +base_model = "alibaba/qwen3-coder-next" + +[cost] +input = 0.5811 +output = 2.3244 diff --git a/providers/opper/models/qwen3.6-35b-a3b.toml b/providers/opper/models/qwen3.6-35b-a3b.toml new file mode 100644 index 00000000000..f7e98234caf --- /dev/null +++ b/providers/opper/models/qwen3.6-35b-a3b.toml @@ -0,0 +1,15 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "alibaba/qwen3.6-35b-a3b" + +reasoning_options = [] + +[cost] +input = 0.248 +output = 1.485 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/qwen3.8-2.4t-a95b.toml b/providers/opper/models/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..db70485f41e --- /dev/null +++ b/providers/opper/models/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,13 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "alibaba/qwen3.8-2.4t-a95b" + +reasoning_options = [] + +[cost] +input = 2.5 +output = 6 +cache_read = 0.63 + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/qwen3.8-27b.toml b/providers/opper/models/qwen3.8-27b.toml new file mode 100644 index 00000000000..8cd1f5f6882 --- /dev/null +++ b/providers/opper/models/qwen3.8-27b.toml @@ -0,0 +1,17 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# Effort: top-level reasoning_effort; levels shared by the pool members. +base_model = "alibaba/qwen3.8-27b" +attachment = false +structured_output = false + +reasoning_options = [{ type = "effort", values = ["low", "medium", "xhigh"] }] + +[cost] +input = 0.5811 +output = 3 + +[modalities] +input = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/qwen3.8-max.toml b/providers/opper/models/qwen3.8-max.toml new file mode 100644 index 00000000000..65905a41b73 --- /dev/null +++ b/providers/opper/models/qwen3.8-max.toml @@ -0,0 +1,20 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide graded effort or toggle: members share only reasoning_effort=none. +base_model = "alibaba/qwen3.8-max" +structured_output = true + +reasoning_options = [] + +[cost] +input = 2 +output = 6 +cache_read = 0.25 + +[limit] +context = 983616 + +[modalities] +input = ["text", "image"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/sonar-pro.toml b/providers/opper/models/sonar-pro.toml new file mode 100644 index 00000000000..de476cca476 --- /dev/null +++ b/providers/opper/models/sonar-pro.toml @@ -0,0 +1,14 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +base_model = "perplexity/sonar-pro" +attachment = false +structured_output = true + +[cost] +input = 3 +output = 15 + +[limit] +output = 8000 + +[modalities] +input = ["text"] diff --git a/providers/opper/models/sonar-reasoning-pro.toml b/providers/opper/models/sonar-reasoning-pro.toml new file mode 100644 index 00000000000..98d5cd92398 --- /dev/null +++ b/providers/opper/models/sonar-reasoning-pro.toml @@ -0,0 +1,17 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +# No pool-wide effort control: at least one member does not forward reasoning_effort. +base_model = "perplexity/sonar-reasoning-pro" +attachment = false +structured_output = true + +reasoning_options = [] + +[cost] +input = 2 +output = 8 + +[modalities] +input = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/opper/models/sonar.toml b/providers/opper/models/sonar.toml new file mode 100644 index 00000000000..7242f639443 --- /dev/null +++ b/providers/opper/models/sonar.toml @@ -0,0 +1,7 @@ +# Pool metadata: https://api.opper.ai/v3/models (2026-09-08). +base_model = "perplexity/sonar" +structured_output = true + +[cost] +input = 1 +output = 1 diff --git a/providers/opper/models/vertexai/gemini-3.7-flash-eu.toml b/providers/opper/models/vertexai/gemini-3.7-flash-eu.toml deleted file mode 100644 index 03e8b29cab7..00000000000 --- a/providers/opper/models/vertexai/gemini-3.7-flash-eu.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "google/gemini-3.7-flash" -name = "Gemini 3.7 Flash (EU)" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 0.75 -output = 3.75 -cache_read = 0.075 diff --git a/providers/opper/models/vertexai/gemini-3.7-flash.toml b/providers/opper/models/vertexai/gemini-3.7-flash.toml deleted file mode 100644 index e9bacad26c9..00000000000 --- a/providers/opper/models/vertexai/gemini-3.7-flash.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3.7-flash" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 0.75 -output = 3.75 -cache_read = 0.075 diff --git a/providers/opper/models/xai/grok-4.3.toml b/providers/opper/models/xai/grok-4.3.toml deleted file mode 100644 index 329b82bb4ac..00000000000 --- a/providers/opper/models/xai/grok-4.3.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "xai/grok-4.3" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high"] - -[cost] -input = 1.25 -output = 2.5 -cache_read = 0.2 - -[[cost.tiers]] -tier = { size = 200_000 } -input = 2.5 -output = 5 -cache_read = 0.4 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/xai/grok-4.5.toml b/providers/opper/models/xai/grok-4.5.toml deleted file mode 100644 index fd10a1caace..00000000000 --- a/providers/opper/models/xai/grok-4.5.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "xai/grok-4.5" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 2 -output = 6 -cache_read = 0.3 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 4 -output = 12 -cache_read = 0.6 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/models/xai/grok-build-0.1.toml b/providers/opper/models/xai/grok-build-0.1.toml deleted file mode 100644 index b2c4aeeafad..00000000000 --- a/providers/opper/models/xai/grok-build-0.1.toml +++ /dev/null @@ -1,17 +0,0 @@ -base_model = "xai/grok-build-0.1" - -reasoning_options = [] - -[cost] -input = 1 -output = 2 -cache_read = 0.2 - -[[cost.tiers]] -tier = { size = 200_000 } -input = 2.00 -output = 4.00 -cache_read = 0.40 - -[interleaved] -field = "reasoning_content" diff --git a/providers/opper/provider.toml b/providers/opper/provider.toml index 7d14eff2de3..7ccd9102a07 100644 --- a/providers/opper/provider.toml +++ b/providers/opper/provider.toml @@ -1,13 +1,20 @@ +# Sources (2026-09-08): https://api.opper.ai/v3/models?limit=0&type=llm +# and authenticated https://api.opper.ai/v3/compat/models?type=model,pool. +# Model IDs are bare Opper pool names; Opper selects the underlying provider. +# Prices are USD/MTok ceilings across pool members, not a fixed rate for every +# request. Cache-read discounts are listed only when all priced members offer +# them. Context tiers include the highest marginal rates and whole-request +# surcharges, so estimates can exceed the actual bill. No currency conversion +# is needed: the catalog already converts non-USD provider prices to USD. +# Limits do not exceed either the lab model or the pool's smallest known limit; +# input modalities are restricted to those shared by the lab and pool members. +# POST /v3/compat/chat/completions accepts top-level reasoning_effort. Native +# thinking toggles and reasoning-token budgets are not exposed on this surface. +# Effort options describe common pool controls; [] means no pool-wide control, +# including pools where some adapters ignore effort. Reasoning output, when +# present, uses message.reasoning_content or streaming delta.reasoning_content. name = "Opper" env = ["OPPER_API_KEY"] npm = "@ai-sdk/openai-compatible" api = "https://api.opper.ai/v3/compat" -# Reasoning HTTP format (measured against POST /v3/compat/chat/completions, 2026-08-19): -# top-level reasoning_effort takes an effort string that is passed through to the -# upstream model API unchanged, so each model accepts its native set (Anthropic: -# low|medium|high|max; OpenAI GPT-5.x: none|low|medium|high|xhigh and max where the -# model supports it; Gemini: minimal|low|medium|high). Budget-token values are not -# accepted on this surface, and no separate reasoning on/off toggle is exposed. Models with no upstream effort control accept the -# parameter without effect, so those entries use reasoning_options = []. -# Pricing is mirrored from GET /v3/compat/models (USD per token, no gateway markup). doc = "https://opper.ai/models" diff --git a/providers/ovhcloud/models/qwen3-32b.toml b/providers/ovhcloud/models/qwen3-32b.toml deleted file mode 100644 index b117c1bb3b6..00000000000 --- a/providers/ovhcloud/models/qwen3-32b.toml +++ /dev/null @@ -1,26 +0,0 @@ -# Put `/no_think` in prompt content to disable the model's default reasoning. -# https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog/qwen-3-32b/ - -name = "Qwen3-32B" -description = "Reasoning model for deliberate analysis, multi-step problem solving, and tool use" -release_date = "2025-07-16" -last_updated = "2025-07-16" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -open_weights = true - -[cost] -input = 0.09 -output = 0.25 - -[limit] -context = 32_768 -output = 32_768 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/pioneer/models/fastino/gliguard-LLMGuardrails-300M.toml b/providers/pioneer/models/fastino/gliguard-LLMGuardrails-300M.toml index de754b40e20..f455ac6d116 100644 --- a/providers/pioneer/models/fastino/gliguard-LLMGuardrails-300M.toml +++ b/providers/pioneer/models/fastino/gliguard-LLMGuardrails-300M.toml @@ -3,11 +3,18 @@ description = "Tool-capable chat model for instruction following and agentic app release_date = "2026-04-30" last_updated = "2026-04-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliguard-PII-multi.toml b/providers/pioneer/models/fastino/gliguard-PII-multi.toml new file mode 100644 index 00000000000..cec904e782a --- /dev/null +++ b/providers/pioneer/models/fastino/gliguard-PII-multi.toml @@ -0,0 +1,36 @@ +# Sources: +# - Pioneer live catalog GET https://api.pioneer.ai/v1/models (id fastino/gliguard-PII-multi: display_name, max_input_tokens 8192, max_tokens 8192, prices 0.15, capabilities all false) +# - Pioneer base-models GET https://api.pioneer.ai/base-models (label GLiNER2-Guardrails-PII-Multi, context_window 8192, license Apache-2.0, release_month Jul 2026) +# - Fastino release blog https://fastino.ai/blog/gliner2-guardrails-pii-multi-safety-moderation-privacy-filtering-small-language-model (released July 8 2026, 0.3B param, single forward pass, Apache 2.0) +# - Fastino model page https://fastino.ai/models/gliner2-guardrails-pii-multi (300M-param multilingual safety moderation + PII detection, open weights) +# - Hugging Face https://huggingface.co/fastino/GLiNER2-Guardrails-PII-Multi (Apache-2.0, 0.3B params, unified safety moderation + PII detection) +name = "GLiNER2-Guardrails-PII-Multi" +description = "A 300M-parameter multilingual model that runs LLM safety moderation and PII detection in a single forward pass." +release_date = "2026-07-08" +last_updated = "2026-07-08" +attachment = false +reasoning = true +temperature = true +tool_call = true +open_weights = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 0.15 +output = 0.15 +cache_read = 0.15 +cache_write = 0.15 + +[limit] +context = 8_192 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2-base-v1.toml b/providers/pioneer/models/fastino/gliner2-base-v1.toml index ef23d31a6ad..060cd035d56 100644 --- a/providers/pioneer/models/fastino/gliner2-base-v1.toml +++ b/providers/pioneer/models/fastino/gliner2-base-v1.toml @@ -3,11 +3,18 @@ description = "Tool-capable chat model for instruction following and agentic app release_date = "2025-06-30" last_updated = "2025-06-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2-large-v1.toml b/providers/pioneer/models/fastino/gliner2-large-v1.toml index 086e826101d..93f13f5d9c0 100644 --- a/providers/pioneer/models/fastino/gliner2-large-v1.toml +++ b/providers/pioneer/models/fastino/gliner2-large-v1.toml @@ -3,11 +3,18 @@ description = "Flagship model for demanding analysis, coding, and production age release_date = "2025-06-30" last_updated = "2025-06-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2-multi-large-v1.toml b/providers/pioneer/models/fastino/gliner2-multi-large-v1.toml index 74367df5b6c..5e3d3050671 100644 --- a/providers/pioneer/models/fastino/gliner2-multi-large-v1.toml +++ b/providers/pioneer/models/fastino/gliner2-multi-large-v1.toml @@ -3,11 +3,18 @@ description = "Flagship model for demanding analysis, coding, and production age release_date = "2025-11-30" last_updated = "2025-11-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2-multi-v1.toml b/providers/pioneer/models/fastino/gliner2-multi-v1.toml index 70b2d9553eb..76f6cdfd84a 100644 --- a/providers/pioneer/models/fastino/gliner2-multi-v1.toml +++ b/providers/pioneer/models/fastino/gliner2-multi-v1.toml @@ -3,11 +3,18 @@ description = "Tool-capable chat model for instruction following and agentic app release_date = "2025-11-30" last_updated = "2025-11-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2-privacy-filter-PII-multi.toml b/providers/pioneer/models/fastino/gliner2-privacy-filter-PII-multi.toml index 707d407fa98..af0a2ddc10a 100644 --- a/providers/pioneer/models/fastino/gliner2-privacy-filter-PII-multi.toml +++ b/providers/pioneer/models/fastino/gliner2-privacy-filter-PII-multi.toml @@ -3,11 +3,18 @@ description = "Tool-capable chat model for instruction following and agentic app release_date = "2026-04-30" last_updated = "2026-04-30" attachment = false -reasoning = false +reasoning = true temperature = true tool_call = true open_weights = false +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + [cost] input = 0.15 output = 0.15 @@ -16,7 +23,7 @@ cache_write = 0.15 [limit] context = 8_192 -output = 4_096 +output = 8_192 [modalities] input = ["text"] diff --git a/providers/pioneer/models/fastino/gliner2.5-multi-v1.toml b/providers/pioneer/models/fastino/gliner2.5-multi-v1.toml new file mode 100644 index 00000000000..932fc2e0d70 --- /dev/null +++ b/providers/pioneer/models/fastino/gliner2.5-multi-v1.toml @@ -0,0 +1,35 @@ +# Sources: +# - Pioneer live catalog GET https://api.pioneer.ai/base-models (id fastino/gliner2.5-multi-v1: label "GLiNER 2.5 Multi", description "Multilingual boundary NER and span extraction; non-trainable encoder.", context_window 4096, input/output $0.15, cache $0.15, license Apache-2.0, release Aug 2026, is_chat_model true) +# - Model card https://huggingface.co/fastino/gliner2.5-multi-v1 (287M mDeBERTa-v3-base multilingual boundary checkpoint, max_len 4096, Apache-2.0) +# - Release announcement https://fastino.ai/blog/gliner2-5-span-free-information-extraction (released Aug 24, 2026; three Apache-2.0 variants) +# Pioneer is the first-party Fastino inference API (https://pioneer.ai/, https://www.prnewswire.com/news-releases/fastino-launches-pioneer-the-first-agent-for-fine-tuning-and-inference-of-llms-302748105.html); full inline entry follows existing providers/pioneer/models/fastino/* pattern. +name = "GLiNER 2.5 Multi" +description = "Multilingual boundary NER and span extraction; non-trainable encoder." +release_date = "2026-08-24" +last_updated = "2026-08-24" +attachment = false +reasoning = true +temperature = true +tool_call = true +open_weights = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 0.15 +output = 0.15 +cache_read = 0.15 +cache_write = 0.15 + +[limit] +context = 4_096 +output = 4_096 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/requesty/models/claude-opus-5-5.toml b/providers/requesty/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..1c8fca5fbe5 --- /dev/null +++ b/providers/requesty/models/claude-opus-5-5.toml @@ -0,0 +1,15 @@ +base_model = "anthropic/claude-opus-5-5" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/requesty/models/claude-opus-5-5@eu.toml b/providers/requesty/models/claude-opus-5-5@eu.toml new file mode 100644 index 00000000000..eaa19cb4ca0 --- /dev/null +++ b/providers/requesty/models/claude-opus-5-5@eu.toml @@ -0,0 +1,16 @@ +base_model = "anthropic/claude-opus-5-5" +name = "Claude Opus 5.5 (EU)" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 5.5 +output = 27.5 +cache_read = 0.55 +cache_write = 6.875 diff --git a/providers/requesty/models/gpt-6-luna.toml b/providers/requesty/models/gpt-6-luna.toml new file mode 100644 index 00000000000..a11e192297d --- /dev/null +++ b/providers/requesty/models/gpt-6-luna.toml @@ -0,0 +1,19 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.2 +output = 0.75 +cache_read = 0.02 diff --git a/providers/requesty/models/gpt-6-luna@eu.toml b/providers/requesty/models/gpt-6-luna@eu.toml new file mode 100644 index 00000000000..b0d9e112fcc --- /dev/null +++ b/providers/requesty/models/gpt-6-luna@eu.toml @@ -0,0 +1,20 @@ +base_model = "openai/gpt-6-luna" +name = "GPT-6 Luna (EU)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.12 +output = 0.6 +cache_read = 0.012 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.24 +output = 0.9 +cache_read = 0.024 diff --git a/providers/requesty/models/gpt-6-sol.toml b/providers/requesty/models/gpt-6-sol.toml new file mode 100644 index 00000000000..fccceb9056e --- /dev/null +++ b/providers/requesty/models/gpt-6-sol.toml @@ -0,0 +1,19 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 2 +output = 10 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 15 +cache_read = 0.4 diff --git a/providers/requesty/models/gpt-6-sol@eu.toml b/providers/requesty/models/gpt-6-sol@eu.toml new file mode 100644 index 00000000000..357ef21ce6c --- /dev/null +++ b/providers/requesty/models/gpt-6-sol@eu.toml @@ -0,0 +1,20 @@ +base_model = "openai/gpt-6-sol" +name = "GPT-6 Sol (EU)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 2.4 +output = 12 +cache_read = 0.24 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4.8 +output = 18 +cache_read = 0.48 diff --git a/providers/siliconflow/models/MiniMaxAI/MiniMax-M3.toml b/providers/siliconflow/models/MiniMaxAI/MiniMax-M3.toml new file mode 100644 index 00000000000..748a1991b69 --- /dev/null +++ b/providers/siliconflow/models/MiniMaxAI/MiniMax-M3.toml @@ -0,0 +1,26 @@ +# Budget: thinking_budget (integer reasoning tokens, 128..32768) — SiliconFlow documents +# thinking_budget for all reasoning models; enable_thinking is not enumerated for MiniMax +# models on this host, so no toggle. +# Sources: https://www.siliconflow.com/models (2026-09-08), +# https://docs.siliconflow.com/en/api-reference/chat-completions/chat-completions +base_model = "minimax/MiniMax-M3" + + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.06 + +[limit] +output = 131_000 + +# SiliconFlow serves image input but not video for this model +# (model page "Support image input: Yes"): https://www.siliconflow.com/models/minimax-m3 +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/siliconflow/models/Qwen/Qwen3-235B-A22B-Thinking-2507.toml b/providers/siliconflow/models/Qwen/Qwen3-235B-A22B-Thinking-2507.toml deleted file mode 100644 index fb446e7d261..00000000000 --- a/providers/siliconflow/models/Qwen/Qwen3-235B-A22B-Thinking-2507.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Qwen/Qwen3-235B-A22B-Thinking-2507" -description = "Qwen reasoning model for deliberate problem solving, math, and coding" -family = "qwen" -release_date = "2025-07-28" -last_updated = "2025-11-25" -attachment = false -reasoning = true -reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.13 -output = 0.6 - -[limit] -context = 262_000 -output = 262_000 - -[modalities] -input = ["text"] -output = ["text"] \ No newline at end of file diff --git a/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Instruct.toml b/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Instruct.toml deleted file mode 100644 index 8c74b4bceca..00000000000 --- a/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Instruct.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "Qwen/Qwen3-VL-235B-A22B-Instruct" -description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" -family = "qwen" -release_date = "2025-10-04" -last_updated = "2025-11-25" -attachment = true -reasoning = false -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.3 -output = 1.5 - -[limit] -context = 262_000 -output = 262_000 - -[modalities] -input = ["text", "image"] -output = ["text"] \ No newline at end of file diff --git a/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Thinking.toml b/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Thinking.toml deleted file mode 100644 index 03bb1766276..00000000000 --- a/providers/siliconflow/models/Qwen/Qwen3-VL-235B-A22B-Thinking.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Qwen/Qwen3-VL-235B-A22B-Thinking" -description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" -family = "qwen" -release_date = "2025-10-04" -last_updated = "2025-11-25" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.45 -output = 3.5 - -[limit] -context = 262_000 -output = 262_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/siliconflow/models/Qwen/Qwen3.8-2.4T-A95B.toml b/providers/siliconflow/models/Qwen/Qwen3.8-2.4T-A95B.toml new file mode 100644 index 00000000000..86601f6baa1 --- /dev/null +++ b/providers/siliconflow/models/Qwen/Qwen3.8-2.4T-A95B.toml @@ -0,0 +1,16 @@ +# Budget: thinking_budget (integer reasoning tokens, 128..32768) — SiliconFlow exposes no +# reasoning_effort parameter; thinking_budget is documented for all reasoning models. +# Sources: https://www.siliconflow.com/models (2026-09-08), +# https://docs.siliconflow.com/en/api-reference/chat-completions/chat-completions +base_model = "alibaba/qwen3.8-2.4t-a95b" + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[cost] +input = 2.0 +output = 6.0 +cache_read = 0.25 + +[limit] +context = 1_049_000 +output = 131_000 diff --git a/providers/siliconflow/models/baidu/ERNIE-4.5-300B-A47B.toml b/providers/siliconflow/models/baidu/ERNIE-4.5-300B-A47B.toml deleted file mode 100644 index 5309d4c0402..00000000000 --- a/providers/siliconflow/models/baidu/ERNIE-4.5-300B-A47B.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "baidu/ERNIE-4.5-300B-A47B" -description = "Tool-capable chat model for instruction following and agentic application workflows" -family = "ernie" -release_date = "2025-07-02" -last_updated = "2025-11-25" -attachment = false -reasoning = false -temperature = true -tool_call = true -structured_output = true -open_weights = false - -[cost] -input = 0.28 -output = 1.1 - -[limit] -context = 131_000 -output = 131_000 - -[modalities] -input = ["text"] -output = ["text"] \ No newline at end of file diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V3.2.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V3.2.toml index a05b736c528..84bd5c7b342 100644 --- a/providers/siliconflow/models/deepseek-ai/DeepSeek-V3.2.toml +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V3.2.toml @@ -14,6 +14,7 @@ open_weights = false [cost] input = 0.27 output = 0.42 +cache_read = 0.135 [limit] context = 164_000 diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-0731.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-0731.toml new file mode 100644 index 00000000000..be8c8bf0744 --- /dev/null +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-0731.toml @@ -0,0 +1,12 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "deepseek/deepseek-v4-flash-0731" + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.22 +output = 0.66 +cache_read = 0.014 diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp.toml new file mode 100644 index 00000000000..674c8102988 --- /dev/null +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp.toml @@ -0,0 +1,12 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "deepseek/deepseek-v4-flash-vision-exp" + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.44 +output = 1.32 +cache_read = 0.028 diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash.toml index 744b022d3f0..14ac45494a9 100644 --- a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash.toml +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Flash.toml @@ -6,6 +6,6 @@ reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] field = "reasoning_content" [cost] -input = 0.14 +input = 0.13 output = 0.28 cache_read = 0.028 diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml new file mode 100644 index 00000000000..dd2c092e9dc --- /dev/null +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro-0813.toml @@ -0,0 +1,12 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "deepseek/deepseek-v4-pro-0813" + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1.32 +output = 3.96 +cache_read = 0.044 diff --git a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro.toml b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro.toml index a573cf5c2af..9994892360a 100644 --- a/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro.toml +++ b/providers/siliconflow/models/deepseek-ai/DeepSeek-V4-Pro.toml @@ -1,3 +1,5 @@ +# Cost values are the published USD list prices on SiliconFlow international +# (https://www.siliconflow.com/pricing, accessed 2026-09-08), not FX conversions by this PR. base_model = "deepseek/deepseek-v4-pro" reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] @@ -6,6 +8,6 @@ reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] field = "reasoning_content" [cost] -input = 1.74 -output = 3.48 -cache_read = 0.145 +input = 1.50162 +output = 3.135 +cache_read = 0.135 diff --git a/providers/siliconflow/models/google/gemma-4-12B-it.toml b/providers/siliconflow/models/google/gemma-4-12B-it.toml new file mode 100644 index 00000000000..05329524ad5 --- /dev/null +++ b/providers/siliconflow/models/google/gemma-4-12B-it.toml @@ -0,0 +1,16 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "google/gemma-4-12b-it" +attachment = false +reasoning = false +open_weights = false + +[cost] +input = 0.10 +output = 0.30 + +[limit] +output = 262_144 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/siliconflow/models/meituan-longcat/LongCat-2.0.toml b/providers/siliconflow/models/meituan-longcat/LongCat-2.0.toml new file mode 100644 index 00000000000..dfe567a7db0 --- /dev/null +++ b/providers/siliconflow/models/meituan-longcat/LongCat-2.0.toml @@ -0,0 +1,18 @@ +# Budget: thinking_budget (integer reasoning tokens, 128..32768) — SiliconFlow documents +# thinking_budget for all reasoning models and enumerates enable_thinking only for specific +# models (LongCat not among them), so no toggle on this host. +# Sources: https://www.siliconflow.com/models (2026-09-08), +# https://docs.siliconflow.com/en/api-reference/chat-completions/chat-completions +base_model = "meituan/longcat-2.0" +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.75 +output = 2.95 +cache_read = 0.015 + +[limit] +context = 1_049_000 diff --git a/providers/siliconflow/models/moonshotai/Kimi-K2.6.toml b/providers/siliconflow/models/moonshotai/Kimi-K2.6.toml index 73e5e28af8b..48e6aaf149e 100644 --- a/providers/siliconflow/models/moonshotai/Kimi-K2.6.toml +++ b/providers/siliconflow/models/moonshotai/Kimi-K2.6.toml @@ -16,8 +16,8 @@ field = "reasoning_content" [cost] input = 0.77 -output = 4.0 -cache_read = 0.2 +output = 3.4 +cache_read = 0.14 [limit] context = 262_000 diff --git a/providers/siliconflow/models/moonshotai/Kimi-K2.7-Code.toml b/providers/siliconflow/models/moonshotai/Kimi-K2.7-Code.toml new file mode 100644 index 00000000000..f70f197f92d --- /dev/null +++ b/providers/siliconflow/models/moonshotai/Kimi-K2.7-Code.toml @@ -0,0 +1,21 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "moonshotai/kimi-k2.7-code" + + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.85916 +output = 3.8 +cache_read = 0.17993 + +# Cost values are the published USD list prices on SiliconFlow international +# (https://www.siliconflow.com/pricing), not FX conversions by this PR. +# SiliconFlow serves image input but not video for this model +# (model page "Support image input: Yes"): https://www.siliconflow.com/models/kimi-k2-7-code +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/siliconflow/models/moonshotai/Kimi-K3.toml b/providers/siliconflow/models/moonshotai/Kimi-K3.toml new file mode 100644 index 00000000000..daa54bb61f7 --- /dev/null +++ b/providers/siliconflow/models/moonshotai/Kimi-K3.toml @@ -0,0 +1,22 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "moonshotai/kimi-k3" + + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 2.7 +output = 13.5 +cache_read = 0.27 + +[limit] +output = 262_000 + +# SiliconFlow serves image input but not video for this model +# (model page "Support image input: Yes"): https://www.siliconflow.com/models/kimi-k3 +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/siliconflow/models/nex-agi/Nex-N2-Pro.toml b/providers/siliconflow/models/nex-agi/Nex-N2-Pro.toml new file mode 100644 index 00000000000..f160fdf834d --- /dev/null +++ b/providers/siliconflow/models/nex-agi/Nex-N2-Pro.toml @@ -0,0 +1,15 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "nex-agi/nex-n2-pro" + +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.5 +output = 2.5 +cache_read = 0.25 + +[limit] +output = 256_000 diff --git a/providers/siliconflow/models/tencent/Hy3-preview.toml b/providers/siliconflow/models/tencent/Hy3-preview.toml deleted file mode 100644 index fddc3766f5a..00000000000 --- a/providers/siliconflow/models/tencent/Hy3-preview.toml +++ /dev/null @@ -1,18 +0,0 @@ -base_model = "tencent/hy3-preview" -attachment = false -reasoning = true -reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] -open_weights = false - -[cost] -input = 0.066 -cache_read = 0.029 -output = 0.26 - -[limit] -context = 262_144 -output = 262_144 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/siliconflow/models/tencent/Hy3.toml b/providers/siliconflow/models/tencent/Hy3.toml new file mode 100644 index 00000000000..f304cbd9060 --- /dev/null +++ b/providers/siliconflow/models/tencent/Hy3.toml @@ -0,0 +1,15 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +# Limits below are this host's real deltas vs the lab entry (256K/128K): +# SiliconFlow advertises 262K total context and 262K max output +# (https://www.siliconflow.com/models/hy3). +base_model = "tencent/hy3" +reasoning_options = [{ type = "budget_tokens", min = 128, max = 32_768 }] + +[cost] +input = 0.132 +cache_read = 0.033 +output = 0.528 + +[limit] +context = 262_144 +output = 262_144 diff --git a/providers/siliconflow/models/zai-org/GLM-5.1.toml b/providers/siliconflow/models/zai-org/GLM-5.1.toml index 5a341072c0d..248e165981b 100644 --- a/providers/siliconflow/models/zai-org/GLM-5.1.toml +++ b/providers/siliconflow/models/zai-org/GLM-5.1.toml @@ -1,3 +1,6 @@ +# Prices verified against SiliconFlow international, GLM-5.1 model page +# (https://www.siliconflow.com/models/glm-5-1, accessed 2026-09-08): +# input $1.19, cached input $0.6, output $3.74 per M tokens. name = "zai-org/GLM-5.1" description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" family = "glm" @@ -15,14 +18,14 @@ open_weights = true field = "reasoning_content" [cost] -input = 1.4 -output = 4.4 -cache_read = 0.26 +input = 1.19 +output = 3.74 +cache_read = 0.6 cache_write = 0 [limit] context = 205_000 -output = 205_000 +output = 131_000 [modalities] input = ["text"] diff --git a/providers/siliconflow/models/zai-org/GLM-5.2.toml b/providers/siliconflow/models/zai-org/GLM-5.2.toml index 1a6b9e26746..4f4b4d198e8 100644 --- a/providers/siliconflow/models/zai-org/GLM-5.2.toml +++ b/providers/siliconflow/models/zai-org/GLM-5.2.toml @@ -1,3 +1,6 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +# Cost values are the published USD list prices on SiliconFlow international +# (https://www.siliconflow.com/pricing, accessed 2026-09-08), not FX conversions by this PR. base_model = "zhipuai/glm-5.2" reasoning_options = [{ type = "effort", values = ["high", "max"] }] @@ -5,8 +8,8 @@ reasoning_options = [{ type = "effort", values = ["high", "max"] }] field = "reasoning_content" [cost] -input = 1.4 -output = 4.4 +input = 1.302 +output = 4.092 cache_read = 0.26 cache_write = 0 diff --git a/providers/siliconflow/models/zai-org/GLM-5.3-Flash.toml b/providers/siliconflow/models/zai-org/GLM-5.3-Flash.toml new file mode 100644 index 00000000000..afb2e9853b8 --- /dev/null +++ b/providers/siliconflow/models/zai-org/GLM-5.3-Flash.toml @@ -0,0 +1,24 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "zhipuai/glm-5.3-flash" + + +reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.03 +cache_write = 0 + +[limit] +context = 1_049_000 +output = 262_000 + +# SiliconFlow serves image input but not video/pdf for this model +# (model page "Support image input: Yes"): https://www.siliconflow.com/models/glm-5-3-flash +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/siliconflow/models/zai-org/GLM-5.3.toml b/providers/siliconflow/models/zai-org/GLM-5.3.toml new file mode 100644 index 00000000000..3de6268387c --- /dev/null +++ b/providers/siliconflow/models/zai-org/GLM-5.3.toml @@ -0,0 +1,16 @@ +# Sources: https://www.siliconflow.com/models (2026-09-08) +base_model = "zhipuai/glm-5.3" +reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1.4 +output = 4.4 +cache_read = 0.26 +cache_write = 0 + +[limit] +context = 1_049_000 +output = 262_000 diff --git a/providers/stepfun-ai-step-plan/models/step-5-preview.toml b/providers/stepfun-ai-step-plan/models/step-5-preview.toml new file mode 100644 index 00000000000..674a616d307 --- /dev/null +++ b/providers/stepfun-ai-step-plan/models/step-5-preview.toml @@ -0,0 +1,11 @@ +# Chat `reasoning_effort` and Messages `output_config.effort` accept +# low/medium/high (accessed 2026-09-20). +# https://platform.stepfun.ai/docs/en/step-plan/integrations/reasoning-api +base_model = "stepfun/step-5-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/stepfun-ai-step-plan/provider.toml b/providers/stepfun-ai-step-plan/provider.toml index 15e9fb3b704..f357f0a530f 100644 --- a/providers/stepfun-ai-step-plan/provider.toml +++ b/providers/stepfun-ai-step-plan/provider.toml @@ -1,12 +1,12 @@ name = "StepFun Step Plan (Global)" env = ["STEPFUN_API_KEY"] npm = "@ai-sdk/openai-compatible" -# Reasoning HTTP format (accessed 2026-06-25): +# Reasoning HTTP format (accessed 2026-09-20): # Step Plan exposes POST /step_plan/v1/chat/completions with top-level # `reasoning_effort` and POST /step_plan/v1/messages with -# `output_config.effort`. step-3.7-flash accepts low/medium/high; -# step-3.5-flash and step-3.5-flash-2603 accept low/high. No plan Responses -# endpoint is listed. +# `output_config.effort`. step-5-preview and step-3.7-flash accept +# low/medium/high; step-3.5-flash and step-3.5-flash-2603 accept low/high. +# No plan Responses endpoint is listed. # Source: # https://platform.stepfun.ai/docs/en/step-plan/integrations/reasoning-api doc = "https://platform.stepfun.ai/docs/en/step-plan/integrations/reasoning-api" diff --git a/providers/stepfun-ai/models/step-5-preview.toml b/providers/stepfun-ai/models/step-5-preview.toml new file mode 100644 index 00000000000..c7a645ba40a --- /dev/null +++ b/providers/stepfun-ai/models/step-5-preview.toml @@ -0,0 +1,21 @@ +# Chat `reasoning_effort`, Messages `output_config.effort`, and Responses +# `reasoning.effort` accept low/medium/high (accessed 2026-09-21). +# https://platform.stepfun.ai/docs/en/guides/models/step-5-preview +# Verified live: GET https://api.stepfun.ai/v1/models lists step-5-preview +# with reasoning_effort_support_list ["low", "medium", "high"] and +# max_input_tokens 1024000; POST /v1/chat/completions succeeds. +# Pricing (USD/MTok): input 1.00, output 2.70, cache_read 0.05 (95% cache +# discount), matching existing mirrors and vendor reporting. +base_model = "stepfun/step-5-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1 +output = 2.7 +cache_read = 0.05 diff --git a/providers/stepfun-ai/provider.toml b/providers/stepfun-ai/provider.toml index 0c84a7aaa5d..84b890902aa 100644 --- a/providers/stepfun-ai/provider.toml +++ b/providers/stepfun-ai/provider.toml @@ -4,8 +4,8 @@ npm = "@ai-sdk/openai-compatible" # Reasoning HTTP format (accessed 2026-06-29): # POST /v1/chat/completions uses top-level `reasoning_effort`; POST /v1/messages # uses `output_config.effort`; POST /v1/responses uses `reasoning.effort`. -# Values are low/medium/high for step-3.7-flash; step-3.5-flash-2603 accepts -# low/high. Responses supports only step-3.7-flash. Chat returns reasoning at +# Values are low/medium/high for step-3.7-flash and step-5-preview; step-3.5-flash-2603 accepts +# low/high. Responses supports step-3.7-flash and step-5-preview. Chat returns reasoning at # `choices[].message.reasoning` or streamed `choices[].delta.reasoning`; # `reasoning_format` is general (default), the latter using `reasoning_content`. # Responses streams `response.reasoning_text.delta`. diff --git a/providers/stepfun-step-plan/models/step-5-preview.toml b/providers/stepfun-step-plan/models/step-5-preview.toml new file mode 100644 index 00000000000..684d9f67224 --- /dev/null +++ b/providers/stepfun-step-plan/models/step-5-preview.toml @@ -0,0 +1,11 @@ +# Chat `reasoning_effort` accepts low/medium/high (accessed 2026-09-20). +# https://platform.stepfun.com/docs/zh/guides/models/step-5-preview +# Step Plan is subscription/credit based with no public per-token price, so no [cost]. +base_model = "stepfun/step-5-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/stepfun-step-plan/provider.toml b/providers/stepfun-step-plan/provider.toml index ebcf390c442..1c0546449a3 100644 --- a/providers/stepfun-step-plan/provider.toml +++ b/providers/stepfun-step-plan/provider.toml @@ -4,7 +4,7 @@ npm = "@ai-sdk/openai-compatible" # Reasoning HTTP format (accessed 2026-06-25): # Step Plan exposes POST /step_plan/v1/chat/completions with top-level # `reasoning_effort` and POST /step_plan/v1/messages with -# `output_config.effort`. step-3.7-flash accepts low/medium/high; +# `output_config.effort`. step-3.7-flash and step-5-preview accept low/medium/high; # step-3.5-flash and step-3.5-flash-2603 accept low/high. No plan Responses # endpoint is listed. # Source: diff --git a/providers/stepfun/models/step-5-preview.toml b/providers/stepfun/models/step-5-preview.toml new file mode 100644 index 00000000000..94aa0f9f5f1 --- /dev/null +++ b/providers/stepfun/models/step-5-preview.toml @@ -0,0 +1,20 @@ +# Chat `reasoning_effort`, Messages `output_config.effort`, and Responses +# `reasoning.effort` accept low/medium/high (accessed 2026-09-20). +# https://platform.stepfun.com/docs/zh/guides/models/step-5-preview +# Cost converted from CNY list price (¥7 / ¥20 / ¥0.35 per 1M tokens) at the +# same 7.2973 CNY/USD rate used by the other stepfun entries (accessed +# 2026-09-20). +# https://platform.stepfun.com/docs/zh/guides/pricing/details +base_model = "stepfun/step-5-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.959 +output = 2.741 +cache_read = 0.048 diff --git a/providers/stepfun/provider.toml b/providers/stepfun/provider.toml index 0de8764f300..b687e0750f1 100644 --- a/providers/stepfun/provider.toml +++ b/providers/stepfun/provider.toml @@ -1,11 +1,12 @@ name = "StepFun (China)" env = ["STEPFUN_API_KEY"] npm = "@ai-sdk/openai-compatible" -# Reasoning HTTP format (accessed 2026-06-25): +# Reasoning HTTP format (accessed 2026-09-20): # POST /v1/chat/completions uses top-level `reasoning_effort`; POST /v1/messages # uses `output_config.effort`; POST /v1/responses uses `reasoning.effort`. -# Values are low/medium/high for step-3.7-flash; step-3.5-flash-2603 accepts -# low/high. Responses supports only step-3.7-flash. Chat returns reasoning at +# Values are low/medium/high for step-5-preview and step-3.7-flash; +# step-3.5-flash-2603 accepts low/high. Responses supports only +# step-5-preview and step-3.7-flash. Chat returns reasoning at # `choices[].message.reasoning` or streamed `choices[].delta.reasoning`; # `reasoning_format` is general (default) or deepseek-style, the latter using # `reasoning_content`. Responses streams `response.reasoning_text.delta`. diff --git a/providers/tempr/logo.svg b/providers/tempr/logo.svg new file mode 100644 index 00000000000..a9fb18e6892 --- /dev/null +++ b/providers/tempr/logo.svg @@ -0,0 +1,4 @@ + + + + diff --git a/providers/tempr/models/anthropic/claude-fable-5-1.toml b/providers/tempr/models/anthropic/claude-fable-5-1.toml new file mode 100644 index 00000000000..78d27e34265 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-fable-5-1.toml @@ -0,0 +1,16 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = low|medium|high|xhigh|max +base_model = "anthropic/claude-fable-5-1" +structured_output = true + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 10 +output = 50 +cache_read = 0.25 +cache_write = 12.5 diff --git a/providers/tempr/models/anthropic/claude-fable-5.toml b/providers/tempr/models/anthropic/claude-fable-5.toml new file mode 100644 index 00000000000..03847400952 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-fable-5.toml @@ -0,0 +1,16 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = low|medium|high|xhigh|max +base_model = "anthropic/claude-fable-5" +structured_output = true + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 diff --git a/providers/tempr/models/anthropic/claude-haiku-4-5-20251001.toml b/providers/tempr/models/anthropic/claude-haiku-4-5-20251001.toml new file mode 100644 index 00000000000..c2502f29a34 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-haiku-4-5-20251001.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high (sized to a thinking budget) +base_model = "anthropic/claude-haiku-4-5-20251001" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 1 +output = 5 +cache_read = 0.1 +cache_write = 1.25 diff --git a/providers/tempr/models/anthropic/claude-haiku-4-5.toml b/providers/tempr/models/anthropic/claude-haiku-4-5.toml new file mode 100644 index 00000000000..102da98f92e --- /dev/null +++ b/providers/tempr/models/anthropic/claude-haiku-4-5.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high (sized to a thinking budget) +base_model = "anthropic/claude-haiku-4-5" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 1 +output = 5 +cache_read = 0.1 +cache_write = 1.25 diff --git a/providers/tempr/models/anthropic/claude-opus-4-5-20251101.toml b/providers/tempr/models/anthropic/claude-opus-4-5-20251101.toml new file mode 100644 index 00000000000..1ff845fa823 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-4-5-20251101.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high (alias: top-level reasoning_effort) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high +base_model = "anthropic/claude-opus-4-5-20251101" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high"] }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-opus-4-5.toml b/providers/tempr/models/anthropic/claude-opus-4-5.toml new file mode 100644 index 00000000000..35b1bae9829 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-4-5.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high (alias: top-level reasoning_effort) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high +base_model = "anthropic/claude-opus-4-5" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high"] }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-opus-4-6.toml b/providers/tempr/models/anthropic/claude-opus-4-6.toml new file mode 100644 index 00000000000..bd4664a3a5f --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-4-6.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high|max (alias: top-level reasoning_effort) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high|max; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high|max +base_model = "anthropic/claude-opus-4-6" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high", "max"] }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-opus-4-7.toml b/providers/tempr/models/anthropic/claude-opus-4-7.toml new file mode 100644 index 00000000000..5796ecdbe51 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-4-7.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = none|low|medium|high|xhigh|max +base_model = "anthropic/claude-opus-4-7" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-opus-4-8.toml b/providers/tempr/models/anthropic/claude-opus-4-8.toml new file mode 100644 index 00000000000..23f9c6ab18e --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-4-8.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = none|low|medium|high|xhigh|max +base_model = "anthropic/claude-opus-4-8" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-opus-5.toml b/providers/tempr/models/anthropic/claude-opus-5.toml new file mode 100644 index 00000000000..ca74b5d1101 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-opus-5.toml @@ -0,0 +1,16 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = low|medium|high|xhigh|max +base_model = "anthropic/claude-opus-5" +structured_output = true + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/tempr/models/anthropic/claude-sonnet-4-5-20250929.toml b/providers/tempr/models/anthropic/claude-sonnet-4-5-20250929.toml new file mode 100644 index 00000000000..8cda1b84187 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-sonnet-4-5-20250929.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high (sized to a thinking budget) +base_model = "anthropic/claude-sonnet-4-5-20250929" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 diff --git a/providers/tempr/models/anthropic/claude-sonnet-4-5.toml b/providers/tempr/models/anthropic/claude-sonnet-4-5.toml new file mode 100644 index 00000000000..aba63782362 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-sonnet-4-5.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high (sized to a thinking budget) +base_model = "anthropic/claude-sonnet-4-5" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 diff --git a/providers/tempr/models/anthropic/claude-sonnet-4-6.toml b/providers/tempr/models/anthropic/claude-sonnet-4-6.toml new file mode 100644 index 00000000000..396e2f09c38 --- /dev/null +++ b/providers/tempr/models/anthropic/claude-sonnet-4-6.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high|max (alias: top-level reasoning_effort) +# Budget: reasoning.max_tokens (integer reasoning tokens) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high|max; thinking.budget_tokens +# /v1/responses: reasoning.effort = none|low|medium|high|max +base_model = "anthropic/claude-sonnet-4-6" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high", "max"] }, + { type = "budget_tokens", min = 1024 }, +] + +[cost] +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 diff --git a/providers/tempr/models/anthropic/claude-sonnet-5.toml b/providers/tempr/models/anthropic/claude-sonnet-5.toml new file mode 100644 index 00000000000..a7908c1fc7c --- /dev/null +++ b/providers/tempr/models/anthropic/claude-sonnet-5.toml @@ -0,0 +1,18 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# Effort: reasoning.effort = none|low|medium|high|xhigh|max (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = low|medium|high|xhigh|max +# /v1/responses: reasoning.effort = none|low|medium|high|xhigh|max +base_model = "anthropic/claude-sonnet-5" +structured_output = true + +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }, +] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 diff --git a/providers/tempr/models/google/gemini-3-flash-preview.toml b/providers/tempr/models/google/gemini-3-flash-preview.toml new file mode 100644 index 00000000000..af36c2e66df --- /dev/null +++ b/providers/tempr/models/google/gemini-3-flash-preview.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-3-flash-preview" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 0.5 +output = 3 +cache_read = 0.05 +input_audio = 1 diff --git a/providers/tempr/models/google/gemini-3.1-flash-lite.toml b/providers/tempr/models/google/gemini-3.1-flash-lite.toml new file mode 100644 index 00000000000..56979e347f7 --- /dev/null +++ b/providers/tempr/models/google/gemini-3.1-flash-lite.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-3.1-flash-lite" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 0.25 +output = 1.5 +cache_read = 0.025 +input_audio = 0.5 diff --git a/providers/tempr/models/google/gemini-3.1-pro-preview-customtools.toml b/providers/tempr/models/google/gemini-3.1-pro-preview-customtools.toml new file mode 100644 index 00000000000..cc288f0d5ca --- /dev/null +++ b/providers/tempr/models/google/gemini-3.1-pro-preview-customtools.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high +# /v1/responses: reasoning.effort = low|medium|high +base_model = "google/gemini-3.1-pro-preview-customtools" + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high"] }, +] + +[cost] +input = 2 +output = 12 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 200000 } +input = 4 +output = 18 +cache_read = 0.4 diff --git a/providers/tempr/models/google/gemini-3.1-pro-preview.toml b/providers/tempr/models/google/gemini-3.1-pro-preview.toml new file mode 100644 index 00000000000..e08f204c81c --- /dev/null +++ b/providers/tempr/models/google/gemini-3.1-pro-preview.toml @@ -0,0 +1,20 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high +# /v1/responses: reasoning.effort = low|medium|high +base_model = "google/gemini-3.1-pro-preview" + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high"] }, +] + +[cost] +input = 2 +output = 12 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 200000 } +input = 4 +output = 18 +cache_read = 0.4 diff --git a/providers/tempr/models/google/gemini-3.5-flash-lite.toml b/providers/tempr/models/google/gemini-3.5-flash-lite.toml new file mode 100644 index 00000000000..6863ddd6378 --- /dev/null +++ b/providers/tempr/models/google/gemini-3.5-flash-lite.toml @@ -0,0 +1,14 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-3.5-flash-lite" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 0.3 +output = 2.5 +cache_read = 0.03 diff --git a/providers/tempr/models/google/gemini-3.5-flash.toml b/providers/tempr/models/google/gemini-3.5-flash.toml new file mode 100644 index 00000000000..bb1b51b4d5e --- /dev/null +++ b/providers/tempr/models/google/gemini-3.5-flash.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-3.5-flash" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 1.5 +output = 9 +cache_read = 0.15 +input_audio = 1.5 diff --git a/providers/tempr/models/google/gemini-3.6-flash.toml b/providers/tempr/models/google/gemini-3.6-flash.toml new file mode 100644 index 00000000000..aa1d7ce7069 --- /dev/null +++ b/providers/tempr/models/google/gemini-3.6-flash.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-3.6-flash" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 +input_audio = 0.75 diff --git a/providers/tempr/models/google/gemini-3.7-flash.toml b/providers/tempr/models/google/gemini-3.7-flash.toml new file mode 100644 index 00000000000..135e4ffe07a --- /dev/null +++ b/providers/tempr/models/google/gemini-3.7-flash.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high +# /v1/responses: reasoning.effort = low|medium|high +base_model = "google/gemini-3.7-flash" + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high"] }, +] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 +input_audio = 0.75 diff --git a/providers/tempr/models/google/gemini-3.8-flash.toml b/providers/tempr/models/google/gemini-3.8-flash.toml new file mode 100644 index 00000000000..44ee2192f92 --- /dev/null +++ b/providers/tempr/models/google/gemini-3.8-flash.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high +# /v1/responses: reasoning.effort = low|medium|high +base_model = "google/gemini-3.8-flash" + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high"] }, +] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 +input_audio = 0.75 diff --git a/providers/tempr/models/google/gemini-embedding-001.toml b/providers/tempr/models/google/gemini-embedding-001.toml new file mode 100644 index 00000000000..eb9ce08109f --- /dev/null +++ b/providers/tempr/models/google/gemini-embedding-001.toml @@ -0,0 +1,5 @@ +base_model = "google/gemini-embedding-001" + +[cost] +input = 0.15 +output = 0 diff --git a/providers/tempr/models/google/gemini-embedding-2.toml b/providers/tempr/models/google/gemini-embedding-2.toml new file mode 100644 index 00000000000..a9b158bc61f --- /dev/null +++ b/providers/tempr/models/google/gemini-embedding-2.toml @@ -0,0 +1,15 @@ +# Tempr serves this model on /v1/embeddings, which takes OpenAI's embeddings shape, and +# Gemini gets text input only there: image, audio, video and PDF input isn't reachable. +base_model = "google/gemini-embedding-2" +attachment = false + +[cost] +input = 0.2 +output = 0 + +[limit] +output = 1 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/tempr/models/google/gemini-flash-latest.toml b/providers/tempr/models/google/gemini-flash-latest.toml new file mode 100644 index 00000000000..2aa8e6ce3bd --- /dev/null +++ b/providers/tempr/models/google/gemini-flash-latest.toml @@ -0,0 +1,15 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|medium|high +# /v1/responses: reasoning.effort = low|medium|high +base_model = "google/gemini-flash-latest" + +reasoning_options = [ + { type = "effort", values = ["low", "medium", "high"] }, +] + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 +input_audio = 0.75 diff --git a/providers/tempr/models/google/gemini-flash-lite-latest.toml b/providers/tempr/models/google/gemini-flash-lite-latest.toml new file mode 100644 index 00000000000..0c9e9f3e9c8 --- /dev/null +++ b/providers/tempr/models/google/gemini-flash-lite-latest.toml @@ -0,0 +1,14 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = minimal|low|medium|high (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = minimal|low|medium|high +# /v1/responses: reasoning.effort = minimal|low|medium|high +base_model = "google/gemini-flash-lite-latest" + +reasoning_options = [ + { type = "effort", values = ["minimal", "low", "medium", "high"] }, +] + +[cost] +input = 0.3 +output = 2.5 +cache_read = 0.03 diff --git a/providers/tempr/models/google/gemma-4-26b-a4b-it.toml b/providers/tempr/models/google/gemma-4-26b-a4b-it.toml new file mode 100644 index 00000000000..89c5a30b22b --- /dev/null +++ b/providers/tempr/models/google/gemma-4-26b-a4b-it.toml @@ -0,0 +1,9 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# /v1/messages: thinking.type = enabled|disabled +# /v1/responses: reasoning.effort = "none" turns it off, any other level turns it on +base_model = "google/gemma-4-26b-a4b-it" + +reasoning_options = [ + { type = "toggle" }, +] diff --git a/providers/tempr/models/google/gemma-4-31b-it.toml b/providers/tempr/models/google/gemma-4-31b-it.toml new file mode 100644 index 00000000000..94549c23965 --- /dev/null +++ b/providers/tempr/models/google/gemma-4-31b-it.toml @@ -0,0 +1,9 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Toggle: reasoning.enabled = true|false (reasoning.effort = "none" also turns it off) +# /v1/messages: thinking.type = enabled|disabled +# /v1/responses: reasoning.effort = "none" turns it off, any other level turns it on +base_model = "google/gemma-4-31b-it" + +reasoning_options = [ + { type = "toggle" }, +] diff --git a/providers/tempr/provider.toml b/providers/tempr/provider.toml new file mode 100644 index 00000000000..70ceee60f53 --- /dev/null +++ b/providers/tempr/provider.toml @@ -0,0 +1,5 @@ +name = "Tempr" +npm = "@ai-sdk/openai-compatible" +env = ["TEMPR_API_KEY"] +api = "https://api.temprhq.io/v1" +doc = "https://temprhq.io/docs/gateway-reference.html" diff --git a/providers/tensorx/models/deepseek/deepseek-chat-v3.1.toml b/providers/tensorx/models/deepseek/deepseek-chat-v3.1.toml deleted file mode 100644 index 93143b88c68..00000000000 --- a/providers/tensorx/models/deepseek/deepseek-chat-v3.1.toml +++ /dev/null @@ -1,30 +0,0 @@ -name = "DeepSeek Chat V3.1" -description = "DeepSeek chat model for instruction following, coding, and analysis" -family = "deepseek" -release_date = "2025-08-21" -last_updated = "2025-08-21" -attachment = false -reasoning = true -temperature = true -tool_call = true -knowledge = "2024-11" -open_weights = true - - -[[reasoning_options]] -type = "effort" # API: {"reasoning_effort": }; "none" disables reasoning -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.2 -output = 0.8 -cache_read = 0.05 -cache_write = 0.25 - -[limit] -context = 164000 -output = 163840 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/tensorx/models/deepseek/deepseek-v4-flash.toml b/providers/tensorx/models/deepseek/deepseek-v4-flash.toml deleted file mode 100644 index fdea8b9a03d..00000000000 --- a/providers/tensorx/models/deepseek/deepseek-v4-flash.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "deepseek/deepseek-v4-flash" - -[[reasoning_options]] -type = "toggle" # API: {"chat_template_kwargs": {"thinking": true}} (default off) - -[cost] -input = 0.15 -output = 0.3 -cache_read = 0.0375 -cache_write = 0.1875 - -[limit] -context = 1048576 diff --git a/providers/tensorx/models/deepseek/deepseek-v4-pro-0813.toml b/providers/tensorx/models/deepseek/deepseek-v4-pro-0813.toml new file mode 100644 index 00000000000..82492863b2c --- /dev/null +++ b/providers/tensorx/models/deepseek/deepseek-v4-pro-0813.toml @@ -0,0 +1,19 @@ +base_model = "deepseek/deepseek-v4-pro-0813" +# DeepSeek V4 reasons off by default; toggle chat_template_kwargs.thinking, effort high|max when on. +# https://docs.tensorx.ai/api-reference/reasoning + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["high", "max"] + +[cost] +input = 2.0 +output = 4.0 +cache_read = 0.5 + +[limit] +context = 1048576 +output = 64000 \ No newline at end of file diff --git a/providers/tensorx/models/nvidia/nemotron-3-super-120b-a12b.toml b/providers/tensorx/models/nvidia/nemotron-3-super-120b-a12b.toml deleted file mode 100644 index 279eb86ea6a..00000000000 --- a/providers/tensorx/models/nvidia/nemotron-3-super-120b-a12b.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "nvidia/nemotron-3-super-120b-a12b" - -[[reasoning_options]] -type = "effort" # API: {"reasoning_effort": }; "none" disables reasoning -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.3 -output = 0.9 -cache_read = 0.075 -cache_write = 0.375 diff --git a/providers/tensorx/models/openai/gpt-oss-120b.toml b/providers/tensorx/models/openai/gpt-oss-120b.toml deleted file mode 100644 index b34a24534e3..00000000000 --- a/providers/tensorx/models/openai/gpt-oss-120b.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/gpt-oss-120b" -knowledge = "2024-10" - -[[reasoning_options]] -type = "effort" # API: {"reasoning_effort": }; reasoning is mandatory, "none" is rejected -values = ["minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.04 -output = 0.2 -cache_read = 0.01 -cache_write = 0.05 diff --git a/providers/tensorx/models/qwen/qwen3-coder-30b-a3b-instruct.toml b/providers/tensorx/models/qwen/qwen3-coder-30b-a3b-instruct.toml deleted file mode 100644 index bf2ca2550b6..00000000000 --- a/providers/tensorx/models/qwen/qwen3-coder-30b-a3b-instruct.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "alibaba/qwen3-coder-30b-a3b-instruct" - -[cost] -input = 0.06 -output = 0.25 -cache_read = 0.015 -cache_write = 0.075 - -[limit] -context = 262000 diff --git a/providers/tensorx/models/qwen/qwen3-vl-235b-a22b-instruct.toml b/providers/tensorx/models/qwen/qwen3-vl-235b-a22b-instruct.toml deleted file mode 100644 index 63f03d54a44..00000000000 --- a/providers/tensorx/models/qwen/qwen3-vl-235b-a22b-instruct.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Qwen3 VL 235B-A22B Instruct" -description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" -family = "qwen" -release_date = "2025-09-23" -last_updated = "2025-09-23" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-03-31" -open_weights = true - -[cost] -input = 0.21 -output = 1.9 -cache_read = 0.0525 -cache_write = 0.2625 - -[limit] -context = 131000 -output = 131072 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/tensorx/models/qwen/qwen3.8-2.4t-a95b.toml b/providers/tensorx/models/qwen/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..70a027487ce --- /dev/null +++ b/providers/tensorx/models/qwen/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,15 @@ +base_model = "alibaba/qwen3.8-2.4t-a95b" +# Qwen3.8 always reasons; effort xhigh (default) | medium | low. +# https://docs.tensorx.ai/api-reference/reasoning + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 2.5 +output = 6.0 +cache_read = 0.63 + +[limit] +output = 64000 \ No newline at end of file diff --git a/providers/tensorx/models/qwen/qwen3.8-27b.toml b/providers/tensorx/models/qwen/qwen3.8-27b.toml new file mode 100644 index 00000000000..552bc1978f7 --- /dev/null +++ b/providers/tensorx/models/qwen/qwen3.8-27b.toml @@ -0,0 +1,11 @@ +base_model = "alibaba/qwen3.8-27b" +# Thinking is always on; reasoning_effort low|medium|xhigh (default xhigh). + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 0.4 +output = 2.4 +cache_read = 0.1 \ No newline at end of file diff --git a/providers/tensorx/models/qwen/qwen3.8-flash-next.toml b/providers/tensorx/models/qwen/qwen3.8-flash-next.toml new file mode 100644 index 00000000000..7df5eb110ec --- /dev/null +++ b/providers/tensorx/models/qwen/qwen3.8-flash-next.toml @@ -0,0 +1,15 @@ +base_model = "alibaba/qwen3.8-flash-next" +# Qwen3.8 always reasons; effort xhigh (default) | medium | low. +# https://docs.tensorx.ai/api-reference/reasoning + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 0.2 +output = 0.5 +cache_read = 0.05 + +[limit] +output = 64000 \ No newline at end of file diff --git a/providers/tensorx/models/z-ai/glm-4.7.toml b/providers/tensorx/models/z-ai/glm-4.7.toml deleted file mode 100644 index 931c963e6a7..00000000000 --- a/providers/tensorx/models/z-ai/glm-4.7.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "zhipuai/glm-4.7" - -[[reasoning_options]] -type = "effort" # API: {"reasoning_effort": }; "none" disables reasoning -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.6 -output = 2.2 -cache_read = 0.15 -cache_write = 0.75 - -[limit] -context = 200000 -output = 200000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/tensorx/models/z-ai/glm-5.3-flash.toml b/providers/tensorx/models/z-ai/glm-5.3-flash.toml new file mode 100644 index 00000000000..ad6a7edac8e --- /dev/null +++ b/providers/tensorx/models/z-ai/glm-5.3-flash.toml @@ -0,0 +1,20 @@ +base_model = "zhipuai/glm-5.3-flash" +# GLM-5.3-Flash reasons by default; toggle enable_thinking, effort low|high|max (default max). +# Unlike first-party Z.AI, TensorX allows thinking to be disabled (enable_thinking). +# https://docs.tensorx.ai/api-reference/reasoning + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.2 +output = 0.5 +cache_read = 0.05 + +[limit] +context = 1048576 +output = 64000 \ No newline at end of file diff --git a/providers/tensorx/models/z-ai/glm-5.3.toml b/providers/tensorx/models/z-ai/glm-5.3.toml new file mode 100644 index 00000000000..85587ef550d --- /dev/null +++ b/providers/tensorx/models/z-ai/glm-5.3.toml @@ -0,0 +1,20 @@ +base_model = "zhipuai/glm-5.3" +# GLM-5.3 reasons by default; toggle enable_thinking, effort low|high|max (default max). +# Unlike first-party Z.AI, TensorX allows thinking to be disabled (enable_thinking). +# https://docs.tensorx.ai/api-reference/reasoning + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 1.75 +output = 4.5 +cache_read = 0.44 + +[limit] +context = 1048576 +output = 64000 \ No newline at end of file diff --git a/providers/venice/models/claude-opus-5-5.toml b/providers/venice/models/claude-opus-5-5.toml new file mode 100644 index 00000000000..ff60fa3a6ef --- /dev/null +++ b/providers/venice/models/claude-opus-5-5.toml @@ -0,0 +1,17 @@ +base_model = "anthropic/claude-opus-5-5" +description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" +release_date = "2026-09-18" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4.8 +output = 24 +cache_read = 0.24 +cache_write = 6 + +[modalities] +input = ["text", "image"] diff --git a/providers/venice/models/grok-4-7.toml b/providers/venice/models/grok-4-7.toml new file mode 100644 index 00000000000..e405114dc5c --- /dev/null +++ b/providers/venice/models/grok-4-7.toml @@ -0,0 +1,24 @@ +base_model = "xai/grok-4.7" +description = "Grok model for agentic tool use, reasoning, coding, and live assistance" +release_date = "2026-09-16" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2.27 +output = 6.8 +cache_read = 0.57 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4.53 +output = 13.6 +cache_read = 1.13 + +[limit] +output = 200_000 + +[modalities] +input = ["text", "image"] diff --git a/providers/venice/models/openai-gpt-56-sol-pro.toml b/providers/venice/models/openai-gpt-56-sol-pro.toml index 36093d01914..1b837b0d610 100644 --- a/providers/venice/models/openai-gpt-56-sol-pro.toml +++ b/providers/venice/models/openai-gpt-56-sol-pro.toml @@ -8,18 +8,18 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 2.5 -output = 12.5 -cache_read = 0.25 -cache_write = 3.125 - -[[cost.tiers]] -tier = { type = "context", size = 272_000 } input = 5 -output = 18.75 +output = 25 cache_read = 0.5 cache_write = 6.25 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 10 +output = 37.5 +cache_read = 1 +cache_write = 12.5 + [limit] context = 1_000_000 diff --git a/providers/venice/models/openai-gpt-56-sol.toml b/providers/venice/models/openai-gpt-56-sol.toml index dedfc528097..a0a89340131 100644 --- a/providers/venice/models/openai-gpt-56-sol.toml +++ b/providers/venice/models/openai-gpt-56-sol.toml @@ -7,18 +7,18 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 2.5 -output = 12.5 -cache_read = 0.25 -cache_write = 3.125 - -[[cost.tiers]] -tier = { type = "context", size = 272_000 } input = 5 -output = 18.75 +output = 25 cache_read = 0.5 cache_write = 6.25 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 10 +output = 37.5 +cache_read = 1 +cache_write = 12.5 + [limit] context = 1_000_000 diff --git a/providers/venice/models/openai-gpt-6-luna.toml b/providers/venice/models/openai-gpt-6-luna.toml new file mode 100644 index 00000000000..df056dfdc5c --- /dev/null +++ b/providers/venice/models/openai-gpt-6-luna.toml @@ -0,0 +1,22 @@ +base_model = "openai/gpt-6-luna" +description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.125 +output = 0.625 +cache_read = 0.0125 +cache_write = 0.15625 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.25 +output = 0.9375 +cache_read = 0.025 +cache_write = 0.3125 + +[modalities] +input = ["text", "image"] diff --git a/providers/venice/models/openai-gpt-6-sol.toml b/providers/venice/models/openai-gpt-6-sol.toml new file mode 100644 index 00000000000..5a130d788ba --- /dev/null +++ b/providers/venice/models/openai-gpt-6-sol.toml @@ -0,0 +1,22 @@ +base_model = "openai/gpt-6-sol" +description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2.5 +output = 12.5 +cache_read = 0.25 +cache_write = 3.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 5 +output = 18.75 +cache_read = 0.5 +cache_write = 6.25 + +[modalities] +input = ["text", "image"] diff --git a/providers/vercel/models/alibaba/qwen-3.6-max-preview.toml b/providers/vercel/models/alibaba/qwen-3.6-max-preview.toml index 2978e3e117c..df7e9852286 100644 --- a/providers/vercel/models/alibaba/qwen-3.6-max-preview.toml +++ b/providers/vercel/models/alibaba/qwen-3.6-max-preview.toml @@ -23,6 +23,13 @@ output = 7.8 cache_read = 0.13 cache_write = 1.625 +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 2 +output = 12 +cache_read = 0.2 +cache_write = 2.5 + [limit] context = 240_000 output = 64_000 diff --git a/providers/vercel/models/alibaba/qwen3-coder-plus.toml b/providers/vercel/models/alibaba/qwen3-coder-plus.toml index 06f877de2eb..656712e8607 100644 --- a/providers/vercel/models/alibaba/qwen3-coder-plus.toml +++ b/providers/vercel/models/alibaba/qwen3-coder-plus.toml @@ -5,5 +5,23 @@ input = 1 output = 5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 32_001 } +input = 1.8 +output = 9 +cache_read = 0.36 + +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 3 +output = 15 +cache_read = 0.6 + +[[cost.tiers]] +tier = { type = "context", size = 256_001 } +input = 6 +output = 60 +cache_read = 1.2 + [limit] context = 1_000_000 diff --git a/providers/vercel/models/alibaba/qwen3-coder.toml b/providers/vercel/models/alibaba/qwen3-coder.toml index e61dd0d67dd..f5fe46e649f 100644 --- a/providers/vercel/models/alibaba/qwen3-coder.toml +++ b/providers/vercel/models/alibaba/qwen3-coder.toml @@ -16,6 +16,18 @@ input = 1.5 output = 7.5 cache_read = 0.3 +[[cost.tiers]] +tier = { type = "context", size = 32_001 } +input = 2.7 +output = 13.5 +cache_read = 0.54 + +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 4.5 +output = 22.5 +cache_read = 0.9 + [limit] context = 262_144 output = 65_536 diff --git a/providers/vercel/models/alibaba/qwen3-max-preview.toml b/providers/vercel/models/alibaba/qwen3-max-preview.toml index 94b902c8984..69a704070b5 100644 --- a/providers/vercel/models/alibaba/qwen3-max-preview.toml +++ b/providers/vercel/models/alibaba/qwen3-max-preview.toml @@ -15,6 +15,18 @@ input = 1.2 output = 6 cache_read = 0.24 +[[cost.tiers]] +tier = { type = "context", size = 32_001 } +input = 2.4 +output = 12 +cache_read = 0.48 + +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 3 +output = 15 +cache_read = 0.6 + [limit] context = 262_144 output = 32_768 diff --git a/providers/vercel/models/alibaba/qwen3-max-thinking.toml b/providers/vercel/models/alibaba/qwen3-max-thinking.toml index db97c26cd56..b00cd1597a9 100644 --- a/providers/vercel/models/alibaba/qwen3-max-thinking.toml +++ b/providers/vercel/models/alibaba/qwen3-max-thinking.toml @@ -5,17 +5,33 @@ release_date = "2026-01-23" last_updated = "2025-01" attachment = false reasoning = true -reasoning_options = [{ type = "budget_tokens", min = 1, max = 81_920 }] temperature = true tool_call = true knowledge = "2025-01" open_weights = true +[[reasoning_options]] +type = "budget_tokens" +min = 1 +max = 81_920 + [cost] input = 1.2 output = 6 cache_read = 0.24 +[[cost.tiers]] +tier = { type = "context", size = 32_001 } +input = 2.4 +output = 12 +cache_read = 0.48 + +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 3 +output = 15 +cache_read = 0.6 + [limit] context = 256_000 output = 65_536 diff --git a/providers/vercel/models/alibaba/qwen3-max.toml b/providers/vercel/models/alibaba/qwen3-max.toml index 74f80d21720..556b71930db 100644 --- a/providers/vercel/models/alibaba/qwen3-max.toml +++ b/providers/vercel/models/alibaba/qwen3-max.toml @@ -5,5 +5,17 @@ input = 1.2 output = 6 cache_read = 0.24 +[[cost.tiers]] +tier = { type = "context", size = 32_001 } +input = 2.4 +output = 12 +cache_read = 0.48 + +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 3 +output = 15 +cache_read = 0.6 + [limit] output = 32_768 diff --git a/providers/vercel/models/alibaba/qwen3.5-plus.toml b/providers/vercel/models/alibaba/qwen3.5-plus.toml index fab6054e8ca..3f4c69a844a 100644 --- a/providers/vercel/models/alibaba/qwen3.5-plus.toml +++ b/providers/vercel/models/alibaba/qwen3.5-plus.toml @@ -16,6 +16,13 @@ output = 2.5 cache_read = 0.04 cache_write = 0.5 +[[cost.tiers]] +tier = { type = "context", size = 256_001 } +input = 0.5 +output = 3 +cache_read = 0.05 +cache_write = 0.625 + [limit] output = 64_000 diff --git a/providers/vercel/models/alibaba/qwen3.6-plus.toml b/providers/vercel/models/alibaba/qwen3.6-plus.toml index 1b29656202b..e87a6735aae 100644 --- a/providers/vercel/models/alibaba/qwen3.6-plus.toml +++ b/providers/vercel/models/alibaba/qwen3.6-plus.toml @@ -15,6 +15,13 @@ output = 3 cache_read = 0.05 cache_write = 0.625 +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 2 +output = 6 +cache_read = 0.2 +cache_write = 2.5 + [limit] output = 64_000 diff --git a/providers/vercel/models/alibaba/qwen3.7-flash.toml b/providers/vercel/models/alibaba/qwen3.7-flash.toml index 774d5fe725d..8aaf3abadf1 100644 --- a/providers/vercel/models/alibaba/qwen3.7-flash.toml +++ b/providers/vercel/models/alibaba/qwen3.7-flash.toml @@ -10,6 +10,20 @@ output = 0.13 cache_read = 0.006 cache_write = 0.038 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.1 +output = 0.4 +cache_read = 0.02 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.2 +output = 0.8 +cache_read = 0.04 +cache_write = 0.25 + [limit] context = 991_000 output = 64_000 diff --git a/providers/vercel/models/alibaba/qwen3.7-plus.toml b/providers/vercel/models/alibaba/qwen3.7-plus.toml index 819ab65c9e3..4511b24804c 100644 --- a/providers/vercel/models/alibaba/qwen3.7-plus.toml +++ b/providers/vercel/models/alibaba/qwen3.7-plus.toml @@ -24,5 +24,12 @@ output = 1.6 cache_read = 0.08 cache_write = 0.5 +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 1.2 +output = 4.8 +cache_read = 0.24 +cache_write = 1.5 + [modalities] input = ["text", "image", "pdf"] diff --git a/providers/vercel/models/anthropic/claude-opus-4.toml b/providers/vercel/models/anthropic/claude-opus-4.toml index a80b0442665..2553fa1f326 100644 --- a/providers/vercel/models/anthropic/claude-opus-4.toml +++ b/providers/vercel/models/anthropic/claude-opus-4.toml @@ -6,6 +6,3 @@ input = 15 output = 75 cache_read = 1.5 cache_write = 18.75 - -[limit] -output = 8_192 diff --git a/providers/vercel/models/anthropic/claude-opus-5.5-fast.toml b/providers/vercel/models/anthropic/claude-opus-5.5-fast.toml new file mode 100644 index 00000000000..9e516c8d185 --- /dev/null +++ b/providers/vercel/models/anthropic/claude-opus-5.5-fast.toml @@ -0,0 +1,9 @@ +base_model = "anthropic/claude-opus-5-5" +name = "Claude Opus 5.5 (Fast)" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 8 +output = 40 +cache_read = 0.4 +cache_write = 10 diff --git a/providers/vercel/models/anthropic/claude-opus-5.5.toml b/providers/vercel/models/anthropic/claude-opus-5.5.toml new file mode 100644 index 00000000000..8c3442bfe4b --- /dev/null +++ b/providers/vercel/models/anthropic/claude-opus-5.5.toml @@ -0,0 +1,8 @@ +base_model = "anthropic/claude-opus-5-5" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.2 +cache_write = 5 diff --git a/providers/vercel/models/anthropic/claude-sonnet-4.toml b/providers/vercel/models/anthropic/claude-sonnet-4.toml index df6a061b4b1..7e189ae395a 100644 --- a/providers/vercel/models/anthropic/claude-sonnet-4.toml +++ b/providers/vercel/models/anthropic/claude-sonnet-4.toml @@ -7,6 +7,12 @@ output = 15 cache_read = 0.3 cache_write = 3.75 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 6 +output = 22.5 +cache_read = 0.6 +cache_write = 7.5 + [limit] context = 1_000_000 -output = 8_192 diff --git a/providers/vercel/models/bytedance/seed-1.6.toml b/providers/vercel/models/bytedance/seed-1.6.toml index c0346917f50..bc03000e407 100644 --- a/providers/vercel/models/bytedance/seed-1.6.toml +++ b/providers/vercel/models/bytedance/seed-1.6.toml @@ -18,6 +18,12 @@ input = 0.25 output = 2 cache_read = 0.05 +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 0.5 +output = 4 +cache_read = 0.05 + [limit] context = 256_000 output = 32_000 diff --git a/providers/vercel/models/bytedance/seed-1.8.toml b/providers/vercel/models/bytedance/seed-1.8.toml index 1fa077191ce..7e02aaccd52 100644 --- a/providers/vercel/models/bytedance/seed-1.8.toml +++ b/providers/vercel/models/bytedance/seed-1.8.toml @@ -5,17 +5,29 @@ release_date = "2025-09-01" last_updated = "2025-10" attachment = false reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] temperature = true tool_call = true knowledge = "2024-10" open_weights = false +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + [cost] input = 0.25 output = 2 cache_read = 0.05 +[[cost.tiers]] +tier = { type = "context", size = 128_001 } +input = 0.5 +output = 4 +cache_read = 0.05 + [limit] context = 256_000 output = 64_000 diff --git a/providers/vercel/models/deepseek/deepseek-v4.1-flash.toml b/providers/vercel/models/deepseek/deepseek-v4.1-flash.toml index c185e36af68..5d9ec77c911 100644 --- a/providers/vercel/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/vercel/models/deepseek/deepseek-v4.1-flash.toml @@ -12,7 +12,7 @@ values = ["high", "xhigh"] [cost] input = 0.3 output = 1.2 -cache_read = 0.03 +cache_read = 0.007 [limit] context = 1_048_576 diff --git a/providers/vercel/models/fish-audio/s1-free.toml b/providers/vercel/models/fish-audio/s1-free.toml deleted file mode 100644 index 2eb0f456def..00000000000 --- a/providers/vercel/models/fish-audio/s1-free.toml +++ /dev/null @@ -1,16 +0,0 @@ -name = "S1 (Free)" -description = "Speech generation model for controllable voice, narration, and audio delivery" -release_date = "2025-10-20" -last_updated = "2025-10-20" -attachment = false -reasoning = false -tool_call = false -open_weights = false - -[limit] -context = 0 -output = 0 - -[modalities] -input = ["text"] -output = ["audio"] diff --git a/providers/vercel/models/fish-audio/s2-pro-free.toml b/providers/vercel/models/fish-audio/s2-pro-free.toml deleted file mode 100644 index c2bbbb724c8..00000000000 --- a/providers/vercel/models/fish-audio/s2-pro-free.toml +++ /dev/null @@ -1,16 +0,0 @@ -name = "S2 Pro (Free)" -description = "Speech generation model for controllable voice, narration, and audio delivery" -release_date = "2026-03-09" -last_updated = "2026-03-09" -attachment = false -reasoning = false -tool_call = false -open_weights = false - -[limit] -context = 0 -output = 0 - -[modalities] -input = ["text"] -output = ["audio"] diff --git a/providers/vercel/models/fish-audio/s2.1-pro-free.toml b/providers/vercel/models/fish-audio/s2.1-pro-free.toml deleted file mode 100644 index c580e92cd34..00000000000 --- a/providers/vercel/models/fish-audio/s2.1-pro-free.toml +++ /dev/null @@ -1,16 +0,0 @@ -name = "S2.1 Pro (Free)" -description = "Speech generation model for controllable voice, narration, and audio delivery" -release_date = "2026-07-28" -last_updated = "2026-07-28" -attachment = false -reasoning = false -tool_call = false -open_weights = false - -[limit] -context = 0 -output = 0 - -[modalities] -input = ["text"] -output = ["audio"] diff --git a/providers/vercel/models/fish-audio/transcribe-1-free.toml b/providers/vercel/models/fish-audio/transcribe-1-free.toml deleted file mode 100644 index e21df2298c0..00000000000 --- a/providers/vercel/models/fish-audio/transcribe-1-free.toml +++ /dev/null @@ -1,16 +0,0 @@ -name = "Transcribe-1 (Free)" -description = "Speech transcription model for accurate audio-to-text and captioning workflows" -release_date = "2026-03-01" -last_updated = "2026-03-01" -attachment = false -reasoning = false -tool_call = false -open_weights = false - -[limit] -context = 0 -output = 0 - -[modalities] -input = ["audio"] -output = ["text"] diff --git a/providers/vercel/models/google/gemini-2.5-flash-image.toml b/providers/vercel/models/google/gemini-2.5-flash-image.toml index 1ea71885c13..115f8bd626c 100644 --- a/providers/vercel/models/google/gemini-2.5-flash-image.toml +++ b/providers/vercel/models/google/gemini-2.5-flash-image.toml @@ -2,7 +2,6 @@ base_model = "google/gemini-2.5-flash-image" name = "Nano Banana (Gemini 2.5 Flash Image)" attachment = false reasoning = false -knowledge = "2025-01" [cost] input = 0.3 @@ -10,4 +9,4 @@ output = 2.5 cache_read = 0.03 [limit] -output = 65_536 +output = 65_535 diff --git a/providers/vercel/models/google/gemini-2.5-flash-lite.toml b/providers/vercel/models/google/gemini-2.5-flash-lite.toml index 9dbd105b8d4..c77c8d9fb11 100644 --- a/providers/vercel/models/google/gemini-2.5-flash-lite.toml +++ b/providers/vercel/models/google/gemini-2.5-flash-lite.toml @@ -1,12 +1,21 @@ base_model = "google/gemini-2.5-flash-lite" -reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", min = 512, max = 24_576 }] name = "Gemini 2.5 Flash Lite" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" +min = 512 +max = 24_576 + [cost] input = 0.1 output = 0.4 cache_read = 0.01 +[limit] +output = 65_535 + [modalities] input = ["text", "image", "pdf"] -output = ["text"] diff --git a/providers/vercel/models/google/gemini-3.1-pro-preview.toml b/providers/vercel/models/google/gemini-3.1-pro-preview.toml index 31fde20cebe..742f9779a46 100644 --- a/providers/vercel/models/google/gemini-3.1-pro-preview.toml +++ b/providers/vercel/models/google/gemini-3.1-pro-preview.toml @@ -9,6 +9,12 @@ input = 2 output = 12 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 4 +output = 18 +cache_read = 0.4 + [limit] context = 1_000_000 output = 64_000 diff --git a/providers/vercel/models/google/gemini-3.7-flash.toml b/providers/vercel/models/google/gemini-3.7-flash.toml index 9a978ed0a3b..c75cf2fdf40 100644 --- a/providers/vercel/models/google/gemini-3.7-flash.toml +++ b/providers/vercel/models/google/gemini-3.7-flash.toml @@ -12,6 +12,7 @@ cache_read = 0.075 [limit] context = 1_000_000 +output = 65_535 [modalities] input = ["text", "image", "pdf"] diff --git a/providers/vercel/models/google/gemini-3.8-flash.toml b/providers/vercel/models/google/gemini-3.8-flash.toml index 32fddda9360..39eb1bf1658 100644 --- a/providers/vercel/models/google/gemini-3.8-flash.toml +++ b/providers/vercel/models/google/gemini-3.8-flash.toml @@ -8,6 +8,7 @@ cache_read = 0.075 [limit] context = 1_000_000 +output = 65_535 [modalities] input = ["text", "image", "pdf"] diff --git a/providers/vercel/models/mixedbread/toast-1.toml b/providers/vercel/models/mixedbread/toast-1.toml new file mode 100644 index 00000000000..b50f8e55367 --- /dev/null +++ b/providers/vercel/models/mixedbread/toast-1.toml @@ -0,0 +1,6 @@ +base_model = "mixedbread/toast-1" + +[cost] +input = 0.3 +output = 0.72 +cache_read = 0.036 diff --git a/providers/vercel/models/openai/gpt-5.4-pro.toml b/providers/vercel/models/openai/gpt-5.4-pro.toml index 802e2b01e7d..8f49caeaef4 100644 --- a/providers/vercel/models/openai/gpt-5.4-pro.toml +++ b/providers/vercel/models/openai/gpt-5.4-pro.toml @@ -10,5 +10,10 @@ values = ["medium", "high", "xhigh"] input = 30 output = 180 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 60 +output = 270 + [modalities] input = ["text", "image", "pdf"] diff --git a/providers/vercel/models/openai/gpt-5.4.toml b/providers/vercel/models/openai/gpt-5.4.toml index 117744507a5..e80519bdd16 100644 --- a/providers/vercel/models/openai/gpt-5.4.toml +++ b/providers/vercel/models/openai/gpt-5.4.toml @@ -1,9 +1,17 @@ base_model = "openai/gpt-5.4" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] name = "GPT 5.4" -temperature = true + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 2.5 output = 15 cache_read = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 5 +output = 22.5 +cache_read = 0.5 diff --git a/providers/vercel/models/openai/gpt-5.5-pro.toml b/providers/vercel/models/openai/gpt-5.5-pro.toml index effd8f55a5a..b124e744fda 100644 --- a/providers/vercel/models/openai/gpt-5.5-pro.toml +++ b/providers/vercel/models/openai/gpt-5.5-pro.toml @@ -10,6 +10,11 @@ values = ["medium", "high", "xhigh"] input = 30 output = 180 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 60 +output = 270 + [limit] context = 1_000_000 input = 872_000 diff --git a/providers/vercel/models/openai/gpt-5.5.toml b/providers/vercel/models/openai/gpt-5.5.toml index 4d37b9c3632..6fa3472701a 100644 --- a/providers/vercel/models/openai/gpt-5.5.toml +++ b/providers/vercel/models/openai/gpt-5.5.toml @@ -11,6 +11,12 @@ input = 5 output = 30 cache_read = 0.5 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 10 +output = 45 +cache_read = 1 + [limit] context = 1_000_000 input = 872_000 diff --git a/providers/vercel/models/openai/gpt-5.6-luna-fast.toml b/providers/vercel/models/openai/gpt-5.6-luna-fast.toml index 0677db18d76..74dd96c0e31 100644 --- a/providers/vercel/models/openai/gpt-5.6-luna-fast.toml +++ b/providers/vercel/models/openai/gpt-5.6-luna-fast.toml @@ -10,3 +10,10 @@ input = 0.4 output = 2.4 cache_read = 0.04 cache_write = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.8 +output = 3.6 +cache_read = 0.08 +cache_write = 1 diff --git a/providers/vercel/models/openai/gpt-5.6-luna.toml b/providers/vercel/models/openai/gpt-5.6-luna.toml index f33d0f6f9b5..e48b3985f6c 100644 --- a/providers/vercel/models/openai/gpt-5.6-luna.toml +++ b/providers/vercel/models/openai/gpt-5.6-luna.toml @@ -12,3 +12,10 @@ input = 0.2 output = 1.2 cache_read = 0.02 cache_write = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.4 +output = 1.8 +cache_read = 0.04 +cache_write = 0.5 diff --git a/providers/vercel/models/openai/gpt-5.6-sol-fast.toml b/providers/vercel/models/openai/gpt-5.6-sol-fast.toml index adf0da2fcb0..d989b000e02 100644 --- a/providers/vercel/models/openai/gpt-5.6-sol-fast.toml +++ b/providers/vercel/models/openai/gpt-5.6-sol-fast.toml @@ -6,7 +6,14 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 4 -output = 20 -cache_read = 0.4 -cache_write = 5 +input = 8 +output = 40 +cache_read = 0.8 +cache_write = 10 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 16 +output = 60 +cache_read = 1.6 +cache_write = 20 diff --git a/providers/vercel/models/openai/gpt-5.6-sol.toml b/providers/vercel/models/openai/gpt-5.6-sol.toml index 8b88a743a65..ed5b73a9699 100644 --- a/providers/vercel/models/openai/gpt-5.6-sol.toml +++ b/providers/vercel/models/openai/gpt-5.6-sol.toml @@ -8,7 +8,14 @@ type = "effort" values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 2 -output = 10 -cache_read = 0.2 -cache_write = 2.5 +input = 4 +output = 20 +cache_read = 0.4 +cache_write = 5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 8 +output = 30 +cache_read = 0.8 +cache_write = 10 diff --git a/providers/vercel/models/openai/gpt-5.6-terra-fast.toml b/providers/vercel/models/openai/gpt-5.6-terra-fast.toml index c5e03cad536..e2caf6406e9 100644 --- a/providers/vercel/models/openai/gpt-5.6-terra-fast.toml +++ b/providers/vercel/models/openai/gpt-5.6-terra-fast.toml @@ -10,3 +10,10 @@ input = 4 output = 24 cache_read = 0.4 cache_write = 5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 8 +output = 36 +cache_read = 0.8 +cache_write = 10 diff --git a/providers/vercel/models/openai/gpt-5.6-terra.toml b/providers/vercel/models/openai/gpt-5.6-terra.toml index 7c036e01d1c..cc47098412f 100644 --- a/providers/vercel/models/openai/gpt-5.6-terra.toml +++ b/providers/vercel/models/openai/gpt-5.6-terra.toml @@ -12,3 +12,10 @@ input = 2 output = 12 cache_read = 0.2 cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 18 +cache_read = 0.4 +cache_write = 5 diff --git a/providers/vercel/models/openai/gpt-6-astra-fast.toml b/providers/vercel/models/openai/gpt-6-astra-fast.toml index ad67cd2908a..bbb8660be6c 100644 --- a/providers/vercel/models/openai/gpt-6-astra-fast.toml +++ b/providers/vercel/models/openai/gpt-6-astra-fast.toml @@ -16,4 +16,4 @@ tier = { type = "context", size = 272_001 } input = 40 output = 150 cache_read = 4 -cache_write = 25 +cache_write = 50 diff --git a/providers/vercel/models/openai/gpt-6-luna-fast.toml b/providers/vercel/models/openai/gpt-6-luna-fast.toml new file mode 100644 index 00000000000..5087080071f --- /dev/null +++ b/providers/vercel/models/openai/gpt-6-luna-fast.toml @@ -0,0 +1,19 @@ +base_model = "openai/gpt-6-luna" +name = "GPT-6 Luna (Fast)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.2 +output = 1 +cache_read = 0.02 +cache_write = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 0.4 +output = 1.5 +cache_read = 0.04 +cache_write = 0.5 diff --git a/providers/vercel/models/openai/gpt-6-luna.toml b/providers/vercel/models/openai/gpt-6-luna.toml new file mode 100644 index 00000000000..c9115379e71 --- /dev/null +++ b/providers/vercel/models/openai/gpt-6-luna.toml @@ -0,0 +1,18 @@ +base_model = "openai/gpt-6-luna" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.1 +output = 0.5 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 0.2 +output = 0.75 +cache_read = 0.02 +cache_write = 0.25 diff --git a/providers/vercel/models/openai/gpt-6-sol-fast.toml b/providers/vercel/models/openai/gpt-6-sol-fast.toml new file mode 100644 index 00000000000..8b33bf6106a --- /dev/null +++ b/providers/vercel/models/openai/gpt-6-sol-fast.toml @@ -0,0 +1,19 @@ +base_model = "openai/gpt-6-sol" +name = "GPT-6 Sol (Fast)" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.4 +cache_write = 5 + +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 8 +output = 30 +cache_read = 0.8 +cache_write = 10 diff --git a/providers/vercel/models/openai/gpt-6-sol.toml b/providers/vercel/models/openai/gpt-6-sol.toml new file mode 100644 index 00000000000..39b8ede6517 --- /dev/null +++ b/providers/vercel/models/openai/gpt-6-sol.toml @@ -0,0 +1,18 @@ +base_model = "openai/gpt-6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 4 +output = 15 +cache_read = 0.4 +cache_write = 5 diff --git a/providers/vercel/models/quiverai/arrow-2-telos.toml b/providers/vercel/models/quiverai/arrow-2-telos.toml new file mode 100644 index 00000000000..59dae7b0880 --- /dev/null +++ b/providers/vercel/models/quiverai/arrow-2-telos.toml @@ -0,0 +1,13 @@ +# https://docs.quiver.ai/api-reference/openresponses/createopenresponse +base_model = "quiverai/arrow-2-telos" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] + +[cost] +input = 6 +output = 30 +cache_read = 0.6 +cache_write = 7.5 + +[limit] +context = 131_072 +output = 131_072 diff --git a/providers/vercel/models/quiverai/arrow-2.toml b/providers/vercel/models/quiverai/arrow-2.toml new file mode 100644 index 00000000000..2b7145c9081 --- /dev/null +++ b/providers/vercel/models/quiverai/arrow-2.toml @@ -0,0 +1,12 @@ +# https://docs.quiver.ai/api-reference/openresponses/createopenresponse +base_model = "quiverai/arrow-2" +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] + +[cost] +input = 4 +output = 20 +cache_read = 0.4 +cache_write = 5 + +[limit] +output = 131_072 diff --git a/providers/vercel/models/sakana/fugu-ultra-v2.toml b/providers/vercel/models/sakana/fugu-ultra-v2.toml index f77055d8e6e..3ada0b1935d 100644 --- a/providers/vercel/models/sakana/fugu-ultra-v2.toml +++ b/providers/vercel/models/sakana/fugu-ultra-v2.toml @@ -9,13 +9,22 @@ attachment = true reasoning = true tool_call = true open_weights = false -reasoning_options = [{ type = "effort", values = ["high", "xhigh", "max"] }] + +[[reasoning_options]] +type = "effort" +values = ["high", "xhigh", "max"] [cost] input = 5 output = 30 cache_read = 0.5 +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 10 +output = 45 +cache_read = 1 + [limit] context = 1_000_000 output = 1_000_000 diff --git a/providers/vercel/models/sakana/fugu-ultra.toml b/providers/vercel/models/sakana/fugu-ultra.toml index 099d441fbe6..f36e9177ff8 100644 --- a/providers/vercel/models/sakana/fugu-ultra.toml +++ b/providers/vercel/models/sakana/fugu-ultra.toml @@ -8,5 +8,11 @@ input = 5 output = 30 cache_read = 0.5 +[[cost.tiers]] +tier = { type = "context", size = 272_001 } +input = 10 +output = 45 +cache_read = 1 + [limit] output = 1_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-multi-agent-beta.toml b/providers/vercel/models/spacexai/grok-4.20-multi-agent-beta.toml index ed37aa69ee5..d45eab3701e 100644 --- a/providers/vercel/models/spacexai/grok-4.20-multi-agent-beta.toml +++ b/providers/vercel/models/spacexai/grok-4.20-multi-agent-beta.toml @@ -14,6 +14,12 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-multi-agent.toml b/providers/vercel/models/spacexai/grok-4.20-multi-agent.toml index 37a06048cc0..a90774a3c3f 100644 --- a/providers/vercel/models/spacexai/grok-4.20-multi-agent.toml +++ b/providers/vercel/models/spacexai/grok-4.20-multi-agent.toml @@ -14,6 +14,12 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-non-reasoning-beta.toml b/providers/vercel/models/spacexai/grok-4.20-non-reasoning-beta.toml index acd189124d0..ac3eafaaecd 100644 --- a/providers/vercel/models/spacexai/grok-4.20-non-reasoning-beta.toml +++ b/providers/vercel/models/spacexai/grok-4.20-non-reasoning-beta.toml @@ -13,6 +13,12 @@ input = 1.25 output = 2.5 cache_read = 0.4 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-non-reasoning.toml b/providers/vercel/models/spacexai/grok-4.20-non-reasoning.toml index 468d77b85b2..0e9a2baf86f 100644 --- a/providers/vercel/models/spacexai/grok-4.20-non-reasoning.toml +++ b/providers/vercel/models/spacexai/grok-4.20-non-reasoning.toml @@ -13,6 +13,12 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-reasoning-beta.toml b/providers/vercel/models/spacexai/grok-4.20-reasoning-beta.toml index 19522d5f419..691eded5322 100644 --- a/providers/vercel/models/spacexai/grok-4.20-reasoning-beta.toml +++ b/providers/vercel/models/spacexai/grok-4.20-reasoning-beta.toml @@ -14,6 +14,12 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.20-reasoning.toml b/providers/vercel/models/spacexai/grok-4.20-reasoning.toml index 8fb2843f07d..e6c7e7ae17b 100644 --- a/providers/vercel/models/spacexai/grok-4.20-reasoning.toml +++ b/providers/vercel/models/spacexai/grok-4.20-reasoning.toml @@ -14,6 +14,12 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] context = 2_000_000 output = 2_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.3.toml b/providers/vercel/models/spacexai/grok-4.3.toml index d379c871d7f..1ce53c49412 100644 --- a/providers/vercel/models/spacexai/grok-4.3.toml +++ b/providers/vercel/models/spacexai/grok-4.3.toml @@ -1,5 +1,6 @@ base_model = "xai/grok-4.3" description = "Grok model for agentic tool use, reasoning, coding, and live assistance" + [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] @@ -9,5 +10,11 @@ input = 1.25 output = 2.5 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.5 +output = 5 +cache_read = 0.4 + [limit] output = 1_000_000 diff --git a/providers/vercel/models/spacexai/grok-4.5.toml b/providers/vercel/models/spacexai/grok-4.5.toml index 2eb4ebf0d6e..ec9457541d2 100644 --- a/providers/vercel/models/spacexai/grok-4.5.toml +++ b/providers/vercel/models/spacexai/grok-4.5.toml @@ -1,5 +1,6 @@ base_model = "xai/grok-4.5" description = "Grok model for agentic tool use, reasoning, coding, and live assistance" + [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] @@ -9,5 +10,11 @@ input = 2 output = 6 cache_read = 0.3 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 4 +output = 12 +cache_read = 0.6 + [modalities] input = ["text", "image", "pdf"] diff --git a/providers/vercel/models/spacexai/grok-4.6.toml b/providers/vercel/models/spacexai/grok-4.6.toml index 713ac05e4b3..58cede40b86 100644 --- a/providers/vercel/models/spacexai/grok-4.6.toml +++ b/providers/vercel/models/spacexai/grok-4.6.toml @@ -1,5 +1,6 @@ base_model = "xai/grok-4.6" description = "Grok model for agentic tool use, reasoning, coding, and live assistance" + [[reasoning_options]] type = "effort" values = ["low", "medium", "high"] @@ -8,3 +9,9 @@ values = ["low", "medium", "high"] input = 2 output = 6 cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 4 +output = 12 +cache_read = 1 diff --git a/providers/vercel/models/spacexai/grok-4.7.toml b/providers/vercel/models/spacexai/grok-4.7.toml new file mode 100644 index 00000000000..95c1c2ce193 --- /dev/null +++ b/providers/vercel/models/spacexai/grok-4.7.toml @@ -0,0 +1,19 @@ +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1.2 +output = 3.6 +cache_read = 0.3 + +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2.4 +output = 7.2 +cache_read = 0.6 + +[modalities] +input = ["text", "image"] diff --git a/providers/vercel/models/spacexai/grok-build-0.1.toml b/providers/vercel/models/spacexai/grok-build-0.1.toml index dfcdb207ca7..ec232622716 100644 --- a/providers/vercel/models/spacexai/grok-build-0.1.toml +++ b/providers/vercel/models/spacexai/grok-build-0.1.toml @@ -7,5 +7,11 @@ input = 1 output = 2 cache_read = 0.2 +[[cost.tiers]] +tier = { type = "context", size = 200_001 } +input = 2 +output = 4 +cache_read = 0.4 + [modalities] input = ["text", "image"] diff --git a/providers/vercel/models/stepfun/step-5-preview.toml b/providers/vercel/models/stepfun/step-5-preview.toml new file mode 100644 index 00000000000..ce090c9a924 --- /dev/null +++ b/providers/vercel/models/stepfun/step-5-preview.toml @@ -0,0 +1,12 @@ +base_model = "stepfun/step-5-preview" +description = "StepFun flash model for efficient multimodal reasoning, coding, and tool use" +reasoning = false +tool_call = false + +[cost] +input = 1 +output = 2.7 +cache_read = 0.05 + +[modalities] +input = ["text", "image"] diff --git a/providers/vercel/models/typesafe-ai/jev.toml b/providers/vercel/models/typesafe-ai/jev.toml index e3c1d0652ad..f18f60fb34b 100644 --- a/providers/vercel/models/typesafe-ai/jev.toml +++ b/providers/vercel/models/typesafe-ai/jev.toml @@ -3,3 +3,6 @@ base_model = "typesafe/jev-latest" [cost] input = 0.042 output = 0 + +[limit] +context = 32_000 diff --git a/providers/vercel/models/xiaomi/mimo-v2.6-flash.toml b/providers/vercel/models/xiaomi/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..bb5496a5085 --- /dev/null +++ b/providers/vercel/models/xiaomi/mimo-v2.6-flash.toml @@ -0,0 +1,13 @@ +# Effort: reasoning.effort = none|minimal|low|medium|high|xhigh|max +# https://mimo.mi.com/docs/en-US/api/chat/responses +base_model = "xiaomi/mimo-v2.6-flash" +name = "MiMo V2.6 Flash" +reasoning_options = [{ type = "effort", values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[modalities] +input = ["text", "image"] diff --git a/providers/vercel/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml b/providers/vercel/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..dc1eb9fa8fa --- /dev/null +++ b/providers/vercel/models/xiaomi/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,13 @@ +# Effort: reasoning.effort = none|minimal|low|medium|high|xhigh|max +# https://mimo.mi.com/docs/en-US/api/chat/responses +base_model = "xiaomi/mimo-v2.6-pro-ultraspeed" +name = "MiMo V2.6 Pro UltraSpeed" +reasoning_options = [{ type = "effort", values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 4.35 +output = 8.7 +cache_read = 0.036 + +[modalities] +input = ["text", "image"] diff --git a/providers/vercel/models/xiaomi/mimo-v2.6-pro.toml b/providers/vercel/models/xiaomi/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..11f32cb1976 --- /dev/null +++ b/providers/vercel/models/xiaomi/mimo-v2.6-pro.toml @@ -0,0 +1,13 @@ +# Effort: reasoning.effort = none|minimal|low|medium|high|xhigh|max +# https://mimo.mi.com/docs/en-US/api/chat/responses +base_model = "xiaomi/mimo-v2.6-pro" +name = "MiMo V2.6 Pro" +reasoning_options = [{ type = "effort", values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 + +[modalities] +input = ["text", "image"] diff --git a/providers/vercel/models/zai/glm-5.3-flashx.toml b/providers/vercel/models/zai/glm-5.3-flashx.toml new file mode 100644 index 00000000000..c7fc6f5e326 --- /dev/null +++ b/providers/vercel/models/zai/glm-5.3-flashx.toml @@ -0,0 +1,16 @@ +# GLM-5.3-FlashX uses the same model and forced-thinking effort levels as Flash. +# https://docs.z.ai/guides/vlm/glm-5.3-flash +base_model = "zhipuai/glm-5.3-flash" +name = "GLM 5.3 FlashX" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.37 +output = 1.25 +cache_read = 0.075 + +[modalities] +input = ["text", "image"] diff --git a/providers/vivgrid/models/claude-fable-5-1.toml b/providers/vivgrid/models/claude-fable-5-1.toml index 471e3cb571d..756c74d20a3 100644 --- a/providers/vivgrid/models/claude-fable-5-1.toml +++ b/providers/vivgrid/models/claude-fable-5-1.toml @@ -1,7 +1,6 @@ # Pricing, context window and max output: # https://docs.vivgrid.com/models/claude-fable-5-1 (accessed 2026-09-07) -# Cached input is $0.50/MTok. Vivgrid publishes no cache-write rate and states -# pricing matches the original provider, so Anthropic's $12.50/MTok applies. +# Effort: output_config.effort = low|medium|high|xhigh|max base_model = "anthropic/claude-fable-5-1" structured_output = true reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] @@ -13,4 +12,4 @@ cache_read = 0.5 cache_write = 12.5 [provider] -npm = "@ai-sdk/openai-compatible" \ No newline at end of file +npm = "@ai-sdk/anthropic" diff --git a/providers/vivgrid/models/claude-fable-5.toml b/providers/vivgrid/models/claude-fable-5.toml index 912e2ff029a..31f1362efa1 100644 --- a/providers/vivgrid/models/claude-fable-5.toml +++ b/providers/vivgrid/models/claude-fable-5.toml @@ -1,7 +1,6 @@ # Pricing, context window and max output: # https://docs.vivgrid.com/models/claude-fable-5 (accessed 2026-09-07) -# Cached input is $1.25/MTok. Vivgrid publishes no cache-write rate and states -# pricing matches the original provider, so Anthropic's $12.50/MTok applies. +# Effort: output_config.effort = low|medium|high|xhigh|max base_model = "anthropic/claude-fable-5" structured_output = true reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] @@ -13,4 +12,4 @@ cache_read = 1.25 cache_write = 12.5 [provider] -npm = "@ai-sdk/openai-compatible" \ No newline at end of file +npm = "@ai-sdk/anthropic" diff --git a/providers/vivgrid/models/claude-opus-5.toml b/providers/vivgrid/models/claude-opus-5.toml new file mode 100644 index 00000000000..fbc0e693efc --- /dev/null +++ b/providers/vivgrid/models/claude-opus-5.toml @@ -0,0 +1,14 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/claude-opus-5 (accessed 2026-09-07) +base_model = "anthropic/claude-opus-5" +structured_output = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + +[cost] +input = 5.0 +output = 25.0 +cache_read = 0.50 +cache_write = 6.25 + +[provider] +npm = "@ai-sdk/anthropic" diff --git a/providers/vivgrid/models/claude-sonnet-5.toml b/providers/vivgrid/models/claude-sonnet-5.toml new file mode 100644 index 00000000000..44da13b30e5 --- /dev/null +++ b/providers/vivgrid/models/claude-sonnet-5.toml @@ -0,0 +1,16 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/claude-sonnet-5 (accessed 2026-09-07) +# Toggle: thinking.type = adaptive|disabled +# Effort: output_config.effort = low|medium|high|xhigh|max +base_model = "anthropic/claude-sonnet-5" +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +structured_output = true + +[cost] +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[provider] +npm = "@ai-sdk/anthropic" \ No newline at end of file diff --git a/providers/vivgrid/models/deepseek-v4-flash.toml b/providers/vivgrid/models/deepseek-v4-flash.toml index 38f0598e8fe..464e7b70d2d 100644 --- a/providers/vivgrid/models/deepseek-v4-flash.toml +++ b/providers/vivgrid/models/deepseek-v4-flash.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/deepseek-v4-flash (accessed 2026-09-18) base_model = "deepseek/deepseek-v4-flash-0731" name = "DeepSeek V4 Flash" diff --git a/providers/vivgrid/models/deepseek-v4.1-flash.toml b/providers/vivgrid/models/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..276f6360406 --- /dev/null +++ b/providers/vivgrid/models/deepseek-v4.1-flash.toml @@ -0,0 +1,20 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/deepseek-v4.1-flash (accessed 2026-09-21) +base_model = "deepseek/deepseek-v4.1-flash" + +# Toggle: thinking.type = enabled|disabled +# Effort: reasoning_effort = low|high|max +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.31 +output = 1.23 +cache_read = 0.01 diff --git a/providers/vivgrid/models/glm-5.2.toml b/providers/vivgrid/models/glm-5.2.toml index ad841133eeb..2ba2cd2bf76 100644 --- a/providers/vivgrid/models/glm-5.2.toml +++ b/providers/vivgrid/models/glm-5.2.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/glm-5.2 (accessed 2026-09-18) base_model = "zhipuai/glm-5.2" [[reasoning_options]] diff --git a/providers/vivgrid/models/glm-5.3.toml b/providers/vivgrid/models/glm-5.3.toml index 704b2199e03..da960ec71fa 100644 --- a/providers/vivgrid/models/glm-5.3.toml +++ b/providers/vivgrid/models/glm-5.3.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/glm-5.3 (accessed 2026-09-18) base_model = "zhipuai/glm-5.3" reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] diff --git a/providers/vivgrid/models/gpt-5.1-codex-max.toml b/providers/vivgrid/models/gpt-5.1-codex-max.toml index b96035a695e..68ec50cbee8 100644 --- a/providers/vivgrid/models/gpt-5.1-codex-max.toml +++ b/providers/vivgrid/models/gpt-5.1-codex-max.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.1-codex-max (accessed 2026-09-18) name = "GPT-5.1 Codex Max" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" family = "gpt-codex" diff --git a/providers/vivgrid/models/gpt-5.1-codex.toml b/providers/vivgrid/models/gpt-5.1-codex.toml index c8b1158f20c..f6e6f2fa8ee 100644 --- a/providers/vivgrid/models/gpt-5.1-codex.toml +++ b/providers/vivgrid/models/gpt-5.1-codex.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.1-codex (accessed 2026-09-18) name = "GPT-5.1 Codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" family = "gpt-codex" diff --git a/providers/vivgrid/models/gpt-5.2-codex.toml b/providers/vivgrid/models/gpt-5.2-codex.toml index e28e266ee10..7319407808d 100644 --- a/providers/vivgrid/models/gpt-5.2-codex.toml +++ b/providers/vivgrid/models/gpt-5.2-codex.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.2-codex (accessed 2026-09-18) name = "GPT-5.2 Codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" family = "gpt-codex" diff --git a/providers/vivgrid/models/gpt-5.3-codex.toml b/providers/vivgrid/models/gpt-5.3-codex.toml index 715972403fe..454bc79f662 100644 --- a/providers/vivgrid/models/gpt-5.3-codex.toml +++ b/providers/vivgrid/models/gpt-5.3-codex.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.3-codex (accessed 2026-09-18) name = "GPT-5.3 Codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" family = "gpt-codex" diff --git a/providers/vivgrid/models/gpt-5.6-luna.toml b/providers/vivgrid/models/gpt-5.6-luna.toml index fb81c98d331..9c4852173d2 100644 --- a/providers/vivgrid/models/gpt-5.6-luna.toml +++ b/providers/vivgrid/models/gpt-5.6-luna.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.6-luna (accessed 2026-09-18) base_model = "openai/gpt-5.6-luna" name = "GPT 5.6 Luna" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" diff --git a/providers/vivgrid/models/gpt-5.6-sol.toml b/providers/vivgrid/models/gpt-5.6-sol.toml index df3250cf9b6..65eb349ce72 100644 --- a/providers/vivgrid/models/gpt-5.6-sol.toml +++ b/providers/vivgrid/models/gpt-5.6-sol.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.6-sol (accessed 2026-09-18) base_model = "openai/gpt-5.6-sol" name = "GPT 5.6 Sol" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" diff --git a/providers/vivgrid/models/gpt-5.6-terra.toml b/providers/vivgrid/models/gpt-5.6-terra.toml index 385029c25bd..371995729d7 100644 --- a/providers/vivgrid/models/gpt-5.6-terra.toml +++ b/providers/vivgrid/models/gpt-5.6-terra.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-5.6-terra (accessed 2026-09-18) base_model = "openai/gpt-5.6-terra" name = "GPT 5.6 Terra" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" diff --git a/providers/vivgrid/models/gpt-6-astra.toml b/providers/vivgrid/models/gpt-6-astra.toml index afdc38e7f28..2a246582bfd 100644 --- a/providers/vivgrid/models/gpt-6-astra.toml +++ b/providers/vivgrid/models/gpt-6-astra.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/gpt-6-astra (accessed 2026-09-18) base_model = "openai/gpt-6-astra" reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] diff --git a/providers/vivgrid/models/jev.toml b/providers/vivgrid/models/jev.toml new file mode 100644 index 00000000000..f499745e785 --- /dev/null +++ b/providers/vivgrid/models/jev.toml @@ -0,0 +1,7 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/jev +base_model = "typesafe/jev-latest" + +[cost] +input = 0.042 +output = 0 diff --git a/providers/vivgrid/models/kimi-k3.toml b/providers/vivgrid/models/kimi-k3.toml index 7e11b4c39c2..1ca4fad3636 100644 --- a/providers/vivgrid/models/kimi-k3.toml +++ b/providers/vivgrid/models/kimi-k3.toml @@ -1,3 +1,5 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/kimi-k3 (accessed 2026-09-18) base_model = "moonshotai/kimi-k3" description = "Kimi multimodal agent model for visual understanding, coding, and planning" reasoning_options = [] diff --git a/providers/vivgrid/models/viv-fast.toml b/providers/vivgrid/models/viv-fast.toml new file mode 100644 index 00000000000..354c69bf873 --- /dev/null +++ b/providers/vivgrid/models/viv-fast.toml @@ -0,0 +1,10 @@ +# Pricing, context window and max output: +# https://docs.vivgrid.com/models/viv-fast (accessed 2026-09-21) +base_model = "vivgrid/viv-fast" +# Effort: reasoning_effort = low|high|max +reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] + +[cost] +input = 0.13 +output = 0.40 +cache_read = 0.05 diff --git a/providers/volcengine-coding-plan/models/kimi-k3.toml b/providers/volcengine-coding-plan/models/kimi-k3.toml index 18f097450bd..7b7fb0c9e7e 100644 --- a/providers/volcengine-coding-plan/models/kimi-k3.toml +++ b/providers/volcengine-coding-plan/models/kimi-k3.toml @@ -2,7 +2,7 @@ # (accessed 2026-09-11), listed alongside kimi-k2.7-code. Subscription tier: # no per-token price. # reasoning_options mirror the lab (providers/moonshotai/models/kimi-k3.toml) -# and the kimi-for-coding relay peer. Wire fields on this provider's OpenAI +# and the kimi-code-plan-cn relay peer. Wire fields on this provider's OpenAI # path (provider.toml): POST /api/coding/v3/chat/completions takes # thinking.type = enabled|disabled plus reasoning_effort = low|high|max # (`adaptive` is the Moonshot lab surface, not accepted here); the separate diff --git a/providers/wandb/models/deepseek-ai/DeepSeek-V4.1-Flash.toml b/providers/wandb/models/deepseek-ai/DeepSeek-V4.1-Flash.toml new file mode 100644 index 00000000000..0c25c1c3eec --- /dev/null +++ b/providers/wandb/models/deepseek-ai/DeepSeek-V4.1-Flash.toml @@ -0,0 +1,15 @@ +base_model = "deepseek/deepseek-v4.1-flash" +description = "DeepSeek V4.1 Flash is a multimodal MoE model for coding, reasoning, and agentic workloads with long contexts." +family = "deepseek" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.2 +output = 0.65 +cache_read = 0.03 + +[limit] +context = 1_048_576 +output = 1_048_576 diff --git a/providers/wandb/models/google/gemma-4-26B-A4B-it.toml b/providers/wandb/models/google/gemma-4-26B-A4B-it.toml new file mode 100644 index 00000000000..8014a72bac7 --- /dev/null +++ b/providers/wandb/models/google/gemma-4-26B-A4B-it.toml @@ -0,0 +1,14 @@ +base_model = "google/gemma-4-26b-a4b-it" +name = "Gemma 4 26B A4B" +description = "Gemma 4 26B A4B is a multimodal MoE model with LoRA support and function calling for agentic workflows." + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.1 +output = 0.3 +cache_read = 0.05 + +[limit] +output = 262_144 diff --git a/providers/xai/models/grok-4.7.toml b/providers/xai/models/grok-4.7.toml new file mode 100644 index 00000000000..c205cfd8209 --- /dev/null +++ b/providers/xai/models/grok-4.7.toml @@ -0,0 +1,18 @@ +# Sources: https://docs.x.ai/developers/models/grok-4.7, https://docs.x.ai/developers/pricing, and https://docs.x.ai/developers/model-capabilities/text/reasoning +base_model = "xai/grok-4.7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 + diff --git a/providers/xai/models/grok-imagine-image-quality.toml b/providers/xai/models/grok-imagine-image-quality.toml new file mode 100644 index 00000000000..1be5068f356 --- /dev/null +++ b/providers/xai/models/grok-imagine-image-quality.toml @@ -0,0 +1,11 @@ +# Sources: +# - https://docs.x.ai/docs/models +# - https://docs.x.ai/developers/models/grok-imagine-image-quality +# - https://docs.x.ai/developers/pricing +# Pricing: $0.05/image (1K), $0.07/image (2K); image input $0.01/image (not token-based; no [cost] authored) +# Aliases: grok-imagine-image-quality-20260403, grok-imagine-image-quality-latest, grok-imagine-image-pro +base_model = "xai/grok-imagine-image-quality" + +[modalities] +input = ["text", "image", "pdf"] +output = ["image", "pdf"] diff --git a/providers/xiaomi-token-plan-ams/models/mimo-v2.6-flash.toml b/providers/xiaomi-token-plan-ams/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..4129697bc51 --- /dev/null +++ b/providers/xiaomi-token-plan-ams/models/mimo-v2.6-flash.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/en-US/tokenplan/Token%20Plan/quick-access +base_model = "xiaomi/mimo-v2.6-flash" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi-token-plan-ams/models/mimo-v2.6-pro.toml b/providers/xiaomi-token-plan-ams/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..62ef14092f6 --- /dev/null +++ b/providers/xiaomi-token-plan-ams/models/mimo-v2.6-pro.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/en-US/tokenplan/Token%20Plan/quick-access +base_model = "xiaomi/mimo-v2.6-pro" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi-token-plan-cn/models/mimo-v2.6-flash.toml b/providers/xiaomi-token-plan-cn/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..091c178cc51 --- /dev/null +++ b/providers/xiaomi-token-plan-cn/models/mimo-v2.6-flash.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/zh-CN/api/chat/openai-api +base_model = "xiaomi/mimo-v2.6-flash" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi-token-plan-cn/models/mimo-v2.6-pro.toml b/providers/xiaomi-token-plan-cn/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..27c7d777207 --- /dev/null +++ b/providers/xiaomi-token-plan-cn/models/mimo-v2.6-pro.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/zh-CN/api/chat/openai-api +base_model = "xiaomi/mimo-v2.6-pro" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-flash.toml b/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..4129697bc51 --- /dev/null +++ b/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-flash.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/en-US/tokenplan/Token%20Plan/quick-access +base_model = "xiaomi/mimo-v2.6-flash" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-pro.toml b/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..62ef14092f6 --- /dev/null +++ b/providers/xiaomi-token-plan-sgp/models/mimo-v2.6-pro.toml @@ -0,0 +1,12 @@ +# Toggle: thinking.type = enabled|disabled +# Source: https://mimo.mi.com/docs/en-US/tokenplan/Token%20Plan/quick-access +base_model = "xiaomi/mimo-v2.6-pro" +reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0 +output = 0 +cache_read = 0 diff --git a/providers/xiaomi/models/mimo-v2.6-flash.toml b/providers/xiaomi/models/mimo-v2.6-flash.toml new file mode 100644 index 00000000000..8a98ed174da --- /dev/null +++ b/providers/xiaomi/models/mimo-v2.6-flash.toml @@ -0,0 +1,31 @@ +# Toggle: thinking.type = enabled|disabled +# Sources: https://mimo.mi.com/models/zh-CN/mimo-v2.6-flash and https://mimo.mi.com/docs/zh-CN/api/chat/openai-api +name = "MiMo-V2.6-Flash" +description = "MiMo Flash model for multimodal coding agents and long-context automation" +family = "mimo" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.14 +output = 0.28 +cache_read = 0.0028 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "audio", "video"] +output = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml b/providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml new file mode 100644 index 00000000000..400e80cc7c5 --- /dev/null +++ b/providers/xiaomi/models/mimo-v2.6-pro-ultraspeed.toml @@ -0,0 +1,31 @@ +# Toggle: thinking.type = enabled|disabled +# Sources: https://mimo.mi.com/models/zh-CN/mimo-v2.6-pro-ultraspeed and https://mimo.mi.com/docs/zh-CN/api/chat/openai-api +name = "MiMo-V2.6-Pro-UltraSpeed" +description = "MiMo pro model for strong multimodal reasoning and agent execution" +family = "mimo" +release_date = "2026-09-21" +last_updated = "2026-09-21" +attachment = true +reasoning = true +temperature = true +tool_call = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 4.35 +output = 8.7 +cache_read = 0.036 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "audio", "video"] +output = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/xiaomi/models/mimo-v2.6-pro.toml b/providers/xiaomi/models/mimo-v2.6-pro.toml new file mode 100644 index 00000000000..d3b71b59d4c --- /dev/null +++ b/providers/xiaomi/models/mimo-v2.6-pro.toml @@ -0,0 +1,31 @@ +# Toggle: thinking.type = enabled|disabled +# Sources: https://mimo.mi.com/models/zh-CN/mimo-v2.6-pro and https://mimo.mi.com/docs/zh-CN/api/chat/openai-api +name = "MiMo-V2.6-Pro" +description = "MiMo Pro model for multimodal coding agents and long-context automation" +family = "mimo" +release_date = "2026-09-22" +last_updated = "2026-09-22" +attachment = true +reasoning = true +temperature = true +tool_call = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.435 +output = 0.87 +cache_read = 0.0036 + +[limit] +context = 1_048_576 +output = 131_072 + +[modalities] +input = ["text", "image", "audio", "video"] +output = ["text"] + +[interleaved] +field = "reasoning_content" diff --git a/providers/zai/models/glm-4.6v-flash.toml b/providers/zai/models/glm-4.6v-flash.toml new file mode 100644 index 00000000000..d9793eab5e8 --- /dev/null +++ b/providers/zai/models/glm-4.6v-flash.toml @@ -0,0 +1,14 @@ +base_model = "zhipuai/glm-4.6v-flash" +# Free Flash tier on Z.AI; cost shape mirrors providers/zai/models/glm-4.5-flash.toml. +# Lab metadata (limits/modalities) lives in models/zhipuai/glm-4.6v-flash.toml. +# https://z.ai/blog/glm-4.6v +# https://huggingface.co/zai-org/GLM-4.6V-Flash + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 +cache_read = 0 +cache_write = 0 diff --git a/providers/zai/models/glm-5.3-flash.toml b/providers/zai/models/glm-5.3-flash.toml index 59029d6336a..5850a4727dc 100644 --- a/providers/zai/models/glm-5.3-flash.toml +++ b/providers/zai/models/glm-5.3-flash.toml @@ -2,6 +2,7 @@ base_model = "zhipuai/glm-5.3-flash" # GLM-5.3-Flash reasons by default; effort levels low|high|max, and thinking can # be suppressed with {"thinking": {"type": "disabled"}}. # https://z.ai/blog/glm-5.3-flash +# Pricing: https://docs.z.ai/guides/overview/pricing (accessed 2026-09-19) # Served by https://api.z.ai/api/paas/v4 (verified against GET /models 2026-08-26). [[reasoning_options]] @@ -12,7 +13,7 @@ values = ["low", "high", "max"] field = "reasoning_content" [cost] -input = 0.075 -output = 0.25 -cache_read = 0.015 +input = 0.15 +output = 0.50 +cache_read = 0.03 cache_write = 0 diff --git a/providers/zai/models/glm-5.3-flashx.toml b/providers/zai/models/glm-5.3-flashx.toml new file mode 100644 index 00000000000..059cb836933 --- /dev/null +++ b/providers/zai/models/glm-5.3-flashx.toml @@ -0,0 +1,21 @@ +# GLM-5.3-FlashX is the high-speed serving option for GLM-5.3-Flash. +# https://docs.z.ai/guides/vlm/glm-5.3-flash +# Pricing: https://docs.z.ai/guides/overview/pricing (accessed 2026-09-19) +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3-FlashX" +description = "High-speed GLM-5.3-Flash serving option for coding and agent workflows" +release_date = "2026-09-18" +last_updated = "2026-09-18" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.37 +output = 1.25 +cache_read = 0.075 +cache_write = 0 diff --git a/providers/zenmux/models/x-ai/grok-4-fast.toml b/providers/zenmux/models/x-ai/grok-4-fast.toml deleted file mode 100644 index f722ac7ac71..00000000000 --- a/providers/zenmux/models/x-ai/grok-4-fast.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Grok 4 Fast" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2025-09-19" -last_updated = "2025-09-19" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.20 -output = 0.50 -cache_read = 0.05 - -[limit] -context = 2000_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/zenmux/models/x-ai/grok-4.1-fast-non-reasoning.toml b/providers/zenmux/models/x-ai/grok-4.1-fast-non-reasoning.toml deleted file mode 100644 index 997935a9745..00000000000 --- a/providers/zenmux/models/x-ai/grok-4.1-fast-non-reasoning.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Grok 4.1 Fast Non Reasoning" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2025-11-20" -last_updated = "2025-11-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.20 -output = 0.50 -cache_read = 0.05 - -[limit] -context = 2000_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/zenmux/models/x-ai/grok-4.1-fast.toml b/providers/zenmux/models/x-ai/grok-4.1-fast.toml deleted file mode 100644 index 6ddced52aef..00000000000 --- a/providers/zenmux/models/x-ai/grok-4.1-fast.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Grok 4.1 Fast" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2025-11-20" -last_updated = "2025-11-20" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.20 -output = 0.50 -cache_read = 0.05 - -[limit] -context = 2000_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/zenmux/models/x-ai/grok-4.2-fast-non-reasoning.toml b/providers/zenmux/models/x-ai/grok-4.2-fast-non-reasoning.toml index 5fd0f203453..b521181860f 100644 --- a/providers/zenmux/models/x-ai/grok-4.2-fast-non-reasoning.toml +++ b/providers/zenmux/models/x-ai/grok-4.2-fast-non-reasoning.toml @@ -1,22 +1,20 @@ +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "xai/grok-4.20-0309-non-reasoning" name = "Grok 4.2 Fast Non Reasoning" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false [cost] -input = 3.00 -output = 9.00 +input = 2 +output = 6 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 4 +output = 12 +cache_read = 0.2 [limit] context = 2_000_000 -output = 30_000 [modalities] -input = ["text", "image", "video"] -output = ["text"] +input = ["text", "image"] diff --git a/providers/zenmux/models/x-ai/grok-4.2-fast.toml b/providers/zenmux/models/x-ai/grok-4.2-fast.toml index c768fe47135..fc524864b65 100644 --- a/providers/zenmux/models/x-ai/grok-4.2-fast.toml +++ b/providers/zenmux/models/x-ai/grok-4.2-fast.toml @@ -1,23 +1,25 @@ +# ZenMux's 2026-09-19 page has supports_reasoning=0 and both live OpenAI- and +# Anthropic-compatible entries have capabilities.reasoning=false for this id. +# This route therefore uses the reasoning checkpoint identity but disables the +# reasoning surface as an explicit host delta. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "xai/grok-4.20-0309-reasoning" name = "Grok 4.2 Fast" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false +reasoning = false [cost] -input = 3.00 -output = 9.00 +input = 2 +output = 6 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 4 +output = 12 +cache_read = 0.2 [limit] context = 2_000_000 -output = 30_000 [modalities] -input = ["text", "image", "video"] -output = ["text"] +input = ["text", "image"] diff --git a/providers/zenmux/models/x-ai/grok-4.3.toml b/providers/zenmux/models/x-ai/grok-4.3.toml index 455779665fb..ab8ed82e9a7 100644 --- a/providers/zenmux/models/x-ai/grok-4.3.toml +++ b/providers/zenmux/models/x-ai/grok-4.3.toml @@ -1,19 +1,22 @@ +# Effort: reasoning_effort = none|low|medium|high +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "xai/grok-4.3" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] [cost] input = 1.25 -output = 2.50 -cache_read = 0.20 -cache_write = 0 +output = 2.5 +cache_read = 0.2 [[cost.tiers]] -tier = { size = 200_000 } -input = 2.50 -output = 5.00 -cache_read = 0.40 -cache_write = 0 +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 5 +cache_read = 0.4 [limit] -context = 1_000_000 output = 1_000_000 diff --git a/providers/zenmux/models/x-ai/grok-4.6.toml b/providers/zenmux/models/x-ai/grok-4.6.toml new file mode 100644 index 00000000000..7b40b61706e --- /dev/null +++ b/providers/zenmux/models/x-ai/grok-4.6.toml @@ -0,0 +1,19 @@ +# Effort: reasoning_effort = low|medium|high|xhigh +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "xai/grok-4.6" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 12 +cache_read = 1 diff --git a/providers/zenmux/models/x-ai/grok-4.toml b/providers/zenmux/models/x-ai/grok-4.toml deleted file mode 100644 index 9a3bd9acea0..00000000000 --- a/providers/zenmux/models/x-ai/grok-4.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Grok 4" -description = "Grok model for agentic tool use, reasoning, coding, and live assistance" -release_date = "2025-07-09" -last_updated = "2025-07-09" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 3.00 -output = 15.00 -cache_read = 0.75 - -[limit] -context = 256_000 -output = 64_000 - -[modalities] -input = ["image", "text"] -output = ["text"] diff --git a/providers/zenmux/models/x-ai/grok-build-0.1.toml b/providers/zenmux/models/x-ai/grok-build-0.1.toml index 59480b618c9..f00530ac65f 100644 --- a/providers/zenmux/models/x-ai/grok-build-0.1.toml +++ b/providers/zenmux/models/x-ai/grok-build-0.1.toml @@ -1,7 +1,13 @@ +# ZenMux's 2026-09-19 page has supports_reasoning=0 and both live OpenAI- and +# Anthropic-compatible entries have capabilities.reasoning=false for this id. +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "xai/grok-build-0.1" -reasoning_options = [] +reasoning = false [cost] -input = 1.00 -output = 2.00 -cache_read = 0.20 +input = 1 +output = 2 +cache_read = 0.2 + +[modalities] +input = ["text", "image"] diff --git a/providers/zenmux/models/x-ai/grok-code-fast-1.toml b/providers/zenmux/models/x-ai/grok-code-fast-1.toml deleted file mode 100644 index 5a692a5e5d9..00000000000 --- a/providers/zenmux/models/x-ai/grok-code-fast-1.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Grok Code Fast 1" -description = "Fast Grok model for responsive chat, reasoning, and tool-assisted work" -release_date = "2025-08-26" -last_updated = "2025-08-26" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.20 -output = 1.50 -cache_read = 0.02 - -[limit] -context = 256_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/x-ai/grok-imagine-image-2.0.toml b/providers/zenmux/models/x-ai/grok-imagine-image-2.0.toml new file mode 100644 index 00000000000..05abc6d0374 --- /dev/null +++ b/providers/zenmux/models/x-ai/grok-imagine-image-2.0.toml @@ -0,0 +1,7 @@ +# Host context: page and OpenAI list both report 66000 tokens. +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "xai/grok-imagine-image-2.0" + +[limit] +context = 66_000 diff --git a/providers/zenmux/models/x-ai/grok-voice-stt-1.0.toml b/providers/zenmux/models/x-ai/grok-voice-stt-1.0.toml new file mode 100644 index 00000000000..c4bf10390e9 --- /dev/null +++ b/providers/zenmux/models/x-ai/grok-voice-stt-1.0.toml @@ -0,0 +1,3 @@ +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "xai/grok-voice-stt-1.0" diff --git a/providers/zenmux/models/x-ai/grok-voice-tts-1.0.toml b/providers/zenmux/models/x-ai/grok-voice-tts-1.0.toml new file mode 100644 index 00000000000..27f8f70b699 --- /dev/null +++ b/providers/zenmux/models/x-ai/grok-voice-tts-1.0.toml @@ -0,0 +1,3 @@ +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "xai/grok-voice-tts-1.0" diff --git a/providers/zenmux/models/z-ai/glm-4.5-air.toml b/providers/zenmux/models/z-ai/glm-4.5-air.toml index 756ca62b321..3797e830f0e 100644 --- a/providers/zenmux/models/z-ai/glm-4.5-air.toml +++ b/providers/zenmux/models/z-ai/glm-4.5-air.toml @@ -1,24 +1,23 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.5-air" name = "GLM 4.5 Air" -description = "Efficient GLM model for fast reasoning, coding, and agent workflows" -release_date = "2025-07-25" -last_updated = "2025-07-25" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.11 -output = 0.56 -cache_read = 0.02 +input = 0.1165 +output = 0.2911 +cache_read = 0.0233 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.1747 +output = 1.1645 +cache_read = 0.0349 [limit] context = 128_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 96_000 diff --git a/providers/zenmux/models/z-ai/glm-4.5.toml b/providers/zenmux/models/z-ai/glm-4.5.toml index b2f7197b924..fffd19cacc1 100644 --- a/providers/zenmux/models/z-ai/glm-4.5.toml +++ b/providers/zenmux/models/z-ai/glm-4.5.toml @@ -1,24 +1,24 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.5" name = "GLM 4.5" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -release_date = "2025-07-25" -last_updated = "2025-07-25" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.35 -output = 1.54 -cache_read = 0.07 +input = 0.2911 +output = 1.1645 +cache_read = 0.0582 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.5823 +output = 2.3291 +cache_read = 0.1165 [limit] context = 128_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 96_000 diff --git a/providers/zenmux/models/z-ai/glm-4.6.toml b/providers/zenmux/models/z-ai/glm-4.6.toml index e5f9d2ca1a4..d4d4456a446 100644 --- a/providers/zenmux/models/z-ai/glm-4.6.toml +++ b/providers/zenmux/models/z-ai/glm-4.6.toml @@ -1,24 +1,24 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.6" name = "GLM 4.6" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -release_date = "2025-09-30" -last_updated = "2025-09-30" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.35 -output = 1.54 -cache_read = 0.07 +input = 0.2911 +output = 1.1645 +cache_read = 0.0582 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.5823 +output = 2.3291 +cache_read = 0.1165 [limit] context = 200_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-4.6v-flash-free.toml b/providers/zenmux/models/z-ai/glm-4.6v-flash-free.toml index 33b4db29d44..18b34502274 100644 --- a/providers/zenmux/models/z-ai/glm-4.6v-flash-free.toml +++ b/providers/zenmux/models/z-ai/glm-4.6v-flash-free.toml @@ -1,23 +1,27 @@ +# Toggle: reasoning.enabled = true|false +# Host modalities: the live free route includes file/PDF input; the paid route omits it. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.6v-flash" name = "GLM 4.6V Flash (Free)" -description = "GLM vision model for visual reasoning, documents, and multimodal agents" -release_date = "2025-12-08" -last_updated = "2025-12-08" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.00 -output = 0.00 +input = 0 +output = 0 +cache_read = 0 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0 +output = 0 +cache_read = 0 [limit] context = 200_000 -output = 64_000 +output = 128_000 [modalities] -input = ["text", "image", "video"] -output = ["text"] +input = ["text", "image", "video", "pdf"] diff --git a/providers/zenmux/models/z-ai/glm-4.6v-flash.toml b/providers/zenmux/models/z-ai/glm-4.6v-flash.toml index 00a8fa34b67..6c53afd51b0 100644 --- a/providers/zenmux/models/z-ai/glm-4.6v-flash.toml +++ b/providers/zenmux/models/z-ai/glm-4.6v-flash.toml @@ -1,24 +1,24 @@ +# Toggle: reasoning.enabled = true|false +# Host naming: the 2026-09-19 live route z-ai/glm-4.6v-flash is labeled "GLM 4.6V FlashX". +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.6v-flash" name = "GLM 4.6V FlashX" -description = "GLM vision model for visual reasoning, documents, and multimodal agents" -release_date = "2025-12-08" -last_updated = "2025-12-08" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.02 -output = 0.21 -cache_read = 0.0043 +input = 0.0218 +output = 0.2184 +cache_read = 0.0044 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.0437 +output = 0.4367 +cache_read = 0.0044 [limit] context = 200_000 -output = 64_000 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-4.6v.toml b/providers/zenmux/models/z-ai/glm-4.6v.toml index 73690ef9e4f..e8c345954c8 100644 --- a/providers/zenmux/models/z-ai/glm-4.6v.toml +++ b/providers/zenmux/models/z-ai/glm-4.6v.toml @@ -1,24 +1,23 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.6v" name = "GLM 4.6V" -description = "GLM vision model for visual reasoning, documents, and multimodal agents" -release_date = "2025-12-08" -last_updated = "2025-12-08" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.14 -output = 0.42 -cache_read = 0.03 +input = 0.1456 +output = 0.4367 +cache_read = 0.0291 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.2911 +output = 0.8734 +cache_read = 0.0582 [limit] context = 200_000 -output = 64_000 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-4.7-flash-free.toml b/providers/zenmux/models/z-ai/glm-4.7-flash-free.toml index b2ddbce96ec..df0e92ac02a 100644 --- a/providers/zenmux/models/z-ai/glm-4.7-flash-free.toml +++ b/providers/zenmux/models/z-ai/glm-4.7-flash-free.toml @@ -1,26 +1,19 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.7-flash" name = "GLM 4.7 Flash (Free)" -description = "Efficient GLM model for fast reasoning, coding, and agent workflows" -release_date = "2026-01-19" -last_updated = "2026-01-19" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -[cost] -input = 0.00 -output = 0.00 - -[limit] -context = 200_000 -output = 64_000 [interleaved] field = "reasoning_content" -[modalities] -input = ["text"] -output = ["text"] +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 +cache_read = 0 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-4.7-flashx.toml b/providers/zenmux/models/z-ai/glm-4.7-flashx.toml index eba199ff720..b131a75aa2a 100644 --- a/providers/zenmux/models/z-ai/glm-4.7-flashx.toml +++ b/providers/zenmux/models/z-ai/glm-4.7-flashx.toml @@ -1,27 +1,19 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.7-flashx" name = "GLM 4.7 FlashX" -description = "Efficient GLM model for fast reasoning, coding, and agent workflows" -release_date = "2026-01-19" -last_updated = "2026-01-19" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -[cost] -input = 0.07 -output = 0.42 -cache_read = 0.01 - -[limit] -context = 200_000 -output = 64_000 [interleaved] field = "reasoning_content" -[modalities] -input = ["text"] -output = ["text"] +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.0728 +output = 0.4367 +cache_read = 0.0146 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-4.7.toml b/providers/zenmux/models/z-ai/glm-4.7.toml index 266005dda7e..c32d76030a6 100644 --- a/providers/zenmux/models/z-ai/glm-4.7.toml +++ b/providers/zenmux/models/z-ai/glm-4.7.toml @@ -1,27 +1,27 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-4.7" name = "GLM 4.7" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -release_date = "2025-12-23" -last_updated = "2025-12-23" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.28 -output = 1.14 -cache_read = 0.06 +input = 0.2911 +output = 1.1645 +cache_read = 0.0582 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.5823 +output = 2.3291 +cache_read = 0.1165 [limit] context = 200_000 -output = 64_000 - -[interleaved] -field = "reasoning_content" - -[modalities] -input = ["text"] -output = ["text"] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5-turbo.toml b/providers/zenmux/models/z-ai/glm-5-turbo.toml index ef6438e58ac..ecc0fbae95d 100644 --- a/providers/zenmux/models/z-ai/glm-5-turbo.toml +++ b/providers/zenmux/models/z-ai/glm-5-turbo.toml @@ -1,23 +1,25 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5-turbo" name = "GLM 5 Turbo" -description = "Efficient GLM model for fast reasoning, coding, and agent workflows" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.88 -output = 3.48 +input = 0.73 +output = 3.19 +cache_read = 0.174 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 1.02 +output = 3.77 +cache_read = 0.261 [limit] -context = 200_000 output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/z-ai/glm-5.1.toml b/providers/zenmux/models/z-ai/glm-5.1.toml index 3b4c0bec049..3a64eedea4a 100644 --- a/providers/zenmux/models/z-ai/glm-5.1.toml +++ b/providers/zenmux/models/z-ai/glm-5.1.toml @@ -1,27 +1,25 @@ -name = "GLM-5.1" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -release_date = "2026-04-03" -last_updated = "2026-04-03" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -open_weights = false +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5.1" +name = "GLM 5.1" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.8781 output = 3.5126 cache_read = 0.1903 -[limit] -context = 200000 -output = 131072 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 1.1709 +output = 4.098 +cache_read = 0.2927 -[modalities] -input = ["text"] -output = ["text"] +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5.2-free.toml b/providers/zenmux/models/z-ai/glm-5.2-free.toml deleted file mode 100644 index 561b079445e..00000000000 --- a/providers/zenmux/models/z-ai/glm-5.2-free.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "zhipuai/glm-5.2" -name = "GLM 5.2 (Free)" - -[cost] -input = 0 -output = 0 -cache_read = 0 - -[[reasoning_options]] -type = "effort" -values = ["high", "max"] diff --git a/providers/zenmux/models/z-ai/glm-5.2.toml b/providers/zenmux/models/z-ai/glm-5.2.toml index 7d3e1a5ebf4..09a38c80a35 100644 --- a/providers/zenmux/models/z-ai/glm-5.2.toml +++ b/providers/zenmux/models/z-ai/glm-5.2.toml @@ -1,11 +1,20 @@ +# Effort: reasoning_effort = high|max +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "zhipuai/glm-5.2" name = "GLM 5.2" -[cost] -input = 1.40 -output = 4.50 -cache_read = 0.26 +[interleaved] +field = "reasoning_content" [[reasoning_options]] type = "effort" values = ["high", "max"] + +[cost] +input = 0.98 +output = 3.08 +cache_read = 0.182 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5.3-flash.toml b/providers/zenmux/models/z-ai/glm-5.3-flash.toml new file mode 100644 index 00000000000..3bd77e65bd9 --- /dev/null +++ b/providers/zenmux/models/z-ai/glm-5.3-flash.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = low|high|max +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5.3-flash" +name = "GLM 5.3 Flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.03 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5.3-flashx.toml b/providers/zenmux/models/z-ai/glm-5.3-flashx.toml new file mode 100644 index 00000000000..50ad6eb5ea5 --- /dev/null +++ b/providers/zenmux/models/z-ai/glm-5.3-flashx.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = low|high|max +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5.3-flash" +name = "GLM 5.3 FlashX" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.375 +output = 1.25 +cache_read = 0.075 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5.3.toml b/providers/zenmux/models/z-ai/glm-5.3.toml new file mode 100644 index 00000000000..18b45ff89ad --- /dev/null +++ b/providers/zenmux/models/z-ai/glm-5.3.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = low|high|max +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5.3" +name = "GLM 5.3" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 1.4 +output = 4.4 +cache_read = 0.26 + +[limit] +output = 128_000 diff --git a/providers/zenmux/models/z-ai/glm-5.toml b/providers/zenmux/models/z-ai/glm-5.toml index 3be16bf629a..f2ee297f1d0 100644 --- a/providers/zenmux/models/z-ai/glm-5.toml +++ b/providers/zenmux/models/z-ai/glm-5.toml @@ -1,27 +1,27 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5" name = "GLM 5" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -release_date = "2026-02-12" -last_updated = "2026-02-12" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = true +structured_output = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.58 output = 2.6 cache_read = 0.14 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.87 +output = 3.18 +cache_read = 0.22 + [limit] context = 200_000 output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/z-ai/glm-5v-turbo.toml b/providers/zenmux/models/z-ai/glm-5v-turbo.toml index 00fc2f71901..0e75e57e194 100644 --- a/providers/zenmux/models/z-ai/glm-5v-turbo.toml +++ b/providers/zenmux/models/z-ai/glm-5v-turbo.toml @@ -1,26 +1,25 @@ +# Toggle: reasoning.enabled = true|false +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "zhipuai/glm-5v-turbo" name = "GLM 5V Turbo" -description = "GLM vision model for visual reasoning, documents, and multimodal agents" -release_date = "2026-04-01" -last_updated = "2026-04-01" -attachment = true -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.726 output = 3.1946 cache_read = 0.1743 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 1.0165 +output = 3.7754 +cache_read = 0.2614 + [limit] -context = 200_000 output = 128_000 - -[modalities] -input = ["text", "image", "video", "pdf"] -output = ["text"] diff --git a/providers/zenmux/models/z-ai/glm-image.toml b/providers/zenmux/models/z-ai/glm-image.toml new file mode 100644 index 00000000000..79c6282b394 --- /dev/null +++ b/providers/zenmux/models/z-ai/glm-image.toml @@ -0,0 +1,9 @@ +# ZenMux's 2026-09-19 page and live Google-compatible list expose text input only; +# image editing exists in the lab model but is unavailable on this hosted route. +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "zhipuai/glm-image" +attachment = false + +[modalities] +input = ["text"] diff --git a/providers/zhipuai/models/glm-5.3-flash.toml b/providers/zhipuai/models/glm-5.3-flash.toml index 84efd9a9cef..d2ff44820e7 100644 --- a/providers/zhipuai/models/glm-5.3-flash.toml +++ b/providers/zhipuai/models/glm-5.3-flash.toml @@ -1,8 +1,7 @@ # GLM-5.3-Flash always reasons (thinking cannot be disabled); text params # match GLM-5.3 — effort low|high|max with default max. # https://docs.bigmodel.cn/cn/guide/models/vlm/glm-5.3-flash -# Cost: https://docs.z.ai/guides/overview/pricing (accessed 2026-08-26) -# GLM-5.3-Flash 50% promo ends 2026-09-09 24:00 UTC+8; list $0.15/$0.03/$0.50. +# Pricing: https://docs.z.ai/guides/overview/pricing (accessed 2026-09-19) base_model = "zhipuai/glm-5.3-flash" [[reasoning_options]] @@ -13,7 +12,7 @@ values = ["low", "high", "max"] field = "reasoning_content" [cost] -input = 0.075 -output = 0.25 -cache_read = 0.015 +input = 0.15 +output = 0.50 +cache_read = 0.03 cache_write = 0 diff --git a/providers/zhipuai/models/glm-5.3-flashx.toml b/providers/zhipuai/models/glm-5.3-flashx.toml new file mode 100644 index 00000000000..851a6969e41 --- /dev/null +++ b/providers/zhipuai/models/glm-5.3-flashx.toml @@ -0,0 +1,21 @@ +# GLM-5.3-FlashX is the high-speed serving option for GLM-5.3-Flash. +# https://docs.bigmodel.cn/cn/guide/models/vlm/glm-5.3-flash +# Pricing: https://docs.z.ai/guides/overview/pricing (accessed 2026-09-19) +base_model = "zhipuai/glm-5.3-flash" +name = "GLM-5.3-FlashX" +description = "High-speed GLM-5.3-Flash serving option for coding and agent workflows" +release_date = "2026-09-18" +last_updated = "2026-09-18" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.37 +output = 1.25 +cache_read = 0.075 +cache_write = 0 diff --git a/sync.md b/sync.md index 01d484bcdeb..356da1af5a5 100644 --- a/sync.md +++ b/sync.md @@ -294,6 +294,7 @@ Nebul is implemented in `packages/core/src/sync/providers/nebul.ts`. - Whole-catalog faults fail closed: an empty response, or one where nothing matches the chat-model filter (`model_type: "llm"`, `mode: "chat"`), throws in `parseModels` before any file is written or deleted — mirroring the ingest-side guards. - Per-model robustness is the inverse: an existing entry survives transient null pricing, a missing context limit, or a served alias that no longer resolves to lab metadata, keeping its authored `base_model`/`cost`/`limit`; only brand-new models require a fully-priced, resolvable source entry. Embeddings and rerankers are filtered by mode/model_type; specialized document-OCR models (name-scoped) and entries the host flags via `display_tags` (`Guard Model`, `Content Safety`, `Private`, `Internal`) are out of catalog scope; entries naming a successor via `model_info.superseded_by_model_name` are excluded as superseded. All of these skip silently. - `reasoning_options` come from the endpoint's per-model `reasoning_efforts` list, Nebul's only documented reasoning control. A reasoner with neither advertised efforts nor authored controls fails sync for manual authoring rather than writing an empty control set. +- An authored `reasoning = false` is the escape hatch for a served ID whose lab model reasons but which this host runs with thinking disabled (catalog reports `supports_reasoning = false` and no `reasoning_efforts`). The override is kept, the missing-reasoning-options guard does not fire, and any authored `reasoning_options`/`interleaved` are dropped from the synced file. - `interleaved`, `status`, and other fields the endpoint does not expose are preserved from existing files; the reasoning trace channel (`reasoning_content` vs `message.reasoning` vs inline ``) is family-specific per Nebul's docs. - `BASE_MODEL_ALIASES` in the module bridges served IDs to models.dev's canonical metadata naming where they differ (e.g. `mistralai/Mistral-Medium-3.5-128B` → `mistral/mistral-medium-2604`). Canonical paths are provider-agnostic — often release-date slugs rather than the host's internal model name — so an alias does not mean the wrong model is referenced.