From ea97ff1c4be991a282dfc27a415e64d1bf3eace6 Mon Sep 17 00:00:00 2001 From: chenxue Date: Wed, 9 Sep 2026 11:14:59 +0800 Subject: [PATCH 01/20] fix(sync): write per-tier audio pricing `CostTier` extends `Cost`, so `input_audio` and `output_audio` are valid on a tier, but `formatToml` only emitted them for the top-level `[cost]` table. Any sync that rewrote a model with tiered audio rates silently dropped them. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/index.ts | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index 7f88aacb5bb..1f2596dd057 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -6,6 +6,7 @@ import { z } from "zod"; import { AuthoredModel, AuthoredModelShape, ModelMetadata } from "../schema.js"; import { openMissingModelIssues } from "./missing-issues.js"; import { MissingReasoningOptionsError } from "./missing-reasoning-options.js"; +import { aihubmix } from "./providers/aihubmix.js"; import { ambient } from "./providers/ambient.js"; import { anthropic } from "./providers/anthropic.js"; import { baseten } from "./providers/baseten.js"; @@ -130,6 +131,7 @@ export interface SyncResult { } export const providers: { + aihubmix: SyncProvider; ambient: SyncProvider; anthropic: SyncProvider; baseten: SyncProvider; @@ -165,6 +167,7 @@ export const providers: { wandb: SyncProvider; xai: SyncProvider; } = { + aihubmix, ambient, anthropic, baseten, @@ -203,6 +206,7 @@ export const providers: { export const groups = { aggregators: [ + "aihubmix", "crossmodel", "edenai", "empiriolabs", @@ -1055,6 +1059,12 @@ export function formatToml(model: z.infer) { if (tier.reasoning !== undefined) lines.push(`reasoning = ${formatNumber(tier.reasoning)}`); if (tier.cache_read !== undefined) lines.push(`cache_read = ${formatNumber(tier.cache_read)}`); if (tier.cache_write !== undefined) lines.push(`cache_write = ${formatNumber(tier.cache_write)}`); + if (tier.input_audio !== undefined) { + lines.push(`input_audio = ${formatNumber(tier.input_audio)}`); + } + if (tier.output_audio !== undefined) { + lines.push(`output_audio = ${formatNumber(tier.output_audio)}`); + } } } From 3cfdf627ddfb7920c61d38d76149a47a4f58a7e5 Mon Sep 17 00:00:00 2001 From: chenxue Date: Wed, 9 Sep 2026 11:15:02 +0800 Subject: [PATCH 02/20] feat(aihubmix): sync pricing and deprecation from the public catalog AIHubMix has had no sync module, so its 77 models were only ever refreshed by hand-written PRs. The last one landed 2026-08-31, which is why prices have drifted and new relays never arrive on their own. The endpoint (`https://aihubmix.com/api/v1/models?type=llm`, no auth) is authoritative for pricing and deprecation status only, matching the Ofox scope. Token limits and modalities are deliberately not synced: the endpoint reports each relay's conservative defaults rather than the upstream model's capabilities. It caps `context_length` per relay (Claude Opus 4.6 is listed at 200K against its 1M window), quotes `max_output` per default request, and never lists `pdf` even for models that accept PDFs. `cache_read` is ignored when it equals `input`: the endpoint echoes the input price for models with no cached rate configured, which covers 35 of the 301 priced entries at a nonzero price (plus 51 free models reporting 0 across the board, where the guard is a no-op). Taking the echoed value literally would have set Gemini 3.1 Flash Lite to $0.25 against the $0.025 that 26 other providers list. The first run updates 17 models. Beyond precision refinements it corrects real drift: GPT-5.6 Luna to OpenAI's own $0.20/$1.20 (was $1/$6), Sol and Terra to their current cuts, Gemini 3.5 Flash's `cache_read` from $1.50 to $0.15 (the authored value had the same echoed-input bug), and the Coding MiMo v2.5 output rates onto Xiaomi's actual 2:1 ratio. New relays are not created automatically (`skipCreates`) since AIHubMix serves roughly 400 upstream models against this hand-verified subset; each missing ID opens a deduped issue instead. Routing aliases such as `alicloud-glm-5.1` are served but unlisted, so local files absent from the response are retained. The Gemini 2.5 Flash thinking-budget comment moves to the file header, which is the only comment block `formatToml` preserves. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 135 ++++++++++++++++++ packages/core/test/sync.test.ts | 79 ++++++++++ .../models/coding-xiaomi-mimo-v2.5-pro.toml | 8 +- .../models/coding-xiaomi-mimo-v2.5.toml | 8 +- .../models/doubao-seed-2-0-code-preview.toml | 16 ++- .../models/doubao-seed-2-0-lite-260428.toml | 18 ++- .../models/doubao-seed-2-0-mini-260428.toml | 16 ++- .../aihubmix/models/doubao-seed-2-0-pro.toml | 16 ++- .../aihubmix/models/gemini-2.5-flash.toml | 19 ++- .../aihubmix/models/gemini-3.5-flash.toml | 7 +- providers/aihubmix/models/gpt-5.1.toml | 11 +- providers/aihubmix/models/gpt-5.6-luna.toml | 13 +- providers/aihubmix/models/gpt-5.6-sol.toml | 13 +- providers/aihubmix/models/gpt-5.6-terra.toml | 13 +- providers/aihubmix/models/kimi-k2.5.toml | 6 +- providers/aihubmix/models/kimi-k2.6.toml | 8 +- providers/aihubmix/models/minimax-m2.7.toml | 8 +- providers/aihubmix/models/qwen3.6-flash.toml | 10 +- .../aihubmix/models/qwen3.6-max-preview.toml | 10 +- providers/aihubmix/models/qwen3.6-plus.toml | 10 +- sync.md | 12 ++ 21 files changed, 363 insertions(+), 73 deletions(-) create mode 100644 packages/core/src/sync/providers/aihubmix.ts diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts new file mode 100644 index 00000000000..36134bdab72 --- /dev/null +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -0,0 +1,135 @@ +import { z } from "zod"; + +import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js"; + +const API_ENDPOINT = "https://aihubmix.com/api/v1/models?type=llm"; + +/** AIHubMix quotes USD per 1M tokens directly, matching the catalog unit. */ +const Pricing = z + .object({ + input: z.number().nullish(), + output: z.number().nullish(), + cache_read: z.number().nullish(), + cache_write: z.number().nullish(), + }) + .passthrough(); + +export const AihubmixModel = z + .object({ + model_id: z.string().min(1), + model_name: z.string().nullish(), + pricing: Pricing.nullish(), + retire_stage: z.string().nullish(), + }) + .passthrough(); + +export const AihubmixResponse = z + .object({ + success: z.boolean().nullish(), + data: z.array(AihubmixModel).min(1), + }) + .passthrough(); + +export type AihubmixModel = z.infer; + +/** + * AIHubMix relays ~400 upstream models while the catalog curates a much smaller + * hand-verified subset, so this sync only updates existing TOMLs + * (`skipCreates`) and treats the AIHubMix endpoint as authoritative for pricing + * and deprecation status only. + * + * Everything else in the authored TOMLs is preserved as hand-authored, because + * the endpoint reports the relay's own conservative defaults rather than the + * upstream model's real capabilities: `context_length` is capped per relay + * (Claude Opus 4.6 reports 200K against its 1M window), `max_output` is quoted + * per default request rather than per model, and `input_modalities` never lists + * `pdf` even for models the provider does accept PDFs for. Its free-text + * `features` list likewise mixes synonyms (`thinking` vs `reasoning`, `tools` + * vs `tool_calling`) and never exposes accepted reasoning effort levels, so + * capability flags, `reasoning_options`, `base_model` inheritance, and the + * per-model `[provider]` protocol overrides all stay authored. + * + * Routing aliases such as `alicloud-glm-5.1` and `deep-deepseek-v4-pro` are + * served but not listed by the endpoint, so local files missing from the + * response are retained (`deleteMissing: false`). + */ +export const aihubmix = { + id: "aihubmix", + name: "AIHubMix", + modelsDir: "providers/aihubmix/models", + skipCreates: true, + trackMissingModels: true, + deleteMissing: false, + sourceID(model) { + // Deprecated relays are not catalog additions worth an issue. + return model.retire_stage === "deprecated" ? undefined : model.model_id; + }, + missingNotice(paths) { + return paths.map( + (file) => + `AIHubMix no longer lists ${file}; confirm it is still a served routing alias or deprecate it.`, + ); + }, + async fetchModels() { + const response = await fetch(process.env.AIHUBMIX_MODELS_URL ?? API_ENDPOINT); + if (!response.ok) { + throw new Error(`AIHubMix models request failed: ${response.status} ${response.statusText}`); + } + return response.json(); + }, + parseModels(raw) { + return AihubmixResponse.parse(raw).data; + }, + translateModel(model, context) { + const authored = context.authored(model.model_id); + if (authored === undefined) return undefined; + return { + id: model.model_id, + model: buildAihubmixModel(model, authored), + }; + }, +} satisfies SyncProvider; + +export function buildAihubmixModel(model: AihubmixModel, authored: ExistingModel): SyncedModel { + const { id: _id, ...preserved } = authored; + return { + ...preserved, + cost: buildCost(model.pricing, authored.cost), + status: model.retire_stage === "deprecated" ? "deprecated" : authored.status, + } as SyncedModel; +} + +/** + * AIHubMix omits a price field when the model has no such rate, but a partial + * authored `[cost]` (tiers, reasoning, audio rates) still carries values the + * endpoint does not model, so those are kept. + */ +function buildCost( + pricing: AihubmixModel["pricing"], + authored: ExistingModel["cost"], +): SyncedFullModel["cost"] { + if (pricing == null) return authored; + const input = price(pricing.input); + const output = price(pricing.output); + if (input === undefined || output === undefined) return authored; + + const cacheRead = price(pricing.cache_read); + return { + ...authored, + input, + output, + // AIHubMix echoes the input price into `cache_read` for models it has no + // cached rate for (35 of the 301 priced entries carry a nonzero price this + // way; 51 more are free models reporting 0 across the board, where this is + // a no-op), so an equal value means "not quoted" rather than "cached reads + // cost full price" — taking it literally would overstate e.g. Gemini 3.1 + // Flash Lite by 10x against the $0.025 every other provider lists. + cache_read: cacheRead === input ? authored?.cache_read : (cacheRead ?? authored?.cache_read), + cache_write: price(pricing.cache_write) ?? authored?.cache_write, + }; +} + +function price(value: number | null | undefined) { + if (value == null || !Number.isFinite(value) || value < 0) return undefined; + return Math.round(value * 1_000_000) / 1_000_000; +} diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 0e51e85cfe1..d55d0be470c 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -4,6 +4,11 @@ import { tmpdir } from "node:os"; import path from "node:path"; import { formatToml, preserveReasoningOptions, syncProvider, type ExistingModel, type SyncProvider } from "../src/sync/index.js"; +import { + aihubmix, + buildAihubmixModel, + type AihubmixModel, +} from "../src/sync/providers/aihubmix.js"; import { anthropic, buildAnthropicModel, @@ -1185,6 +1190,8 @@ test("tracks missing models except for unreliable first-party inventories", () = expect(openai.trackMissingModels).toBe(false); expect(pioneer.skipCreates).toBe(true); expect(pioneer.trackMissingModels).toBe(true); + expect(aihubmix.skipCreates).toBe(true); + expect(aihubmix.trackMissingModels).toBe(true); expect(ofox.skipCreates).toBe(true); expect(ofox.trackMissingModels).toBe(true); expect(tinfoil.skipCreates).toBe(true); @@ -5014,3 +5021,75 @@ test("rejects synced model paths that differ only in case", async () => { await rm(root, { recursive: true, force: true }); } }); + +function aihubmixModel(overrides: Partial = {}): AihubmixModel { + return { + model_id: "gemini-3.1-flash-lite", + model_name: "Gemini 3.1 Flash Lite", + pricing: { input: 0.25, output: 1.5, cache_read: 0.025 }, + ...overrides, + }; +} + +const aihubmixAuthored: ExistingModel = { + id: "gemini-3.1-flash-lite", + name: "Gemini 3.1 Flash Lite", + attachment: true, + reasoning: true, + reasoning_options: [{ type: "effort", values: ["minimal", "low", "medium", "high"] }], + cost: { input: 0.25, output: 1.5, cache_read: 0.025, cache_write: 1 }, + limit: { context: 1_048_576, output: 65_536 }, + modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, +}; + +test("syncs AIHubMix pricing while preserving hand-authored capabilities", () => { + const model = buildAihubmixModel( + aihubmixModel({ pricing: { input: 0.2, output: 1.2, cache_read: 0.02, cache_write: 0.25 } }), + aihubmixAuthored, + ); + expect(model.cost).toMatchObject({ input: 0.2, output: 1.2, cache_read: 0.02, cache_write: 0.25 }); + // The endpoint reports relay defaults for these, so authored values win. + expect(model.limit).toEqual({ context: 1_048_576, output: 65_536 }); + expect(model.modalities).toEqual({ + input: ["text", "image", "audio", "video", "pdf"], + output: ["text"], + }); + expect(model.reasoning_options).toEqual([ + { type: "effort", values: ["minimal", "low", "medium", "high"] }, + ]); +}); + +test("ignores AIHubMix cache_read that merely echoes the input price", () => { + const model = buildAihubmixModel( + aihubmixModel({ pricing: { input: 0.25, output: 1.5, cache_read: 0.25 } }), + aihubmixAuthored, + ); + expect(model.cost?.cache_read).toBe(0.025); +}); + +test("keeps authored AIHubMix pricing when the endpoint quotes no rate", () => { + const model = buildAihubmixModel(aihubmixModel({ pricing: null }), aihubmixAuthored); + expect(model.cost).toEqual(aihubmixAuthored.cost); +}); + +test("marks retired AIHubMix relays deprecated and stops tracking them", () => { + const retired = aihubmixModel({ retire_stage: "deprecated" }); + expect(buildAihubmixModel(retired, aihubmixAuthored).status).toBe("deprecated"); + expect(aihubmix.sourceID?.(retired)).toBeUndefined(); + expect(aihubmix.sourceID?.(aihubmixModel())).toBe("gemini-3.1-flash-lite"); +}); + +test("writes per-tier audio pricing", () => { + const toml = formatToml({ + name: "Doubao Seed 2.0 Lite", + cost: { + input: 0.09041, + output: 0.54246, + input_audio: 1.269, + tiers: [ + { tier: { type: "context", size: 32_000 }, input: 0.13, output: 0.76, input_audio: 1.902 }, + ], + }, + } as never); + expect(toml).toContain("input_audio = 1.902"); +}); diff --git a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml index a106666530c..ccb1203b612 100644 --- a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml +++ b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml @@ -1,5 +1,4 @@ base_model = "xiaomi/mimo-v2.5-pro" -reasoning_options = [{ type = "toggle" }] name = "Coding Xiaomi MiMo-V2.5-Pro" family = "mimo-v2.5-pro" last_updated = "2026-05-13" @@ -7,10 +6,13 @@ last_updated = "2026-05-13" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.2 -output = 0.6 -cache_read = 0.04 +output = 0.4 +cache_read = 0.0016 [[cost.tiers]] tier = { type = "context", size = 256_000 } diff --git a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml index fa49278c22d..23149a9b62b 100644 --- a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml +++ b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml @@ -1,5 +1,4 @@ base_model = "xiaomi/mimo-v2.5" -reasoning_options = [{ type = "toggle" }] name = "Coding Xiaomi MiMo-V2.5" family = "mimo-v2.5" last_updated = "2026-05-13" @@ -7,10 +6,13 @@ last_updated = "2026-05-13" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.08 -output = 0.4 -cache_read = 0.016 +output = 0.16 +cache_read = 0.0016 [[cost.tiers]] tier = { type = "context", size = 256_000 } diff --git a/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml b/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml index 4eb6666263d..60d3e2decbb 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml @@ -5,7 +5,6 @@ release_date = "2026-02-14" last_updated = "2026-02-14" attachment = true reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] temperature = true tool_call = true structured_output = true @@ -14,19 +13,26 @@ open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + [cost] -input = 0.48 -output = 2.41 +input = 0.4822 +output = 2.411 cache_read = 0.09644 [[cost.tiers]] -tier = { size = 32_000 } +tier = { type = "context", size = 32_000 } input = 0.72 output = 3.62 cache_read = 0.144656 [[cost.tiers]] -tier = { size = 128_000 } +tier = { type = "context", size = 128_000 } input = 1.45 output = 7.23 cache_read = 0.28932 diff --git a/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml b/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml index dc89d4bccc6..7dd55b32212 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml @@ -5,7 +5,6 @@ release_date = "2026-04-28" last_updated = "2026-04-28" attachment = true reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] temperature = true tool_call = true structured_output = true @@ -14,21 +13,28 @@ open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + [cost] -input = 0.08 -output = 0.51 -cache_read = 0.01692 +input = 0.09041 +output = 0.54246 +cache_read = 0.018082 input_audio = 1.269 [[cost.tiers]] -tier = { size = 32_000 } +tier = { type = "context", size = 32_000 } input = 0.13 output = 0.76 cache_read = 0.02536 input_audio = 1.902 [[cost.tiers]] -tier = { size = 128_000 } +tier = { type = "context", size = 128_000 } input = 0.25 output = 1.52 cache_read = 0.05072 diff --git a/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml b/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml index 2f14132485f..f83044c0ef6 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml @@ -5,7 +5,6 @@ release_date = "2026-04-28" last_updated = "2026-04-28" attachment = true reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] temperature = true tool_call = true structured_output = true @@ -14,21 +13,28 @@ open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + [cost] -input = 0.03 -output = 0.28 +input = 0.0282 +output = 0.282 cache_read = 0.00564 input_audio = 0.423 [[cost.tiers]] -tier = { size = 32_000 } +tier = { type = "context", size = 32_000 } input = 0.06 output = 0.56 cache_read = 0.01128 input_audio = 0.846 [[cost.tiers]] -tier = { size = 128_000 } +tier = { type = "context", size = 128_000 } input = 0.11 output = 1.13 cache_read = 0.02256 diff --git a/providers/aihubmix/models/doubao-seed-2-0-pro.toml b/providers/aihubmix/models/doubao-seed-2-0-pro.toml index a3f4582f729..f51fb455088 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-pro.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-pro.toml @@ -5,7 +5,6 @@ release_date = "2026-02-14" last_updated = "2026-02-14" attachment = true reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] temperature = true tool_call = true structured_output = true @@ -14,19 +13,26 @@ open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + [cost] -input = 0.48 -output = 2.41 +input = 0.4822 +output = 2.411 cache_read = 0.09644 [[cost.tiers]] -tier = { size = 32_000 } +tier = { type = "context", size = 32_000 } input = 0.72 output = 3.62 cache_read = 0.144656 [[cost.tiers]] -tier = { size = 128_000 } +tier = { type = "context", size = 128_000 } input = 1.45 output = 7.23 cache_read = 0.28932 diff --git a/providers/aihubmix/models/gemini-2.5-flash.toml b/providers/aihubmix/models/gemini-2.5-flash.toml index 23f238b714b..1b342075259 100644 --- a/providers/aihubmix/models/gemini-2.5-flash.toml +++ b/providers/aihubmix/models/gemini-2.5-flash.toml @@ -1,3 +1,4 @@ +# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: 0 disables, -1 is dynamic, and manual budgets are 1..24576. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) name = "Gemini 2.5 Flash" description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost" family = "gemini-flash" @@ -5,19 +6,29 @@ release_date = "2025-03-20" last_updated = "2025-06-05" attachment = true reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 0, max = 24_576 }] -# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: 0 disables, -1 is dynamic, and manual budgets are 1..24576. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) temperature = true tool_call = true structured_output = true knowledge = "2025-01" open_weights = false +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" +min = 0 +max = 24_576 + [cost] input = 0.3 -output = 2.50 +output = 2.499 cache_read = 0.03 -input_audio = 1.00 +input_audio = 1 [limit] context = 1_048_576 diff --git a/providers/aihubmix/models/gemini-3.5-flash.toml b/providers/aihubmix/models/gemini-3.5-flash.toml index fb3e8ecd08a..577253bb682 100644 --- a/providers/aihubmix/models/gemini-3.5-flash.toml +++ b/providers/aihubmix/models/gemini-3.5-flash.toml @@ -1,10 +1,13 @@ base_model = "google/gemini-3.5-flash" -reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] [cost] input = 1.5 output = 9 -cache_read = 1.5 +cache_read = 0.15 [limit] context = 1_000_000 diff --git a/providers/aihubmix/models/gpt-5.1.toml b/providers/aihubmix/models/gpt-5.1.toml index d707b927063..28234deacd6 100644 --- a/providers/aihubmix/models/gpt-5.1.toml +++ b/providers/aihubmix/models/gpt-5.1.toml @@ -5,17 +5,20 @@ release_date = "2025-11-13" last_updated = "2025-11-13" attachment = true reasoning = true -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] temperature = false -knowledge = "2024-09-30" tool_call = true structured_output = true +knowledge = "2024-09-30" open_weights = false +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + [cost] input = 1.25 -output = 10.00 -cache_read = 0.13 +output = 10 +cache_read = 0.125 [limit] context = 400_000 diff --git a/providers/aihubmix/models/gpt-5.6-luna.toml b/providers/aihubmix/models/gpt-5.6-luna.toml index 7213c25165e..3bfd7a75d89 100644 --- a/providers/aihubmix/models/gpt-5.6-luna.toml +++ b/providers/aihubmix/models/gpt-5.6-luna.toml @@ -1,12 +1,15 @@ base_model = "openai/gpt-5.6-luna" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] temperature = false +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + [cost] -input = 1 -output = 6 -cache_read = 0.1 -cache_write = 1.25 +input = 0.2 +output = 1.2 +cache_read = 0.02 +cache_write = 0.25 [limit] context = 1_050_000 diff --git a/providers/aihubmix/models/gpt-5.6-sol.toml b/providers/aihubmix/models/gpt-5.6-sol.toml index 2d0fb532e7e..4150cc42d58 100644 --- a/providers/aihubmix/models/gpt-5.6-sol.toml +++ b/providers/aihubmix/models/gpt-5.6-sol.toml @@ -1,12 +1,15 @@ base_model = "openai/gpt-5.6-sol" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] temperature = false +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + [cost] -input = 5 -output = 30 -cache_read = 0.5 -cache_write = 6.25 +input = 4 +output = 20 +cache_read = 0.4 +cache_write = 5 [limit] context = 1_050_000 diff --git a/providers/aihubmix/models/gpt-5.6-terra.toml b/providers/aihubmix/models/gpt-5.6-terra.toml index 1c68d95d455..7a13fcf9dc4 100644 --- a/providers/aihubmix/models/gpt-5.6-terra.toml +++ b/providers/aihubmix/models/gpt-5.6-terra.toml @@ -1,12 +1,15 @@ base_model = "openai/gpt-5.6-terra" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] temperature = false +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + [cost] -input = 2.5 -output = 15 -cache_read = 0.25 -cache_write = 3.125 +input = 2 +output = 12 +cache_read = 0.2 +cache_write = 2.5 [limit] context = 1_050_000 diff --git a/providers/aihubmix/models/kimi-k2.5.toml b/providers/aihubmix/models/kimi-k2.5.toml index ce3eca895df..280040aed77 100644 --- a/providers/aihubmix/models/kimi-k2.5.toml +++ b/providers/aihubmix/models/kimi-k2.5.toml @@ -5,7 +5,6 @@ release_date = "2026-01" last_updated = "2026-01" attachment = true reasoning = true -reasoning_options = [{ type = "toggle" }] temperature = false tool_call = true structured_output = true @@ -15,10 +14,13 @@ open_weights = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.6 output = 3 -cache_read = 0.10 +cache_read = 0.105 [limit] context = 262_144 diff --git a/providers/aihubmix/models/kimi-k2.6.toml b/providers/aihubmix/models/kimi-k2.6.toml index ddb87134c56..29b985faca7 100644 --- a/providers/aihubmix/models/kimi-k2.6.toml +++ b/providers/aihubmix/models/kimi-k2.6.toml @@ -5,7 +5,6 @@ release_date = "2026-04-21" last_updated = "2026-04-21" attachment = true reasoning = true -reasoning_options = [{ type = "toggle" }] temperature = false tool_call = true structured_output = true @@ -15,10 +14,13 @@ open_weights = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.95 -output = 4 -cache_read = 0.16 +output = 3.9995 +cache_read = 0.160835 [limit] context = 262_144 diff --git a/providers/aihubmix/models/minimax-m2.7.toml b/providers/aihubmix/models/minimax-m2.7.toml index f00a935f81f..e91777c895b 100644 --- a/providers/aihubmix/models/minimax-m2.7.toml +++ b/providers/aihubmix/models/minimax-m2.7.toml @@ -5,19 +5,19 @@ release_date = "2026-03-18" last_updated = "2026-03-18" attachment = false reasoning = true -reasoning_options = [] temperature = true tool_call = true structured_output = true open_weights = true +reasoning_options = [] [interleaved] field = "reasoning_content" [cost] -input = 0.3 -output = 1.2 -cache_read = 0.06 +input = 0.2958 +output = 1.1832 +cache_read = 0.05916 cache_write = 0.375 [limit] diff --git a/providers/aihubmix/models/qwen3.6-flash.toml b/providers/aihubmix/models/qwen3.6-flash.toml index 2371f400435..3da56465a7a 100644 --- a/providers/aihubmix/models/qwen3.6-flash.toml +++ b/providers/aihubmix/models/qwen3.6-flash.toml @@ -5,7 +5,6 @@ release_date = "2026-04-02" last_updated = "2026-04-02" attachment = true reasoning = true -reasoning_options = [{ type = "toggle" }] temperature = true tool_call = true structured_output = true @@ -15,14 +14,17 @@ open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] -input = 0.17 -output = 1.01 +input = 0.169 +output = 1.014 cache_read = 0.0169 cache_write = 0.21125 [[cost.tiers]] -tier = { size = 256_000 } +tier = { type = "context", size = 256_000 } input = 0.68 output = 4.06 cache_read = 0.0676 diff --git a/providers/aihubmix/models/qwen3.6-max-preview.toml b/providers/aihubmix/models/qwen3.6-max-preview.toml index 32c2d9f4bfe..299b558c20a 100644 --- a/providers/aihubmix/models/qwen3.6-max-preview.toml +++ b/providers/aihubmix/models/qwen3.6-max-preview.toml @@ -5,7 +5,6 @@ release_date = "2026-05-09" last_updated = "2026-05-09" attachment = false reasoning = true -reasoning_options = [{ type = "toggle" }] temperature = true tool_call = true structured_output = true @@ -15,14 +14,17 @@ open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] -input = 1.27 -output = 7.61 +input = 1.268 +output = 7.608 cache_read = 0.1268 cache_write = 1.585 [[cost.tiers]] -tier = { size = 128_000 } +tier = { type = "context", size = 128_000 } input = 2.11 output = 12.67 cache_read = 0.2112 diff --git a/providers/aihubmix/models/qwen3.6-plus.toml b/providers/aihubmix/models/qwen3.6-plus.toml index 722d03774fc..33b3dfc9bf9 100644 --- a/providers/aihubmix/models/qwen3.6-plus.toml +++ b/providers/aihubmix/models/qwen3.6-plus.toml @@ -5,7 +5,6 @@ release_date = "2026-05-09" last_updated = "2026-05-09" attachment = true reasoning = true -reasoning_options = [{ type = "toggle" }] temperature = true tool_call = true structured_output = true @@ -15,14 +14,17 @@ open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] -input = 0.28 -output = 1.69 +input = 0.282 +output = 1.692 cache_read = 0.0282 cache_write = 0.3525 [[cost.tiers]] -tier = { size = 256_000 } +tier = { type = "context", size = 256_000 } input = 1.13 output = 6.77 cache_read = 0.1128 diff --git a/sync.md b/sync.md index 7595a924dc4..ff3ad5bbe69 100644 --- a/sync.md +++ b/sync.md @@ -248,6 +248,18 @@ xAI is implemented in `packages/core/src/sync/providers/xai.ts`. - Existing xAI models are updated from API-authoritative fields while local metadata is preserved for fields the API does not expose, especially output token limits and some feature/capability flags. - New xAI API models are not created automatically (`skipCreates`); each missing ID opens a deduped GitHub issue. Alias IDs of models already cataloged under their canonical ID are skipped silently and never reported as missing. +## AIHubMix Notes + +- AIHubMix is implemented in `packages/core/src/sync/providers/aihubmix.ts`. +- Source endpoint: `https://aihubmix.com/api/v1/models?type=llm`. +- No authentication is required; the catalog is public. +- The endpoint is authoritative for pricing and deprecation status only. Everything else in the authored TOMLs is preserved. +- Token limits and modalities are deliberately **not** synced: the endpoint reports the relay's conservative defaults rather than the upstream model's capabilities. It caps `context_length` per relay (Claude Opus 4.6 is listed at 200K against its 1M window), quotes `max_output` per default request, and never lists `pdf` in `input_modalities` even for models that accept PDFs. +- `cache_read` is ignored when it equals `input`. The endpoint echoes the input price for models with no cached rate configured: 35 of the 301 priced entries carry a nonzero price this way, and 51 more are free models reporting 0 across the board. Taking the echoed value literally would overstate Gemini 3.1 Flash Lite tenfold against the $0.025 every other provider lists. +- The free-text `features` list mixes synonyms (`thinking` vs `reasoning`, `tools` vs `tool_calling`) and never exposes accepted reasoning effort levels, so capability flags and `reasoning_options` stay hand-authored. +- AIHubMix relays roughly 400 upstream models against a much smaller hand-verified subset here, so new IDs are not created automatically (`skipCreates`); each missing ID opens a deduped GitHub issue. +- Routing aliases such as `alicloud-glm-5.1` and `deep-deepseek-v4-pro` are served but not listed by the endpoint, so local files missing from the response are retained (`deleteMissing: false`). + ## Tinfoil Notes - Tinfoil is implemented in `packages/core/src/sync/providers/tinfoil.ts`. From ba56a476aa5e38467c2143dbef53d44e8aa4ad5e Mon Sep 17 00:00:00 2001 From: chenxue Date: Thu, 10 Sep 2026 12:31:58 +0800 Subject: [PATCH 03/20] feat(aihubmix): sync the full model catalog from the endpoint AIHubMix now serves capabilities, limits, modalities and reasoning controls alongside pricing, so the adapter reads all of them instead of treating the endpoint as authoritative for cost and status alone. Relays are factored onto the lab metadata they serve: `developer_id` maps a relay to its lab, and routing prefixes (`coding-`, `alicloud-`) and suffixes (`-free`, `-think`, `-nothink`) select a mode rather than a different model, so they are stripped when resolving the base. A relay then records only what it actually changes. With bases resolving, new IDs no longer need to be held back, so `skipCreates` is dropped and 155 relays are created. Three source quirks are handled in translation rather than written through: `reasoning_options[]` carries an AIHubMix-only `default` key the strict schema rejects, two effort levels are spelled `no_think` and `instant`, and `max_output: 0` means "unknown" rather than a real ceiling for 102 of 415 models. A relay with neither resolvable lab metadata nor the release_date and open_weights a standalone entry requires is reported rather than written with invented values. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 323 +++++++++++++++--- packages/core/test/sync.test.ts | 116 +++++-- .../models/claude-3-haiku-20240307.toml | 9 + .../aihubmix/models/claude-fable-5-1.toml | 15 + providers/aihubmix/models/claude-fable-5.toml | 17 +- .../aihubmix/models/claude-haiku-4-5.toml | 14 + .../aihubmix/models/claude-opus-4-1.toml | 15 + .../models/claude-opus-4-5-think.toml | 18 + .../aihubmix/models/claude-opus-4-5.toml | 18 + .../models/claude-opus-4-6-think.toml | 33 +- .../aihubmix/models/claude-opus-4-6.toml | 35 +- .../models/claude-opus-4-7-think.toml | 30 +- .../aihubmix/models/claude-opus-4-7.toml | 32 +- .../models/claude-opus-4-8-think.toml | 18 +- .../aihubmix/models/claude-opus-4-8.toml | 18 +- providers/aihubmix/models/claude-opus-5.toml | 13 +- .../models/claude-sonnet-4-5-think.toml | 21 ++ .../aihubmix/models/claude-sonnet-4-5.toml | 21 ++ .../models/claude-sonnet-4-6-think.toml | 44 ++- .../aihubmix/models/claude-sonnet-4-6.toml | 46 ++- .../aihubmix/models/claude-sonnet-5.toml | 18 +- .../aihubmix/models/coding-glm-4.6-free.toml | 12 + providers/aihubmix/models/coding-glm-4.6.toml | 13 + .../aihubmix/models/coding-glm-4.7-free.toml | 12 + providers/aihubmix/models/coding-glm-4.7.toml | 13 + .../aihubmix/models/coding-glm-5-free.toml | 12 + .../models/coding-glm-5-turbo-free.toml | 11 + .../aihubmix/models/coding-glm-5-turbo.toml | 11 + .../aihubmix/models/coding-glm-5.1-free.toml | 22 +- providers/aihubmix/models/coding-glm-5.1.toml | 23 +- .../aihubmix/models/coding-glm-5.2-free.toml | 12 + providers/aihubmix/models/coding-glm-5.2.toml | 12 + .../aihubmix/models/coding-glm-5.3-free.toml | 15 + providers/aihubmix/models/coding-glm-5.3.toml | 16 + providers/aihubmix/models/coding-glm-5.toml | 12 + .../aihubmix/models/coding-kimi-k3-free.toml | 15 + providers/aihubmix/models/coding-kimi-k3.toml | 16 + .../models/coding-minimax-m2.7-free.toml | 12 +- .../models/coding-minimax-m2.7-highspeed.toml | 10 +- .../aihubmix/models/coding-minimax-m2.7.toml | 10 +- .../models/coding-xiaomi-mimo-v2.5-pro.toml | 6 +- .../models/coding-xiaomi-mimo-v2.5.toml | 6 +- .../aihubmix/models/command-a-03-2025.toml | 11 + .../models/command-a-plus-05-2026.toml | 15 + .../aihubmix/models/command-r-08-2024.toml | 6 + .../models/command-r-plus-08-2024.toml | 6 + .../aihubmix/models/deepseek-v3.2-think.toml | 12 + providers/aihubmix/models/deepseek-v3.2.toml | 12 + .../models/deepseek-v4-flash-vision-exp.toml | 13 + .../aihubmix/models/deepseek-v4-flash.toml | 13 + .../aihubmix/models/deepseek-v4-pro-0813.toml | 8 +- .../aihubmix/models/deepseek-v4-pro.toml | 13 + .../models/doubao-seed-2-0-code-preview.toml | 10 +- .../models/doubao-seed-2-0-lite-260428.toml | 15 +- .../models/doubao-seed-2-0-mini-260428.toml | 15 +- .../aihubmix/models/doubao-seed-2-0-pro.toml | 10 +- .../models/gemini-2.5-flash-image.toml | 10 + .../models/gemini-2.5-flash-lite-nothink.toml | 13 + .../models/gemini-2.5-flash-lite.toml | 13 + .../models/gemini-2.5-flash-nothink.toml | 16 + .../models/gemini-2.5-flash-search.toml | 16 + .../aihubmix/models/gemini-2.5-flash.toml | 27 +- .../models/gemini-2.5-pro-search.toml | 19 ++ providers/aihubmix/models/gemini-2.5-pro.toml | 35 +- .../models/gemini-3-flash-preview-free.toml | 12 + .../models/gemini-3-flash-preview-search.toml | 13 + .../models/gemini-3-flash-preview.toml | 34 +- .../models/gemini-3-pro-image-preview.toml | 7 + .../aihubmix/models/gemini-3-pro-image.toml | 12 + .../gemini-3.1-flash-image-preview.toml | 10 + .../models/gemini-3.1-flash-image.toml | 12 + .../models/gemini-3.1-flash-lite-image.toml | 11 + .../models/gemini-3.1-flash-lite-nothink.toml | 13 + .../models/gemini-3.1-flash-lite.toml | 29 +- .../gemini-3.1-pro-preview-customtools.toml | 30 +- .../models/gemini-3.1-pro-preview-search.toml | 19 ++ .../models/gemini-3.1-pro-preview.toml | 36 +- .../models/gemini-3.5-flash-lite-free.toml | 9 + .../models/gemini-3.5-flash-lite.toml | 10 + .../aihubmix/models/gemini-3.5-flash.toml | 11 +- .../models/gemini-3.6-flash-free.toml | 9 + .../aihubmix/models/gemini-3.6-flash.toml | 10 + .../models/gemini-3.7-flash-free.toml | 12 + .../aihubmix/models/gemini-3.7-flash.toml | 9 +- .../models/gemini-3.8-flash-free.toml | 12 + .../aihubmix/models/gemini-3.8-flash.toml | 13 + .../models/gemma-4-26b-a4b-it-free.toml | 15 + .../aihubmix/models/gemma-4-26b-a4b-it.toml | 19 ++ .../aihubmix/models/gemma-4-31b-it-free.toml | 15 + providers/aihubmix/models/gemma-4-31b-it.toml | 19 ++ providers/aihubmix/models/glm-4.5v.toml | 21 ++ providers/aihubmix/models/glm-4.6.toml | 19 ++ providers/aihubmix/models/glm-4.6v.toml | 23 ++ .../aihubmix/models/glm-4.7-flash-free.toml | 9 + providers/aihubmix/models/glm-4.7.toml | 19 ++ providers/aihubmix/models/glm-5-turbo.toml | 12 + providers/aihubmix/models/glm-5.1.toml | 9 + providers/aihubmix/models/glm-5.2.toml | 19 +- providers/aihubmix/models/glm-5.3-flash.toml | 11 +- providers/aihubmix/models/glm-5.3.toml | 10 +- providers/aihubmix/models/glm-5.toml | 13 + providers/aihubmix/models/glm-5v-turbo.toml | 22 +- providers/aihubmix/models/gpt-4.1-free.toml | 8 + .../aihubmix/models/gpt-4.1-mini-free.toml | 8 + providers/aihubmix/models/gpt-4.1-mini.toml | 9 + .../aihubmix/models/gpt-4.1-nano-free.toml | 5 + providers/aihubmix/models/gpt-4.1-nano.toml | 6 + providers/aihubmix/models/gpt-4.1.toml | 9 + .../aihubmix/models/gpt-4o-2024-11-20.toml | 7 + providers/aihubmix/models/gpt-4o-free.toml | 8 + providers/aihubmix/models/gpt-4o-mini.toml | 10 + providers/aihubmix/models/gpt-4o.toml | 9 + .../aihubmix/models/gpt-5-chat-latest.toml | 12 + providers/aihubmix/models/gpt-5-codex.toml | 8 + providers/aihubmix/models/gpt-5-mini.toml | 7 + providers/aihubmix/models/gpt-5-nano.toml | 7 + providers/aihubmix/models/gpt-5-pro.toml | 9 + .../aihubmix/models/gpt-5.1-chat-latest.toml | 7 + .../aihubmix/models/gpt-5.1-codex-max.toml | 7 + .../aihubmix/models/gpt-5.1-codex-mini.toml | 28 +- providers/aihubmix/models/gpt-5.1-codex.toml | 28 +- providers/aihubmix/models/gpt-5.1.toml | 20 +- .../aihubmix/models/gpt-5.2-chat-latest.toml | 7 + providers/aihubmix/models/gpt-5.2-codex.toml | 27 +- providers/aihubmix/models/gpt-5.2-pro.toml | 11 + providers/aihubmix/models/gpt-5.2.toml | 27 +- .../aihubmix/models/gpt-5.3-chat-latest.toml | 6 + providers/aihubmix/models/gpt-5.3-codex.toml | 26 +- providers/aihubmix/models/gpt-5.4-mini.toml | 31 +- providers/aihubmix/models/gpt-5.4-nano.toml | 10 + providers/aihubmix/models/gpt-5.4-pro.toml | 15 + providers/aihubmix/models/gpt-5.4.toml | 40 +-- providers/aihubmix/models/gpt-5.5-free.toml | 12 + providers/aihubmix/models/gpt-5.5-pro.toml | 17 + providers/aihubmix/models/gpt-5.5.toml | 43 +-- providers/aihubmix/models/gpt-5.6-luna.toml | 11 +- .../aihubmix/models/gpt-5.6-sol-disc.toml | 21 ++ providers/aihubmix/models/gpt-5.6-sol.toml | 11 +- providers/aihubmix/models/gpt-5.6-terra.toml | 11 +- providers/aihubmix/models/gpt-5.toml | 10 + providers/aihubmix/models/gpt-6-astra.toml | 21 ++ .../aihubmix/models/gpt-image-2-free.toml | 8 + providers/aihubmix/models/gpt-oss-120b.toml | 6 + .../aihubmix/models/gpt-oss-20b-free.toml | 12 + providers/aihubmix/models/gpt-oss-20b.toml | 10 + providers/aihubmix/models/grok-4.3.toml | 22 +- providers/aihubmix/models/grok-4.5.toml | 17 +- providers/aihubmix/models/grok-4.6.toml | 11 +- providers/aihubmix/models/grok-build-0.1.toml | 9 +- providers/aihubmix/models/hy3-free.toml | 13 + providers/aihubmix/models/hy3-preview.toml | 25 +- providers/aihubmix/models/hy3.toml | 14 + providers/aihubmix/models/hy4-preview.toml | 20 ++ .../aihubmix/models/kimi-k2-thinking.toml | 8 + providers/aihubmix/models/kimi-k2.5.toml | 18 +- providers/aihubmix/models/kimi-k2.6.toml | 16 +- .../models/kimi-k2.7-code-highspeed.toml | 9 +- providers/aihubmix/models/kimi-k2.7-code.toml | 9 +- providers/aihubmix/models/kimi-k3.toml | 13 +- .../models/llama-3.3-70b-instruct.toml | 10 + providers/aihubmix/models/longcat-2.0.toml | 9 + .../aihubmix/models/mimo-v2-flash-free.toml | 11 + providers/aihubmix/models/mimo-v2-flash.toml | 18 + providers/aihubmix/models/mimo-v2-omni.toml | 14 + providers/aihubmix/models/mimo-v2-pro.toml | 17 + providers/aihubmix/models/mimo-v2.5-pro.toml | 13 + providers/aihubmix/models/mimo-v2.5.toml | 13 + providers/aihubmix/models/minimax-m2.7.toml | 11 +- .../models/nemotron-3-nano-30b-a3b-free.toml | 15 + ...on-3-nano-omni-30b-a3b-reasoning-free.toml | 14 + .../nemotron-3-super-120b-a12b-free.toml | 11 + .../nemotron-3-ultra-550b-a55b-free.toml | 14 + .../nemotron-3.5-content-safety-free.toml | 11 + .../models/nemotron-3.5-lightning-free.toml | 11 + .../models/nemotron-nano-12b-v2-vl-free.toml | 11 + .../models/nemotron-nano-9b-v2-free.toml | 11 + providers/aihubmix/models/o1-preview.toml | 11 + providers/aihubmix/models/o1-pro.toml | 12 + providers/aihubmix/models/o1.toml | 12 + providers/aihubmix/models/o3-mini.toml | 12 + providers/aihubmix/models/o3-pro.toml | 7 + providers/aihubmix/models/o3.toml | 13 + providers/aihubmix/models/o4-mini.toml | 10 + providers/aihubmix/models/qwen-turbo.toml | 8 + .../models/qwen3-235b-a22b-instruct-2507.toml | 13 + .../aihubmix/models/qwen3-235b-a22b.toml | 11 + .../models/qwen3-coder-30b-a3b-instruct.toml | 20 ++ .../qwen3-coder-480b-a35b-instruct.toml | 20 ++ .../aihubmix/models/qwen3-coder-flash.toml | 24 ++ .../aihubmix/models/qwen3-coder-next.toml | 15 + .../aihubmix/models/qwen3-coder-plus.toml | 22 ++ .../aihubmix/models/qwen3-max-preview.toml | 23 ++ providers/aihubmix/models/qwen3-max.toml | 26 ++ .../models/qwen3-next-80b-a3b-instruct.toml | 14 + .../models/qwen3-next-80b-a3b-thinking.toml | 15 + .../models/qwen3-vl-235b-a22b-instruct.toml | 12 + .../models/qwen3-vl-235b-a22b-thinking.toml | 13 + providers/aihubmix/models/qwen3-vl-plus.toml | 26 ++ .../aihubmix/models/qwen3.5-122b-a10b.toml | 23 ++ providers/aihubmix/models/qwen3.5-27b.toml | 23 ++ .../aihubmix/models/qwen3.5-35b-a3b.toml | 23 ++ .../aihubmix/models/qwen3.5-397b-a17b.toml | 23 ++ providers/aihubmix/models/qwen3.5-flash.toml | 31 ++ providers/aihubmix/models/qwen3.5-plus.toml | 33 ++ providers/aihubmix/models/qwen3.6-27b.toml | 11 + .../aihubmix/models/qwen3.6-35b-a3b.toml | 18 + providers/aihubmix/models/qwen3.6-flash.toml | 31 +- .../aihubmix/models/qwen3.6-max-preview.toml | 26 +- .../models/qwen3.6-plus-preview-free.toml | 14 + providers/aihubmix/models/qwen3.6-plus.toml | 30 +- providers/aihubmix/models/qwen3.7-flash.toml | 33 +- providers/aihubmix/models/qwen3.7-max.toml | 18 +- providers/aihubmix/models/qwen3.7-plus.toml | 25 +- .../aihubmix/models/qwen3.8-2.4t-a95b.toml | 19 +- providers/aihubmix/models/qwen3.8-flash.toml | 17 + .../aihubmix/models/qwen3.8-max-preview.toml | 18 + providers/aihubmix/models/qwen3.8-max.toml | 22 +- providers/aihubmix/models/solar-pro4.toml | 8 + providers/aihubmix/models/step-3.5-flash.toml | 9 + providers/aihubmix/models/step-3.7-flash.toml | 14 + .../models/xiaomi-mimo-v2-omni-free.toml | 13 + .../models/xiaomi-mimo-v2-pro-free.toml | 10 + .../models/xiaomi-mimo-v2.5-free.toml | 10 +- .../models/xiaomi-mimo-v2.5-pro-free.toml | 10 +- .../aihubmix/models/zai-glm-5-turbo.toml | 13 + sync.md | 13 +- 226 files changed, 3019 insertions(+), 922 deletions(-) create mode 100644 providers/aihubmix/models/claude-3-haiku-20240307.toml create mode 100644 providers/aihubmix/models/claude-fable-5-1.toml create mode 100644 providers/aihubmix/models/claude-haiku-4-5.toml create mode 100644 providers/aihubmix/models/claude-opus-4-1.toml create mode 100644 providers/aihubmix/models/claude-opus-4-5-think.toml create mode 100644 providers/aihubmix/models/claude-opus-4-5.toml create mode 100644 providers/aihubmix/models/claude-sonnet-4-5-think.toml create mode 100644 providers/aihubmix/models/claude-sonnet-4-5.toml create mode 100644 providers/aihubmix/models/coding-glm-4.6-free.toml create mode 100644 providers/aihubmix/models/coding-glm-4.6.toml create mode 100644 providers/aihubmix/models/coding-glm-4.7-free.toml create mode 100644 providers/aihubmix/models/coding-glm-4.7.toml create mode 100644 providers/aihubmix/models/coding-glm-5-free.toml create mode 100644 providers/aihubmix/models/coding-glm-5-turbo-free.toml create mode 100644 providers/aihubmix/models/coding-glm-5-turbo.toml create mode 100644 providers/aihubmix/models/coding-glm-5.2-free.toml create mode 100644 providers/aihubmix/models/coding-glm-5.2.toml create mode 100644 providers/aihubmix/models/coding-glm-5.3-free.toml create mode 100644 providers/aihubmix/models/coding-glm-5.3.toml create mode 100644 providers/aihubmix/models/coding-glm-5.toml create mode 100644 providers/aihubmix/models/coding-kimi-k3-free.toml create mode 100644 providers/aihubmix/models/coding-kimi-k3.toml create mode 100644 providers/aihubmix/models/command-a-03-2025.toml create mode 100644 providers/aihubmix/models/command-a-plus-05-2026.toml create mode 100644 providers/aihubmix/models/command-r-08-2024.toml create mode 100644 providers/aihubmix/models/command-r-plus-08-2024.toml create mode 100644 providers/aihubmix/models/deepseek-v3.2-think.toml create mode 100644 providers/aihubmix/models/deepseek-v3.2.toml create mode 100644 providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml create mode 100644 providers/aihubmix/models/deepseek-v4-flash.toml create mode 100644 providers/aihubmix/models/deepseek-v4-pro.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-image.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-lite-nothink.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-lite.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-nothink.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-search.toml create mode 100644 providers/aihubmix/models/gemini-2.5-pro-search.toml create mode 100644 providers/aihubmix/models/gemini-3-flash-preview-free.toml create mode 100644 providers/aihubmix/models/gemini-3-flash-preview-search.toml create mode 100644 providers/aihubmix/models/gemini-3-pro-image-preview.toml create mode 100644 providers/aihubmix/models/gemini-3-pro-image.toml create mode 100644 providers/aihubmix/models/gemini-3.1-flash-image-preview.toml create mode 100644 providers/aihubmix/models/gemini-3.1-flash-image.toml create mode 100644 providers/aihubmix/models/gemini-3.1-flash-lite-image.toml create mode 100644 providers/aihubmix/models/gemini-3.1-flash-lite-nothink.toml create mode 100644 providers/aihubmix/models/gemini-3.1-pro-preview-search.toml create mode 100644 providers/aihubmix/models/gemini-3.5-flash-lite-free.toml create mode 100644 providers/aihubmix/models/gemini-3.5-flash-lite.toml create mode 100644 providers/aihubmix/models/gemini-3.6-flash-free.toml create mode 100644 providers/aihubmix/models/gemini-3.6-flash.toml create mode 100644 providers/aihubmix/models/gemini-3.7-flash-free.toml create mode 100644 providers/aihubmix/models/gemini-3.8-flash-free.toml create mode 100644 providers/aihubmix/models/gemini-3.8-flash.toml create mode 100644 providers/aihubmix/models/gemma-4-26b-a4b-it-free.toml create mode 100644 providers/aihubmix/models/gemma-4-26b-a4b-it.toml create mode 100644 providers/aihubmix/models/gemma-4-31b-it-free.toml create mode 100644 providers/aihubmix/models/gemma-4-31b-it.toml create mode 100644 providers/aihubmix/models/glm-4.5v.toml create mode 100644 providers/aihubmix/models/glm-4.6.toml create mode 100644 providers/aihubmix/models/glm-4.6v.toml create mode 100644 providers/aihubmix/models/glm-4.7-flash-free.toml create mode 100644 providers/aihubmix/models/glm-4.7.toml create mode 100644 providers/aihubmix/models/glm-5-turbo.toml create mode 100644 providers/aihubmix/models/glm-5.1.toml create mode 100644 providers/aihubmix/models/glm-5.toml create mode 100644 providers/aihubmix/models/gpt-4.1-free.toml create mode 100644 providers/aihubmix/models/gpt-4.1-mini-free.toml create mode 100644 providers/aihubmix/models/gpt-4.1-mini.toml create mode 100644 providers/aihubmix/models/gpt-4.1-nano-free.toml create mode 100644 providers/aihubmix/models/gpt-4.1-nano.toml create mode 100644 providers/aihubmix/models/gpt-4.1.toml create mode 100644 providers/aihubmix/models/gpt-4o-2024-11-20.toml create mode 100644 providers/aihubmix/models/gpt-4o-free.toml create mode 100644 providers/aihubmix/models/gpt-4o-mini.toml create mode 100644 providers/aihubmix/models/gpt-4o.toml create mode 100644 providers/aihubmix/models/gpt-5-chat-latest.toml create mode 100644 providers/aihubmix/models/gpt-5-codex.toml create mode 100644 providers/aihubmix/models/gpt-5-mini.toml create mode 100644 providers/aihubmix/models/gpt-5-nano.toml create mode 100644 providers/aihubmix/models/gpt-5-pro.toml create mode 100644 providers/aihubmix/models/gpt-5.1-chat-latest.toml create mode 100644 providers/aihubmix/models/gpt-5.1-codex-max.toml create mode 100644 providers/aihubmix/models/gpt-5.2-chat-latest.toml create mode 100644 providers/aihubmix/models/gpt-5.2-pro.toml create mode 100644 providers/aihubmix/models/gpt-5.3-chat-latest.toml create mode 100644 providers/aihubmix/models/gpt-5.4-nano.toml create mode 100644 providers/aihubmix/models/gpt-5.4-pro.toml create mode 100644 providers/aihubmix/models/gpt-5.5-free.toml create mode 100644 providers/aihubmix/models/gpt-5.5-pro.toml create mode 100644 providers/aihubmix/models/gpt-5.6-sol-disc.toml create mode 100644 providers/aihubmix/models/gpt-5.toml create mode 100644 providers/aihubmix/models/gpt-6-astra.toml create mode 100644 providers/aihubmix/models/gpt-image-2-free.toml create mode 100644 providers/aihubmix/models/gpt-oss-120b.toml create mode 100644 providers/aihubmix/models/gpt-oss-20b-free.toml create mode 100644 providers/aihubmix/models/gpt-oss-20b.toml create mode 100644 providers/aihubmix/models/hy3-free.toml create mode 100644 providers/aihubmix/models/hy3.toml create mode 100644 providers/aihubmix/models/hy4-preview.toml create mode 100644 providers/aihubmix/models/kimi-k2-thinking.toml create mode 100644 providers/aihubmix/models/llama-3.3-70b-instruct.toml create mode 100644 providers/aihubmix/models/longcat-2.0.toml create mode 100644 providers/aihubmix/models/mimo-v2-flash-free.toml create mode 100644 providers/aihubmix/models/mimo-v2-flash.toml create mode 100644 providers/aihubmix/models/mimo-v2-omni.toml create mode 100644 providers/aihubmix/models/mimo-v2-pro.toml create mode 100644 providers/aihubmix/models/mimo-v2.5-pro.toml create mode 100644 providers/aihubmix/models/mimo-v2.5.toml create mode 100644 providers/aihubmix/models/nemotron-3-nano-30b-a3b-free.toml create mode 100644 providers/aihubmix/models/nemotron-3-nano-omni-30b-a3b-reasoning-free.toml create mode 100644 providers/aihubmix/models/nemotron-3-super-120b-a12b-free.toml create mode 100644 providers/aihubmix/models/nemotron-3-ultra-550b-a55b-free.toml create mode 100644 providers/aihubmix/models/nemotron-3.5-content-safety-free.toml create mode 100644 providers/aihubmix/models/nemotron-3.5-lightning-free.toml create mode 100644 providers/aihubmix/models/nemotron-nano-12b-v2-vl-free.toml create mode 100644 providers/aihubmix/models/nemotron-nano-9b-v2-free.toml create mode 100644 providers/aihubmix/models/o1-preview.toml create mode 100644 providers/aihubmix/models/o1-pro.toml create mode 100644 providers/aihubmix/models/o1.toml create mode 100644 providers/aihubmix/models/o3-mini.toml create mode 100644 providers/aihubmix/models/o3-pro.toml create mode 100644 providers/aihubmix/models/o3.toml create mode 100644 providers/aihubmix/models/o4-mini.toml create mode 100644 providers/aihubmix/models/qwen-turbo.toml create mode 100644 providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml create mode 100644 providers/aihubmix/models/qwen3-235b-a22b.toml create mode 100644 providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml create mode 100644 providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml create mode 100644 providers/aihubmix/models/qwen3-coder-flash.toml create mode 100644 providers/aihubmix/models/qwen3-coder-next.toml create mode 100644 providers/aihubmix/models/qwen3-coder-plus.toml create mode 100644 providers/aihubmix/models/qwen3-max-preview.toml create mode 100644 providers/aihubmix/models/qwen3-max.toml create mode 100644 providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml create mode 100644 providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml create mode 100644 providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml create mode 100644 providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml create mode 100644 providers/aihubmix/models/qwen3-vl-plus.toml create mode 100644 providers/aihubmix/models/qwen3.5-122b-a10b.toml create mode 100644 providers/aihubmix/models/qwen3.5-27b.toml create mode 100644 providers/aihubmix/models/qwen3.5-35b-a3b.toml create mode 100644 providers/aihubmix/models/qwen3.5-397b-a17b.toml create mode 100644 providers/aihubmix/models/qwen3.5-flash.toml create mode 100644 providers/aihubmix/models/qwen3.5-plus.toml create mode 100644 providers/aihubmix/models/qwen3.6-27b.toml create mode 100644 providers/aihubmix/models/qwen3.6-35b-a3b.toml create mode 100644 providers/aihubmix/models/qwen3.6-plus-preview-free.toml create mode 100644 providers/aihubmix/models/qwen3.8-flash.toml create mode 100644 providers/aihubmix/models/qwen3.8-max-preview.toml create mode 100644 providers/aihubmix/models/solar-pro4.toml create mode 100644 providers/aihubmix/models/step-3.5-flash.toml create mode 100644 providers/aihubmix/models/step-3.7-flash.toml create mode 100644 providers/aihubmix/models/xiaomi-mimo-v2-omni-free.toml create mode 100644 providers/aihubmix/models/xiaomi-mimo-v2-pro-free.toml create mode 100644 providers/aihubmix/models/zai-glm-5-turbo.toml diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index 36134bdab72..f42a1c56c8a 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -1,6 +1,10 @@ +import path from "node:path"; + import { z } from "zod"; +import { describeModel } from "../../describe.js"; import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js"; +import { factorBaseModel } from "./openrouter.js"; const API_ENDPOINT = "https://aihubmix.com/api/v1/models?type=llm"; @@ -11,6 +15,33 @@ const Pricing = z output: z.number().nullish(), cache_read: z.number().nullish(), cache_write: z.number().nullish(), + tiers: z + .array( + z + .object({ + tier: z.object({ + type: z.string().nullish(), + size: z.number(), + }), + input: z.number().nullish(), + output: z.number().nullish(), + cache_read: z.number().nullish(), + cache_write: z.number().nullish(), + }) + .passthrough(), + ) + .nullish(), + }) + .passthrough(); + +/** + * AIHubMix ships `default` alongside `type`/`values`, which the catalog's strict + * ReasoningOption rejects, so the extra key is dropped during translation. + */ +const ReasoningOption = z + .object({ + type: z.string(), + values: z.array(z.string()).nullish(), }) .passthrough(); @@ -18,7 +49,23 @@ export const AihubmixModel = z .object({ model_id: z.string().min(1), model_name: z.string().nullish(), + developer_id: z.number().nullish(), + desc: z.string().nullish(), pricing: Pricing.nullish(), + features: z.string().nullish(), + input_modalities: z.string().nullish(), + output_modalities: z.string().nullish(), + context_length: z.number().nullish(), + max_output: z.number().nullish(), + reasoning: z.boolean().nullish(), + reasoning_options: z.array(ReasoningOption).nullish(), + tool_call: z.boolean().nullish(), + release_date: z.string().nullish(), + last_updated: z.string().nullish(), + // Not served yet; read opportunistically so creates unblock without a code + // change once AIHubMix adds them. + knowledge: z.string().nullish(), + open_weights: z.boolean().nullish(), retire_stage: z.string().nullish(), }) .passthrough(); @@ -33,35 +80,80 @@ export const AihubmixResponse = z export type AihubmixModel = z.infer; /** - * AIHubMix relays ~400 upstream models while the catalog curates a much smaller - * hand-verified subset, so this sync only updates existing TOMLs - * (`skipCreates`) and treats the AIHubMix endpoint as authoritative for pricing - * and deprecation status only. - * - * Everything else in the authored TOMLs is preserved as hand-authored, because - * the endpoint reports the relay's own conservative defaults rather than the - * upstream model's real capabilities: `context_length` is capped per relay - * (Claude Opus 4.6 reports 200K against its 1M window), `max_output` is quoted - * per default request rather than per model, and `input_modalities` never lists - * `pdf` even for models the provider does accept PDFs for. Its free-text - * `features` list likewise mixes synonyms (`thinking` vs `reasoning`, `tools` - * vs `tool_calling`) and never exposes accepted reasoning effort levels, so - * capability flags, `reasoning_options`, `base_model` inheritance, and the - * per-model `[provider]` protocol overrides all stay authored. + * AIHubMix relays upstream models under its own IDs, so a relay is factored onto + * the lab metadata it serves whenever that metadata exists — the relay then only + * records what it actually changes (price, reasoning controls, limits). * - * Routing aliases such as `alicloud-glm-5.1` and `deep-deepseek-v4-pro` are - * served but not listed by the endpoint, so local files missing from the - * response are retained (`deleteMissing: false`). + * `developer_id` is AIHubMix's own lab identifier and is the only reliable way + * back to a catalog namespace: relay IDs carry routing prefixes (`coding-`, + * `alicloud-`) and suffixes (`-free`, `-think`, `-nothink`) that are AIHubMix + * routing modes rather than distinct upstream models. + */ +const LAB_BY_DEVELOPER: Record = { + 2: "anthropic", + 3: "microsoft", + 4: "bytedance-seed", + 5: "zhipuai", + 6: "cohere", + 7: "deepseek", + 8: "google", + 9: "xai", + 10: "mistral", + 11: "meta", + 12: "openai", + 13: "alibaba", + 15: "moonshotai", + 16: "stepfun", + 17: "nvidia", + 18: "minimax", + 24: "tencent", + 28: "meituan", + 29: "inclusionai", + 31: "xiaomi", + 44: "upstage", +}; + +/** Routing prefixes and suffixes that select a mode, not a different model. */ +const ROUTING_PREFIXES = ["coding-", "alicloud-", "deep-", "zai-", "anthropic-", "xiaomi-", "openai-"]; +const ROUTING_SUFFIXES = ["-free", "-think", "-nothink", "-search", "-preview", "-disc", "-exp"]; + +/** Catalog effort levels; AIHubMix spells two of them differently. */ +const EFFORT_ALIASES: Record = { no_think: "none", instant: "minimal" }; +const EFFORT_VALUES = new Set([ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max", + "default", +]); + +let labMetadataIDs: Set | undefined; + +/** + * The catalog rejects a `base_model` that resolves to nothing, so relays are + * only factored onto metadata that is actually present on disk. */ +async function readLabMetadataIDs(modelsDir: string) { + const metadataDir = path.join(path.dirname(path.dirname(path.dirname(modelsDir))), "models"); + const ids = new Set(); + for await (const file of new Bun.Glob("**/*.toml").scan({ cwd: metadataDir, followSymlinks: true })) { + ids.add(file.split(path.sep).join("/").slice(0, -5)); + } + return ids; +} + export const aihubmix = { id: "aihubmix", name: "AIHubMix", modelsDir: "providers/aihubmix/models", - skipCreates: true, trackMissingModels: true, + // Routing aliases such as `alicloud-glm-5.1` are served but unlisted, so a + // local file absent from the response is retained rather than deleted. deleteMissing: false, sourceID(model) { - // Deprecated relays are not catalog additions worth an issue. return model.retire_stage === "deprecated" ? undefined : model.model_id; }, missingNotice(paths) { @@ -70,7 +162,14 @@ export const aihubmix = { `AIHubMix no longer lists ${file}; confirm it is still a served routing alias or deprecate it.`, ); }, + skippedNotice(ids) { + return ids.map( + (id) => + `AIHubMix lists ${id} but the response carries neither a resolvable base model nor the release_date/open_weights a standalone entry needs.`, + ); + }, async fetchModels() { + labMetadataIDs = await readLabMetadataIDs(this.modelsDir); const response = await fetch(process.env.AIHUBMIX_MODELS_URL ?? API_ENDPOINT); if (!response.ok) { throw new Error(`AIHubMix models request failed: ${response.status} ${response.statusText}`); @@ -81,28 +180,141 @@ export const aihubmix = { return AihubmixResponse.parse(raw).data; }, translateModel(model, context) { - const authored = context.authored(model.model_id); - if (authored === undefined) return undefined; - return { - id: model.model_id, - model: buildAihubmixModel(model, authored), - }; + const existing = context.existing(model.model_id); + const built = buildAihubmixModel(model, existing, labMetadataIDs); + if (built === undefined) return undefined; + return { id: model.model_id, model: built }; }, } satisfies SyncProvider; -export function buildAihubmixModel(model: AihubmixModel, authored: ExistingModel): SyncedModel { - const { id: _id, ...preserved } = authored; +export function buildAihubmixModel( + model: AihubmixModel, + existing: ExistingModel | undefined, + labIDs: Set | undefined = labMetadataIDs, +): SyncedModel | undefined { + const input = modalities(model.input_modalities, existing?.modalities?.input ?? ["text"]); + const output = modalities(model.output_modalities, existing?.modalities?.output ?? ["text"]); + const features = new Set((model.features ?? "").split(",").map((value) => value.trim())); + const reasoning = model.reasoning ?? existing?.reasoning ?? false; + const toolCall = model.tool_call ?? existing?.tool_call ?? false; + const structuredOutput = features.has("structured_outputs") || existing?.structured_output; + const name = model.model_name ?? existing?.name; + const limit = { + context: tokens(model.context_length) ?? existing?.limit?.context, + output: tokens(model.max_output) ?? existing?.limit?.output, + }; + const shared = { + attachment: input.some((value) => value !== "text"), + reasoning, + reasoning_options: reasoningOptions(model) ?? existing?.reasoning_options, + tool_call: toolCall, + structured_output: structuredOutput, + // AIHubMix serves no temperature or interleaved flags; keep what was authored. + temperature: existing?.temperature, + interleaved: existing?.interleaved, + status: model.retire_stage === "deprecated" ? ("deprecated" as const) : existing?.status, + modalities: { input, output }, + limit, + cost: buildCost(model.pricing, existing?.cost), + }; + + const base = existing?.base_model ?? resolveBaseModel(model, labIDs); + if (base !== undefined) { + return factorBaseModel( + base, + { name: existing?.name, description: existing?.description, ...shared }, + limit, + existing?.base_model === base ? existing.base_model_omit : undefined, + ); + } + + // A standalone entry must carry every required catalog field itself. AIHubMix + // dates only 52 of its 415 models and serves no open_weights flag, so a relay + // with neither metadata to inherit nor those fields is reported rather than + // written with invented values. + const releaseDate = model.release_date ?? existing?.release_date; + const openWeights = model.open_weights ?? existing?.open_weights; + if (name === undefined || releaseDate === undefined || openWeights === undefined) { + return existing === undefined ? undefined : (existing as SyncedModel); + } + return { - ...preserved, - cost: buildCost(model.pricing, authored.cost), - status: model.retire_stage === "deprecated" ? "deprecated" : authored.status, - } as SyncedModel; + ...shared, + name, + description: + existing?.description ?? + model.desc ?? + describeModel({ + id: model.model_id, + providerId: "aihubmix", + name, + reasoning, + tool_call: toolCall, + structured_output: structuredOutput, + open_weights: openWeights, + limit, + modalities: { input, output }, + }), + family: existing?.family, + release_date: releaseDate, + last_updated: model.last_updated ?? model.release_date ?? existing?.last_updated ?? releaseDate, + knowledge: model.knowledge ?? existing?.knowledge, + open_weights: openWeights, + } as SyncedFullModel; +} + +function resolveBaseModel(model: AihubmixModel, labIDs: Set | undefined) { + const lab = LAB_BY_DEVELOPER[model.developer_id ?? -1]; + if (lab === undefined || labIDs === undefined) return undefined; + for (const candidate of baseCandidates(model.model_id)) { + const id = `${lab}/${candidate}`; + if (labIDs.has(id)) return id; + } + return undefined; +} + +/** Longest match first: strip routing prefixes, then routing suffixes. */ +function baseCandidates(modelID: string) { + const bare = modelID.split("/").at(-1) ?? modelID; + const candidates = new Set([bare]); + for (const prefix of ROUTING_PREFIXES) { + if (bare.startsWith(prefix)) candidates.add(bare.slice(prefix.length)); + } + for (const suffix of ROUTING_SUFFIXES) { + for (const candidate of [...candidates]) { + if (candidate.endsWith(suffix)) candidates.add(candidate.slice(0, -suffix.length)); + } + } + return candidates; +} + +function reasoningOptions(model: AihubmixModel): SyncedFullModel["reasoning_options"] { + if (model.reasoning_options == null) return undefined; + const options = model.reasoning_options.flatMap((option) => { + if (option.type === "toggle" || option.type === "budget_tokens") { + return [{ type: option.type }]; + } + if (option.type !== "effort") return []; + const values = (option.values ?? []) + .map((value) => EFFORT_ALIASES[value] ?? value) + .filter((value) => EFFORT_VALUES.has(value)); + return values.length > 0 ? [{ type: "effort" as const, values }] : []; + }); + return options.length > 0 ? (options as SyncedFullModel["reasoning_options"]) : undefined; +} + +function modalities(value: string | null | undefined, fallback: string[]) { + const parsed = (value ?? "") + .split(",") + .map((entry) => entry.trim()) + .filter((entry) => ["text", "audio", "image", "video", "pdf"].includes(entry)); + return (parsed.length > 0 ? parsed : fallback) as SyncedFullModel["modalities"]["input"]; } /** - * AIHubMix omits a price field when the model has no such rate, but a partial - * authored `[cost]` (tiers, reasoning, audio rates) still carries values the - * endpoint does not model, so those are kept. + * AIHubMix omits a price field when the model has no such rate, so an omitted + * field means "not offered" and an authored value is only kept when the + * endpoint quotes nothing at all for the model. */ function buildCost( pricing: AihubmixModel["pricing"], @@ -113,22 +325,43 @@ function buildCost( const output = price(pricing.output); if (input === undefined || output === undefined) return authored; - const cacheRead = price(pricing.cache_read); return { - ...authored, input, output, - // AIHubMix echoes the input price into `cache_read` for models it has no - // cached rate for (35 of the 301 priced entries carry a nonzero price this - // way; 51 more are free models reporting 0 across the board, where this is - // a no-op), so an equal value means "not quoted" rather than "cached reads - // cost full price" — taking it literally would overstate e.g. Gemini 3.1 - // Flash Lite by 10x against the $0.025 every other provider lists. - cache_read: cacheRead === input ? authored?.cache_read : (cacheRead ?? authored?.cache_read), - cache_write: price(pricing.cache_write) ?? authored?.cache_write, + cache_read: price(pricing.cache_read), + cache_write: price(pricing.cache_write), + tiers: costTiers(pricing) ?? authored?.tiers, }; } +function costTiers(pricing: NonNullable) { + const tiers = (pricing.tiers ?? []).flatMap((tier) => { + const input = price(tier.input); + const output = price(tier.output); + if (input === undefined || output === undefined) return []; + return [ + { + tier: { type: tier.tier.type ?? "context", size: tier.tier.size }, + input, + output, + cache_read: price(tier.cache_read), + cache_write: price(tier.cache_write), + }, + ]; + }); + return tiers.length > 0 ? (tiers as NonNullable["tiers"]) : undefined; +} + +/** + * AIHubMix sends 0 for a limit it does not know rather than omitting the field — + * 102 of 415 models quote `max_output: 0` — so 0 is read as absent. A model that + * truly emitted no tokens would not be servable. + */ +function tokens(value: number | null | undefined) { + if (value == null || !Number.isFinite(value) || value <= 0) return undefined; + return value; +} + function price(value: number | null | undefined) { if (value == null || !Number.isFinite(value) || value < 0) return undefined; return Math.round(value * 1_000_000) / 1_000_000; diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index d55d0be470c..0fa06bc0577 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -1190,7 +1190,7 @@ test("tracks missing models except for unreliable first-party inventories", () = expect(openai.trackMissingModels).toBe(false); expect(pioneer.skipCreates).toBe(true); expect(pioneer.trackMissingModels).toBe(true); - expect(aihubmix.skipCreates).toBe(true); + expect(aihubmix.skipCreates).toBeUndefined(); expect(aihubmix.trackMissingModels).toBe(true); expect(ofox.skipCreates).toBe(true); expect(ofox.trackMissingModels).toBe(true); @@ -5026,6 +5026,7 @@ function aihubmixModel(overrides: Partial = {}): AihubmixModel { return { model_id: "gemini-3.1-flash-lite", model_name: "Gemini 3.1 Flash Lite", + developer_id: 8, pricing: { input: 0.25, output: 1.5, cache_read: 0.025 }, ...overrides, }; @@ -5042,39 +5043,116 @@ const aihubmixAuthored: ExistingModel = { modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, }; -test("syncs AIHubMix pricing while preserving hand-authored capabilities", () => { +const aihubmixLabIDs = new Set(["google/gemini-3.1-flash-lite", "openai/gpt-5.5"]); + +test("factors an AIHubMix relay onto the lab metadata it serves", () => { + const model = buildAihubmixModel( + aihubmixModel({ + model_id: "gemini-3.1-flash-lite-nothink", + context_length: 1_048_576, + max_output: 65_536, + input_modalities: "text,image", + }), + undefined, + aihubmixLabIDs, + ); + expect(model).toMatchObject({ + base_model: "google/gemini-3.1-flash-lite", + cost: { input: 0.25, output: 1.5, cache_read: 0.025 }, + }); + // The relay only records what it actually changes. + expect(model).not.toHaveProperty("release_date"); + expect(model).not.toHaveProperty("open_weights"); +}); + +test("routes AIHubMix prefixes and suffixes back to the upstream lab model", () => { + for (const id of ["coding-gemini-3.1-flash-lite", "gemini-3.1-flash-lite-free"]) { + const model = buildAihubmixModel(aihubmixModel({ model_id: id }), undefined, aihubmixLabIDs); + expect(model).toMatchObject({ base_model: "google/gemini-3.1-flash-lite" }); + } +}); + +test("skips an AIHubMix relay with neither base metadata nor standalone fields", () => { + const model = buildAihubmixModel( + aihubmixModel({ model_id: "house-brand-v1", developer_id: 999 }), + undefined, + aihubmixLabIDs, + ); + expect(model).toBeUndefined(); +}); + +test("normalizes AIHubMix reasoning options to the catalog vocabulary", () => { + const model = buildAihubmixModel( + aihubmixModel({ + reasoning: true, + // `default` is AIHubMix-only, and it spells two efforts differently. + reasoning_options: [ + { type: "effort", values: ["no_think", "instant", "high", "bogus"], default: "high" }, + { type: "toggle", default: true }, + ] as AihubmixModel["reasoning_options"], + }), + undefined, + aihubmixLabIDs, + ); + expect(model?.reasoning_options).toEqual([ + { type: "effort", values: ["none", "minimal", "high"] }, + { type: "toggle" }, + ]); +}); + +test("reads a zero AIHubMix limit as absent rather than a real ceiling", () => { + // 102 of 415 models quote `max_output: 0` for a limit the endpoint does not know. + const zeroed = buildAihubmixModel( + aihubmixModel({ context_length: 262_144, max_output: 0 }), + aihubmixAuthored, + aihubmixLabIDs, + ); + const omitted = buildAihubmixModel( + aihubmixModel({ context_length: 262_144 }), + aihubmixAuthored, + aihubmixLabIDs, + ); + expect(zeroed?.limit).toEqual(omitted?.limit); + expect(zeroed?.limit?.output).not.toBe(0); +}); + +test("syncs AIHubMix pricing over the authored entry", () => { const model = buildAihubmixModel( aihubmixModel({ pricing: { input: 0.2, output: 1.2, cache_read: 0.02, cache_write: 0.25 } }), aihubmixAuthored, + aihubmixLabIDs, ); - expect(model.cost).toMatchObject({ input: 0.2, output: 1.2, cache_read: 0.02, cache_write: 0.25 }); - // The endpoint reports relay defaults for these, so authored values win. - expect(model.limit).toEqual({ context: 1_048_576, output: 65_536 }); - expect(model.modalities).toEqual({ - input: ["text", "image", "audio", "video", "pdf"], - output: ["text"], - }); - expect(model.reasoning_options).toEqual([ - { type: "effort", values: ["minimal", "low", "medium", "high"] }, - ]); + expect(model?.cost).toMatchObject({ + input: 0.2, + output: 1.2, + cache_read: 0.02, + cache_write: 0.25, + }); }); -test("ignores AIHubMix cache_read that merely echoes the input price", () => { +test("drops an authored AIHubMix rate the endpoint no longer quotes", () => { + // An omitted price field means the model has no such rate, not that it is unknown. const model = buildAihubmixModel( - aihubmixModel({ pricing: { input: 0.25, output: 1.5, cache_read: 0.25 } }), + aihubmixModel({ pricing: { input: 0.25, output: 1.5 } }), aihubmixAuthored, + aihubmixLabIDs, ); - expect(model.cost?.cache_read).toBe(0.025); + expect(model?.cost?.cache_read).toBeUndefined(); + expect(model?.cost?.cache_write).toBeUndefined(); }); -test("keeps authored AIHubMix pricing when the endpoint quotes no rate", () => { - const model = buildAihubmixModel(aihubmixModel({ pricing: null }), aihubmixAuthored); - expect(model.cost).toEqual(aihubmixAuthored.cost); +test("keeps authored AIHubMix pricing when the endpoint quotes no rate at all", () => { + const model = buildAihubmixModel( + aihubmixModel({ pricing: null }), + aihubmixAuthored, + aihubmixLabIDs, + ); + expect(model?.cost).toEqual(aihubmixAuthored.cost); }); test("marks retired AIHubMix relays deprecated and stops tracking them", () => { const retired = aihubmixModel({ retire_stage: "deprecated" }); - expect(buildAihubmixModel(retired, aihubmixAuthored).status).toBe("deprecated"); + expect(buildAihubmixModel(retired, aihubmixAuthored, aihubmixLabIDs)?.status).toBe("deprecated"); expect(aihubmix.sourceID?.(retired)).toBeUndefined(); expect(aihubmix.sourceID?.(aihubmixModel())).toBe("gemini-3.1-flash-lite"); }); diff --git a/providers/aihubmix/models/claude-3-haiku-20240307.toml b/providers/aihubmix/models/claude-3-haiku-20240307.toml new file mode 100644 index 00000000000..d404c3f9957 --- /dev/null +++ b/providers/aihubmix/models/claude-3-haiku-20240307.toml @@ -0,0 +1,9 @@ +base_model = "anthropic/claude-3-haiku-20240307" +tool_call = false + +[cost] +input = 0.275 +output = 1.375 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/claude-fable-5-1.toml b/providers/aihubmix/models/claude-fable-5-1.toml new file mode 100644 index 00000000000..4896a255a1f --- /dev/null +++ b/providers/aihubmix/models/claude-fable-5-1.toml @@ -0,0 +1,15 @@ +base_model = "anthropic/claude-fable-5-1" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 11 +output = 55 +cache_read = 0.275 +cache_write = 13.75 diff --git a/providers/aihubmix/models/claude-fable-5.toml b/providers/aihubmix/models/claude-fable-5.toml index afa2221178b..c7e1152cec4 100644 --- a/providers/aihubmix/models/claude-fable-5.toml +++ b/providers/aihubmix/models/claude-fable-5.toml @@ -1,18 +1,17 @@ base_model = "anthropic/claude-fable-5" structured_output = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + interleaved = true +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + [cost] input = 11 output = 55 cache_read = 1.1 cache_write = 13.75 - -[limit] -context = 1_000_000 -output = 128_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/claude-haiku-4-5.toml b/providers/aihubmix/models/claude-haiku-4-5.toml new file mode 100644 index 00000000000..6e5b90ce32e --- /dev/null +++ b/providers/aihubmix/models/claude-haiku-4-5.toml @@ -0,0 +1,14 @@ +base_model = "anthropic/claude-haiku-4-5" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.1 +output = 5.5 +cache_read = 0.11 +cache_write = 1.375 diff --git a/providers/aihubmix/models/claude-opus-4-1.toml b/providers/aihubmix/models/claude-opus-4-1.toml new file mode 100644 index 00000000000..8b33d3e0342 --- /dev/null +++ b/providers/aihubmix/models/claude-opus-4-1.toml @@ -0,0 +1,15 @@ +base_model = "anthropic/claude-opus-4-1" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 16.5 +output = 82.5 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/claude-opus-4-5-think.toml b/providers/aihubmix/models/claude-opus-4-5-think.toml new file mode 100644 index 00000000000..c9dac00196a --- /dev/null +++ b/providers/aihubmix/models/claude-opus-4-5-think.toml @@ -0,0 +1,18 @@ +base_model = "anthropic/claude-opus-4-5" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/aihubmix/models/claude-opus-4-5.toml b/providers/aihubmix/models/claude-opus-4-5.toml new file mode 100644 index 00000000000..c9dac00196a --- /dev/null +++ b/providers/aihubmix/models/claude-opus-4-5.toml @@ -0,0 +1,18 @@ +base_model = "anthropic/claude-opus-4-5" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 diff --git a/providers/aihubmix/models/claude-opus-4-6-think.toml b/providers/aihubmix/models/claude-opus-4-6-think.toml index 58007ca6a42..5226b66dec7 100644 --- a/providers/aihubmix/models/claude-opus-4-6-think.toml +++ b/providers/aihubmix/models/claude-opus-4-6-think.toml @@ -1,20 +1,21 @@ +base_model = "anthropic/claude-opus-4-6" name = "Claude Opus 4.6 Thinking" description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -family = "claude-opus" -release_date = "2026-02-05" -last_updated = "2026-03-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] -temperature = true -tool_call = true structured_output = true -knowledge = "2025-05-31" -open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 5 output = 25 @@ -22,16 +23,8 @@ cache_read = 0.5 cache_write = 6.25 [[cost.tiers]] -tier = { size = 200_000 } +tier = { type = "context", size = 200_000 } input = 10 output = 37.5 -cache_read = 1.0 +cache_read = 1 cache_write = 12.5 - -[limit] -context = 1_000_000 -output = 128_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-4-6.toml b/providers/aihubmix/models/claude-opus-4-6.toml index 57b40db1786..44952c1e2bc 100644 --- a/providers/aihubmix/models/claude-opus-4-6.toml +++ b/providers/aihubmix/models/claude-opus-4-6.toml @@ -1,20 +1,19 @@ -name = "Claude Opus 4.6" +base_model = "anthropic/claude-opus-4-6" description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -family = "claude-opus" -release_date = "2026-02-05" -last_updated = "2026-03-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] -# Native Messages prefers $.thinking.type = "adaptive" with $.output_config.effort = "low"|"medium"|"high"|"max"; enabled budget_tokens >= 1024 is deprecated and must be < $.max_tokens. https://docs.aihubmix.com/cn/api/Claude-Native (accessed 2026-06-25) -temperature = true -tool_call = true structured_output = true -knowledge = "2025-05-31" -open_weights = false interleaved = true +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 5 output = 25 @@ -22,16 +21,8 @@ cache_read = 0.5 cache_write = 6.25 [[cost.tiers]] -tier = { size = 200_000 } +tier = { type = "context", size = 200_000 } input = 10 output = 37.5 -cache_read = 1.0 +cache_read = 1 cache_write = 12.5 - -[limit] -context = 1_000_000 -output = 128_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-4-7-think.toml b/providers/aihubmix/models/claude-opus-4-7-think.toml index e9da40d2fae..d14e3b269be 100644 --- a/providers/aihubmix/models/claude-opus-4-7-think.toml +++ b/providers/aihubmix/models/claude-opus-4-7-think.toml @@ -1,20 +1,18 @@ +base_model = "anthropic/claude-opus-4-7" name = "Claude Opus 4.7 Thinking" description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -family = "claude-opus" -release_date = "2026-04-16" -last_updated = "2026-04-16" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] -temperature = false -tool_call = true structured_output = true -knowledge = "2026-01-31" -open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + [cost] input = 5 output = 25 @@ -22,16 +20,8 @@ cache_read = 0.5 cache_write = 6.25 [[cost.tiers]] -tier = { size = 200_000 } +tier = { type = "context", size = 200_000 } input = 10 output = 37.5 -cache_read = 1.0 +cache_read = 1 cache_write = 12.5 - -[limit] -context = 1_000_000 -output = 128_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-4-7.toml b/providers/aihubmix/models/claude-opus-4-7.toml index 0fb0b1c89f3..f3267b25d12 100644 --- a/providers/aihubmix/models/claude-opus-4-7.toml +++ b/providers/aihubmix/models/claude-opus-4-7.toml @@ -1,20 +1,16 @@ -name = "Claude Opus 4.7" +base_model = "anthropic/claude-opus-4-7" description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -family = "claude-opus" -release_date = "2026-04-16" -last_updated = "2026-04-16" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] -# Native Messages uses $.thinking.type = "adaptive" and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected. https://docs.aihubmix.com/cn/blogs/Claude-Opus4.7 (accessed 2026-06-25) -temperature = false -tool_call = true structured_output = true -knowledge = "2026-01-31" -open_weights = false interleaved = true +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + [cost] input = 5 output = 25 @@ -22,16 +18,8 @@ cache_read = 0.5 cache_write = 6.25 [[cost.tiers]] -tier = { size = 200_000 } +tier = { type = "context", size = 200_000 } input = 10 output = 37.5 -cache_read = 1.0 +cache_read = 1 cache_write = 12.5 - -[limit] -context = 1_000_000 -output = 128_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-4-8-think.toml b/providers/aihubmix/models/claude-opus-4-8-think.toml index 52f226453d2..af514c8d231 100644 --- a/providers/aihubmix/models/claude-opus-4-8-think.toml +++ b/providers/aihubmix/models/claude-opus-4-8-think.toml @@ -1,20 +1,18 @@ base_model = "anthropic/claude-opus-4-8" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] -# Anthropic-compatible /v1/messages: $.thinking.type = "enabled"|"disabled"|"adaptive" (disabled turns reasoning off = toggle) and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected on this Opus tier. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message (accessed 2026-07-02) +structured_output = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + [cost] input = 5 output = 25 cache_read = 0.5 cache_write = 6.25 - -[limit] -context = 200_000 -output = 32_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-4-8.toml b/providers/aihubmix/models/claude-opus-4-8.toml index e8d612cf002..7605942029d 100644 --- a/providers/aihubmix/models/claude-opus-4-8.toml +++ b/providers/aihubmix/models/claude-opus-4-8.toml @@ -1,19 +1,17 @@ base_model = "anthropic/claude-opus-4-8" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] -# Anthropic-compatible /v1/messages: $.thinking.type = "enabled"|"disabled"|"adaptive" (disabled turns reasoning off = toggle) and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected on this Opus tier. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message (accessed 2026-07-02) +structured_output = true interleaved = true +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + [cost] input = 5 output = 25 cache_read = 0.5 cache_write = 6.25 - -[limit] -context = 200_000 -output = 32_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-5.toml b/providers/aihubmix/models/claude-opus-5.toml index 6566c208cd6..70accd1aad8 100644 --- a/providers/aihubmix/models/claude-opus-5.toml +++ b/providers/aihubmix/models/claude-opus-5.toml @@ -1,15 +1,18 @@ # AIHubMix Anthropic-compatible /v1/messages: $.thinking.type = "disabled"|"adaptive" (toggle) and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; verified live 2026-08-11. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message base_model = "anthropic/claude-opus-5" structured_output = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + interleaved = true +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + [cost] input = 5 output = 25 cache_read = 0.5 cache_write = 6.25 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/claude-sonnet-4-5-think.toml b/providers/aihubmix/models/claude-sonnet-4-5-think.toml new file mode 100644 index 00000000000..c0d6579cc19 --- /dev/null +++ b/providers/aihubmix/models/claude-sonnet-4-5-think.toml @@ -0,0 +1,21 @@ +base_model = "anthropic/claude-sonnet-4-5" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 3.3 +output = 16.5 +cache_read = 0.33 +cache_write = 4.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 6.6 +output = 24.75 +cache_read = 0.66 +cache_write = 8.25 diff --git a/providers/aihubmix/models/claude-sonnet-4-5.toml b/providers/aihubmix/models/claude-sonnet-4-5.toml new file mode 100644 index 00000000000..c0d6579cc19 --- /dev/null +++ b/providers/aihubmix/models/claude-sonnet-4-5.toml @@ -0,0 +1,21 @@ +base_model = "anthropic/claude-sonnet-4-5" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 3.3 +output = 16.5 +cache_read = 0.33 +cache_write = 4.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 6.6 +output = 24.75 +cache_read = 0.66 +cache_write = 8.25 diff --git a/providers/aihubmix/models/claude-sonnet-4-6-think.toml b/providers/aihubmix/models/claude-sonnet-4-6-think.toml index 9e3fc461f4e..4e000db48bd 100644 --- a/providers/aihubmix/models/claude-sonnet-4-6-think.toml +++ b/providers/aihubmix/models/claude-sonnet-4-6-think.toml @@ -1,37 +1,33 @@ +base_model = "anthropic/claude-sonnet-4-6" name = "Claude Sonnet 4.6 Thinking" description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" -family = "claude-sonnet" -release_date = "2026-02-17" -last_updated = "2026-03-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] -temperature = true -tool_call = true structured_output = true -knowledge = "2025-08-31" -open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 +input = 3 +output = 15 +cache_read = 0.3 cache_write = 3.75 [[cost.tiers]] -tier = { size = 200_000 } -input = 6.00 -output = 22.50 -cache_read = 0.60 -cache_write = 7.50 +tier = { type = "context", size = 200_000 } +input = 6 +output = 22.5 +cache_read = 0.6 +cache_write = 7.5 [limit] -context = 1_000_000 -output = 64_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] +output = 128_000 diff --git a/providers/aihubmix/models/claude-sonnet-4-6.toml b/providers/aihubmix/models/claude-sonnet-4-6.toml index b252b0f52b3..df8ab08fbfa 100644 --- a/providers/aihubmix/models/claude-sonnet-4-6.toml +++ b/providers/aihubmix/models/claude-sonnet-4-6.toml @@ -1,37 +1,31 @@ -name = "Claude Sonnet 4.6" +base_model = "anthropic/claude-sonnet-4-6" description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" -family = "claude-sonnet" -release_date = "2026-02-17" -last_updated = "2026-03-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] -# Native Messages prefers $.thinking.type = "adaptive" with $.output_config.effort = "low"|"medium"|"high"; enabled budget_tokens >= 1024 is deprecated and must be < $.max_tokens. Chat effort "max" maps to native "high". https://docs.aihubmix.com/cn/api/Claude-Native (accessed 2026-06-25) -temperature = true -tool_call = true structured_output = true -knowledge = "2025-08-31" -open_weights = false interleaved = true +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 +input = 3 +output = 15 +cache_read = 0.3 cache_write = 3.75 [[cost.tiers]] -tier = { size = 200_000 } -input = 6.00 -output = 22.50 -cache_read = 0.60 -cache_write = 7.50 +tier = { type = "context", size = 200_000 } +input = 6 +output = 22.5 +cache_read = 0.6 +cache_write = 7.5 [limit] -context = 1_000_000 -output = 64_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] +output = 128_000 diff --git a/providers/aihubmix/models/claude-sonnet-5.toml b/providers/aihubmix/models/claude-sonnet-5.toml index 880ddcd2f41..6ff86b6a08f 100644 --- a/providers/aihubmix/models/claude-sonnet-5.toml +++ b/providers/aihubmix/models/claude-sonnet-5.toml @@ -1,19 +1,17 @@ base_model = "anthropic/claude-sonnet-5" structured_output = true -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] + interleaved = true -temperature = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] [cost] input = 2 output = 10 cache_read = 0.2 cache_write = 2.5 - -[limit] -context = 1_000_000 -output = 128_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/coding-glm-4.6-free.toml b/providers/aihubmix/models/coding-glm-4.6-free.toml new file mode 100644 index 00000000000..6d567a1b123 --- /dev/null +++ b/providers/aihubmix/models/coding-glm-4.6-free.toml @@ -0,0 +1,12 @@ +base_model = "zhipuai/glm-4.6" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 + +[limit] +context = 200_000 diff --git a/providers/aihubmix/models/coding-glm-4.6.toml b/providers/aihubmix/models/coding-glm-4.6.toml new file mode 100644 index 00000000000..26f0de57cbe --- /dev/null +++ b/providers/aihubmix/models/coding-glm-4.6.toml @@ -0,0 +1,13 @@ +base_model = "zhipuai/glm-4.6" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.06 +output = 0.22 +cache_read = 0.010998 + +[limit] +context = 200_000 diff --git a/providers/aihubmix/models/coding-glm-4.7-free.toml b/providers/aihubmix/models/coding-glm-4.7-free.toml new file mode 100644 index 00000000000..e6f1d52ca64 --- /dev/null +++ b/providers/aihubmix/models/coding-glm-4.7-free.toml @@ -0,0 +1,12 @@ +base_model = "zhipuai/glm-4.7" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 + +[limit] +context = 200_000 diff --git a/providers/aihubmix/models/coding-glm-4.7.toml b/providers/aihubmix/models/coding-glm-4.7.toml new file mode 100644 index 00000000000..679d37ca186 --- /dev/null +++ b/providers/aihubmix/models/coding-glm-4.7.toml @@ -0,0 +1,13 @@ +base_model = "zhipuai/glm-4.7" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.06 +output = 0.22 +cache_read = 0.010998 + +[limit] +context = 200_000 diff --git a/providers/aihubmix/models/coding-glm-5-free.toml b/providers/aihubmix/models/coding-glm-5-free.toml new file mode 100644 index 00000000000..c955d2f4274 --- /dev/null +++ b/providers/aihubmix/models/coding-glm-5-free.toml @@ -0,0 +1,12 @@ +base_model = "zhipuai/glm-5" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 + +[limit] +context = 200_000 diff --git a/providers/aihubmix/models/coding-glm-5-turbo-free.toml b/providers/aihubmix/models/coding-glm-5-turbo-free.toml new file mode 100644 index 00000000000..29147dbed38 --- /dev/null +++ b/providers/aihubmix/models/coding-glm-5-turbo-free.toml @@ -0,0 +1,11 @@ +base_model = "zhipuai/glm-5-turbo" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 + +[limit] +context = 204_800 diff --git a/providers/aihubmix/models/coding-glm-5-turbo.toml b/providers/aihubmix/models/coding-glm-5-turbo.toml new file mode 100644 index 00000000000..50c916e5495 --- /dev/null +++ b/providers/aihubmix/models/coding-glm-5-turbo.toml @@ -0,0 +1,11 @@ +base_model = "zhipuai/glm-5-turbo" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.06 +output = 0.22 + +[limit] +context = 204_800 diff --git a/providers/aihubmix/models/coding-glm-5.1-free.toml b/providers/aihubmix/models/coding-glm-5.1-free.toml index c44e059e3ad..b3fa0b48e57 100644 --- a/providers/aihubmix/models/coding-glm-5.1-free.toml +++ b/providers/aihubmix/models/coding-glm-5.1-free.toml @@ -1,27 +1,13 @@ +base_model = "zhipuai/glm-5.1" name = "Coding GLM 5.1 (free)" description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -family = "glm-free" -release_date = "2026-04-11" -last_updated = "2026-04-11" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -open_weights = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0 output = 0 - -[limit] -context = 200_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/coding-glm-5.1.toml b/providers/aihubmix/models/coding-glm-5.1.toml index ad67b0aeab4..c6a5d070c82 100644 --- a/providers/aihubmix/models/coding-glm-5.1.toml +++ b/providers/aihubmix/models/coding-glm-5.1.toml @@ -1,28 +1,13 @@ +base_model = "zhipuai/glm-5.1" name = "Coding GLM 5.1" description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -family = "glm" -release_date = "2026-04-11" -last_updated = "2026-04-11" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -open_weights = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.06 output = 0.22 -cache_read = 0.013 - -[limit] -context = 200_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/coding-glm-5.2-free.toml b/providers/aihubmix/models/coding-glm-5.2-free.toml new file mode 100644 index 00000000000..9e3d289059d --- /dev/null +++ b/providers/aihubmix/models/coding-glm-5.2-free.toml @@ -0,0 +1,12 @@ +base_model = "zhipuai/glm-5.2" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/coding-glm-5.2.toml b/providers/aihubmix/models/coding-glm-5.2.toml new file mode 100644 index 00000000000..e5ae0dc0bb7 --- /dev/null +++ b/providers/aihubmix/models/coding-glm-5.2.toml @@ -0,0 +1,12 @@ +base_model = "zhipuai/glm-5.2" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.06 +output = 0.22 diff --git a/providers/aihubmix/models/coding-glm-5.3-free.toml b/providers/aihubmix/models/coding-glm-5.3-free.toml new file mode 100644 index 00000000000..05636bbb0b7 --- /dev/null +++ b/providers/aihubmix/models/coding-glm-5.3-free.toml @@ -0,0 +1,15 @@ +base_model = "zhipuai/glm-5.3" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_048_576 diff --git a/providers/aihubmix/models/coding-glm-5.3.toml b/providers/aihubmix/models/coding-glm-5.3.toml new file mode 100644 index 00000000000..2803d9d5996 --- /dev/null +++ b/providers/aihubmix/models/coding-glm-5.3.toml @@ -0,0 +1,16 @@ +base_model = "zhipuai/glm-5.3" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.06 +output = 0.22 +cache_read = 0.015 + +[limit] +context = 1_048_576 diff --git a/providers/aihubmix/models/coding-glm-5.toml b/providers/aihubmix/models/coding-glm-5.toml new file mode 100644 index 00000000000..43a367a1581 --- /dev/null +++ b/providers/aihubmix/models/coding-glm-5.toml @@ -0,0 +1,12 @@ +base_model = "zhipuai/glm-5" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.06 +output = 0.22 + +[limit] +context = 200_000 diff --git a/providers/aihubmix/models/coding-kimi-k3-free.toml b/providers/aihubmix/models/coding-kimi-k3-free.toml new file mode 100644 index 00000000000..0a47dcba8ec --- /dev/null +++ b/providers/aihubmix/models/coding-kimi-k3-free.toml @@ -0,0 +1,15 @@ +base_model = "moonshotai/kimi-k3" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 + +[limit] +output = 1_048_576 diff --git a/providers/aihubmix/models/coding-kimi-k3.toml b/providers/aihubmix/models/coding-kimi-k3.toml new file mode 100644 index 00000000000..2001fbbeb87 --- /dev/null +++ b/providers/aihubmix/models/coding-kimi-k3.toml @@ -0,0 +1,16 @@ +base_model = "moonshotai/kimi-k3" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.44 +output = 1.61333 +cache_read = 0.066 + +[limit] +output = 1_048_576 diff --git a/providers/aihubmix/models/coding-minimax-m2.7-free.toml b/providers/aihubmix/models/coding-minimax-m2.7-free.toml index de8f6b2b766..0c37e9fdcf6 100644 --- a/providers/aihubmix/models/coding-minimax-m2.7-free.toml +++ b/providers/aihubmix/models/coding-minimax-m2.7-free.toml @@ -1,11 +1,10 @@ -name = "Coding MiniMax M2.7 (Free)" +name = "Coding MiniMax M2.7 (free)" description = "MiniMax model for chat, coding, office work, and agentic tasks" family = "minimax-free" release_date = "2026-03-18" last_updated = "2026-03-18" attachment = false reasoning = true -reasoning_options = [] temperature = true tool_call = true structured_output = true @@ -14,13 +13,20 @@ open_weights = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0 output = 0 [limit] context = 204_800 -output = 128_100 +output = 204_800 [modalities] input = ["text"] diff --git a/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml b/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml index 24b69d7fa7b..bae714ec693 100644 --- a/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml +++ b/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml @@ -5,7 +5,6 @@ release_date = "2026-03-18" last_updated = "2026-03-18" attachment = false reasoning = true -reasoning_options = [] temperature = true tool_call = true structured_output = true @@ -14,13 +13,20 @@ open_weights = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.2 output = 0.2 [limit] context = 204_800 -output = 128_100 +output = 204_800 [modalities] input = ["text"] diff --git a/providers/aihubmix/models/coding-minimax-m2.7.toml b/providers/aihubmix/models/coding-minimax-m2.7.toml index c17cc1f4e4c..58717be4441 100644 --- a/providers/aihubmix/models/coding-minimax-m2.7.toml +++ b/providers/aihubmix/models/coding-minimax-m2.7.toml @@ -5,7 +5,6 @@ release_date = "2026-03-18" last_updated = "2026-03-18" attachment = false reasoning = true -reasoning_options = [] temperature = true tool_call = true structured_output = true @@ -14,13 +13,20 @@ open_weights = true [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.2 output = 0.2 [limit] context = 204_800 -output = 128_100 +output = 204_800 [modalities] input = ["text"] diff --git a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml index ccb1203b612..f53ed8228b2 100644 --- a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml +++ b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml @@ -1,7 +1,6 @@ base_model = "xiaomi/mimo-v2.5-pro" name = "Coding Xiaomi MiMo-V2.5-Pro" -family = "mimo-v2.5-pro" -last_updated = "2026-05-13" +structured_output = true [interleaved] field = "reasoning_content" @@ -19,3 +18,6 @@ tier = { type = "context", size = 256_000 } input = 0.4 output = 1.2 cache_read = 0.08 + +[limit] +context = 1_000_000 diff --git a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml index 23149a9b62b..492102fc78e 100644 --- a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml +++ b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml @@ -1,7 +1,6 @@ base_model = "xiaomi/mimo-v2.5" name = "Coding Xiaomi MiMo-V2.5" -family = "mimo-v2.5" -last_updated = "2026-05-13" +structured_output = true [interleaved] field = "reasoning_content" @@ -19,3 +18,6 @@ tier = { type = "context", size = 256_000 } input = 0.16 output = 0.8 cache_read = 0.032 + +[limit] +context = 1_000_000 diff --git a/providers/aihubmix/models/command-a-03-2025.toml b/providers/aihubmix/models/command-a-03-2025.toml new file mode 100644 index 00000000000..b00d795de92 --- /dev/null +++ b/providers/aihubmix/models/command-a-03-2025.toml @@ -0,0 +1,11 @@ +base_model = "cohere/command-a-03-2025" +reasoning = true +tool_call = false + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 2.5 +output = 10 diff --git a/providers/aihubmix/models/command-a-plus-05-2026.toml b/providers/aihubmix/models/command-a-plus-05-2026.toml new file mode 100644 index 00000000000..562165dc4dc --- /dev/null +++ b/providers/aihubmix/models/command-a-plus-05-2026.toml @@ -0,0 +1,15 @@ +base_model = "cohere/command-a-plus-05-2026" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 2.5 +output = 10 diff --git a/providers/aihubmix/models/command-r-08-2024.toml b/providers/aihubmix/models/command-r-08-2024.toml new file mode 100644 index 00000000000..3eb86afe515 --- /dev/null +++ b/providers/aihubmix/models/command-r-08-2024.toml @@ -0,0 +1,6 @@ +base_model = "cohere/command-r-08-2024" +tool_call = false + +[cost] +input = 0.2 +output = 0.8 diff --git a/providers/aihubmix/models/command-r-plus-08-2024.toml b/providers/aihubmix/models/command-r-plus-08-2024.toml new file mode 100644 index 00000000000..e0e3ee2ad05 --- /dev/null +++ b/providers/aihubmix/models/command-r-plus-08-2024.toml @@ -0,0 +1,6 @@ +base_model = "cohere/command-r-plus-08-2024" +tool_call = false + +[cost] +input = 2.8 +output = 11.2 diff --git a/providers/aihubmix/models/deepseek-v3.2-think.toml b/providers/aihubmix/models/deepseek-v3.2-think.toml new file mode 100644 index 00000000000..b0d1587bebd --- /dev/null +++ b/providers/aihubmix/models/deepseek-v3.2-think.toml @@ -0,0 +1,12 @@ +base_model = "deepseek/deepseek-v3.2" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.302 +output = 0.453 +cache_read = 0.0302 + +[limit] +context = 163_840 diff --git a/providers/aihubmix/models/deepseek-v3.2.toml b/providers/aihubmix/models/deepseek-v3.2.toml new file mode 100644 index 00000000000..b0d1587bebd --- /dev/null +++ b/providers/aihubmix/models/deepseek-v3.2.toml @@ -0,0 +1,12 @@ +base_model = "deepseek/deepseek-v3.2" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.302 +output = 0.453 +cache_read = 0.0302 + +[limit] +context = 163_840 diff --git a/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml b/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml new file mode 100644 index 00000000000..9be732aad47 --- /dev/null +++ b/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml @@ -0,0 +1,13 @@ +base_model = "deepseek/deepseek-v4-flash-vision-exp" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.2324 +output = 0.6972 +cache_read = 0.007746 diff --git a/providers/aihubmix/models/deepseek-v4-flash.toml b/providers/aihubmix/models/deepseek-v4-flash.toml new file mode 100644 index 00000000000..fd5c8bd99ac --- /dev/null +++ b/providers/aihubmix/models/deepseek-v4-flash.toml @@ -0,0 +1,13 @@ +base_model = "deepseek/deepseek-v4-flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.142 +output = 0.284 +cache_read = 0.0284 diff --git a/providers/aihubmix/models/deepseek-v4-pro-0813.toml b/providers/aihubmix/models/deepseek-v4-pro-0813.toml index fbe29803f4c..dab59a28608 100644 --- a/providers/aihubmix/models/deepseek-v4-pro-0813.toml +++ b/providers/aihubmix/models/deepseek-v4-pro-0813.toml @@ -2,11 +2,17 @@ # Effort: reasoning_effort = high|max # AIHubMix effort levels returned HTTP 200 (validated 2026-08-31T04:17:24Z). base_model = "deepseek/deepseek-v4-pro-0813" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["high", "max"] }] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + [cost] input = 0.6918 output = 2.0754 diff --git a/providers/aihubmix/models/deepseek-v4-pro.toml b/providers/aihubmix/models/deepseek-v4-pro.toml new file mode 100644 index 00000000000..097f0ad22a7 --- /dev/null +++ b/providers/aihubmix/models/deepseek-v4-pro.toml @@ -0,0 +1,13 @@ +base_model = "deepseek/deepseek-v4-pro" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 1.69 +output = 3.38 +cache_read = 0.14027 diff --git a/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml b/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml index 60d3e2decbb..b37e7c9b461 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml @@ -18,7 +18,7 @@ type = "toggle" [[reasoning_options]] type = "effort" -values = ["minimal", "low", "medium", "high"] +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 0.4822 @@ -27,14 +27,14 @@ cache_read = 0.09644 [[cost.tiers]] tier = { type = "context", size = 32_000 } -input = 0.72 -output = 3.62 +input = 0.72328 +output = 3.6164 cache_read = 0.144656 [[cost.tiers]] tier = { type = "context", size = 128_000 } -input = 1.45 -output = 7.23 +input = 1.4466 +output = 7.233 cache_read = 0.28932 [limit] diff --git a/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml b/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml index 7dd55b32212..bd50446b1af 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml @@ -18,32 +18,29 @@ type = "toggle" [[reasoning_options]] type = "effort" -values = ["minimal", "low", "medium", "high"] +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 0.09041 output = 0.54246 cache_read = 0.018082 -input_audio = 1.269 [[cost.tiers]] tier = { type = "context", size = 32_000 } -input = 0.13 -output = 0.76 +input = 0.1268 +output = 0.7608 cache_read = 0.02536 -input_audio = 1.902 [[cost.tiers]] tier = { type = "context", size = 128_000 } -input = 0.25 -output = 1.52 +input = 0.2536 +output = 1.5216 cache_read = 0.05072 -input_audio = 3.804 [limit] context = 256_000 output = 128_000 [modalities] -input = ["text", "image", "video"] +input = ["text", "image", "video", "audio"] output = ["text"] diff --git a/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml b/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml index f83044c0ef6..8f929910c37 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml @@ -18,32 +18,29 @@ type = "toggle" [[reasoning_options]] type = "effort" -values = ["minimal", "low", "medium", "high"] +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 0.0282 output = 0.282 cache_read = 0.00564 -input_audio = 0.423 [[cost.tiers]] tier = { type = "context", size = 32_000 } -input = 0.06 -output = 0.56 +input = 0.0564 +output = 0.564 cache_read = 0.01128 -input_audio = 0.846 [[cost.tiers]] tier = { type = "context", size = 128_000 } -input = 0.11 -output = 1.13 +input = 0.1128 +output = 1.128 cache_read = 0.02256 -input_audio = 1.692 [limit] context = 256_000 output = 128_000 [modalities] -input = ["text", "image", "video"] +input = ["text", "image", "video", "audio"] output = ["text"] diff --git a/providers/aihubmix/models/doubao-seed-2-0-pro.toml b/providers/aihubmix/models/doubao-seed-2-0-pro.toml index f51fb455088..dbe58911d3d 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-pro.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-pro.toml @@ -18,7 +18,7 @@ type = "toggle" [[reasoning_options]] type = "effort" -values = ["minimal", "low", "medium", "high"] +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] [cost] input = 0.4822 @@ -27,14 +27,14 @@ cache_read = 0.09644 [[cost.tiers]] tier = { type = "context", size = 32_000 } -input = 0.72 -output = 3.62 +input = 0.72328 +output = 3.6164 cache_read = 0.144656 [[cost.tiers]] tier = { type = "context", size = 128_000 } -input = 1.45 -output = 7.23 +input = 1.4466 +output = 7.233 cache_read = 0.28932 [limit] diff --git a/providers/aihubmix/models/gemini-2.5-flash-image.toml b/providers/aihubmix/models/gemini-2.5-flash-image.toml new file mode 100644 index 00000000000..c46f3c2e4b7 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-image.toml @@ -0,0 +1,10 @@ +base_model = "google/gemini-2.5-flash-image" +reasoning = false + +[cost] +input = 0.3 +output = 2.499 +cache_read = 0.03 + +[limit] +context = 65_536 diff --git a/providers/aihubmix/models/gemini-2.5-flash-lite-nothink.toml b/providers/aihubmix/models/gemini-2.5-flash-lite-nothink.toml new file mode 100644 index 00000000000..f42f82a1953 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-lite-nothink.toml @@ -0,0 +1,13 @@ +base_model = "google/gemini-2.5-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.4 +cache_read = 0.01 diff --git a/providers/aihubmix/models/gemini-2.5-flash-lite.toml b/providers/aihubmix/models/gemini-2.5-flash-lite.toml new file mode 100644 index 00000000000..f42f82a1953 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-lite.toml @@ -0,0 +1,13 @@ +base_model = "google/gemini-2.5-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.4 +cache_read = 0.01 diff --git a/providers/aihubmix/models/gemini-2.5-flash-nothink.toml b/providers/aihubmix/models/gemini-2.5-flash-nothink.toml new file mode 100644 index 00000000000..da2afd88e71 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-nothink.toml @@ -0,0 +1,16 @@ +base_model = "google/gemini-2.5-flash" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.3 +output = 2.499 +cache_read = 0.03 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-search.toml b/providers/aihubmix/models/gemini-2.5-flash-search.toml new file mode 100644 index 00000000000..da2afd88e71 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-search.toml @@ -0,0 +1,16 @@ +base_model = "google/gemini-2.5-flash" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.3 +output = 2.499 +cache_read = 0.03 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-flash.toml b/providers/aihubmix/models/gemini-2.5-flash.toml index 1b342075259..0e43b319839 100644 --- a/providers/aihubmix/models/gemini-2.5-flash.toml +++ b/providers/aihubmix/models/gemini-2.5-flash.toml @@ -1,39 +1,18 @@ # Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: 0 disables, -1 is dynamic, and manual budgets are 1..24576. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) -name = "Gemini 2.5 Flash" +base_model = "google/gemini-2.5-flash" description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost" -family = "gemini-flash" -release_date = "2025-03-20" -last_updated = "2025-06-05" -attachment = true -reasoning = true -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-01" -open_weights = false - -[[reasoning_options]] -type = "toggle" [[reasoning_options]] type = "effort" -values = ["low", "medium", "high"] +values = ["none", "minimal", "low", "medium", "high"] [[reasoning_options]] type = "budget_tokens" -min = 0 -max = 24_576 [cost] input = 0.3 output = 2.499 cache_read = 0.03 -input_audio = 1 - -[limit] -context = 1_048_576 -output = 65_536 [modalities] -input = ["text", "image", "audio", "video", "pdf"] -output = ["text"] +input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-search.toml b/providers/aihubmix/models/gemini-2.5-pro-search.toml new file mode 100644 index 00000000000..a97c9c65bb6 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-pro-search.toml @@ -0,0 +1,19 @@ +base_model = "google/gemini-2.5-pro" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 diff --git a/providers/aihubmix/models/gemini-2.5-pro.toml b/providers/aihubmix/models/gemini-2.5-pro.toml index 8ef984bdbf3..75750cf69ba 100644 --- a/providers/aihubmix/models/gemini-2.5-pro.toml +++ b/providers/aihubmix/models/gemini-2.5-pro.toml @@ -1,17 +1,12 @@ -name = "Gemini 2.5 Pro" +base_model = "google/gemini-2.5-pro" description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" -family = "gemini-pro" -release_date = "2025-03-20" -last_updated = "2025-06-05" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: -1 is dynamic and manual budgets are 128..32768; 0/off is unsupported. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 1.25 @@ -19,15 +14,7 @@ output = 10 cache_read = 0.125 [[cost.tiers]] -tier = { size = 200_000 } -input = 2.50 -output = 15.00 +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 cache_read = 0.25 - -[limit] -context = 1_048_576 -output = 65_536 - -[modalities] -input = ["text", "image", "audio", "video", "pdf"] -output = ["text"] diff --git a/providers/aihubmix/models/gemini-3-flash-preview-free.toml b/providers/aihubmix/models/gemini-3-flash-preview-free.toml new file mode 100644 index 00000000000..274c6a78644 --- /dev/null +++ b/providers/aihubmix/models/gemini-3-flash-preview-free.toml @@ -0,0 +1,12 @@ +base_model = "google/gemini-3-flash-preview" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/gemini-3-flash-preview-search.toml b/providers/aihubmix/models/gemini-3-flash-preview-search.toml new file mode 100644 index 00000000000..522c79650d8 --- /dev/null +++ b/providers/aihubmix/models/gemini-3-flash-preview-search.toml @@ -0,0 +1,13 @@ +base_model = "google/gemini-3-flash-preview" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.5 +output = 3 +cache_read = 0.05 diff --git a/providers/aihubmix/models/gemini-3-flash-preview.toml b/providers/aihubmix/models/gemini-3-flash-preview.toml index 379e0cf258d..991ea923538 100644 --- a/providers/aihubmix/models/gemini-3-flash-preview.toml +++ b/providers/aihubmix/models/gemini-3-flash-preview.toml @@ -1,16 +1,12 @@ -name = "Gemini 3 Flash Preview" +base_model = "google/gemini-3-flash-preview" description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost" -family = "gemini-flash" -release_date = "2025-12-17" -last_updated = "2025-12-17" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 0.5 @@ -18,15 +14,7 @@ output = 3 cache_read = 0.05 [[cost.tiers]] -tier = { size = 200_000 } -input = 0.50 -output = 3.00 +tier = { type = "context", size = 200_000 } +input = 0.5 +output = 3 cache_read = 0.05 - -[limit] -context = 1_048_576 -output = 65_536 - -[modalities] -input = ["text", "image", "audio", "video", "pdf"] -output = ["text"] diff --git a/providers/aihubmix/models/gemini-3-pro-image-preview.toml b/providers/aihubmix/models/gemini-3-pro-image-preview.toml new file mode 100644 index 00000000000..0bf04c931f5 --- /dev/null +++ b/providers/aihubmix/models/gemini-3-pro-image-preview.toml @@ -0,0 +1,7 @@ +base_model = "google/gemini-3-pro-image-preview" +status = "deprecated" +reasoning_options = [] + +[cost] +input = 2 +output = 12 diff --git a/providers/aihubmix/models/gemini-3-pro-image.toml b/providers/aihubmix/models/gemini-3-pro-image.toml new file mode 100644 index 00000000000..bd71f0099f2 --- /dev/null +++ b/providers/aihubmix/models/gemini-3-pro-image.toml @@ -0,0 +1,12 @@ +base_model = "google/gemini-3-pro-image" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 2 +output = 12 diff --git a/providers/aihubmix/models/gemini-3.1-flash-image-preview.toml b/providers/aihubmix/models/gemini-3.1-flash-image-preview.toml new file mode 100644 index 00000000000..2409dc19d4e --- /dev/null +++ b/providers/aihubmix/models/gemini-3.1-flash-image-preview.toml @@ -0,0 +1,10 @@ +base_model = "google/gemini-3.1-flash-image-preview" +status = "deprecated" +reasoning_options = [] + +[cost] +input = 0.5 +output = 3 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gemini-3.1-flash-image.toml b/providers/aihubmix/models/gemini-3.1-flash-image.toml new file mode 100644 index 00000000000..bf9b9becb85 --- /dev/null +++ b/providers/aihubmix/models/gemini-3.1-flash-image.toml @@ -0,0 +1,12 @@ +base_model = "google/gemini-3.1-flash-image" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.5 +output = 3 diff --git a/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml b/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml new file mode 100644 index 00000000000..91587bf22be --- /dev/null +++ b/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml @@ -0,0 +1,11 @@ +base_model = "google/gemini-3.1-flash-lite-image" +tool_call = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.25 +output = 1.5 +cache_read = 0.025 diff --git a/providers/aihubmix/models/gemini-3.1-flash-lite-nothink.toml b/providers/aihubmix/models/gemini-3.1-flash-lite-nothink.toml new file mode 100644 index 00000000000..51db76c6d9d --- /dev/null +++ b/providers/aihubmix/models/gemini-3.1-flash-lite-nothink.toml @@ -0,0 +1,13 @@ +base_model = "google/gemini-3.1-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.25 +output = 1.5 +cache_read = 0.025 diff --git a/providers/aihubmix/models/gemini-3.1-flash-lite.toml b/providers/aihubmix/models/gemini-3.1-flash-lite.toml index 1454eedebea..4adbf5f45d6 100644 --- a/providers/aihubmix/models/gemini-3.1-flash-lite.toml +++ b/providers/aihubmix/models/gemini-3.1-flash-lite.toml @@ -1,27 +1,14 @@ -name = "Gemini 3.1 Flash Lite" +base_model = "google/gemini-3.1-flash-lite" description = "Low-latency Gemini model for high-volume multimodal and agent workloads" -family = "gemini-flash-lite" -release_date = "2026-05-07" -last_updated = "2026-05-07" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 0.25 output = 1.5 cache_read = 0.025 -cache_write = 1.00 - -[limit] -context = 1_048_576 -output = 65_536 - -[modalities] -input = ["text", "image", "audio", "video", "pdf"] -output = ["text"] diff --git a/providers/aihubmix/models/gemini-3.1-pro-preview-customtools.toml b/providers/aihubmix/models/gemini-3.1-pro-preview-customtools.toml index 7a53e5d5345..c0d271ad953 100644 --- a/providers/aihubmix/models/gemini-3.1-pro-preview-customtools.toml +++ b/providers/aihubmix/models/gemini-3.1-pro-preview-customtools.toml @@ -1,16 +1,12 @@ -name = "Gemini 3.1 Pro Preview Custom Tools" +base_model = "google/gemini-3.1-pro-preview-customtools" description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" -family = "gemini-pro" -release_date = "2026-02-19" -last_updated = "2026-02-19" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 2 @@ -18,15 +14,7 @@ output = 12 cache_read = 0.2 [[cost.tiers]] -tier = { size = 200_000 } +tier = { type = "context", size = 200_000 } input = 4 output = 18 cache_read = 0.4 - -[limit] -context = 1_048_576 -output = 65_536 - -[modalities] -input = ["text", "image", "audio", "video", "pdf"] -output = ["text"] diff --git a/providers/aihubmix/models/gemini-3.1-pro-preview-search.toml b/providers/aihubmix/models/gemini-3.1-pro-preview-search.toml new file mode 100644 index 00000000000..6fdb9a70c1b --- /dev/null +++ b/providers/aihubmix/models/gemini-3.1-pro-preview-search.toml @@ -0,0 +1,19 @@ +base_model = "google/gemini-3.1-pro-preview" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 2 +output = 12 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 18 +cache_read = 0.4 diff --git a/providers/aihubmix/models/gemini-3.1-pro-preview.toml b/providers/aihubmix/models/gemini-3.1-pro-preview.toml index c7dc961e75d..559bc725893 100644 --- a/providers/aihubmix/models/gemini-3.1-pro-preview.toml +++ b/providers/aihubmix/models/gemini-3.1-pro-preview.toml @@ -1,16 +1,12 @@ -name = "Gemini 3.1 Pro Preview" +base_model = "google/gemini-3.1-pro-preview" description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" -family = "gemini-pro" -release_date = "2026-02-19" -last_updated = "2026-02-19" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 2 @@ -18,15 +14,7 @@ output = 12 cache_read = 0.2 [[cost.tiers]] -tier = { size = 200_000 } -input = 4.00 -output = 18.00 -cache_read = 0.40 - -[limit] -context = 1_048_576 -output = 65_536 - -[modalities] -input = ["text", "image", "audio", "video", "pdf"] -output = ["text"] +tier = { type = "context", size = 200_000 } +input = 4 +output = 18 +cache_read = 0.4 diff --git a/providers/aihubmix/models/gemini-3.5-flash-lite-free.toml b/providers/aihubmix/models/gemini-3.5-flash-lite-free.toml new file mode 100644 index 00000000000..8ba9484d2bf --- /dev/null +++ b/providers/aihubmix/models/gemini-3.5-flash-lite-free.toml @@ -0,0 +1,9 @@ +base_model = "google/gemini-3.5-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/gemini-3.5-flash-lite.toml b/providers/aihubmix/models/gemini-3.5-flash-lite.toml new file mode 100644 index 00000000000..980fef0620f --- /dev/null +++ b/providers/aihubmix/models/gemini-3.5-flash-lite.toml @@ -0,0 +1,10 @@ +base_model = "google/gemini-3.5-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0.3 +output = 2.499999 +cache_read = 0.03 diff --git a/providers/aihubmix/models/gemini-3.5-flash.toml b/providers/aihubmix/models/gemini-3.5-flash.toml index 577253bb682..1a5b79c9241 100644 --- a/providers/aihubmix/models/gemini-3.5-flash.toml +++ b/providers/aihubmix/models/gemini-3.5-flash.toml @@ -4,15 +4,10 @@ base_model = "google/gemini-3.5-flash" type = "effort" values = ["minimal", "low", "medium", "high"] +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 1.5 output = 9 cache_read = 0.15 - -[limit] -context = 1_000_000 -output = 64_000 - -[modalities] -input = ["text", "image", "audio", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/gemini-3.6-flash-free.toml b/providers/aihubmix/models/gemini-3.6-flash-free.toml new file mode 100644 index 00000000000..76b7a0c3dfa --- /dev/null +++ b/providers/aihubmix/models/gemini-3.6-flash-free.toml @@ -0,0 +1,9 @@ +base_model = "google/gemini-3.6-flash" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/gemini-3.6-flash.toml b/providers/aihubmix/models/gemini-3.6-flash.toml new file mode 100644 index 00000000000..9307549671d --- /dev/null +++ b/providers/aihubmix/models/gemini-3.6-flash.toml @@ -0,0 +1,10 @@ +base_model = "google/gemini-3.6-flash" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 1.5 +output = 7.5 +cache_read = 0.15 diff --git a/providers/aihubmix/models/gemini-3.7-flash-free.toml b/providers/aihubmix/models/gemini-3.7-flash-free.toml new file mode 100644 index 00000000000..7a29ae2e6da --- /dev/null +++ b/providers/aihubmix/models/gemini-3.7-flash-free.toml @@ -0,0 +1,12 @@ +base_model = "google/gemini-3.7-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/gemini-3.7-flash.toml b/providers/aihubmix/models/gemini-3.7-flash.toml index 31d0f319a9a..1e23441b9c9 100644 --- a/providers/aihubmix/models/gemini-3.7-flash.toml +++ b/providers/aihubmix/models/gemini-3.7-flash.toml @@ -1,8 +1,11 @@ base_model = "google/gemini-3.7-flash" -# AIHubMix unified Chat: $.reasoning_effort = low|medium|high -# https://docs.aihubmix.com/cn/api/unified-inference -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 0.75 diff --git a/providers/aihubmix/models/gemini-3.8-flash-free.toml b/providers/aihubmix/models/gemini-3.8-flash-free.toml new file mode 100644 index 00000000000..f4f34ce371c --- /dev/null +++ b/providers/aihubmix/models/gemini-3.8-flash-free.toml @@ -0,0 +1,12 @@ +base_model = "google/gemini-3.8-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/gemini-3.8-flash.toml b/providers/aihubmix/models/gemini-3.8-flash.toml new file mode 100644 index 00000000000..1252ad652d8 --- /dev/null +++ b/providers/aihubmix/models/gemini-3.8-flash.toml @@ -0,0 +1,13 @@ +base_model = "google/gemini-3.8-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.75 +output = 3.75 +cache_read = 0.075 diff --git a/providers/aihubmix/models/gemma-4-26b-a4b-it-free.toml b/providers/aihubmix/models/gemma-4-26b-a4b-it-free.toml new file mode 100644 index 00000000000..fdbe0240df6 --- /dev/null +++ b/providers/aihubmix/models/gemma-4-26b-a4b-it-free.toml @@ -0,0 +1,15 @@ +base_model = "google/gemma-4-26b-a4b-it" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0 +output = 0 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/gemma-4-26b-a4b-it.toml b/providers/aihubmix/models/gemma-4-26b-a4b-it.toml new file mode 100644 index 00000000000..f8ea6d86182 --- /dev/null +++ b/providers/aihubmix/models/gemma-4-26b-a4b-it.toml @@ -0,0 +1,19 @@ +base_model = "google/gemma-4-26b-a4b-it" +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.14 +output = 0.39998 + +[limit] +output = 131_100 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/gemma-4-31b-it-free.toml b/providers/aihubmix/models/gemma-4-31b-it-free.toml new file mode 100644 index 00000000000..4d1d6d38a83 --- /dev/null +++ b/providers/aihubmix/models/gemma-4-31b-it-free.toml @@ -0,0 +1,15 @@ +base_model = "google/gemma-4-31b-it" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0 +output = 0 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/gemma-4-31b-it.toml b/providers/aihubmix/models/gemma-4-31b-it.toml new file mode 100644 index 00000000000..adcf4ac6145 --- /dev/null +++ b/providers/aihubmix/models/gemma-4-31b-it.toml @@ -0,0 +1,19 @@ +base_model = "google/gemma-4-31b-it" +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.14 +output = 0.39998 + +[limit] +output = 131_100 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/glm-4.5v.toml b/providers/aihubmix/models/glm-4.5v.toml new file mode 100644 index 00000000000..93d89be4fbd --- /dev/null +++ b/providers/aihubmix/models/glm-4.5v.toml @@ -0,0 +1,21 @@ +base_model = "zhipuai/glm-4.5v" +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.274 +output = 0.822 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.548 +output = 1.644 + +[limit] +context = 65_536 diff --git a/providers/aihubmix/models/glm-4.6.toml b/providers/aihubmix/models/glm-4.6.toml new file mode 100644 index 00000000000..17741f5557e --- /dev/null +++ b/providers/aihubmix/models/glm-4.6.toml @@ -0,0 +1,19 @@ +base_model = "zhipuai/glm-4.6" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.273974 +output = 1.095896 +cache_read = 0.054795 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.547946 +output = 2.191784 +cache_read = 0.109589 + +[limit] +context = 200_000 diff --git a/providers/aihubmix/models/glm-4.6v.toml b/providers/aihubmix/models/glm-4.6v.toml new file mode 100644 index 00000000000..f3deb7ff53f --- /dev/null +++ b/providers/aihubmix/models/glm-4.6v.toml @@ -0,0 +1,23 @@ +base_model = "zhipuai/glm-4.6v" +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.137 +output = 0.411 +cache_read = 0.0274 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.274 +output = 0.822 +cache_read = 0.0548 + +[limit] +context = 131_072 diff --git a/providers/aihubmix/models/glm-4.7-flash-free.toml b/providers/aihubmix/models/glm-4.7-flash-free.toml new file mode 100644 index 00000000000..c7baa849a9b --- /dev/null +++ b/providers/aihubmix/models/glm-4.7-flash-free.toml @@ -0,0 +1,9 @@ +base_model = "zhipuai/glm-4.7-flash" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/glm-4.7.toml b/providers/aihubmix/models/glm-4.7.toml new file mode 100644 index 00000000000..302a2bd98b4 --- /dev/null +++ b/providers/aihubmix/models/glm-4.7.toml @@ -0,0 +1,19 @@ +base_model = "zhipuai/glm-4.7" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.273974 +output = 1.095896 +cache_read = 0.054795 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.547946 +output = 2.191784 +cache_read = 0.109589 + +[limit] +context = 200_000 diff --git a/providers/aihubmix/models/glm-5-turbo.toml b/providers/aihubmix/models/glm-5-turbo.toml new file mode 100644 index 00000000000..aa300d4a5bd --- /dev/null +++ b/providers/aihubmix/models/glm-5-turbo.toml @@ -0,0 +1,12 @@ +base_model = "zhipuai/glm-5-turbo" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 1.2 +output = 3.9996 +cache_read = 0.24 + +[limit] +context = 204_800 diff --git a/providers/aihubmix/models/glm-5.1.toml b/providers/aihubmix/models/glm-5.1.toml new file mode 100644 index 00000000000..c357a3506f4 --- /dev/null +++ b/providers/aihubmix/models/glm-5.1.toml @@ -0,0 +1,9 @@ +base_model = "zhipuai/glm-5.1" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.845 +output = 3.38 +cache_read = 0.183112 diff --git a/providers/aihubmix/models/glm-5.2.toml b/providers/aihubmix/models/glm-5.2.toml index 6fb0761ec4c..7a62d7aa858 100644 --- a/providers/aihubmix/models/glm-5.2.toml +++ b/providers/aihubmix/models/glm-5.2.toml @@ -1,21 +1,16 @@ base_model = "zhipuai/glm-5.2" -[[reasoning_options]] -type = "effort" -values = ["high", "max"] - [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + [cost] input = 1.1268 output = 3.9438 cache_read = 0.2817 - -[limit] -context = 1_000_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/glm-5.3-flash.toml b/providers/aihubmix/models/glm-5.3-flash.toml index 58250ec9ab4..11124a3e9e9 100644 --- a/providers/aihubmix/models/glm-5.3-flash.toml +++ b/providers/aihubmix/models/glm-5.3-flash.toml @@ -1,17 +1,22 @@ base_model = "zhipuai/glm-5.3-flash" -reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + [cost] input = 0.11268 output = 0.39438 cache_read = 0.02817 [limit] -output = 128_000 +context = 1_048_576 [modalities] input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/glm-5.3.toml b/providers/aihubmix/models/glm-5.3.toml index f9de29a855e..5d67e72462b 100644 --- a/providers/aihubmix/models/glm-5.3.toml +++ b/providers/aihubmix/models/glm-5.3.toml @@ -1,13 +1,19 @@ base_model = "zhipuai/glm-5.3" -reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + [cost] input = 1.1268 output = 3.9438 cache_read = 0.2817 [limit] -output = 128_000 +context = 1_048_576 diff --git a/providers/aihubmix/models/glm-5.toml b/providers/aihubmix/models/glm-5.toml new file mode 100644 index 00000000000..015f06590c4 --- /dev/null +++ b/providers/aihubmix/models/glm-5.toml @@ -0,0 +1,13 @@ +base_model = "zhipuai/glm-5" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.88 +output = 2.816 +cache_read = 0.176 + +[limit] +context = 200_000 diff --git a/providers/aihubmix/models/glm-5v-turbo.toml b/providers/aihubmix/models/glm-5v-turbo.toml index 32398231e9d..e7f491d8196 100644 --- a/providers/aihubmix/models/glm-5v-turbo.toml +++ b/providers/aihubmix/models/glm-5v-turbo.toml @@ -1,28 +1,24 @@ +base_model = "zhipuai/glm-5v-turbo" name = "GLM 5 Vision Turbo" description = "GLM vision model for visual reasoning, documents, and multimodal agents" -family = "glmv" -release_date = "2026-05-09" -last_updated = "2026-05-09" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true structured_output = true -open_weights = false [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.7042 output = 3.09848 cache_read = 0.169008 -[limit] -context = 200_000 -output = 128_000 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.986 +output = 3.662004 +cache_read = 0.253402 [modalities] input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/gpt-4.1-free.toml b/providers/aihubmix/models/gpt-4.1-free.toml new file mode 100644 index 00000000000..7fa52106de5 --- /dev/null +++ b/providers/aihubmix/models/gpt-4.1-free.toml @@ -0,0 +1,8 @@ +base_model = "openai/gpt-4.1" + +[cost] +input = 0 +output = 0 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4.1-mini-free.toml b/providers/aihubmix/models/gpt-4.1-mini-free.toml new file mode 100644 index 00000000000..bba3f71945f --- /dev/null +++ b/providers/aihubmix/models/gpt-4.1-mini-free.toml @@ -0,0 +1,8 @@ +base_model = "openai/gpt-4.1-mini" + +[cost] +input = 0 +output = 0 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4.1-mini.toml b/providers/aihubmix/models/gpt-4.1-mini.toml new file mode 100644 index 00000000000..27124c07c4d --- /dev/null +++ b/providers/aihubmix/models/gpt-4.1-mini.toml @@ -0,0 +1,9 @@ +base_model = "openai/gpt-4.1-mini" + +[cost] +input = 0.4 +output = 1.6 +cache_read = 0.1 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4.1-nano-free.toml b/providers/aihubmix/models/gpt-4.1-nano-free.toml new file mode 100644 index 00000000000..75bcde2c355 --- /dev/null +++ b/providers/aihubmix/models/gpt-4.1-nano-free.toml @@ -0,0 +1,5 @@ +base_model = "openai/gpt-4.1-nano" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/gpt-4.1-nano.toml b/providers/aihubmix/models/gpt-4.1-nano.toml new file mode 100644 index 00000000000..f85c0947701 --- /dev/null +++ b/providers/aihubmix/models/gpt-4.1-nano.toml @@ -0,0 +1,6 @@ +base_model = "openai/gpt-4.1-nano" + +[cost] +input = 0.1 +output = 0.4 +cache_read = 0.025 diff --git a/providers/aihubmix/models/gpt-4.1.toml b/providers/aihubmix/models/gpt-4.1.toml new file mode 100644 index 00000000000..f618ee47235 --- /dev/null +++ b/providers/aihubmix/models/gpt-4.1.toml @@ -0,0 +1,9 @@ +base_model = "openai/gpt-4.1" + +[cost] +input = 2 +output = 8 +cache_read = 0.5 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4o-2024-11-20.toml b/providers/aihubmix/models/gpt-4o-2024-11-20.toml new file mode 100644 index 00000000000..db72e0c8807 --- /dev/null +++ b/providers/aihubmix/models/gpt-4o-2024-11-20.toml @@ -0,0 +1,7 @@ +base_model = "openai/gpt-4o-2024-11-20" +tool_call = false + +[cost] +input = 2.5 +output = 10 +cache_read = 1.25 diff --git a/providers/aihubmix/models/gpt-4o-free.toml b/providers/aihubmix/models/gpt-4o-free.toml new file mode 100644 index 00000000000..f87a9a8d6d3 --- /dev/null +++ b/providers/aihubmix/models/gpt-4o-free.toml @@ -0,0 +1,8 @@ +base_model = "openai/gpt-4o" + +[cost] +input = 0 +output = 0 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4o-mini.toml b/providers/aihubmix/models/gpt-4o-mini.toml new file mode 100644 index 00000000000..d7bafa9cf42 --- /dev/null +++ b/providers/aihubmix/models/gpt-4o-mini.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-4o-mini" +tool_call = false + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.075 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4o.toml b/providers/aihubmix/models/gpt-4o.toml new file mode 100644 index 00000000000..eb571296634 --- /dev/null +++ b/providers/aihubmix/models/gpt-4o.toml @@ -0,0 +1,9 @@ +base_model = "openai/gpt-4o" + +[cost] +input = 2.5 +output = 10 +cache_read = 1.25 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5-chat-latest.toml b/providers/aihubmix/models/gpt-5-chat-latest.toml new file mode 100644 index 00000000000..e3a239e5180 --- /dev/null +++ b/providers/aihubmix/models/gpt-5-chat-latest.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-5-chat-latest" +base_model_omit = ["limit.input"] +reasoning = false + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 + +[limit] +context = 128_000 +output = 16_384 diff --git a/providers/aihubmix/models/gpt-5-codex.toml b/providers/aihubmix/models/gpt-5-codex.toml new file mode 100644 index 00000000000..2b9e666df4d --- /dev/null +++ b/providers/aihubmix/models/gpt-5-codex.toml @@ -0,0 +1,8 @@ +base_model = "openai/gpt-5-codex" +attachment = true +reasoning_options = [] + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 diff --git a/providers/aihubmix/models/gpt-5-mini.toml b/providers/aihubmix/models/gpt-5-mini.toml new file mode 100644 index 00000000000..ec2e0406779 --- /dev/null +++ b/providers/aihubmix/models/gpt-5-mini.toml @@ -0,0 +1,7 @@ +base_model = "openai/gpt-5-mini" +reasoning_options = [] + +[cost] +input = 0.25 +output = 2 +cache_read = 0.025 diff --git a/providers/aihubmix/models/gpt-5-nano.toml b/providers/aihubmix/models/gpt-5-nano.toml new file mode 100644 index 00000000000..35e4f5d9e48 --- /dev/null +++ b/providers/aihubmix/models/gpt-5-nano.toml @@ -0,0 +1,7 @@ +base_model = "openai/gpt-5-nano" +reasoning_options = [] + +[cost] +input = 0.05 +output = 0.4 +cache_read = 0.005 diff --git a/providers/aihubmix/models/gpt-5-pro.toml b/providers/aihubmix/models/gpt-5-pro.toml new file mode 100644 index 00000000000..3d221044701 --- /dev/null +++ b/providers/aihubmix/models/gpt-5-pro.toml @@ -0,0 +1,9 @@ +base_model = "openai/gpt-5-pro" + +[[reasoning_options]] +type = "effort" +values = ["high"] + +[cost] +input = 15 +output = 120 diff --git a/providers/aihubmix/models/gpt-5.1-chat-latest.toml b/providers/aihubmix/models/gpt-5.1-chat-latest.toml new file mode 100644 index 00000000000..d422acc79fc --- /dev/null +++ b/providers/aihubmix/models/gpt-5.1-chat-latest.toml @@ -0,0 +1,7 @@ +base_model = "openai/gpt-5.1-chat-latest" +reasoning = false + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 diff --git a/providers/aihubmix/models/gpt-5.1-codex-max.toml b/providers/aihubmix/models/gpt-5.1-codex-max.toml new file mode 100644 index 00000000000..a137a421adb --- /dev/null +++ b/providers/aihubmix/models/gpt-5.1-codex-max.toml @@ -0,0 +1,7 @@ +base_model = "openai/gpt-5.1-codex-max" +reasoning_options = [] + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 diff --git a/providers/aihubmix/models/gpt-5.1-codex-mini.toml b/providers/aihubmix/models/gpt-5.1-codex-mini.toml index dea686856b0..abc2b111223 100644 --- a/providers/aihubmix/models/gpt-5.1-codex-mini.toml +++ b/providers/aihubmix/models/gpt-5.1-codex-mini.toml @@ -1,27 +1,11 @@ -name = "GPT-5.1 Codex mini" +base_model = "openai/gpt-5.1-codex-mini" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -family = "gpt-codex" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = false -knowledge = "2024-09-30" -tool_call = true -structured_output = true -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 0.25 -output = 2.00 +output = 2 cache_read = 0.025 - -[limit] -context = 400_000 -input = 272_000 -output = 128_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.1-codex.toml b/providers/aihubmix/models/gpt-5.1-codex.toml index be453c3b691..63ed5678cb6 100644 --- a/providers/aihubmix/models/gpt-5.1-codex.toml +++ b/providers/aihubmix/models/gpt-5.1-codex.toml @@ -1,27 +1,11 @@ -name = "GPT-5.1 Codex" +base_model = "openai/gpt-5.1-codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -family = "gpt-codex" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = false -knowledge = "2024-09-30" -tool_call = true -structured_output = true -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 1.25 -output = 10.00 +output = 10 cache_read = 0.125 - -[limit] -context = 400_000 -input = 272_000 -output = 128_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.1.toml b/providers/aihubmix/models/gpt-5.1.toml index 28234deacd6..68263b27fb6 100644 --- a/providers/aihubmix/models/gpt-5.1.toml +++ b/providers/aihubmix/models/gpt-5.1.toml @@ -1,15 +1,6 @@ -name = "GPT-5.1" +base_model = "openai/gpt-5.1" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" -family = "gpt" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = true temperature = false -tool_call = true -structured_output = true -knowledge = "2024-09-30" -open_weights = false [[reasoning_options]] type = "effort" @@ -19,12 +10,3 @@ values = ["none", "low", "medium", "high"] input = 1.25 output = 10 cache_read = 0.125 - -[limit] -context = 400_000 -input = 272_000 -output = 128_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.2-chat-latest.toml b/providers/aihubmix/models/gpt-5.2-chat-latest.toml new file mode 100644 index 00000000000..6218a841ec7 --- /dev/null +++ b/providers/aihubmix/models/gpt-5.2-chat-latest.toml @@ -0,0 +1,7 @@ +base_model = "openai/gpt-5.2-chat-latest" +reasoning = false + +[cost] +input = 1.75 +output = 14 +cache_read = 0.175 diff --git a/providers/aihubmix/models/gpt-5.2-codex.toml b/providers/aihubmix/models/gpt-5.2-codex.toml index 3e67be5f408..1f2eed5fca0 100644 --- a/providers/aihubmix/models/gpt-5.2-codex.toml +++ b/providers/aihubmix/models/gpt-5.2-codex.toml @@ -1,27 +1,14 @@ -name = "GPT-5.2 Codex" +base_model = "openai/gpt-5.2-codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -family = "gpt-codex" -release_date = "2025-12-11" -last_updated = "2025-12-11" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] -temperature = false -knowledge = "2025-08-31" -tool_call = true -structured_output = true -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] [cost] input = 1.75 -output = 14.00 +output = 14 cache_read = 0.175 -[limit] -context = 400_000 -input = 272_000 -output = 128_000 - [modalities] -input = ["text", "image", "pdf"] -output = ["text"] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.2-pro.toml b/providers/aihubmix/models/gpt-5.2-pro.toml new file mode 100644 index 00000000000..ea3515497a5 --- /dev/null +++ b/providers/aihubmix/models/gpt-5.2-pro.toml @@ -0,0 +1,11 @@ +base_model = "openai/gpt-5.2-pro" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["medium", "high", "xhigh"] + +[cost] +input = 21 +output = 168 +cache_read = 2.1 diff --git a/providers/aihubmix/models/gpt-5.2.toml b/providers/aihubmix/models/gpt-5.2.toml index ee922648878..5f6de738b50 100644 --- a/providers/aihubmix/models/gpt-5.2.toml +++ b/providers/aihubmix/models/gpt-5.2.toml @@ -1,27 +1,12 @@ -name = "GPT-5.2" +base_model = "openai/gpt-5.2" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" -family = "gpt" -release_date = "2025-12-11" -last_updated = "2025-12-11" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] temperature = false -knowledge = "2025-08-31" -tool_call = true -structured_output = true -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 1.75 -output = 14.00 +output = 14 cache_read = 0.175 - -[limit] -context = 400_000 -input = 272_000 -output = 128_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.3-chat-latest.toml b/providers/aihubmix/models/gpt-5.3-chat-latest.toml new file mode 100644 index 00000000000..197284ef5f0 --- /dev/null +++ b/providers/aihubmix/models/gpt-5.3-chat-latest.toml @@ -0,0 +1,6 @@ +base_model = "openai/gpt-5.3-chat-latest" + +[cost] +input = 1.75 +output = 14 +cache_read = 0.175 diff --git a/providers/aihubmix/models/gpt-5.3-codex.toml b/providers/aihubmix/models/gpt-5.3-codex.toml index 82ef5c400e3..b4ea92ff965 100644 --- a/providers/aihubmix/models/gpt-5.3-codex.toml +++ b/providers/aihubmix/models/gpt-5.3-codex.toml @@ -1,27 +1,15 @@ -name = "GPT-5.3 Codex" +base_model = "openai/gpt-5.3-codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -family = "gpt-codex" -release_date = "2026-02-05" -last_updated = "2026-02-05" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] temperature = false -knowledge = "2025-08-31" -tool_call = true -structured_output = true -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] [cost] input = 1.75 -output = 14.00 +output = 14 cache_read = 0.175 -[limit] -context = 400_000 -input = 272_000 -output = 128_000 - [modalities] -input = ["text", "image", "pdf"] -output = ["text"] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.4-mini.toml b/providers/aihubmix/models/gpt-5.4-mini.toml index 5697dfff19f..afd01cdd821 100644 --- a/providers/aihubmix/models/gpt-5.4-mini.toml +++ b/providers/aihubmix/models/gpt-5.4-mini.toml @@ -1,31 +1,12 @@ -name = "GPT-5.4 mini" +base_model = "openai/gpt-5.4-mini" description = "Compact GPT model for low-latency assistance and high-volume workloads" -family = "gpt-mini" -release_date = "2026-03-17" -last_updated = "2026-03-17" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] temperature = false -knowledge = "2025-08-31" -tool_call = true -structured_output = true -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 0.75 -output = 4.50 +output = 4.5 cache_read = 0.075 - -[limit] -context = 400_000 -input = 272_000 -output = 128_000 - -[modalities] -input = ["text", "image"] -output = ["text"] - -[experimental.modes.fast] -cost = { input = 1.50, output = 9.00, cache_read = 0.15 } -provider = { body = { service_tier = "priority" } } diff --git a/providers/aihubmix/models/gpt-5.4-nano.toml b/providers/aihubmix/models/gpt-5.4-nano.toml new file mode 100644 index 00000000000..8d6d5218295 --- /dev/null +++ b/providers/aihubmix/models/gpt-5.4-nano.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-5.4-nano" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] + +[cost] +input = 0.2 +output = 1.25 +cache_read = 0.02 diff --git a/providers/aihubmix/models/gpt-5.4-pro.toml b/providers/aihubmix/models/gpt-5.4-pro.toml new file mode 100644 index 00000000000..6765bbb4917 --- /dev/null +++ b/providers/aihubmix/models/gpt-5.4-pro.toml @@ -0,0 +1,15 @@ +base_model = "openai/gpt-5.4-pro" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["medium", "high", "xhigh"] + +[cost] +input = 30 +output = 180 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 60 +output = 270 diff --git a/providers/aihubmix/models/gpt-5.4.toml b/providers/aihubmix/models/gpt-5.4.toml index 1ee0c18fe50..8ae4fe344f4 100644 --- a/providers/aihubmix/models/gpt-5.4.toml +++ b/providers/aihubmix/models/gpt-5.4.toml @@ -1,37 +1,21 @@ -name = "GPT-5.4" +base_model = "openai/gpt-5.4" description = "Frontier GPT model for professional reasoning, coding, and multimodal work" -family = "gpt" -release_date = "2026-03-05" -last_updated = "2026-03-05" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] temperature = false -knowledge = "2025-08-31" -tool_call = true -structured_output = true -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] -input = 2.50 -output = 15.00 +input = 2.5 +output = 15 cache_read = 0.25 [[cost.tiers]] -tier = { size = 272_000 } -input = 5.00 -output = 22.50 -cache_read = 0.50 - -[limit] -context = 1_050_000 -input = 922_000 -output = 128_000 +tier = { type = "context", size = 272_000 } +input = 5 +output = 22.5 +cache_read = 0.5 [modalities] -input = ["text", "image", "pdf"] -output = ["text"] - -[experimental.modes.fast] -cost = { input = 5.00, output = 30.00, cache_read = 0.50 } -provider = { body = { service_tier = "priority" } } +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.5-free.toml b/providers/aihubmix/models/gpt-5.5-free.toml new file mode 100644 index 00000000000..db3d76ed2cc --- /dev/null +++ b/providers/aihubmix/models/gpt-5.5-free.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-5.5" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] + +[cost] +input = 0 +output = 0 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.5-pro.toml b/providers/aihubmix/models/gpt-5.5-pro.toml new file mode 100644 index 00000000000..fbf1ce5c232 --- /dev/null +++ b/providers/aihubmix/models/gpt-5.5-pro.toml @@ -0,0 +1,17 @@ +base_model = "openai/gpt-5.5-pro" + +[[reasoning_options]] +type = "effort" +values = ["medium", "high", "xhigh"] + +[cost] +input = 30 +output = 180 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 60 +output = 270 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.5.toml b/providers/aihubmix/models/gpt-5.5.toml index ad2fe4a56d7..1de661d6511 100644 --- a/providers/aihubmix/models/gpt-5.5.toml +++ b/providers/aihubmix/models/gpt-5.5.toml @@ -1,37 +1,20 @@ -name = "GPT-5.5" +base_model = "openai/gpt-5.5" description = "Frontier GPT model for professional reasoning, coding, and multimodal work" -family = "gpt" -release_date = "2026-04-23" -last_updated = "2026-04-23" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] -temperature = false -knowledge = "2025-12-01" -tool_call = true -structured_output = true -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] -input = 5.00 -output = 30.00 -cache_read = 0.50 +input = 5 +output = 30 +cache_read = 0.5 [[cost.tiers]] -tier = { size = 272_000 } -input = 10.00 -output = 45.00 -cache_read = 1.00 - -[limit] -context = 1_050_000 -input = 922_000 -output = 128_000 +tier = { type = "context", size = 272_000 } +input = 10 +output = 45 +cache_read = 1 [modalities] -input = ["text", "image", "pdf"] -output = ["text"] - -[experimental.modes.fast] -cost = { input = 12.50, output = 75.00, cache_read = 1.25 } -provider = { body = { service_tier = "priority" } } +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.6-luna.toml b/providers/aihubmix/models/gpt-5.6-luna.toml index 3bfd7a75d89..6271edcdca5 100644 --- a/providers/aihubmix/models/gpt-5.6-luna.toml +++ b/providers/aihubmix/models/gpt-5.6-luna.toml @@ -1,5 +1,4 @@ base_model = "openai/gpt-5.6-luna" -temperature = false [[reasoning_options]] type = "effort" @@ -11,10 +10,12 @@ output = 1.2 cache_read = 0.02 cache_write = 0.25 -[limit] -context = 1_050_000 -output = 128_000 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 0.4 +output = 1.8 +cache_read = 0.04 +cache_write = 0.5 [modalities] input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.6-sol-disc.toml b/providers/aihubmix/models/gpt-5.6-sol-disc.toml new file mode 100644 index 00000000000..ceb6dc11d8c --- /dev/null +++ b/providers/aihubmix/models/gpt-5.6-sol-disc.toml @@ -0,0 +1,21 @@ +base_model = "openai/gpt-5.6-sol" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 4 +output = 20 +cache_read = 0.4 +cache_write = 5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 8 +output = 30 +cache_read = 0.8 +cache_write = 10 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.6-sol.toml b/providers/aihubmix/models/gpt-5.6-sol.toml index 4150cc42d58..ceb6dc11d8c 100644 --- a/providers/aihubmix/models/gpt-5.6-sol.toml +++ b/providers/aihubmix/models/gpt-5.6-sol.toml @@ -1,5 +1,4 @@ base_model = "openai/gpt-5.6-sol" -temperature = false [[reasoning_options]] type = "effort" @@ -11,10 +10,12 @@ output = 20 cache_read = 0.4 cache_write = 5 -[limit] -context = 1_050_000 -output = 128_000 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 8 +output = 30 +cache_read = 0.8 +cache_write = 10 [modalities] input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.6-terra.toml b/providers/aihubmix/models/gpt-5.6-terra.toml index 7a13fcf9dc4..2236c6b23c2 100644 --- a/providers/aihubmix/models/gpt-5.6-terra.toml +++ b/providers/aihubmix/models/gpt-5.6-terra.toml @@ -1,5 +1,4 @@ base_model = "openai/gpt-5.6-terra" -temperature = false [[reasoning_options]] type = "effort" @@ -11,10 +10,12 @@ output = 12 cache_read = 0.2 cache_write = 2.5 -[limit] -context = 1_050_000 -output = 128_000 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 4 +output = 18 +cache_read = 0.4 +cache_write = 5 [modalities] input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.toml b/providers/aihubmix/models/gpt-5.toml new file mode 100644 index 00000000000..f1d834f171a --- /dev/null +++ b/providers/aihubmix/models/gpt-5.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-5" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 diff --git a/providers/aihubmix/models/gpt-6-astra.toml b/providers/aihubmix/models/gpt-6-astra.toml new file mode 100644 index 00000000000..3f1fcc6d753 --- /dev/null +++ b/providers/aihubmix/models/gpt-6-astra.toml @@ -0,0 +1,21 @@ +base_model = "openai/gpt-6-astra" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 20 +output = 75 +cache_read = 2 +cache_write = 25 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-image-2-free.toml b/providers/aihubmix/models/gpt-image-2-free.toml new file mode 100644 index 00000000000..63bece6ae44 --- /dev/null +++ b/providers/aihubmix/models/gpt-image-2-free.toml @@ -0,0 +1,8 @@ +base_model = "openai/gpt-image-2" + +[cost] +input = 0 +output = 0 + +[modalities] +output = ["image", "text"] diff --git a/providers/aihubmix/models/gpt-oss-120b.toml b/providers/aihubmix/models/gpt-oss-120b.toml new file mode 100644 index 00000000000..afef8b74204 --- /dev/null +++ b/providers/aihubmix/models/gpt-oss-120b.toml @@ -0,0 +1,6 @@ +base_model = "openai/gpt-oss-120b" +reasoning_options = [] + +[cost] +input = 0.18 +output = 0.9 diff --git a/providers/aihubmix/models/gpt-oss-20b-free.toml b/providers/aihubmix/models/gpt-oss-20b-free.toml new file mode 100644 index 00000000000..9ddc252b4d6 --- /dev/null +++ b/providers/aihubmix/models/gpt-oss-20b-free.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-oss-20b" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 0 +output = 0 + +[limit] +output = 131_072 diff --git a/providers/aihubmix/models/gpt-oss-20b.toml b/providers/aihubmix/models/gpt-oss-20b.toml new file mode 100644 index 00000000000..7e30e0e460f --- /dev/null +++ b/providers/aihubmix/models/gpt-oss-20b.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-oss-20b" +reasoning_options = [] + +[cost] +input = 0.11 +output = 0.55 + +[limit] +context = 128_000 +output = 128_000 diff --git a/providers/aihubmix/models/grok-4.3.toml b/providers/aihubmix/models/grok-4.3.toml index 82cee7b576b..dc49fbfb558 100644 --- a/providers/aihubmix/models/grok-4.3.toml +++ b/providers/aihubmix/models/grok-4.3.toml @@ -1,15 +1,9 @@ -name = "Grok 4.3" +base_model = "xai/grok-4.3" description = "Grok model for agentic tool use, reasoning, coding, and live assistance" -family = "grok" -release_date = "2026-05-01" -last_updated = "2026-05-01" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] -temperature = true -tool_call = true -structured_output = true -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] [cost] input = 1.25 @@ -17,15 +11,13 @@ output = 2.5 cache_read = 0.2 [[cost.tiers]] -tier = { size = 200_000 } +tier = { type = "context", size = 200_000 } input = 2.5 -output = 5.0 +output = 5 cache_read = 0.4 [limit] -context = 1_000_000 output = 1_000_000 [modalities] input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/grok-4.5.toml b/providers/aihubmix/models/grok-4.5.toml index 9545ba4a2d5..1fac2993623 100644 --- a/providers/aihubmix/models/grok-4.5.toml +++ b/providers/aihubmix/models/grok-4.5.toml @@ -1,15 +1,16 @@ base_model = "xai/grok-4.5" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 2 output = 6 cache_read = 0.5 -[limit] -context = 1_000_000 -output = 1_000_000 - -[modalities] -input = ["text", "image"] -output = ["text"] +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4.4 +output = 13.2 +cache_read = 1.1 diff --git a/providers/aihubmix/models/grok-4.6.toml b/providers/aihubmix/models/grok-4.6.toml index 5264268a96f..fd607334f23 100644 --- a/providers/aihubmix/models/grok-4.6.toml +++ b/providers/aihubmix/models/grok-4.6.toml @@ -1,7 +1,16 @@ base_model = "xai/grok-4.6" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh"] [cost] input = 2 output = 6 cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4.4 +output = 13.2 +cache_read = 1.1 diff --git a/providers/aihubmix/models/grok-build-0.1.toml b/providers/aihubmix/models/grok-build-0.1.toml index ff7db61874c..b12f06edb15 100644 --- a/providers/aihubmix/models/grok-build-0.1.toml +++ b/providers/aihubmix/models/grok-build-0.1.toml @@ -6,10 +6,11 @@ input = 1 output = 2 cache_read = 0.2 -[limit] -context = 256_000 -output = 256_000 +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2 +output = 2 +cache_read = 0.4 [modalities] input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/hy3-free.toml b/providers/aihubmix/models/hy3-free.toml new file mode 100644 index 00000000000..0fcbaf489e2 --- /dev/null +++ b/providers/aihubmix/models/hy3-free.toml @@ -0,0 +1,13 @@ +base_model = "tencent/hy3" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high"] + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/hy3-preview.toml b/providers/aihubmix/models/hy3-preview.toml index 0ea62d7c422..91e8ba00aea 100644 --- a/providers/aihubmix/models/hy3-preview.toml +++ b/providers/aihubmix/models/hy3-preview.toml @@ -1,17 +1,30 @@ base_model = "tencent/hy3-preview" name = "Hy3 Preview" structured_output = true -reasoning_options = [{ type = "effort", values = ["none", "low", "high"] }] + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high"] [cost] input = 0.17 output = 0.566661 cache_read = 0.051 +[[cost.tiers]] +tier = { type = "context", size = 16_000 } +input = 0.2254 +output = 0.9016 +cache_read = 0.084525 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.2818 +output = 1.1272 +cache_read = 0.11272 + [limit] -context = 256_000 output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/hy3.toml b/providers/aihubmix/models/hy3.toml new file mode 100644 index 00000000000..38502dbd27a --- /dev/null +++ b/providers/aihubmix/models/hy3.toml @@ -0,0 +1,14 @@ +base_model = "tencent/hy3" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "high"] + +[cost] +input = 0.1562 +output = 0.6248 +cache_read = 0.03905 diff --git a/providers/aihubmix/models/hy4-preview.toml b/providers/aihubmix/models/hy4-preview.toml new file mode 100644 index 00000000000..ab83f489ac5 --- /dev/null +++ b/providers/aihubmix/models/hy4-preview.toml @@ -0,0 +1,20 @@ +base_model = "tencent/hy4-preview" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.845 +output = 2.535 +cache_read = 0.04225 + +[limit] +context = 1_048_576 diff --git a/providers/aihubmix/models/kimi-k2-thinking.toml b/providers/aihubmix/models/kimi-k2-thinking.toml new file mode 100644 index 00000000000..9eb147d8db3 --- /dev/null +++ b/providers/aihubmix/models/kimi-k2-thinking.toml @@ -0,0 +1,8 @@ +base_model = "moonshotai/kimi-k2-thinking" +structured_output = true +reasoning_options = [] + +[cost] +input = 0.548 +output = 2.192 +cache_read = 0.137 diff --git a/providers/aihubmix/models/kimi-k2.5.toml b/providers/aihubmix/models/kimi-k2.5.toml index 280040aed77..1923fec867c 100644 --- a/providers/aihubmix/models/kimi-k2.5.toml +++ b/providers/aihubmix/models/kimi-k2.5.toml @@ -1,15 +1,5 @@ -name = "Kimi K2.5" +base_model = "moonshotai/kimi-k2.5" description = "Kimi multimodal agent model for visual understanding, coding, and planning" -family = "kimi-k2" -release_date = "2026-01" -last_updated = "2026-01" -attachment = true -reasoning = true -temperature = false -tool_call = true -structured_output = true -knowledge = "2025-01" -open_weights = true [interleaved] field = "reasoning_content" @@ -23,9 +13,5 @@ output = 3 cache_read = 0.105 [limit] -context = 262_144 +context = 256_000 output = 32_768 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2.6.toml b/providers/aihubmix/models/kimi-k2.6.toml index 29b985faca7..b352a4b2103 100644 --- a/providers/aihubmix/models/kimi-k2.6.toml +++ b/providers/aihubmix/models/kimi-k2.6.toml @@ -1,15 +1,6 @@ -name = "Kimi K2.6" +base_model = "moonshotai/kimi-k2.6" description = "Kimi multimodal agent model for visual understanding, coding, and planning" -family = "kimi-k2" -release_date = "2026-04-21" -last_updated = "2026-04-21" -attachment = true -reasoning = true temperature = false -tool_call = true -structured_output = true -knowledge = "2025-01" -open_weights = true [interleaved] field = "reasoning_content" @@ -23,9 +14,4 @@ output = 3.9995 cache_read = 0.160835 [limit] -context = 262_144 output = 32_768 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2.7-code-highspeed.toml b/providers/aihubmix/models/kimi-k2.7-code-highspeed.toml index a64a84d4cd5..eb3684dc6f4 100644 --- a/providers/aihubmix/models/kimi-k2.7-code-highspeed.toml +++ b/providers/aihubmix/models/kimi-k2.7-code-highspeed.toml @@ -1,18 +1,15 @@ base_model = "moonshotai/kimi-k2.7-code-highspeed" -reasoning_options = [{ type = "toggle" }] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 1.9 output = 7.999 cache_read = 0.32167 [limit] -context = 262_144 output = 32_768 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2.7-code.toml b/providers/aihubmix/models/kimi-k2.7-code.toml index 529b52a2d8b..edaca21f35b 100644 --- a/providers/aihubmix/models/kimi-k2.7-code.toml +++ b/providers/aihubmix/models/kimi-k2.7-code.toml @@ -1,18 +1,15 @@ base_model = "moonshotai/kimi-k2.7-code" -reasoning_options = [{ type = "toggle" }] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0.95 output = 3.9995 cache_read = 0.160835 [limit] -context = 262_144 output = 32_768 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/kimi-k3.toml b/providers/aihubmix/models/kimi-k3.toml index cc93d685469..21e178e7339 100644 --- a/providers/aihubmix/models/kimi-k3.toml +++ b/providers/aihubmix/models/kimi-k3.toml @@ -2,9 +2,7 @@ # (text, vision, video), and a 1M-token context window. # Source accessed 2026-07-27: # https://aihubmix.com/model/kimi-k3 - base_model = "moonshotai/kimi-k3" -last_updated = "2026-07-27" [interleaved] field = "reasoning_content" @@ -17,10 +15,9 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 +input = 3 +output = 15 +cache_read = 0.3 -[modalities] -input = ["text", "image", "video"] -output = ["text"] \ No newline at end of file +[limit] +output = 1_048_576 diff --git a/providers/aihubmix/models/llama-3.3-70b-instruct.toml b/providers/aihubmix/models/llama-3.3-70b-instruct.toml new file mode 100644 index 00000000000..88ae0b0bfd2 --- /dev/null +++ b/providers/aihubmix/models/llama-3.3-70b-instruct.toml @@ -0,0 +1,10 @@ +base_model = "meta/llama-3.3-70b-instruct" +attachment = false +tool_call = false + +[cost] +input = 0.6 +output = 1.2 + +[limit] +context = 131_072 diff --git a/providers/aihubmix/models/longcat-2.0.toml b/providers/aihubmix/models/longcat-2.0.toml new file mode 100644 index 00000000000..62f5ab0db00 --- /dev/null +++ b/providers/aihubmix/models/longcat-2.0.toml @@ -0,0 +1,9 @@ +base_model = "meituan/longcat-2.0" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.7746 +output = 3.0984 +cache_read = 0.015492 diff --git a/providers/aihubmix/models/mimo-v2-flash-free.toml b/providers/aihubmix/models/mimo-v2-flash-free.toml new file mode 100644 index 00000000000..fa19dce2036 --- /dev/null +++ b/providers/aihubmix/models/mimo-v2-flash-free.toml @@ -0,0 +1,11 @@ +base_model = "xiaomi/mimo-v2-flash" +reasoning = false +tool_call = false + +[cost] +input = 0 +output = 0 + +[limit] +context = 256_000 +output = 256_000 diff --git a/providers/aihubmix/models/mimo-v2-flash.toml b/providers/aihubmix/models/mimo-v2-flash.toml new file mode 100644 index 00000000000..e367b1430ed --- /dev/null +++ b/providers/aihubmix/models/mimo-v2-flash.toml @@ -0,0 +1,18 @@ +base_model = "xiaomi/mimo-v2-flash" +attachment = true +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.1918 +output = 0.5754 +cache_read = 0.03836 + +[limit] +context = 1_000_000 +output = 131_072 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/mimo-v2-omni.toml b/providers/aihubmix/models/mimo-v2-omni.toml new file mode 100644 index 00000000000..a873d045063 --- /dev/null +++ b/providers/aihubmix/models/mimo-v2-omni.toml @@ -0,0 +1,14 @@ +base_model = "xiaomi/mimo-v2-omni" +reasoning = false +tool_call = false + +[cost] +input = 0.44 +output = 2.2 +cache_read = 0.088 + +[limit] +context = 256_000 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/mimo-v2-pro.toml b/providers/aihubmix/models/mimo-v2-pro.toml new file mode 100644 index 00000000000..71418db87b0 --- /dev/null +++ b/providers/aihubmix/models/mimo-v2-pro.toml @@ -0,0 +1,17 @@ +base_model = "xiaomi/mimo-v2-pro" +reasoning = false +tool_call = false + +[cost] +input = 1.1 +output = 3.3 +cache_read = 0.22 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 2.2 +output = 6.6 +cache_read = 0.44 + +[limit] +context = 1_000_000 diff --git a/providers/aihubmix/models/mimo-v2.5-pro.toml b/providers/aihubmix/models/mimo-v2.5-pro.toml new file mode 100644 index 00000000000..3dfd0499613 --- /dev/null +++ b/providers/aihubmix/models/mimo-v2.5-pro.toml @@ -0,0 +1,13 @@ +base_model = "xiaomi/mimo-v2.5-pro" +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.48 +output = 0.96 +cache_read = 0.00384 + +[limit] +context = 1_000_000 diff --git a/providers/aihubmix/models/mimo-v2.5.toml b/providers/aihubmix/models/mimo-v2.5.toml new file mode 100644 index 00000000000..43355271bdc --- /dev/null +++ b/providers/aihubmix/models/mimo-v2.5.toml @@ -0,0 +1,13 @@ +base_model = "xiaomi/mimo-v2.5" +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.155 +output = 0.31 +cache_read = 0.0031 + +[limit] +context = 1_000_000 diff --git a/providers/aihubmix/models/minimax-m2.7.toml b/providers/aihubmix/models/minimax-m2.7.toml index e91777c895b..399fe8d146a 100644 --- a/providers/aihubmix/models/minimax-m2.7.toml +++ b/providers/aihubmix/models/minimax-m2.7.toml @@ -9,20 +9,25 @@ temperature = true tool_call = true structured_output = true open_weights = true -reasoning_options = [] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.2958 output = 1.1832 cache_read = 0.05916 -cache_write = 0.375 [limit] context = 204_800 -output = 128_000 +output = 204_800 [modalities] input = ["text"] diff --git a/providers/aihubmix/models/nemotron-3-nano-30b-a3b-free.toml b/providers/aihubmix/models/nemotron-3-nano-30b-a3b-free.toml new file mode 100644 index 00000000000..d00601f3ca4 --- /dev/null +++ b/providers/aihubmix/models/nemotron-3-nano-30b-a3b-free.toml @@ -0,0 +1,15 @@ +base_model = "nvidia/nemotron-3-nano-30b-a3b" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_048_576 +output = 1_048_576 diff --git a/providers/aihubmix/models/nemotron-3-nano-omni-30b-a3b-reasoning-free.toml b/providers/aihubmix/models/nemotron-3-nano-omni-30b-a3b-reasoning-free.toml new file mode 100644 index 00000000000..1cfeb8d900c --- /dev/null +++ b/providers/aihubmix/models/nemotron-3-nano-omni-30b-a3b-reasoning-free.toml @@ -0,0 +1,14 @@ +base_model = "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 + +[limit] +context = 262_144 diff --git a/providers/aihubmix/models/nemotron-3-super-120b-a12b-free.toml b/providers/aihubmix/models/nemotron-3-super-120b-a12b-free.toml new file mode 100644 index 00000000000..bbb11b6100c --- /dev/null +++ b/providers/aihubmix/models/nemotron-3-super-120b-a12b-free.toml @@ -0,0 +1,11 @@ +base_model = "nvidia/nemotron-3-super-120b-a12b" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_048_576 diff --git a/providers/aihubmix/models/nemotron-3-ultra-550b-a55b-free.toml b/providers/aihubmix/models/nemotron-3-ultra-550b-a55b-free.toml new file mode 100644 index 00000000000..9be7df33a6e --- /dev/null +++ b/providers/aihubmix/models/nemotron-3-ultra-550b-a55b-free.toml @@ -0,0 +1,14 @@ +base_model = "nvidia/nemotron-3-ultra-550b-a55b" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_048_576 diff --git a/providers/aihubmix/models/nemotron-3.5-content-safety-free.toml b/providers/aihubmix/models/nemotron-3.5-content-safety-free.toml new file mode 100644 index 00000000000..da1f6cacd6b --- /dev/null +++ b/providers/aihubmix/models/nemotron-3.5-content-safety-free.toml @@ -0,0 +1,11 @@ +base_model = "nvidia/nemotron-3.5-content-safety" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 + +[limit] +context = 131_072 diff --git a/providers/aihubmix/models/nemotron-3.5-lightning-free.toml b/providers/aihubmix/models/nemotron-3.5-lightning-free.toml new file mode 100644 index 00000000000..030e52ea0f4 --- /dev/null +++ b/providers/aihubmix/models/nemotron-3.5-lightning-free.toml @@ -0,0 +1,11 @@ +base_model = "nvidia/nemotron-3.5-lightning" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_048_576 diff --git a/providers/aihubmix/models/nemotron-nano-12b-v2-vl-free.toml b/providers/aihubmix/models/nemotron-nano-12b-v2-vl-free.toml new file mode 100644 index 00000000000..947e56253a9 --- /dev/null +++ b/providers/aihubmix/models/nemotron-nano-12b-v2-vl-free.toml @@ -0,0 +1,11 @@ +base_model = "nvidia/nemotron-nano-12b-v2-vl" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 + +[limit] +context = 131_072 diff --git a/providers/aihubmix/models/nemotron-nano-9b-v2-free.toml b/providers/aihubmix/models/nemotron-nano-9b-v2-free.toml new file mode 100644 index 00000000000..bc0512d2657 --- /dev/null +++ b/providers/aihubmix/models/nemotron-nano-9b-v2-free.toml @@ -0,0 +1,11 @@ +base_model = "nvidia/nemotron-nano-9b-v2" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/o1-preview.toml b/providers/aihubmix/models/o1-preview.toml new file mode 100644 index 00000000000..df9fac6088e --- /dev/null +++ b/providers/aihubmix/models/o1-preview.toml @@ -0,0 +1,11 @@ +base_model = "openai/o1" +tool_call = false +reasoning_options = [] + +[cost] +input = 15 +output = 60 +cache_read = 7.5 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/o1-pro.toml b/providers/aihubmix/models/o1-pro.toml new file mode 100644 index 00000000000..960adb68778 --- /dev/null +++ b/providers/aihubmix/models/o1-pro.toml @@ -0,0 +1,12 @@ +base_model = "openai/o1-pro" +attachment = false +tool_call = false +reasoning_options = [] + +[cost] +input = 170 +output = 680 +cache_read = 170 + +[modalities] +input = ["text"] diff --git a/providers/aihubmix/models/o1.toml b/providers/aihubmix/models/o1.toml new file mode 100644 index 00000000000..e560a52586b --- /dev/null +++ b/providers/aihubmix/models/o1.toml @@ -0,0 +1,12 @@ +base_model = "openai/o1" +attachment = false +tool_call = false +reasoning_options = [] + +[cost] +input = 15 +output = 60 +cache_read = 7.5 + +[modalities] +input = ["text"] diff --git a/providers/aihubmix/models/o3-mini.toml b/providers/aihubmix/models/o3-mini.toml new file mode 100644 index 00000000000..1a0d9b8945a --- /dev/null +++ b/providers/aihubmix/models/o3-mini.toml @@ -0,0 +1,12 @@ +base_model = "openai/o3-mini" +attachment = true +tool_call = false +reasoning_options = [] + +[cost] +input = 1.1 +output = 4.4 +cache_read = 0.55 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/o3-pro.toml b/providers/aihubmix/models/o3-pro.toml new file mode 100644 index 00000000000..75813e2bcd7 --- /dev/null +++ b/providers/aihubmix/models/o3-pro.toml @@ -0,0 +1,7 @@ +base_model = "openai/o3-pro" +reasoning_options = [] + +[cost] +input = 20 +output = 80 +cache_read = 20 diff --git a/providers/aihubmix/models/o3.toml b/providers/aihubmix/models/o3.toml new file mode 100644 index 00000000000..e8f09e33452 --- /dev/null +++ b/providers/aihubmix/models/o3.toml @@ -0,0 +1,13 @@ +base_model = "openai/o3" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 2 +output = 8 +cache_read = 0.5 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/o4-mini.toml b/providers/aihubmix/models/o4-mini.toml new file mode 100644 index 00000000000..75807e417c1 --- /dev/null +++ b/providers/aihubmix/models/o4-mini.toml @@ -0,0 +1,10 @@ +base_model = "openai/o4-mini" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1.1 +output = 4.4 +cache_read = 0.275 diff --git a/providers/aihubmix/models/qwen-turbo.toml b/providers/aihubmix/models/qwen-turbo.toml new file mode 100644 index 00000000000..3b191626bcc --- /dev/null +++ b/providers/aihubmix/models/qwen-turbo.toml @@ -0,0 +1,8 @@ +base_model = "alibaba/qwen-turbo" +reasoning = false +tool_call = false + +[cost] +input = 0.046 +output = 0.092 +cache_read = 0.0092 diff --git a/providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml b/providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml new file mode 100644 index 00000000000..b5282afe4b1 --- /dev/null +++ b/providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml @@ -0,0 +1,13 @@ +base_model = "alibaba/qwen3-235b-a22b-instruct-2507" +attachment = true +structured_output = true + +[cost] +input = 0.28 +output = 1.12 + +[limit] +output = 262_144 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-235b-a22b.toml b/providers/aihubmix/models/qwen3-235b-a22b.toml new file mode 100644 index 00000000000..f4b5bbbc406 --- /dev/null +++ b/providers/aihubmix/models/qwen3-235b-a22b.toml @@ -0,0 +1,11 @@ +base_model = "alibaba/qwen3-235b-a22b" +reasoning = false +structured_output = true + +[cost] +input = 0.28 +output = 1.12 + +[limit] +context = 131_100 +output = 128_000 diff --git a/providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml b/providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml new file mode 100644 index 00000000000..9b930792aa9 --- /dev/null +++ b/providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml @@ -0,0 +1,20 @@ +base_model = "alibaba/qwen3-coder-30b-a3b-instruct" +structured_output = true + +[cost] +input = 0.2 +output = 0.8 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.308218 +output = 1.232872 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 1.027396 +output = 5.13698 + +[limit] +context = 2_000_000 +output = 262_000 diff --git a/providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml b/providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml new file mode 100644 index 00000000000..c16a2996cd5 --- /dev/null +++ b/providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml @@ -0,0 +1,20 @@ +base_model = "alibaba/qwen3-coder-480b-a35b-instruct" +structured_output = true + +[cost] +input = 0.82 +output = 3.28 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 1.232876 +output = 4.931504 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 2.054794 +output = 8.219176 + +[limit] +context = 262_000 +output = 262_000 diff --git a/providers/aihubmix/models/qwen3-coder-flash.toml b/providers/aihubmix/models/qwen3-coder-flash.toml new file mode 100644 index 00000000000..9443089d686 --- /dev/null +++ b/providers/aihubmix/models/qwen3-coder-flash.toml @@ -0,0 +1,24 @@ +base_model = "alibaba/qwen3-coder-flash" +structured_output = true + +[cost] +input = 0.136 +output = 0.544 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.205478 +output = 0.821912 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.342464 +output = 1.369856 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.68493 +output = 3.42465 + +[limit] +context = 256_000 diff --git a/providers/aihubmix/models/qwen3-coder-next.toml b/providers/aihubmix/models/qwen3-coder-next.toml new file mode 100644 index 00000000000..7c3121eb26d --- /dev/null +++ b/providers/aihubmix/models/qwen3-coder-next.toml @@ -0,0 +1,15 @@ +base_model = "alibaba/qwen3-coder-next" + +[cost] +input = 0.137 +output = 0.548 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.2054 +output = 0.8216 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.3424 +output = 1.3696 diff --git a/providers/aihubmix/models/qwen3-coder-plus.toml b/providers/aihubmix/models/qwen3-coder-plus.toml new file mode 100644 index 00000000000..de385311d0a --- /dev/null +++ b/providers/aihubmix/models/qwen3-coder-plus.toml @@ -0,0 +1,22 @@ +base_model = "alibaba/qwen3-coder-plus" +structured_output = true + +[cost] +input = 0.54 +output = 2.16 +cache_read = 0.108 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.821916 +output = 3.287664 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 1.369862 +output = 5.479448 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 2.739726 +output = 27.39726 diff --git a/providers/aihubmix/models/qwen3-max-preview.toml b/providers/aihubmix/models/qwen3-max-preview.toml new file mode 100644 index 00000000000..bec3493a846 --- /dev/null +++ b/providers/aihubmix/models/qwen3-max-preview.toml @@ -0,0 +1,23 @@ +base_model = "alibaba/qwen3-max" +attachment = true +structured_output = true + +[cost] +input = 0.846 +output = 3.384 +cache_read = 0.1692 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 1.408 +output = 5.632 +cache_read = 0.2816 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 2.1128 +output = 8.4512 +cache_read = 0.42256 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-max.toml b/providers/aihubmix/models/qwen3-max.toml new file mode 100644 index 00000000000..4e0eae21c31 --- /dev/null +++ b/providers/aihubmix/models/qwen3-max.toml @@ -0,0 +1,26 @@ +base_model = "alibaba/qwen3-max" +attachment = true +structured_output = true + +[cost] +input = 0.4508 +output = 1.8032 +cache_read = 0.09016 +cache_write = 0.5635 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.902 +output = 5.412 +cache_read = 0.1804 +cache_write = 1.1275 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 1.3522 +output = 8.1132 +cache_read = 0.27044 +cache_write = 1.69025 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml b/providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml new file mode 100644 index 00000000000..db435e1b4f6 --- /dev/null +++ b/providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml @@ -0,0 +1,14 @@ +base_model = "alibaba/qwen3-next-80b-a3b-instruct" +attachment = true +structured_output = true + +[cost] +input = 0.138 +output = 0.552 + +[limit] +context = 256_000 +output = 256_000 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml b/providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml new file mode 100644 index 00000000000..66b7e4a3283 --- /dev/null +++ b/providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml @@ -0,0 +1,15 @@ +base_model = "alibaba/qwen3-next-80b-a3b-thinking" +attachment = true +structured_output = true +reasoning_options = [] + +[cost] +input = 0.142 +output = 1.42 + +[limit] +context = 256_000 +output = 256_000 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml b/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml new file mode 100644 index 00000000000..69a0206bbb9 --- /dev/null +++ b/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml @@ -0,0 +1,12 @@ +base_model = "alibaba/qwen3-vl-235b-a22b-instruct" + +[cost] +input = 0.274 +output = 1.096 + +[limit] +context = 131_000 +output = 33_000 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml b/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml new file mode 100644 index 00000000000..6e56da6e1ce --- /dev/null +++ b/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml @@ -0,0 +1,13 @@ +base_model = "alibaba/qwen3-vl-235b-a22b-thinking" +reasoning_options = [] + +[cost] +input = 0.274 +output = 2.74 + +[limit] +context = 131_000 +output = 33_000 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3-vl-plus.toml b/providers/aihubmix/models/qwen3-vl-plus.toml new file mode 100644 index 00000000000..7a0ac27b37b --- /dev/null +++ b/providers/aihubmix/models/qwen3-vl-plus.toml @@ -0,0 +1,26 @@ +base_model = "alibaba/qwen3-vl-plus" +attachment = true +reasoning = false +structured_output = true + +[cost] +input = 0.137 +output = 1.37 +cache_read = 0.0274 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.2054 +output = 2.054 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.411 +output = 4.11 + +[limit] +context = 256_000 +output = 32_000 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.5-122b-a10b.toml b/providers/aihubmix/models/qwen3.5-122b-a10b.toml new file mode 100644 index 00000000000..aa89331fc32 --- /dev/null +++ b/providers/aihubmix/models/qwen3.5-122b-a10b.toml @@ -0,0 +1,23 @@ +base_model = "alibaba/qwen3.5-122b-a10b" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1126 +output = 0.9008 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.2818 +output = 2.2544 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.5-27b.toml b/providers/aihubmix/models/qwen3.5-27b.toml new file mode 100644 index 00000000000..afe6ada6bd0 --- /dev/null +++ b/providers/aihubmix/models/qwen3.5-27b.toml @@ -0,0 +1,23 @@ +base_model = "alibaba/qwen3.5-27b" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.0846 +output = 0.6768 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.2536 +output = 2.0288 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.5-35b-a3b.toml b/providers/aihubmix/models/qwen3.5-35b-a3b.toml new file mode 100644 index 00000000000..cbef8c4b6a0 --- /dev/null +++ b/providers/aihubmix/models/qwen3.5-35b-a3b.toml @@ -0,0 +1,23 @@ +base_model = "alibaba/qwen3.5-35b-a3b" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.0564 +output = 0.4512 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.2254 +output = 1.8032 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.5-397b-a17b.toml b/providers/aihubmix/models/qwen3.5-397b-a17b.toml new file mode 100644 index 00000000000..d57dea0957d --- /dev/null +++ b/providers/aihubmix/models/qwen3.5-397b-a17b.toml @@ -0,0 +1,23 @@ +base_model = "alibaba/qwen3.5-397b-a17b" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1644 +output = 0.9864 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.411 +output = 2.466 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.5-flash.toml b/providers/aihubmix/models/qwen3.5-flash.toml new file mode 100644 index 00000000000..f7a2e526d20 --- /dev/null +++ b/providers/aihubmix/models/qwen3.5-flash.toml @@ -0,0 +1,31 @@ +base_model = "alibaba/qwen3.5-flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.0282 +output = 0.282 +cache_read = 0.00282 +cache_write = 0.03525 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.1126 +output = 1.126 +cache_read = 0.01126 +cache_write = 0.14075 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.169 +output = 1.69 +cache_read = 0.0169 +cache_write = 0.21125 diff --git a/providers/aihubmix/models/qwen3.5-plus.toml b/providers/aihubmix/models/qwen3.5-plus.toml new file mode 100644 index 00000000000..a6279667d80 --- /dev/null +++ b/providers/aihubmix/models/qwen3.5-plus.toml @@ -0,0 +1,33 @@ +base_model = "alibaba/qwen3.5-plus" +attachment = true +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1096 +output = 0.6576 +cache_read = 0.01096 +cache_write = 0.137 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.274 +output = 1.644 +cache_read = 0.0274 +cache_write = 0.3425 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.548 +output = 3.288 +cache_read = 0.0548 +cache_write = 0.685 diff --git a/providers/aihubmix/models/qwen3.6-27b.toml b/providers/aihubmix/models/qwen3.6-27b.toml new file mode 100644 index 00000000000..b7bcc334bc0 --- /dev/null +++ b/providers/aihubmix/models/qwen3.6-27b.toml @@ -0,0 +1,11 @@ +base_model = "alibaba/qwen3.6-27b" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.422 +output = 2.532 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.6-35b-a3b.toml b/providers/aihubmix/models/qwen3.6-35b-a3b.toml new file mode 100644 index 00000000000..5516b701f13 --- /dev/null +++ b/providers/aihubmix/models/qwen3.6-35b-a3b.toml @@ -0,0 +1,18 @@ +base_model = "alibaba/qwen3.6-35b-a3b" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.254 +output = 1.524 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.6-flash.toml b/providers/aihubmix/models/qwen3.6-flash.toml index 3da56465a7a..bf52da4cbcf 100644 --- a/providers/aihubmix/models/qwen3.6-flash.toml +++ b/providers/aihubmix/models/qwen3.6-flash.toml @@ -1,15 +1,5 @@ -name = "Qwen3.6 Flash" +base_model = "alibaba/qwen3.6-flash" description = "Multimodal reasoning model for visual analysis, planning, and tool use" -family = "qwen3.6" -release_date = "2026-04-02" -last_updated = "2026-04-02" -attachment = true -reasoning = true -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-04" -open_weights = false [interleaved] field = "reasoning_content" @@ -17,6 +7,13 @@ field = "reasoning_content" [[reasoning_options]] type = "toggle" +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.169 output = 1.014 @@ -25,15 +22,7 @@ cache_write = 0.21125 [[cost.tiers]] tier = { type = "context", size = 256_000 } -input = 0.68 -output = 4.06 +input = 0.676 +output = 4.056 cache_read = 0.0676 cache_write = 0.845 - -[limit] -context = 991_000 -output = 64_000 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/qwen3.6-max-preview.toml b/providers/aihubmix/models/qwen3.6-max-preview.toml index 299b558c20a..bab3477e100 100644 --- a/providers/aihubmix/models/qwen3.6-max-preview.toml +++ b/providers/aihubmix/models/qwen3.6-max-preview.toml @@ -1,15 +1,6 @@ -name = "Qwen3.6 Max Preview" +base_model = "alibaba/qwen3.6-max-preview" description = "Flagship model for demanding analysis, coding, and production agent workflows" -family = "qwen3.6" -release_date = "2026-05-09" -last_updated = "2026-05-09" -attachment = false -reasoning = true -temperature = true -tool_call = true structured_output = true -knowledge = "2025-04" -open_weights = false [interleaved] field = "reasoning_content" @@ -17,6 +8,9 @@ field = "reasoning_content" [[reasoning_options]] type = "toggle" +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 1.268 output = 7.608 @@ -25,15 +19,7 @@ cache_write = 1.585 [[cost.tiers]] tier = { type = "context", size = 128_000 } -input = 2.11 -output = 12.67 +input = 2.112 +output = 12.672 cache_read = 0.2112 cache_write = 2.64 - -[limit] -context = 240_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/qwen3.6-plus-preview-free.toml b/providers/aihubmix/models/qwen3.6-plus-preview-free.toml new file mode 100644 index 00000000000..c0771064bae --- /dev/null +++ b/providers/aihubmix/models/qwen3.6-plus-preview-free.toml @@ -0,0 +1,14 @@ +base_model = "alibaba/qwen3.6-plus" +attachment = false +tool_call = false +reasoning_options = [] + +[cost] +input = 0 +output = 0 + +[limit] +output = 65_535 + +[modalities] +input = ["text"] diff --git a/providers/aihubmix/models/qwen3.6-plus.toml b/providers/aihubmix/models/qwen3.6-plus.toml index 33b3dfc9bf9..800cb342527 100644 --- a/providers/aihubmix/models/qwen3.6-plus.toml +++ b/providers/aihubmix/models/qwen3.6-plus.toml @@ -1,15 +1,6 @@ -name = "Qwen3.6 Plus" +base_model = "alibaba/qwen3.6-plus" description = "Multimodal reasoning model for visual analysis, planning, and tool use" -family = "qwen3.6" -release_date = "2026-05-09" -last_updated = "2026-05-09" -attachment = true -reasoning = true -temperature = true -tool_call = true structured_output = true -knowledge = "2025-04" -open_weights = false [interleaved] field = "reasoning_content" @@ -17,6 +8,13 @@ field = "reasoning_content" [[reasoning_options]] type = "toggle" +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.282 output = 1.692 @@ -25,15 +23,7 @@ cache_write = 0.3525 [[cost.tiers]] tier = { type = "context", size = 256_000 } -input = 1.13 -output = 6.77 +input = 1.128 +output = 6.768 cache_read = 0.1128 cache_write = 1.41 - -[limit] -context = 991_000 -output = 64_000 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/qwen3.7-flash.toml b/providers/aihubmix/models/qwen3.7-flash.toml index 3f34f7fd74f..a97f1ac9ee4 100644 --- a/providers/aihubmix/models/qwen3.7-flash.toml +++ b/providers/aihubmix/models/qwen3.7-flash.toml @@ -1,22 +1,39 @@ # Toggle: enable_thinking = true|false # Budget: thinking_budget = integer reasoning tokens base_model = "alibaba/qwen3.7-flash" -attachment = false -reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", max = 262_144 }] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.0282 output = 0.1128 cache_read = 0.00564 cache_write = 0.03525 -[limit] -context = 991_000 -output = 64_000 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.0845 +output = 0.338 +cache_read = 0.0169 +cache_write = 0.105625 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.169 +output = 0.676 +cache_read = 0.0338 +cache_write = 0.21125 -[modalities] -input = ["text"] -output = ["text"] +[limit] +output = 131_072 diff --git a/providers/aihubmix/models/qwen3.7-max.toml b/providers/aihubmix/models/qwen3.7-max.toml index 3370cc1f374..6aa08a6f013 100644 --- a/providers/aihubmix/models/qwen3.7-max.toml +++ b/providers/aihubmix/models/qwen3.7-max.toml @@ -1,10 +1,19 @@ base_model = "alibaba/qwen3.7-max" structured_output = true -reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", max = 262_144 }] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 1.69 output = 5.07 @@ -12,9 +21,4 @@ cache_read = 0.169 cache_write = 2.1125 [limit] -context = 991_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 131_072 diff --git a/providers/aihubmix/models/qwen3.7-plus.toml b/providers/aihubmix/models/qwen3.7-plus.toml index d797f8535c7..c92a5926949 100644 --- a/providers/aihubmix/models/qwen3.7-plus.toml +++ b/providers/aihubmix/models/qwen3.7-plus.toml @@ -1,20 +1,31 @@ base_model = "alibaba/qwen3.7-plus" structured_output = true -reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", max = 262_144 }] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.282 output = 1.128 cache_read = 0.0564 cache_write = 0.3525 -[limit] -context = 991_000 -output = 64_000 +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.845 +output = 3.38 +cache_read = 0.169 +cache_write = 1.05625 -[modalities] -input = ["text"] -output = ["text"] +[limit] +output = 131_072 diff --git a/providers/aihubmix/models/qwen3.8-2.4t-a95b.toml b/providers/aihubmix/models/qwen3.8-2.4t-a95b.toml index 717b12b3b19..2d7c0a8782a 100644 --- a/providers/aihubmix/models/qwen3.8-2.4t-a95b.toml +++ b/providers/aihubmix/models/qwen3.8-2.4t-a95b.toml @@ -1,21 +1,24 @@ # AIHubMix Models API reports text,image input (queried 2026-08-31T04:06:29Z). # OpenAI-compatible HTTPS image input returned HTTP 200 (validated 2026-08-31T04:08:19Z). base_model = "alibaba/qwen3.8-2.4t-a95b" -attachment = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "xhigh"] }] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 2 output = 6 cache_read = 0.5 [limit] -context = 262_000 -output = 262_000 - -[modalities] -input = ["text", "image"] -output = ["text"] +context = 1_000_000 diff --git a/providers/aihubmix/models/qwen3.8-flash.toml b/providers/aihubmix/models/qwen3.8-flash.toml new file mode 100644 index 00000000000..c4be8d4c10b --- /dev/null +++ b/providers/aihubmix/models/qwen3.8-flash.toml @@ -0,0 +1,17 @@ +base_model = "alibaba/qwen3.8-flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1126 +output = 0.380025 +cache_read = 0.014075 +cache_write = 0.175937 diff --git a/providers/aihubmix/models/qwen3.8-max-preview.toml b/providers/aihubmix/models/qwen3.8-max-preview.toml new file mode 100644 index 00000000000..e160a294c1a --- /dev/null +++ b/providers/aihubmix/models/qwen3.8-max-preview.toml @@ -0,0 +1,18 @@ +base_model = "alibaba/qwen3.8-max-preview" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.338 +output = 1.014 +cache_read = 0.0676 +cache_write = 0.4225 diff --git a/providers/aihubmix/models/qwen3.8-max.toml b/providers/aihubmix/models/qwen3.8-max.toml index b9bd3e40b33..202ce3e10cc 100644 --- a/providers/aihubmix/models/qwen3.8-max.toml +++ b/providers/aihubmix/models/qwen3.8-max.toml @@ -1,26 +1,26 @@ # AIHubMix OpenAI-compatible /v1/chat/completions: $.enable_thinking = true|false (toggle) and $.reasoning_effort = "low"|"medium"|"xhigh"; verified live 2026-08-11. https://docs.aihubmix.com/cn/api/unified-inference # AIHubMix Models API (queried 2026-08-11T09:45:11Z): input 1.69, output 5.07, cache_read 0.169, cache_write 2.1125 USD/MTok. https://aihubmix.com/api/v1/models?model=qwen3.8-max base_model = "alibaba/qwen3.8-max" -attachment = false structured_output = true -reasoning_options = [ - { type = "toggle" }, - { type = "effort", values = ["low", "medium", "xhigh"] }, -] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 1.69 output = 5.07 cache_read = 0.169 cache_write = 2.1125 -[limit] -context = 991_000 -output = 128_000 - [modalities] -input = ["text"] -output = ["text"] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/solar-pro4.toml b/providers/aihubmix/models/solar-pro4.toml new file mode 100644 index 00000000000..7ea7766cc3e --- /dev/null +++ b/providers/aihubmix/models/solar-pro4.toml @@ -0,0 +1,8 @@ +base_model = "upstage/solar-pro4" +reasoning = false +tool_call = false + +[cost] +input = 0.3 +output = 1.2 +cache_read = 0.06 diff --git a/providers/aihubmix/models/step-3.5-flash.toml b/providers/aihubmix/models/step-3.5-flash.toml new file mode 100644 index 00000000000..3ecb188643b --- /dev/null +++ b/providers/aihubmix/models/step-3.5-flash.toml @@ -0,0 +1,9 @@ +base_model = "stepfun/step-3.5-flash" +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.11 +output = 0.33 diff --git a/providers/aihubmix/models/step-3.7-flash.toml b/providers/aihubmix/models/step-3.7-flash.toml new file mode 100644 index 00000000000..4cae5014cae --- /dev/null +++ b/providers/aihubmix/models/step-3.7-flash.toml @@ -0,0 +1,14 @@ +base_model = "stepfun/step-3.7-flash" +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 0.22 +output = 1.32 +cache_read = 0.044 diff --git a/providers/aihubmix/models/xiaomi-mimo-v2-omni-free.toml b/providers/aihubmix/models/xiaomi-mimo-v2-omni-free.toml new file mode 100644 index 00000000000..c2f827b8f3e --- /dev/null +++ b/providers/aihubmix/models/xiaomi-mimo-v2-omni-free.toml @@ -0,0 +1,13 @@ +base_model = "xiaomi/mimo-v2-omni" +reasoning = false +tool_call = false + +[cost] +input = 0 +output = 0 + +[limit] +context = 256_000 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/xiaomi-mimo-v2-pro-free.toml b/providers/aihubmix/models/xiaomi-mimo-v2-pro-free.toml new file mode 100644 index 00000000000..823d3fb6c43 --- /dev/null +++ b/providers/aihubmix/models/xiaomi-mimo-v2-pro-free.toml @@ -0,0 +1,10 @@ +base_model = "xiaomi/mimo-v2-pro" +reasoning = false +tool_call = false + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_000_000 diff --git a/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml b/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml index 5d9852914e9..fa578c977e6 100644 --- a/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml +++ b/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml @@ -1,13 +1,15 @@ base_model = "xiaomi/mimo-v2.5" -reasoning_options = [{ type = "toggle" }] name = "Xiaomi MiMo-V2.5 (free)" -family = "mimo-v2.5" -last_updated = "2026-05-13" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0 output = 0 -cache_read = 0 + +[limit] +context = 1_000_000 diff --git a/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml b/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml index a6277d9bb9c..17492a3e0f5 100644 --- a/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml +++ b/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml @@ -1,13 +1,15 @@ base_model = "xiaomi/mimo-v2.5-pro" -reasoning_options = [{ type = "toggle" }] name = "Xiaomi MiMo-V2.5-Pro (free)" -family = "mimo-v2.5-pro" -last_updated = "2026-05-13" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + [cost] input = 0 output = 0 -cache_read = 0 + +[limit] +context = 1_000_000 diff --git a/providers/aihubmix/models/zai-glm-5-turbo.toml b/providers/aihubmix/models/zai-glm-5-turbo.toml new file mode 100644 index 00000000000..486cfe4ab8c --- /dev/null +++ b/providers/aihubmix/models/zai-glm-5-turbo.toml @@ -0,0 +1,13 @@ +base_model = "zhipuai/glm-5-turbo" +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 1.2 +output = 3.9996 +cache_read = 0.24 + +[limit] +context = 204_800 diff --git a/sync.md b/sync.md index ff3ad5bbe69..363befcf6e0 100644 --- a/sync.md +++ b/sync.md @@ -253,12 +253,13 @@ xAI is implemented in `packages/core/src/sync/providers/xai.ts`. - AIHubMix is implemented in `packages/core/src/sync/providers/aihubmix.ts`. - Source endpoint: `https://aihubmix.com/api/v1/models?type=llm`. - No authentication is required; the catalog is public. -- The endpoint is authoritative for pricing and deprecation status only. Everything else in the authored TOMLs is preserved. -- Token limits and modalities are deliberately **not** synced: the endpoint reports the relay's conservative defaults rather than the upstream model's capabilities. It caps `context_length` per relay (Claude Opus 4.6 is listed at 200K against its 1M window), quotes `max_output` per default request, and never lists `pdf` in `input_modalities` even for models that accept PDFs. -- `cache_read` is ignored when it equals `input`. The endpoint echoes the input price for models with no cached rate configured: 35 of the 301 priced entries carry a nonzero price this way, and 51 more are free models reporting 0 across the board. Taking the echoed value literally would overstate Gemini 3.1 Flash Lite tenfold against the $0.025 every other provider lists. -- The free-text `features` list mixes synonyms (`thinking` vs `reasoning`, `tools` vs `tool_calling`) and never exposes accepted reasoning effort levels, so capability flags and `reasoning_options` stay hand-authored. -- AIHubMix relays roughly 400 upstream models against a much smaller hand-verified subset here, so new IDs are not created automatically (`skipCreates`); each missing ID opens a deduped GitHub issue. -- Routing aliases such as `alicloud-glm-5.1` and `deep-deepseek-v4-pro` are served but not listed by the endpoint, so local files missing from the response are retained (`deleteMissing: false`). +- The endpoint now serves capabilities, limits, modalities, reasoning controls and pricing, so it is authoritative for all of them. Fields it does not serve (`family`, `temperature`, `interleaved`, `knowledge`) keep whatever was authored. +- AIHubMix relays upstream models under its own IDs, so a relay is factored onto the lab metadata it serves (`base_model`) whenever that metadata exists, and then records only what it actually changes. `developer_id` maps a relay to its lab; routing prefixes (`coding-`, `alicloud-`) and suffixes (`-free`, `-think`, `-nothink`) select a mode rather than a different model and are stripped when resolving the base. +- A relay with neither resolvable lab metadata nor the `release_date`/`open_weights` a standalone entry requires is reported rather than written with invented values. AIHubMix dates 52 of its 415 models and serves no `open_weights` flag, so 200-odd relays are skipped on that basis today. +- `max_output: 0` is read as absent, not as a real ceiling: 102 of 415 models quote 0 for a limit the endpoint does not know. +- `reasoning_options[]` entries carry an AIHubMix-only `default` key that the strict `ReasoningOption` schema rejects, and spell two effort levels differently (`no_think`, `instant`), so translation drops the extra key and maps those onto `none` and `minimal`. +- A price field is omitted when the model has no such rate, so an omitted `cache_read`/`cache_write` clears an authored one; authored pricing survives only when the endpoint quotes nothing at all for the model. +- Routing aliases such as `alicloud-glm-5.1` and `deep-deepseek-v4-pro` are served but not listed by the endpoint, so local files missing from the response are retained (`deleteMissing: false`) and each opens a deduped GitHub issue. ## Tinfoil Notes From 63a9b6bf93e5fb32d242539975c7ae36ca899543 Mon Sep 17 00:00:00 2001 From: chenxue Date: Thu, 10 Sep 2026 15:38:55 +0800 Subject: [PATCH 04/20] fix(aihubmix): resolve relay casing and read backfilled limits as absent AIHubMix lowercases every relay ID while labs keep their own casing, so `minimax-m2` never matched `minimax/MiniMax-M2` and the whole MiniMax line fell through to the standalone path. The lab index is now case-folded, and `nvidia-`/`bai-` join the routing prefixes with `-highspeed`, `-fast` and `-latest` joining the suffixes. 24 relays that previously had no resolvable base now factor onto one. The endpoint signals an unknown output ceiling three ways: 0, the value of `context_length` (51 of 415 models, which would leave no room for the prompt), and a value above the window (6 models, up to 10x). All three are read as absent so the base model's real ceiling shows through. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 50 ++++++++++++++----- packages/core/test/sync.test.ts | 17 ++++++- .../aihubmix/models/DeepSeek-V3.1-Think.toml | 13 +++++ providers/aihubmix/models/DeepSeek-V3.toml | 9 ++++ providers/aihubmix/models/Qwen/QwQ-32B.toml | 7 +++ .../bai-qwen3-vl-235b-a22b-instruct.toml | 10 ++++ .../models/coding-minimax-m2-free.toml | 13 +++++ .../models/coding-minimax-m2.1-free.toml | 13 +++++ .../aihubmix/models/coding-minimax-m2.1.toml | 13 +++++ .../models/coding-minimax-m2.5-free.toml | 13 +++++ .../models/coding-minimax-m2.5-highspeed.toml | 13 +++++ .../aihubmix/models/coding-minimax-m2.5.toml | 13 +++++ .../models/coding-minimax-m2.7-free.toml | 14 +----- .../models/coding-minimax-m2.7-highspeed.toml | 14 +----- .../aihubmix/models/coding-minimax-m2.7.toml | 14 +----- .../aihubmix/models/coding-minimax-m2.toml | 13 +++++ .../models/coding-minimax-m3-free.toml | 16 ++++++ .../aihubmix/models/coding-minimax-m3.toml | 16 ++++++ .../models/coding-xiaomi-mimo-v2.5-pro.toml | 3 -- .../models/coding-xiaomi-mimo-v2.5.toml | 3 -- .../models/deepseek-v4-flash-0731-fast.toml | 13 +++++ .../models/deepseek-v4-flash-vision-exp.toml | 6 +-- .../models/gemini-2.5-flash-nothink.toml | 3 -- .../models/gemini-2.5-flash-search.toml | 3 -- .../aihubmix/models/gemini-2.5-flash.toml | 3 -- .../aihubmix/models/glm-5.2-fast-preview.toml | 13 +++++ providers/aihubmix/models/gpt-5.2-codex.toml | 3 -- providers/aihubmix/models/gpt-5.3-codex.toml | 3 -- providers/aihubmix/models/gpt-5.4.toml | 3 -- providers/aihubmix/models/gpt-5.5-free.toml | 3 -- providers/aihubmix/models/gpt-5.5.toml | 3 -- providers/aihubmix/models/kimi-k2.5.toml | 1 - providers/aihubmix/models/mimo-v2.5-pro.toml | 3 -- providers/aihubmix/models/mimo-v2.5.toml | 3 -- providers/aihubmix/models/minimax-m2.1.toml | 13 +++++ .../models/minimax-m2.5-highspeed.toml | 13 +++++ providers/aihubmix/models/minimax-m2.5.toml | 13 +++++ .../aihubmix/models/minimax-m2.7-free.toml | 13 +++++ providers/aihubmix/models/minimax-m2.7.toml | 14 +----- providers/aihubmix/models/minimax-m2.toml | 13 +++++ providers/aihubmix/models/minimax-m3.toml | 16 ++++++ .../nvidia-nemotron-3-super-120b-a12b.toml | 11 ++++ .../aihubmix/models/qwen-plus-latest.toml | 23 +++++++++ .../aihubmix/models/qwen-turbo-latest.toml | 8 +++ .../models/qwen3-coder-30b-a3b-instruct.toml | 1 - .../aihubmix/models/qwen3-coder-flash.toml | 3 -- .../models/xiaomi-mimo-v2.5-free.toml | 3 -- .../models/xiaomi-mimo-v2.5-pro-free.toml | 3 -- 48 files changed, 371 insertions(+), 116 deletions(-) create mode 100644 providers/aihubmix/models/DeepSeek-V3.1-Think.toml create mode 100644 providers/aihubmix/models/DeepSeek-V3.toml create mode 100644 providers/aihubmix/models/Qwen/QwQ-32B.toml create mode 100644 providers/aihubmix/models/bai-qwen3-vl-235b-a22b-instruct.toml create mode 100644 providers/aihubmix/models/coding-minimax-m2-free.toml create mode 100644 providers/aihubmix/models/coding-minimax-m2.1-free.toml create mode 100644 providers/aihubmix/models/coding-minimax-m2.1.toml create mode 100644 providers/aihubmix/models/coding-minimax-m2.5-free.toml create mode 100644 providers/aihubmix/models/coding-minimax-m2.5-highspeed.toml create mode 100644 providers/aihubmix/models/coding-minimax-m2.5.toml create mode 100644 providers/aihubmix/models/coding-minimax-m2.toml create mode 100644 providers/aihubmix/models/coding-minimax-m3-free.toml create mode 100644 providers/aihubmix/models/coding-minimax-m3.toml create mode 100644 providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml create mode 100644 providers/aihubmix/models/glm-5.2-fast-preview.toml create mode 100644 providers/aihubmix/models/minimax-m2.1.toml create mode 100644 providers/aihubmix/models/minimax-m2.5-highspeed.toml create mode 100644 providers/aihubmix/models/minimax-m2.5.toml create mode 100644 providers/aihubmix/models/minimax-m2.7-free.toml create mode 100644 providers/aihubmix/models/minimax-m2.toml create mode 100644 providers/aihubmix/models/minimax-m3.toml create mode 100644 providers/aihubmix/models/nvidia-nemotron-3-super-120b-a12b.toml create mode 100644 providers/aihubmix/models/qwen-plus-latest.toml create mode 100644 providers/aihubmix/models/qwen-turbo-latest.toml diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index f42a1c56c8a..15531d4ac3e 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -114,8 +114,13 @@ const LAB_BY_DEVELOPER: Record = { }; /** Routing prefixes and suffixes that select a mode, not a different model. */ -const ROUTING_PREFIXES = ["coding-", "alicloud-", "deep-", "zai-", "anthropic-", "xiaomi-", "openai-"]; -const ROUTING_SUFFIXES = ["-free", "-think", "-nothink", "-search", "-preview", "-disc", "-exp"]; +const ROUTING_PREFIXES = [ + "coding-", "alicloud-", "deep-", "zai-", "anthropic-", "xiaomi-", "openai-", "nvidia-", "bai-", +]; +const ROUTING_SUFFIXES = [ + "-free", "-think", "-nothink", "-search", "-preview", "-disc", "-exp", "-highspeed", "-fast", + "-latest", +]; /** Catalog effort levels; AIHubMix spells two of them differently. */ const EFFORT_ALIASES: Record = { no_think: "none", instant: "minimal" }; @@ -130,7 +135,9 @@ const EFFORT_VALUES = new Set([ "default", ]); -let labMetadataIDs: Set | undefined; +type LabMetadataIDs = Map; + +let labMetadataIDs: LabMetadataIDs | undefined; /** * The catalog rejects a `base_model` that resolves to nothing, so relays are @@ -138,9 +145,12 @@ let labMetadataIDs: Set | undefined; */ async function readLabMetadataIDs(modelsDir: string) { const metadataDir = path.join(path.dirname(path.dirname(path.dirname(modelsDir))), "models"); - const ids = new Set(); + const ids = new Map(); for await (const file of new Bun.Glob("**/*.toml").scan({ cwd: metadataDir, followSymlinks: true })) { - ids.add(file.split(path.sep).join("/").slice(0, -5)); + const id = file.split(path.sep).join("/").slice(0, -5); + // AIHubMix lowercases every relay ID while labs keep their own casing + // (`minimax-m2` against `minimax/MiniMax-M2`), so lookups are case-folded. + ids.set(id.toLowerCase(), id); } return ids; } @@ -190,7 +200,7 @@ export const aihubmix = { export function buildAihubmixModel( model: AihubmixModel, existing: ExistingModel | undefined, - labIDs: Set | undefined = labMetadataIDs, + labIDs: LabMetadataIDs | undefined = labMetadataIDs, ): SyncedModel | undefined { const input = modalities(model.input_modalities, existing?.modalities?.input ?? ["text"]); const output = modalities(model.output_modalities, existing?.modalities?.output ?? ["text"]); @@ -199,9 +209,21 @@ export function buildAihubmixModel( const toolCall = model.tool_call ?? existing?.tool_call ?? false; const structuredOutput = features.has("structured_outputs") || existing?.structured_output; const name = model.model_name ?? existing?.name; + const context = tokens(model.context_length); + // AIHubMix backfills an unknown `max_output` from `context_length`, so a value + // equal to the window is read as absent the same way a 0 is — 51 of 415 models + // quote the two as equal, and a model whose output ceiling really is its whole + // context window leaves no room for the prompt. + const quoted = tokens(model.max_output); + // Two ways AIHubMix signals an unknown output ceiling: backfilling it from + // `context_length` (51 of 415 models quote the two as equal, which would leave + // no room for the prompt) and quoting a value above the window (6 models, up to + // 10x). Both are read as absent, the same as the 0 the endpoint also uses. + const maxOutput = + context !== undefined && quoted !== undefined && quoted >= context ? undefined : quoted; const limit = { - context: tokens(model.context_length) ?? existing?.limit?.context, - output: tokens(model.max_output) ?? existing?.limit?.output, + context: context ?? existing?.limit?.context, + output: maxOutput ?? existing?.limit?.output, }; const shared = { attachment: input.some((value) => value !== "text"), @@ -263,12 +285,12 @@ export function buildAihubmixModel( } as SyncedFullModel; } -function resolveBaseModel(model: AihubmixModel, labIDs: Set | undefined) { +function resolveBaseModel(model: AihubmixModel, labIDs: LabMetadataIDs | undefined) { const lab = LAB_BY_DEVELOPER[model.developer_id ?? -1]; if (lab === undefined || labIDs === undefined) return undefined; for (const candidate of baseCandidates(model.model_id)) { - const id = `${lab}/${candidate}`; - if (labIDs.has(id)) return id; + const id = labIDs.get(`${lab}/${candidate}`.toLowerCase()); + if (id !== undefined) return id; } return undefined; } @@ -278,11 +300,13 @@ function baseCandidates(modelID: string) { const bare = modelID.split("/").at(-1) ?? modelID; const candidates = new Set([bare]); for (const prefix of ROUTING_PREFIXES) { - if (bare.startsWith(prefix)) candidates.add(bare.slice(prefix.length)); + if (bare.toLowerCase().startsWith(prefix)) candidates.add(bare.slice(prefix.length)); } for (const suffix of ROUTING_SUFFIXES) { for (const candidate of [...candidates]) { - if (candidate.endsWith(suffix)) candidates.add(candidate.slice(0, -suffix.length)); + if (candidate.toLowerCase().endsWith(suffix)) { + candidates.add(candidate.slice(0, -suffix.length)); + } } } return candidates; diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 0fa06bc0577..8e84487c066 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5043,7 +5043,12 @@ const aihubmixAuthored: ExistingModel = { modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, }; -const aihubmixLabIDs = new Set(["google/gemini-3.1-flash-lite", "openai/gpt-5.5"]); +const aihubmixLabIDs = new Map( + ["google/gemini-3.1-flash-lite", "openai/gpt-5.5", "minimax/MiniMax-M2"].map((id) => [ + id.toLowerCase(), + id, + ]), +); test("factors an AIHubMix relay onto the lab metadata it serves", () => { const model = buildAihubmixModel( @@ -5072,6 +5077,16 @@ test("routes AIHubMix prefixes and suffixes back to the upstream lab model", () } }); +test("resolves an AIHubMix relay against a lab that spells its ID differently", () => { + // AIHubMix lowercases every relay ID; the lab keeps `minimax/MiniMax-M2`. + const model = buildAihubmixModel( + aihubmixModel({ model_id: "coding-minimax-m2-free", developer_id: 18 }), + undefined, + aihubmixLabIDs, + ); + expect(model).toMatchObject({ base_model: "minimax/MiniMax-M2" }); +}); + test("skips an AIHubMix relay with neither base metadata nor standalone fields", () => { const model = buildAihubmixModel( aihubmixModel({ model_id: "house-brand-v1", developer_id: 999 }), diff --git a/providers/aihubmix/models/DeepSeek-V3.1-Think.toml b/providers/aihubmix/models/DeepSeek-V3.1-Think.toml new file mode 100644 index 00000000000..2f19ee69fbc --- /dev/null +++ b/providers/aihubmix/models/DeepSeek-V3.1-Think.toml @@ -0,0 +1,13 @@ +base_model = "deepseek/deepseek-v3.1" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.56 +output = 1.68 + +[limit] +context = 128_000 +output = 32_000 diff --git a/providers/aihubmix/models/DeepSeek-V3.toml b/providers/aihubmix/models/DeepSeek-V3.toml new file mode 100644 index 00000000000..6531beb0ddb --- /dev/null +++ b/providers/aihubmix/models/DeepSeek-V3.toml @@ -0,0 +1,9 @@ +base_model = "deepseek/deepseek-v3" +structured_output = true + +[cost] +input = 0.272 +output = 1.088 + +[limit] +context = 163_840 diff --git a/providers/aihubmix/models/Qwen/QwQ-32B.toml b/providers/aihubmix/models/Qwen/QwQ-32B.toml new file mode 100644 index 00000000000..7cf8bfe98c3 --- /dev/null +++ b/providers/aihubmix/models/Qwen/QwQ-32B.toml @@ -0,0 +1,7 @@ +base_model = "alibaba/qwq-32b" +reasoning = false +structured_output = true + +[cost] +input = 0.14 +output = 0.56 diff --git a/providers/aihubmix/models/bai-qwen3-vl-235b-a22b-instruct.toml b/providers/aihubmix/models/bai-qwen3-vl-235b-a22b-instruct.toml new file mode 100644 index 00000000000..776ec2d52ef --- /dev/null +++ b/providers/aihubmix/models/bai-qwen3-vl-235b-a22b-instruct.toml @@ -0,0 +1,10 @@ +base_model = "alibaba/qwen3-vl-235b-a22b-instruct" +attachment = false +tool_call = false + +[cost] +input = 0.274 +output = 1.096 + +[modalities] +input = ["text"] diff --git a/providers/aihubmix/models/coding-minimax-m2-free.toml b/providers/aihubmix/models/coding-minimax-m2-free.toml new file mode 100644 index 00000000000..b87ca9d6a6b --- /dev/null +++ b/providers/aihubmix/models/coding-minimax-m2-free.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/coding-minimax-m2.1-free.toml b/providers/aihubmix/models/coding-minimax-m2.1-free.toml new file mode 100644 index 00000000000..36a757f4ab4 --- /dev/null +++ b/providers/aihubmix/models/coding-minimax-m2.1-free.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.1" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/coding-minimax-m2.1.toml b/providers/aihubmix/models/coding-minimax-m2.1.toml new file mode 100644 index 00000000000..a853c211e95 --- /dev/null +++ b/providers/aihubmix/models/coding-minimax-m2.1.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.1" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.2 +output = 0.2 diff --git a/providers/aihubmix/models/coding-minimax-m2.5-free.toml b/providers/aihubmix/models/coding-minimax-m2.5-free.toml new file mode 100644 index 00000000000..3a00a80be66 --- /dev/null +++ b/providers/aihubmix/models/coding-minimax-m2.5-free.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.5" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/coding-minimax-m2.5-highspeed.toml b/providers/aihubmix/models/coding-minimax-m2.5-highspeed.toml new file mode 100644 index 00000000000..e35e6b64892 --- /dev/null +++ b/providers/aihubmix/models/coding-minimax-m2.5-highspeed.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.5-highspeed" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.2 +output = 0.2 diff --git a/providers/aihubmix/models/coding-minimax-m2.5.toml b/providers/aihubmix/models/coding-minimax-m2.5.toml new file mode 100644 index 00000000000..b9b6b77d2e0 --- /dev/null +++ b/providers/aihubmix/models/coding-minimax-m2.5.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.5" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.2 +output = 0.2 diff --git a/providers/aihubmix/models/coding-minimax-m2.7-free.toml b/providers/aihubmix/models/coding-minimax-m2.7-free.toml index 0c37e9fdcf6..e62a1236d97 100644 --- a/providers/aihubmix/models/coding-minimax-m2.7-free.toml +++ b/providers/aihubmix/models/coding-minimax-m2.7-free.toml @@ -1,14 +1,7 @@ +base_model = "minimax/MiniMax-M2.7" name = "Coding MiniMax M2.7 (free)" description = "MiniMax model for chat, coding, office work, and agentic tasks" -family = "minimax-free" -release_date = "2026-03-18" -last_updated = "2026-03-18" -attachment = false -reasoning = true -temperature = true -tool_call = true structured_output = true -open_weights = true [interleaved] field = "reasoning_content" @@ -25,9 +18,4 @@ input = 0 output = 0 [limit] -context = 204_800 output = 204_800 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml b/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml index bae714ec693..5defdbe12d7 100644 --- a/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml +++ b/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml @@ -1,14 +1,7 @@ +base_model = "minimax/MiniMax-M2.7-highspeed" name = "Coding MiniMax M2.7 Highspeed" description = "High-speed MiniMax model for low-latency coding and agent workflows" -family = "minimax" -release_date = "2026-03-18" -last_updated = "2026-03-18" -attachment = false -reasoning = true -temperature = true -tool_call = true structured_output = true -open_weights = true [interleaved] field = "reasoning_content" @@ -25,9 +18,4 @@ input = 0.2 output = 0.2 [limit] -context = 204_800 output = 204_800 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/coding-minimax-m2.7.toml b/providers/aihubmix/models/coding-minimax-m2.7.toml index 58717be4441..35ab1809da0 100644 --- a/providers/aihubmix/models/coding-minimax-m2.7.toml +++ b/providers/aihubmix/models/coding-minimax-m2.7.toml @@ -1,14 +1,7 @@ +base_model = "minimax/MiniMax-M2.7" name = "Coding MiniMax M2.7" description = "MiniMax model for chat, coding, office work, and agentic tasks" -family = "minimax" -release_date = "2026-03-18" -last_updated = "2026-03-18" -attachment = false -reasoning = true -temperature = true -tool_call = true structured_output = true -open_weights = true [interleaved] field = "reasoning_content" @@ -25,9 +18,4 @@ input = 0.2 output = 0.2 [limit] -context = 204_800 output = 204_800 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/coding-minimax-m2.toml b/providers/aihubmix/models/coding-minimax-m2.toml new file mode 100644 index 00000000000..b2d95706530 --- /dev/null +++ b/providers/aihubmix/models/coding-minimax-m2.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.2 +output = 0.2 diff --git a/providers/aihubmix/models/coding-minimax-m3-free.toml b/providers/aihubmix/models/coding-minimax-m3-free.toml new file mode 100644 index 00000000000..259a40c1fda --- /dev/null +++ b/providers/aihubmix/models/coding-minimax-m3-free.toml @@ -0,0 +1,16 @@ +base_model = "minimax/MiniMax-M3" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_000_000 +output = 524_288 diff --git a/providers/aihubmix/models/coding-minimax-m3.toml b/providers/aihubmix/models/coding-minimax-m3.toml new file mode 100644 index 00000000000..4ae57e13033 --- /dev/null +++ b/providers/aihubmix/models/coding-minimax-m3.toml @@ -0,0 +1,16 @@ +base_model = "minimax/MiniMax-M3" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.2 +output = 0.2 + +[limit] +context = 1_000_000 +output = 524_288 diff --git a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml index f53ed8228b2..ffbb47916c3 100644 --- a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml +++ b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml @@ -18,6 +18,3 @@ tier = { type = "context", size = 256_000 } input = 0.4 output = 1.2 cache_read = 0.08 - -[limit] -context = 1_000_000 diff --git a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml index 492102fc78e..bf5cb9d4e99 100644 --- a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml +++ b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml @@ -18,6 +18,3 @@ tier = { type = "context", size = 256_000 } input = 0.16 output = 0.8 cache_read = 0.032 - -[limit] -context = 1_000_000 diff --git a/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml b/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml new file mode 100644 index 00000000000..8a68fecf328 --- /dev/null +++ b/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml @@ -0,0 +1,13 @@ +base_model = "deepseek/deepseek-v4-flash-0731" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.28 +output = 1.4 +cache_read = 0.07 diff --git a/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml b/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml index 9be732aad47..53b6813a437 100644 --- a/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml +++ b/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml @@ -8,6 +8,6 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 0.2324 -output = 0.6972 -cache_read = 0.007746 +input = 0.155 +output = 0.62 +cache_read = 0.0031 diff --git a/providers/aihubmix/models/gemini-2.5-flash-nothink.toml b/providers/aihubmix/models/gemini-2.5-flash-nothink.toml index da2afd88e71..5f59e3eef8f 100644 --- a/providers/aihubmix/models/gemini-2.5-flash-nothink.toml +++ b/providers/aihubmix/models/gemini-2.5-flash-nothink.toml @@ -11,6 +11,3 @@ type = "budget_tokens" input = 0.3 output = 2.499 cache_read = 0.03 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-search.toml b/providers/aihubmix/models/gemini-2.5-flash-search.toml index da2afd88e71..5f59e3eef8f 100644 --- a/providers/aihubmix/models/gemini-2.5-flash-search.toml +++ b/providers/aihubmix/models/gemini-2.5-flash-search.toml @@ -11,6 +11,3 @@ type = "budget_tokens" input = 0.3 output = 2.499 cache_read = 0.03 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-flash.toml b/providers/aihubmix/models/gemini-2.5-flash.toml index 0e43b319839..8462b4a9581 100644 --- a/providers/aihubmix/models/gemini-2.5-flash.toml +++ b/providers/aihubmix/models/gemini-2.5-flash.toml @@ -13,6 +13,3 @@ type = "budget_tokens" input = 0.3 output = 2.499 cache_read = 0.03 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/glm-5.2-fast-preview.toml b/providers/aihubmix/models/glm-5.2-fast-preview.toml new file mode 100644 index 00000000000..eb6b0815ff3 --- /dev/null +++ b/providers/aihubmix/models/glm-5.2-fast-preview.toml @@ -0,0 +1,13 @@ +base_model = "zhipuai/glm-5.2" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 2.254 +output = 7.889 +cache_read = 0.5635 diff --git a/providers/aihubmix/models/gpt-5.2-codex.toml b/providers/aihubmix/models/gpt-5.2-codex.toml index 1f2eed5fca0..a8725acddb2 100644 --- a/providers/aihubmix/models/gpt-5.2-codex.toml +++ b/providers/aihubmix/models/gpt-5.2-codex.toml @@ -9,6 +9,3 @@ values = ["low", "medium", "high", "xhigh"] input = 1.75 output = 14 cache_read = 0.175 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.3-codex.toml b/providers/aihubmix/models/gpt-5.3-codex.toml index b4ea92ff965..d58667f384b 100644 --- a/providers/aihubmix/models/gpt-5.3-codex.toml +++ b/providers/aihubmix/models/gpt-5.3-codex.toml @@ -10,6 +10,3 @@ values = ["low", "medium", "high", "xhigh"] input = 1.75 output = 14 cache_read = 0.175 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.4.toml b/providers/aihubmix/models/gpt-5.4.toml index 8ae4fe344f4..52b1382fb53 100644 --- a/providers/aihubmix/models/gpt-5.4.toml +++ b/providers/aihubmix/models/gpt-5.4.toml @@ -16,6 +16,3 @@ tier = { type = "context", size = 272_000 } input = 5 output = 22.5 cache_read = 0.5 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.5-free.toml b/providers/aihubmix/models/gpt-5.5-free.toml index db3d76ed2cc..dc9fead01bb 100644 --- a/providers/aihubmix/models/gpt-5.5-free.toml +++ b/providers/aihubmix/models/gpt-5.5-free.toml @@ -7,6 +7,3 @@ values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 0 output = 0 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.5.toml b/providers/aihubmix/models/gpt-5.5.toml index 1de661d6511..965823dab6d 100644 --- a/providers/aihubmix/models/gpt-5.5.toml +++ b/providers/aihubmix/models/gpt-5.5.toml @@ -15,6 +15,3 @@ tier = { type = "context", size = 272_000 } input = 10 output = 45 cache_read = 1 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/kimi-k2.5.toml b/providers/aihubmix/models/kimi-k2.5.toml index 1923fec867c..59191e20864 100644 --- a/providers/aihubmix/models/kimi-k2.5.toml +++ b/providers/aihubmix/models/kimi-k2.5.toml @@ -13,5 +13,4 @@ output = 3 cache_read = 0.105 [limit] -context = 256_000 output = 32_768 diff --git a/providers/aihubmix/models/mimo-v2.5-pro.toml b/providers/aihubmix/models/mimo-v2.5-pro.toml index 3dfd0499613..5075fe759a9 100644 --- a/providers/aihubmix/models/mimo-v2.5-pro.toml +++ b/providers/aihubmix/models/mimo-v2.5-pro.toml @@ -8,6 +8,3 @@ type = "toggle" input = 0.48 output = 0.96 cache_read = 0.00384 - -[limit] -context = 1_000_000 diff --git a/providers/aihubmix/models/mimo-v2.5.toml b/providers/aihubmix/models/mimo-v2.5.toml index 43355271bdc..86372dd670f 100644 --- a/providers/aihubmix/models/mimo-v2.5.toml +++ b/providers/aihubmix/models/mimo-v2.5.toml @@ -8,6 +8,3 @@ type = "toggle" input = 0.155 output = 0.31 cache_read = 0.0031 - -[limit] -context = 1_000_000 diff --git a/providers/aihubmix/models/minimax-m2.1.toml b/providers/aihubmix/models/minimax-m2.1.toml new file mode 100644 index 00000000000..96e0cd924d1 --- /dev/null +++ b/providers/aihubmix/models/minimax-m2.1.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.1" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.288 +output = 1.152 diff --git a/providers/aihubmix/models/minimax-m2.5-highspeed.toml b/providers/aihubmix/models/minimax-m2.5-highspeed.toml new file mode 100644 index 00000000000..8ee59cf05a4 --- /dev/null +++ b/providers/aihubmix/models/minimax-m2.5-highspeed.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.5-highspeed" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.288 +output = 1.152 diff --git a/providers/aihubmix/models/minimax-m2.5.toml b/providers/aihubmix/models/minimax-m2.5.toml new file mode 100644 index 00000000000..e4e96f80dc6 --- /dev/null +++ b/providers/aihubmix/models/minimax-m2.5.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.5" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.288 +output = 1.152 diff --git a/providers/aihubmix/models/minimax-m2.7-free.toml b/providers/aihubmix/models/minimax-m2.7-free.toml new file mode 100644 index 00000000000..a9b38c992fc --- /dev/null +++ b/providers/aihubmix/models/minimax-m2.7-free.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.7" +tool_call = false + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/minimax-m2.7.toml b/providers/aihubmix/models/minimax-m2.7.toml index 399fe8d146a..097633f4c4d 100644 --- a/providers/aihubmix/models/minimax-m2.7.toml +++ b/providers/aihubmix/models/minimax-m2.7.toml @@ -1,14 +1,7 @@ +base_model = "minimax/MiniMax-M2.7" name = "MiniMax M2.7" description = "MiniMax model for chat, coding, office work, and agentic tasks" -family = "minimax" -release_date = "2026-03-18" -last_updated = "2026-03-18" -attachment = false -reasoning = true -temperature = true -tool_call = true structured_output = true -open_weights = true [interleaved] field = "reasoning_content" @@ -26,9 +19,4 @@ output = 1.1832 cache_read = 0.05916 [limit] -context = 204_800 output = 204_800 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/minimax-m2.toml b/providers/aihubmix/models/minimax-m2.toml new file mode 100644 index 00000000000..91837b885d0 --- /dev/null +++ b/providers/aihubmix/models/minimax-m2.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.288 +output = 1.152 diff --git a/providers/aihubmix/models/minimax-m3.toml b/providers/aihubmix/models/minimax-m3.toml new file mode 100644 index 00000000000..ee1919ca4ef --- /dev/null +++ b/providers/aihubmix/models/minimax-m3.toml @@ -0,0 +1,16 @@ +base_model = "minimax/MiniMax-M3" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.288 +output = 1.152 + +[limit] +context = 1_000_000 +output = 524_288 diff --git a/providers/aihubmix/models/nvidia-nemotron-3-super-120b-a12b.toml b/providers/aihubmix/models/nvidia-nemotron-3-super-120b-a12b.toml new file mode 100644 index 00000000000..20b31b0b7bf --- /dev/null +++ b/providers/aihubmix/models/nvidia-nemotron-3-super-120b-a12b.toml @@ -0,0 +1,11 @@ +base_model = "nvidia/nemotron-3-super-120b-a12b" +structured_output = true +reasoning_options = [] + +[cost] +input = 0.11 +output = 0.55 +cache_read = 0.0275 + +[limit] +context = 1_000_000 diff --git a/providers/aihubmix/models/qwen-plus-latest.toml b/providers/aihubmix/models/qwen-plus-latest.toml new file mode 100644 index 00000000000..1527008c135 --- /dev/null +++ b/providers/aihubmix/models/qwen-plus-latest.toml @@ -0,0 +1,23 @@ +base_model = "alibaba/qwen-plus" +reasoning = false +tool_call = false + +[cost] +input = 0.1126 +output = 1.126 +cache_read = 0.02252 +cache_write = 0.14075 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.338 +output = 3.38 +cache_read = 0.0676 +cache_write = 0.4225 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.676 +output = 9.013311 +cache_read = 0.1352 +cache_write = 0.845 diff --git a/providers/aihubmix/models/qwen-turbo-latest.toml b/providers/aihubmix/models/qwen-turbo-latest.toml new file mode 100644 index 00000000000..3b191626bcc --- /dev/null +++ b/providers/aihubmix/models/qwen-turbo-latest.toml @@ -0,0 +1,8 @@ +base_model = "alibaba/qwen-turbo" +reasoning = false +tool_call = false + +[cost] +input = 0.046 +output = 0.092 +cache_read = 0.0092 diff --git a/providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml b/providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml index 9b930792aa9..751e0fcd3f3 100644 --- a/providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml +++ b/providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml @@ -16,5 +16,4 @@ input = 1.027396 output = 5.13698 [limit] -context = 2_000_000 output = 262_000 diff --git a/providers/aihubmix/models/qwen3-coder-flash.toml b/providers/aihubmix/models/qwen3-coder-flash.toml index 9443089d686..3ed4e460af4 100644 --- a/providers/aihubmix/models/qwen3-coder-flash.toml +++ b/providers/aihubmix/models/qwen3-coder-flash.toml @@ -19,6 +19,3 @@ output = 1.369856 tier = { type = "context", size = 256_000 } input = 0.68493 output = 3.42465 - -[limit] -context = 256_000 diff --git a/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml b/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml index fa578c977e6..8c02f9c85b9 100644 --- a/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml +++ b/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml @@ -10,6 +10,3 @@ type = "toggle" [cost] input = 0 output = 0 - -[limit] -context = 1_000_000 diff --git a/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml b/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml index 17492a3e0f5..12a97b3e0e7 100644 --- a/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml +++ b/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml @@ -10,6 +10,3 @@ type = "toggle" [cost] input = 0 output = 0 - -[limit] -context = 1_000_000 From 4b72fb35dcdc91899a43e46890ee6ab578033f0b Mon Sep 17 00:00:00 2001 From: chenxue Date: Thu, 10 Sep 2026 15:41:48 +0800 Subject: [PATCH 05/20] feat(aihubmix): resolve vanity and upstream-compute routing prefixes `cc-`, `mm-`, `aihubmix-`, `aihub-` and `ahm-` are AIHubMix's own namespaces, and `cloudflare-`/`deepinfra-` name the upstream compute a relay routes to, the same way `alicloud-` already did. Stripping them resolves 16 more relays onto the lab metadata they serve. `cc-minimax-m2` and `cc-MiniMax-M2` are one route under two spellings and would claim filenames differing only in case, so the response is deduplicated on the folded ID, keeping the last record whole. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 12 ++++++++++-- .../models/aihubmix-command-r-08-2024.toml | 6 ++++++ .../models/aihubmix-command-r-plus-08-2024.toml | 6 ++++++ providers/aihubmix/models/cc-MiniMax-M2.toml | 7 +++++++ providers/aihubmix/models/cc-deepseek-v3.1.toml | 7 +++++++ providers/aihubmix/models/cc-glm-5-turbo.toml | 11 +++++++++++ providers/aihubmix/models/cc-glm-5.1.toml | 8 ++++++++ providers/aihubmix/models/cc-glm-5.toml | 12 ++++++++++++ providers/aihubmix/models/cc-minimax-m2.1.toml | 13 +++++++++++++ .../models/cc-minimax-m2.5-highspeed.toml | 13 +++++++++++++ providers/aihubmix/models/cc-minimax-m2.5.toml | 13 +++++++++++++ .../models/cc-minimax-m2.7-highspeed.toml | 13 +++++++++++++ providers/aihubmix/models/cc-minimax-m2.7.toml | 13 +++++++++++++ providers/aihubmix/models/cc-minimax-m3.toml | 16 ++++++++++++++++ .../aihubmix/models/cloudflare-glm-5.2.toml | 14 ++++++++++++++ .../models/deepinfra-gemma-4-26b-a4b-it.toml | 16 ++++++++++++++++ .../models/mm-minimax-m2.7-highspeed.toml | 13 +++++++++++++ 17 files changed, 191 insertions(+), 2 deletions(-) create mode 100644 providers/aihubmix/models/aihubmix-command-r-08-2024.toml create mode 100644 providers/aihubmix/models/aihubmix-command-r-plus-08-2024.toml create mode 100644 providers/aihubmix/models/cc-MiniMax-M2.toml create mode 100644 providers/aihubmix/models/cc-deepseek-v3.1.toml create mode 100644 providers/aihubmix/models/cc-glm-5-turbo.toml create mode 100644 providers/aihubmix/models/cc-glm-5.1.toml create mode 100644 providers/aihubmix/models/cc-glm-5.toml create mode 100644 providers/aihubmix/models/cc-minimax-m2.1.toml create mode 100644 providers/aihubmix/models/cc-minimax-m2.5-highspeed.toml create mode 100644 providers/aihubmix/models/cc-minimax-m2.5.toml create mode 100644 providers/aihubmix/models/cc-minimax-m2.7-highspeed.toml create mode 100644 providers/aihubmix/models/cc-minimax-m2.7.toml create mode 100644 providers/aihubmix/models/cc-minimax-m3.toml create mode 100644 providers/aihubmix/models/cloudflare-glm-5.2.toml create mode 100644 providers/aihubmix/models/deepinfra-gemma-4-26b-a4b-it.toml create mode 100644 providers/aihubmix/models/mm-minimax-m2.7-highspeed.toml diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index 15531d4ac3e..2773e188863 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -115,7 +115,11 @@ const LAB_BY_DEVELOPER: Record = { /** Routing prefixes and suffixes that select a mode, not a different model. */ const ROUTING_PREFIXES = [ - "coding-", "alicloud-", "deep-", "zai-", "anthropic-", "xiaomi-", "openai-", "nvidia-", "bai-", + // Upstream compute the relay routes to. + "alicloud-", "cloudflare-", "deepinfra-", "bai-", "zai-", "anthropic-", "openai-", "nvidia-", + "xiaomi-", "deep-", + // AIHubMix's own routing modes and vanity namespaces. + "coding-", "cc-", "mm-", "aihubmix-", "aihub-", "ahm-", ]; const ROUTING_SUFFIXES = [ "-free", "-think", "-nothink", "-search", "-preview", "-disc", "-exp", "-highspeed", "-fast", @@ -187,7 +191,11 @@ export const aihubmix = { return response.json(); }, parseModels(raw) { - return AihubmixResponse.parse(raw).data; + const data = AihubmixResponse.parse(raw).data; + // `cc-minimax-m2` and `cc-MiniMax-M2` are the same route under two spellings + // and would claim filenames that differ only in case. Keep the last entry + // whole rather than mixing two records. + return [...new Map(data.map((model) => [model.model_id.toLowerCase(), model])).values()]; }, translateModel(model, context) { const existing = context.existing(model.model_id); diff --git a/providers/aihubmix/models/aihubmix-command-r-08-2024.toml b/providers/aihubmix/models/aihubmix-command-r-08-2024.toml new file mode 100644 index 00000000000..3eb86afe515 --- /dev/null +++ b/providers/aihubmix/models/aihubmix-command-r-08-2024.toml @@ -0,0 +1,6 @@ +base_model = "cohere/command-r-08-2024" +tool_call = false + +[cost] +input = 0.2 +output = 0.8 diff --git a/providers/aihubmix/models/aihubmix-command-r-plus-08-2024.toml b/providers/aihubmix/models/aihubmix-command-r-plus-08-2024.toml new file mode 100644 index 00000000000..e0e3ee2ad05 --- /dev/null +++ b/providers/aihubmix/models/aihubmix-command-r-plus-08-2024.toml @@ -0,0 +1,6 @@ +base_model = "cohere/command-r-plus-08-2024" +tool_call = false + +[cost] +input = 2.8 +output = 11.2 diff --git a/providers/aihubmix/models/cc-MiniMax-M2.toml b/providers/aihubmix/models/cc-MiniMax-M2.toml new file mode 100644 index 00000000000..a4debfe801b --- /dev/null +++ b/providers/aihubmix/models/cc-MiniMax-M2.toml @@ -0,0 +1,7 @@ +base_model = "minimax/MiniMax-M2" +reasoning = false +structured_output = true + +[cost] +input = 0.1 +output = 0.1 diff --git a/providers/aihubmix/models/cc-deepseek-v3.1.toml b/providers/aihubmix/models/cc-deepseek-v3.1.toml new file mode 100644 index 00000000000..488e4952a8a --- /dev/null +++ b/providers/aihubmix/models/cc-deepseek-v3.1.toml @@ -0,0 +1,7 @@ +base_model = "deepseek/deepseek-v3.1" +reasoning = false +structured_output = true + +[cost] +input = 0.56 +output = 1.68 diff --git a/providers/aihubmix/models/cc-glm-5-turbo.toml b/providers/aihubmix/models/cc-glm-5-turbo.toml new file mode 100644 index 00000000000..50c916e5495 --- /dev/null +++ b/providers/aihubmix/models/cc-glm-5-turbo.toml @@ -0,0 +1,11 @@ +base_model = "zhipuai/glm-5-turbo" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.06 +output = 0.22 + +[limit] +context = 204_800 diff --git a/providers/aihubmix/models/cc-glm-5.1.toml b/providers/aihubmix/models/cc-glm-5.1.toml new file mode 100644 index 00000000000..ac1e901ddf4 --- /dev/null +++ b/providers/aihubmix/models/cc-glm-5.1.toml @@ -0,0 +1,8 @@ +base_model = "zhipuai/glm-5.1" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.06 +output = 0.22 diff --git a/providers/aihubmix/models/cc-glm-5.toml b/providers/aihubmix/models/cc-glm-5.toml new file mode 100644 index 00000000000..43a367a1581 --- /dev/null +++ b/providers/aihubmix/models/cc-glm-5.toml @@ -0,0 +1,12 @@ +base_model = "zhipuai/glm-5" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.06 +output = 0.22 + +[limit] +context = 200_000 diff --git a/providers/aihubmix/models/cc-minimax-m2.1.toml b/providers/aihubmix/models/cc-minimax-m2.1.toml new file mode 100644 index 00000000000..8314c6dfb6a --- /dev/null +++ b/providers/aihubmix/models/cc-minimax-m2.1.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.1" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.1 diff --git a/providers/aihubmix/models/cc-minimax-m2.5-highspeed.toml b/providers/aihubmix/models/cc-minimax-m2.5-highspeed.toml new file mode 100644 index 00000000000..84c99647716 --- /dev/null +++ b/providers/aihubmix/models/cc-minimax-m2.5-highspeed.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.5-highspeed" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.1 diff --git a/providers/aihubmix/models/cc-minimax-m2.5.toml b/providers/aihubmix/models/cc-minimax-m2.5.toml new file mode 100644 index 00000000000..139bb3236d5 --- /dev/null +++ b/providers/aihubmix/models/cc-minimax-m2.5.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.5" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.1 diff --git a/providers/aihubmix/models/cc-minimax-m2.7-highspeed.toml b/providers/aihubmix/models/cc-minimax-m2.7-highspeed.toml new file mode 100644 index 00000000000..1e66879bd49 --- /dev/null +++ b/providers/aihubmix/models/cc-minimax-m2.7-highspeed.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.7-highspeed" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.1 diff --git a/providers/aihubmix/models/cc-minimax-m2.7.toml b/providers/aihubmix/models/cc-minimax-m2.7.toml new file mode 100644 index 00000000000..79e4e4b2888 --- /dev/null +++ b/providers/aihubmix/models/cc-minimax-m2.7.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.7" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.1 diff --git a/providers/aihubmix/models/cc-minimax-m3.toml b/providers/aihubmix/models/cc-minimax-m3.toml new file mode 100644 index 00000000000..d090ab46f95 --- /dev/null +++ b/providers/aihubmix/models/cc-minimax-m3.toml @@ -0,0 +1,16 @@ +base_model = "minimax/MiniMax-M3" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.1 + +[limit] +context = 1_000_000 +output = 524_288 diff --git a/providers/aihubmix/models/cloudflare-glm-5.2.toml b/providers/aihubmix/models/cloudflare-glm-5.2.toml new file mode 100644 index 00000000000..1ab27569fed --- /dev/null +++ b/providers/aihubmix/models/cloudflare-glm-5.2.toml @@ -0,0 +1,14 @@ +base_model = "zhipuai/glm-5.2" +tool_call = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 1.4 +output = 4.4002 +cache_read = 0.2604 diff --git a/providers/aihubmix/models/deepinfra-gemma-4-26b-a4b-it.toml b/providers/aihubmix/models/deepinfra-gemma-4-26b-a4b-it.toml new file mode 100644 index 00000000000..bd9adb9f07c --- /dev/null +++ b/providers/aihubmix/models/deepinfra-gemma-4-26b-a4b-it.toml @@ -0,0 +1,16 @@ +base_model = "google/gemma-4-26b-a4b-it" +attachment = false +reasoning = false +tool_call = false + +[cost] +input = 0.088 +output = 0.385 +cache_read = 0.011 + +[limit] +context = 262_100 +output = 131_100 + +[modalities] +input = ["text"] diff --git a/providers/aihubmix/models/mm-minimax-m2.7-highspeed.toml b/providers/aihubmix/models/mm-minimax-m2.7-highspeed.toml new file mode 100644 index 00000000000..1e66879bd49 --- /dev/null +++ b/providers/aihubmix/models/mm-minimax-m2.7-highspeed.toml @@ -0,0 +1,13 @@ +base_model = "minimax/MiniMax-M2.7-highspeed" +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.1 From c97eddd37b66b871a422ac61a391884fda0e6481 Mon Sep 17 00:00:00 2001 From: chenxue Date: Thu, 10 Sep 2026 16:28:34 +0800 Subject: [PATCH 06/20] chore(aihubmix): pick up corrected output limits from the endpoint AIHubMix fixed 14 routes that had quoted `max_output` equal to `context_length`, plus two `context_length` values rounded to 131_000. Every corrected value matches what the other providers in the catalog already record for the same model. Nine files change and all nine shrink: the endpoint now agrees with the lab metadata, so the factored entries stop recording an override. Co-Authored-By: Claude Opus 5 --- providers/aihubmix/models/gpt-oss-20b.toml | 1 - providers/aihubmix/models/grok-4.3.toml | 3 --- providers/aihubmix/models/mimo-v2-flash-free.toml | 1 - providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml | 3 --- providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml | 1 - providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml | 1 - providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml | 1 - providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml | 1 - providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml | 1 - 9 files changed, 13 deletions(-) diff --git a/providers/aihubmix/models/gpt-oss-20b.toml b/providers/aihubmix/models/gpt-oss-20b.toml index 7e30e0e460f..e878de3f58e 100644 --- a/providers/aihubmix/models/gpt-oss-20b.toml +++ b/providers/aihubmix/models/gpt-oss-20b.toml @@ -7,4 +7,3 @@ output = 0.55 [limit] context = 128_000 -output = 128_000 diff --git a/providers/aihubmix/models/grok-4.3.toml b/providers/aihubmix/models/grok-4.3.toml index dc49fbfb558..d40e8180f6c 100644 --- a/providers/aihubmix/models/grok-4.3.toml +++ b/providers/aihubmix/models/grok-4.3.toml @@ -16,8 +16,5 @@ input = 2.5 output = 5 cache_read = 0.4 -[limit] -output = 1_000_000 - [modalities] input = ["text", "image"] diff --git a/providers/aihubmix/models/mimo-v2-flash-free.toml b/providers/aihubmix/models/mimo-v2-flash-free.toml index fa19dce2036..6dda24ef887 100644 --- a/providers/aihubmix/models/mimo-v2-flash-free.toml +++ b/providers/aihubmix/models/mimo-v2-flash-free.toml @@ -8,4 +8,3 @@ output = 0 [limit] context = 256_000 -output = 256_000 diff --git a/providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml b/providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml index b5282afe4b1..94dc9e653ce 100644 --- a/providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml +++ b/providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml @@ -6,8 +6,5 @@ structured_output = true input = 0.28 output = 1.12 -[limit] -output = 262_144 - [modalities] input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml b/providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml index c16a2996cd5..7b61078de8e 100644 --- a/providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml +++ b/providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml @@ -17,4 +17,3 @@ output = 8.219176 [limit] context = 262_000 -output = 262_000 diff --git a/providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml b/providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml index db435e1b4f6..7a21444834b 100644 --- a/providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml +++ b/providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml @@ -8,7 +8,6 @@ output = 0.552 [limit] context = 256_000 -output = 256_000 [modalities] input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml b/providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml index 66b7e4a3283..fa6363729f4 100644 --- a/providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml +++ b/providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml @@ -9,7 +9,6 @@ output = 1.42 [limit] context = 256_000 -output = 256_000 [modalities] input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml b/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml index 69a0206bbb9..740e3a625de 100644 --- a/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml +++ b/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml @@ -5,7 +5,6 @@ input = 0.274 output = 1.096 [limit] -context = 131_000 output = 33_000 [modalities] diff --git a/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml b/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml index 6e56da6e1ce..9152a056352 100644 --- a/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml +++ b/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml @@ -6,7 +6,6 @@ input = 0.274 output = 2.74 [limit] -context = 131_000 output = 33_000 [modalities] From 5682db7c951b0efed14d8d84a29c9b26d47843d4 Mon Sep 17 00:00:00 2001 From: chenxue Date: Thu, 10 Sep 2026 17:50:36 +0800 Subject: [PATCH 07/20] feat(aihubmix): peel dated release tags and map two more labs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit AIHubMix pins snapshot dates onto relay IDs (`gemini-2.5-pro-preview-06-05`) while labs name the model itself (`google/gemini-2.5-pro`), so the tag has to come off before the ID can match. Peel routing and date affixes to a fixed point instead of one pass per rule, since they stack — `coding-gemini-2.5-pro- preview-05-06-search` carries three, with the date wedged between two of them. The date patterns are anchored and validate real month and day ranges so `llama2-70b-4096` keeps its context size and `-13-45` stays attached to nothing. The unstripped ID is still tried first, so a lab that genuinely carries a date in its name (`cohere/command-a-03-2025`) still wins. Also map developer_id 34 (muse-spark) and 35 (laguna) to the labs that publish them. 36, 37, 43 and 25 have no lab directory in models/ at all, so mapping them would not resolve anything. 27 relays now resolve to a base model: 299 of 415 source models covered, up from 272. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 38 +++++++++++++++++-- packages/core/test/sync.test.ts | 23 +++++++++++ ....5-flash-lite-preview-09-2025-nothink.toml | 13 +++++++ ...gemini-2.5-flash-lite-preview-09-2025.toml | 13 +++++++ ...emini-2.5-flash-preview-05-20-nothink.toml | 16 ++++++++ ...gemini-2.5-flash-preview-05-20-search.toml | 16 ++++++++ .../gemini-2.5-flash-preview-09-2025.toml | 16 ++++++++ .../models/gemini-2.5-pro-exp-03-25.toml | 16 ++++++++ .../gemini-2.5-pro-preview-03-25-search.toml | 16 ++++++++ .../models/gemini-2.5-pro-preview-03-25.toml | 16 ++++++++ .../gemini-2.5-pro-preview-05-06-search.toml | 17 +++++++++ .../models/gemini-2.5-pro-preview-05-06.toml | 20 ++++++++++ .../gemini-2.5-pro-preview-06-05-search.toml | 16 ++++++++ .../models/gemini-2.5-pro-preview-06-05.toml | 16 ++++++++ .../models/gpt-4o-mini-2024-07-18.toml | 10 +++++ .../models/gpt-4o-mini-search-preview.toml | 9 +++++ .../models/gpt-4o-search-preview.toml | 9 +++++ .../aihubmix/models/laguna-s-2.1-free.toml | 11 ++++++ .../aihubmix/models/laguna-xs-2.1-free.toml | 8 ++++ providers/aihubmix/models/muse-spark-1.1.toml | 15 ++++++++ providers/aihubmix/models/muse-spark-1.2.toml | 12 ++++++ providers/aihubmix/models/muse-spark-1.3.toml | 16 ++++++++ providers/aihubmix/models/o1-2024-12-17.toml | 11 ++++++ .../aihubmix/models/qwen-plus-2025-04-28.toml | 23 +++++++++++ .../models/qwen-turbo-2024-11-01.toml | 7 ++++ .../models/qwen-turbo-2025-04-28.toml | 7 ++++ .../models/qwen3-coder-plus-2025-07-22.toml | 25 ++++++++++++ .../aihubmix/models/qwen3-max-2026-01-23.toml | 28 ++++++++++++++ .../models/qwen3.8-max-2026-09-02.toml | 21 ++++++++++ 29 files changed, 460 insertions(+), 4 deletions(-) create mode 100644 providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025-nothink.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-preview-05-20-nothink.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml create mode 100644 providers/aihubmix/models/gemini-2.5-pro-exp-03-25.toml create mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-03-25-search.toml create mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-03-25.toml create mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-05-06-search.toml create mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml create mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-06-05-search.toml create mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-06-05.toml create mode 100644 providers/aihubmix/models/gpt-4o-mini-2024-07-18.toml create mode 100644 providers/aihubmix/models/gpt-4o-mini-search-preview.toml create mode 100644 providers/aihubmix/models/gpt-4o-search-preview.toml create mode 100644 providers/aihubmix/models/laguna-s-2.1-free.toml create mode 100644 providers/aihubmix/models/laguna-xs-2.1-free.toml create mode 100644 providers/aihubmix/models/muse-spark-1.1.toml create mode 100644 providers/aihubmix/models/muse-spark-1.2.toml create mode 100644 providers/aihubmix/models/muse-spark-1.3.toml create mode 100644 providers/aihubmix/models/o1-2024-12-17.toml create mode 100644 providers/aihubmix/models/qwen-plus-2025-04-28.toml create mode 100644 providers/aihubmix/models/qwen-turbo-2024-11-01.toml create mode 100644 providers/aihubmix/models/qwen-turbo-2025-04-28.toml create mode 100644 providers/aihubmix/models/qwen3-coder-plus-2025-07-22.toml create mode 100644 providers/aihubmix/models/qwen3-max-2026-01-23.toml create mode 100644 providers/aihubmix/models/qwen3.8-max-2026-09-02.toml diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index 2773e188863..e0b665ec1c5 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -108,6 +108,8 @@ const LAB_BY_DEVELOPER: Record = { 18: "minimax", 24: "tencent", 28: "meituan", + 34: "meta", + 35: "poolside", 29: "inclusionai", 31: "xiaomi", 44: "upstage", @@ -125,6 +127,19 @@ const ROUTING_SUFFIXES = [ "-free", "-think", "-nothink", "-search", "-preview", "-disc", "-exp", "-highspeed", "-fast", "-latest", ]; +/** + * Dated release tags AIHubMix pins onto a relay ID. Labs name the model itself + * (`google/gemini-2.5-pro`), so the tag has to come off before the ID can match — + * but only when the digits really are a date, or `llama2-70b-4096` would lose its + * context size. Tried after the unstripped ID so a lab that genuinely carries a + * date in its name (`cohere/command-a-03-2025`) still wins. + */ +const DATE_SUFFIXES = [ + /-(?:19|20)\d{2}-(?:0[1-9]|1[0-2])-(?:0[1-9]|[12]\d|3[01])$/, // -2026-01-23 + /-(?:0[1-9]|1[0-2])-(?:19|20)\d{2}$/, // -09-2025 + /-(?:0[1-9]|1[0-2])-(?:0[1-9]|[12]\d|3[01])$/, // -05-20 + /-\d{2}(?:0[1-9]|1[0-2])(?:0[1-9]|[12]\d|3[01])$/, // -260215 +]; /** Catalog effort levels; AIHubMix spells two of them differently. */ const EFFORT_ALIASES: Record = { no_think: "none", instant: "minimal" }; @@ -303,17 +318,32 @@ function resolveBaseModel(model: AihubmixModel, labIDs: LabMetadataIDs | undefin return undefined; } -/** Longest match first: strip routing prefixes, then routing suffixes. */ +/** Peel routing prefixes, then routing and date suffixes, to a fixed point. */ function baseCandidates(modelID: string) { const bare = modelID.split("/").at(-1) ?? modelID; const candidates = new Set([bare]); for (const prefix of ROUTING_PREFIXES) { if (bare.toLowerCase().startsWith(prefix)) candidates.add(bare.slice(prefix.length)); } - for (const suffix of ROUTING_SUFFIXES) { + // Affixes stack — `gemini-2.5-pro-preview-05-06-search` carries three — and a + // date tag can sit between two of them, so peel to a fixed point rather than + // making one pass per rule. + const add = (candidate: string) => { + const before = candidates.size; + candidates.add(candidate); + return candidates.size > before; + }; + for (let growing = true; growing;) { + growing = false; for (const candidate of [...candidates]) { - if (candidate.toLowerCase().endsWith(suffix)) { - candidates.add(candidate.slice(0, -suffix.length)); + for (const suffix of ROUTING_SUFFIXES) { + if (candidate.toLowerCase().endsWith(suffix)) { + growing = add(candidate.slice(0, -suffix.length)) || growing; + } + } + for (const pattern of DATE_SUFFIXES) { + const stripped = candidate.replace(pattern, ""); + if (stripped !== candidate) growing = add(stripped) || growing; } } } diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 8e84487c066..b8895b3ffb6 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5077,6 +5077,29 @@ test("routes AIHubMix prefixes and suffixes back to the upstream lab model", () } }); +test("strips dated release tags from an AIHubMix relay ID", () => { + // Labs name the model, AIHubMix pins the snapshot: three affixes plus a date. + for (const id of [ + "gemini-3.1-flash-lite-preview-05-20", + "gemini-3.1-flash-lite-preview-09-2025", + "gemini-3.1-flash-lite-2026-01-23", + "gemini-3.1-flash-lite-260215", + "coding-gemini-3.1-flash-lite-preview-05-06-search", + ]) { + const model = buildAihubmixModel(aihubmixModel({ model_id: id }), undefined, aihubmixLabIDs); + expect(model).toMatchObject({ base_model: "google/gemini-3.1-flash-lite" }); + } +}); + +test("keeps digits that only look like a date on an AIHubMix relay ID", () => { + // `-4096` is a context size and `-13-45` is not a month and day; stripping + // either would attach the relay to a model it is not a snapshot of. + for (const id of ["gemini-3.1-flash-lite-4096", "gemini-3.1-flash-lite-13-45"]) { + const model = buildAihubmixModel(aihubmixModel({ model_id: id }), undefined, aihubmixLabIDs); + expect(model).toBeUndefined(); + } +}); + test("resolves an AIHubMix relay against a lab that spells its ID differently", () => { // AIHubMix lowercases every relay ID; the lab keeps `minimax/MiniMax-M2`. const model = buildAihubmixModel( diff --git a/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025-nothink.toml b/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025-nothink.toml new file mode 100644 index 00000000000..f42f82a1953 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025-nothink.toml @@ -0,0 +1,13 @@ +base_model = "google/gemini-2.5-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.4 +cache_read = 0.01 diff --git a/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml b/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml new file mode 100644 index 00000000000..f42f82a1953 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml @@ -0,0 +1,13 @@ +base_model = "google/gemini-2.5-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.4 +cache_read = 0.01 diff --git a/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-nothink.toml b/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-nothink.toml new file mode 100644 index 00000000000..da2afd88e71 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-nothink.toml @@ -0,0 +1,16 @@ +base_model = "google/gemini-2.5-flash" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.3 +output = 2.499 +cache_read = 0.03 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml b/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml new file mode 100644 index 00000000000..da2afd88e71 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml @@ -0,0 +1,16 @@ +base_model = "google/gemini-2.5-flash" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.3 +output = 2.499 +cache_read = 0.03 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml b/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml new file mode 100644 index 00000000000..da2afd88e71 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml @@ -0,0 +1,16 @@ +base_model = "google/gemini-2.5-flash" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.3 +output = 2.499 +cache_read = 0.03 + +[modalities] +input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-exp-03-25.toml b/providers/aihubmix/models/gemini-2.5-pro-exp-03-25.toml new file mode 100644 index 00000000000..0e70abe3ea1 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-pro-exp-03-25.toml @@ -0,0 +1,16 @@ +base_model = "google/gemini-2.5-pro" +reasoning = false + +[cost] +input = 1.25 +output = 5 +cache_read = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 + +[modalities] +input = ["text", "image", "audio", "video"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-03-25-search.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-03-25-search.toml new file mode 100644 index 00000000000..7e32f11ea3d --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-pro-preview-03-25-search.toml @@ -0,0 +1,16 @@ +base_model = "google/gemini-2.5-pro" +reasoning_options = [] + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 + +[modalities] +input = ["text", "image", "audio", "video"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-03-25.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-03-25.toml new file mode 100644 index 00000000000..03f51ca6491 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-pro-preview-03-25.toml @@ -0,0 +1,16 @@ +base_model = "google/gemini-2.5-pro" +reasoning_options = [] + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-05-06-search.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-05-06-search.toml new file mode 100644 index 00000000000..395de9d2fd0 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-pro-preview-05-06-search.toml @@ -0,0 +1,17 @@ +base_model = "google/gemini-2.5-pro" +tool_call = false +reasoning_options = [] + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 + +[modalities] +input = ["text", "image", "audio", "video"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml new file mode 100644 index 00000000000..cdc8ebb1103 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml @@ -0,0 +1,20 @@ +base_model = "google/gemini-2.5-pro" +tool_call = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-06-05-search.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-06-05-search.toml new file mode 100644 index 00000000000..7e32f11ea3d --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-pro-preview-06-05-search.toml @@ -0,0 +1,16 @@ +base_model = "google/gemini-2.5-pro" +reasoning_options = [] + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 + +[modalities] +input = ["text", "image", "audio", "video"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-06-05.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-06-05.toml new file mode 100644 index 00000000000..7e32f11ea3d --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-pro-preview-06-05.toml @@ -0,0 +1,16 @@ +base_model = "google/gemini-2.5-pro" +reasoning_options = [] + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 + +[modalities] +input = ["text", "image", "audio", "video"] diff --git a/providers/aihubmix/models/gpt-4o-mini-2024-07-18.toml b/providers/aihubmix/models/gpt-4o-mini-2024-07-18.toml new file mode 100644 index 00000000000..d7bafa9cf42 --- /dev/null +++ b/providers/aihubmix/models/gpt-4o-mini-2024-07-18.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-4o-mini" +tool_call = false + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.075 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4o-mini-search-preview.toml b/providers/aihubmix/models/gpt-4o-mini-search-preview.toml new file mode 100644 index 00000000000..be2e1c63a52 --- /dev/null +++ b/providers/aihubmix/models/gpt-4o-mini-search-preview.toml @@ -0,0 +1,9 @@ +base_model = "openai/gpt-4o-mini" + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.075 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4o-search-preview.toml b/providers/aihubmix/models/gpt-4o-search-preview.toml new file mode 100644 index 00000000000..eb571296634 --- /dev/null +++ b/providers/aihubmix/models/gpt-4o-search-preview.toml @@ -0,0 +1,9 @@ +base_model = "openai/gpt-4o" + +[cost] +input = 2.5 +output = 10 +cache_read = 1.25 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/laguna-s-2.1-free.toml b/providers/aihubmix/models/laguna-s-2.1-free.toml new file mode 100644 index 00000000000..98bd8eb9936 --- /dev/null +++ b/providers/aihubmix/models/laguna-s-2.1-free.toml @@ -0,0 +1,11 @@ +base_model = "poolside/laguna-s-2.1" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 + +[limit] +output = 262_144 diff --git a/providers/aihubmix/models/laguna-xs-2.1-free.toml b/providers/aihubmix/models/laguna-xs-2.1-free.toml new file mode 100644 index 00000000000..80068092f9a --- /dev/null +++ b/providers/aihubmix/models/laguna-xs-2.1-free.toml @@ -0,0 +1,8 @@ +base_model = "poolside/laguna-xs-2.1" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 diff --git a/providers/aihubmix/models/muse-spark-1.1.toml b/providers/aihubmix/models/muse-spark-1.1.toml new file mode 100644 index 00000000000..58eb68641cf --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.1.toml @@ -0,0 +1,15 @@ +base_model = "meta/muse-spark-1.1" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.375 +output = 4.675 + +[modalities] +input = ["text", "image", "video", "audio", "pdf"] diff --git a/providers/aihubmix/models/muse-spark-1.2.toml b/providers/aihubmix/models/muse-spark-1.2.toml new file mode 100644 index 00000000000..86615948cd1 --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.2.toml @@ -0,0 +1,12 @@ +base_model = "meta/muse-spark-1.2" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.375 +output = 4.675 diff --git a/providers/aihubmix/models/muse-spark-1.3.toml b/providers/aihubmix/models/muse-spark-1.3.toml new file mode 100644 index 00000000000..a985b6a7e3a --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.3.toml @@ -0,0 +1,16 @@ +base_model = "meta/muse-spark-1.3" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.375 +output = 4.675 +cache_read = 0.165 + +[limit] +output = 1_000_000 diff --git a/providers/aihubmix/models/o1-2024-12-17.toml b/providers/aihubmix/models/o1-2024-12-17.toml new file mode 100644 index 00000000000..df9fac6088e --- /dev/null +++ b/providers/aihubmix/models/o1-2024-12-17.toml @@ -0,0 +1,11 @@ +base_model = "openai/o1" +tool_call = false +reasoning_options = [] + +[cost] +input = 15 +output = 60 +cache_read = 7.5 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen-plus-2025-04-28.toml b/providers/aihubmix/models/qwen-plus-2025-04-28.toml new file mode 100644 index 00000000000..1527008c135 --- /dev/null +++ b/providers/aihubmix/models/qwen-plus-2025-04-28.toml @@ -0,0 +1,23 @@ +base_model = "alibaba/qwen-plus" +reasoning = false +tool_call = false + +[cost] +input = 0.1126 +output = 1.126 +cache_read = 0.02252 +cache_write = 0.14075 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.338 +output = 3.38 +cache_read = 0.0676 +cache_write = 0.4225 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.676 +output = 9.013311 +cache_read = 0.1352 +cache_write = 0.845 diff --git a/providers/aihubmix/models/qwen-turbo-2024-11-01.toml b/providers/aihubmix/models/qwen-turbo-2024-11-01.toml new file mode 100644 index 00000000000..8a06870fc91 --- /dev/null +++ b/providers/aihubmix/models/qwen-turbo-2024-11-01.toml @@ -0,0 +1,7 @@ +base_model = "alibaba/qwen-turbo" +reasoning = false +tool_call = false + +[cost] +input = 0.046 +output = 0.092 diff --git a/providers/aihubmix/models/qwen-turbo-2025-04-28.toml b/providers/aihubmix/models/qwen-turbo-2025-04-28.toml new file mode 100644 index 00000000000..8a06870fc91 --- /dev/null +++ b/providers/aihubmix/models/qwen-turbo-2025-04-28.toml @@ -0,0 +1,7 @@ +base_model = "alibaba/qwen-turbo" +reasoning = false +tool_call = false + +[cost] +input = 0.046 +output = 0.092 diff --git a/providers/aihubmix/models/qwen3-coder-plus-2025-07-22.toml b/providers/aihubmix/models/qwen3-coder-plus-2025-07-22.toml new file mode 100644 index 00000000000..5df1810f377 --- /dev/null +++ b/providers/aihubmix/models/qwen3-coder-plus-2025-07-22.toml @@ -0,0 +1,25 @@ +base_model = "alibaba/qwen3-coder-plus" +structured_output = true + +[cost] +input = 0.54 +output = 2.16 +cache_read = 0.108 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.821916 +output = 3.287664 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 1.369862 +output = 5.479448 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 2.739726 +output = 27.39726 + +[limit] +context = 128_000 diff --git a/providers/aihubmix/models/qwen3-max-2026-01-23.toml b/providers/aihubmix/models/qwen3-max-2026-01-23.toml new file mode 100644 index 00000000000..50676a9bac9 --- /dev/null +++ b/providers/aihubmix/models/qwen3-max-2026-01-23.toml @@ -0,0 +1,28 @@ +base_model = "alibaba/qwen3-max" +reasoning = true +structured_output = true +reasoning_options = [] + +[cost] +input = 0.4508 +output = 1.8032 +cache_read = 0.09016 +cache_write = 0.5635 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.902 +output = 3.608 +cache_read = 0.1804 +cache_write = 1.1275 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 1.3522 +output = 5.4088 +cache_read = 0.27044 +cache_write = 1.69025 + +[limit] +context = 252_000 +output = 32_000 diff --git a/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml b/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml new file mode 100644 index 00000000000..4b9e46342fe --- /dev/null +++ b/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml @@ -0,0 +1,21 @@ +base_model = "alibaba/qwen3.8-max" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.69 +output = 5.07 +cache_read = 0.169 +cache_write = 2.1125 + +[modalities] +input = ["text", "image", "video"] From 2f9320b74f89dcecfcdc507db67f2c3b7eb5ab7c Mon Sep 17 00:00:00 2001 From: chenxue Date: Thu, 10 Sep 2026 17:52:23 +0800 Subject: [PATCH 08/20] docs(sync): describe how AIHubMix relay IDs resolve to lab models The notes predated case-folded lookups, vanity prefixes, the second limit sentinel and the date-tag rules, and quoted counts from an older snapshot of the endpoint. Co-Authored-By: Claude Opus 5 --- sync.md | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/sync.md b/sync.md index 363befcf6e0..c8f6c81f9b9 100644 --- a/sync.md +++ b/sync.md @@ -253,10 +253,12 @@ xAI is implemented in `packages/core/src/sync/providers/xai.ts`. - AIHubMix is implemented in `packages/core/src/sync/providers/aihubmix.ts`. - Source endpoint: `https://aihubmix.com/api/v1/models?type=llm`. - No authentication is required; the catalog is public. -- The endpoint now serves capabilities, limits, modalities, reasoning controls and pricing, so it is authoritative for all of them. Fields it does not serve (`family`, `temperature`, `interleaved`, `knowledge`) keep whatever was authored. -- AIHubMix relays upstream models under its own IDs, so a relay is factored onto the lab metadata it serves (`base_model`) whenever that metadata exists, and then records only what it actually changes. `developer_id` maps a relay to its lab; routing prefixes (`coding-`, `alicloud-`) and suffixes (`-free`, `-think`, `-nothink`) select a mode rather than a different model and are stripped when resolving the base. -- A relay with neither resolvable lab metadata nor the `release_date`/`open_weights` a standalone entry requires is reported rather than written with invented values. AIHubMix dates 52 of its 415 models and serves no `open_weights` flag, so 200-odd relays are skipped on that basis today. -- `max_output: 0` is read as absent, not as a real ceiling: 102 of 415 models quote 0 for a limit the endpoint does not know. +- The endpoint serves capabilities, limits, modalities, reasoning controls and pricing, so it is authoritative for all of them. Fields it does not serve (`family`, `temperature`, `interleaved`, `knowledge`) keep whatever was authored. +- AIHubMix relays upstream models under its own IDs, so a relay is factored onto the lab metadata it serves (`base_model`) whenever that metadata exists, and then records only what it actually changes. `developer_id` maps a relay to its lab. +- Resolving a relay to a lab model peels three kinds of affix to a fixed point, because they stack: routing prefixes that name the upstream compute or an AIHubMix namespace (`alicloud-`, `coding-`, `ahm-`), routing suffixes that select a mode rather than a different model (`-free`, `-nothink`, `-search`), and dated release tags AIHubMix pins onto the ID (`-preview-05-06`, `-2026-01-23`, `-260215`). The date patterns are anchored and validate real month and day ranges, so `llama2-70b-4096` keeps its context size. The unstripped ID is tried first, so a lab that genuinely carries a date in its name (`cohere/command-a-03-2025`) still wins over a stripped candidate. +- Lab IDs are matched case-insensitively: AIHubMix spells `minimax-m2` where the lab spells `MiniMax-M2`, and the same model can arrive under several casings, so relays are deduped on the case-folded ID. +- A relay with neither resolvable lab metadata nor the `release_date` and `open_weights` a standalone entry requires is skipped rather than written with invented values. The endpoint dates 306 of its 415 models and serves no `open_weights` field at all, which is what the remaining gap is made of. +- Two limit sentinels are read as absent rather than as real ceilings: `max_output: 0` (107 of 415 models) and a `max_output` at or above `context_length` (37), which is the context window quoted a second time. - `reasoning_options[]` entries carry an AIHubMix-only `default` key that the strict `ReasoningOption` schema rejects, and spell two effort levels differently (`no_think`, `instant`), so translation drops the extra key and maps those onto `none` and `minimal`. - A price field is omitted when the model has no such rate, so an omitted `cache_read`/`cache_write` clears an authored one; authored pricing survives only when the endpoint quotes nothing at all for the model. - Routing aliases such as `alicloud-glm-5.1` and `deep-deepseek-v4-pro` are served but not listed by the endpoint, so local files missing from the response are retained (`deleteMissing: false`) and each opens a deduped GitHub issue. From 71596f7bdb851181fad8f6fd0376a771673c5cc1 Mon Sep 17 00:00:00 2001 From: chenxue Date: Thu, 10 Sep 2026 18:00:26 +0800 Subject: [PATCH 09/20] Revert "feat(aihubmix): peel dated release tags and map two more labs" This reverts commit 5682db7c951b0efed14d8d84a29c9b26d47843d4. --- packages/core/src/sync/providers/aihubmix.ts | 38 ++----------------- packages/core/test/sync.test.ts | 23 ----------- ....5-flash-lite-preview-09-2025-nothink.toml | 13 ------- ...gemini-2.5-flash-lite-preview-09-2025.toml | 13 ------- ...emini-2.5-flash-preview-05-20-nothink.toml | 16 -------- ...gemini-2.5-flash-preview-05-20-search.toml | 16 -------- .../gemini-2.5-flash-preview-09-2025.toml | 16 -------- .../models/gemini-2.5-pro-exp-03-25.toml | 16 -------- .../gemini-2.5-pro-preview-03-25-search.toml | 16 -------- .../models/gemini-2.5-pro-preview-03-25.toml | 16 -------- .../gemini-2.5-pro-preview-05-06-search.toml | 17 --------- .../models/gemini-2.5-pro-preview-05-06.toml | 20 ---------- .../gemini-2.5-pro-preview-06-05-search.toml | 16 -------- .../models/gemini-2.5-pro-preview-06-05.toml | 16 -------- .../models/gpt-4o-mini-2024-07-18.toml | 10 ----- .../models/gpt-4o-mini-search-preview.toml | 9 ----- .../models/gpt-4o-search-preview.toml | 9 ----- .../aihubmix/models/laguna-s-2.1-free.toml | 11 ------ .../aihubmix/models/laguna-xs-2.1-free.toml | 8 ---- providers/aihubmix/models/muse-spark-1.1.toml | 15 -------- providers/aihubmix/models/muse-spark-1.2.toml | 12 ------ providers/aihubmix/models/muse-spark-1.3.toml | 16 -------- providers/aihubmix/models/o1-2024-12-17.toml | 11 ------ .../aihubmix/models/qwen-plus-2025-04-28.toml | 23 ----------- .../models/qwen-turbo-2024-11-01.toml | 7 ---- .../models/qwen-turbo-2025-04-28.toml | 7 ---- .../models/qwen3-coder-plus-2025-07-22.toml | 25 ------------ .../aihubmix/models/qwen3-max-2026-01-23.toml | 28 -------------- .../models/qwen3.8-max-2026-09-02.toml | 21 ---------- 29 files changed, 4 insertions(+), 460 deletions(-) delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025-nothink.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-preview-05-20-nothink.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-pro-exp-03-25.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-03-25-search.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-03-25.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-05-06-search.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-06-05-search.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-06-05.toml delete mode 100644 providers/aihubmix/models/gpt-4o-mini-2024-07-18.toml delete mode 100644 providers/aihubmix/models/gpt-4o-mini-search-preview.toml delete mode 100644 providers/aihubmix/models/gpt-4o-search-preview.toml delete mode 100644 providers/aihubmix/models/laguna-s-2.1-free.toml delete mode 100644 providers/aihubmix/models/laguna-xs-2.1-free.toml delete mode 100644 providers/aihubmix/models/muse-spark-1.1.toml delete mode 100644 providers/aihubmix/models/muse-spark-1.2.toml delete mode 100644 providers/aihubmix/models/muse-spark-1.3.toml delete mode 100644 providers/aihubmix/models/o1-2024-12-17.toml delete mode 100644 providers/aihubmix/models/qwen-plus-2025-04-28.toml delete mode 100644 providers/aihubmix/models/qwen-turbo-2024-11-01.toml delete mode 100644 providers/aihubmix/models/qwen-turbo-2025-04-28.toml delete mode 100644 providers/aihubmix/models/qwen3-coder-plus-2025-07-22.toml delete mode 100644 providers/aihubmix/models/qwen3-max-2026-01-23.toml delete mode 100644 providers/aihubmix/models/qwen3.8-max-2026-09-02.toml diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index e0b665ec1c5..2773e188863 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -108,8 +108,6 @@ const LAB_BY_DEVELOPER: Record = { 18: "minimax", 24: "tencent", 28: "meituan", - 34: "meta", - 35: "poolside", 29: "inclusionai", 31: "xiaomi", 44: "upstage", @@ -127,19 +125,6 @@ const ROUTING_SUFFIXES = [ "-free", "-think", "-nothink", "-search", "-preview", "-disc", "-exp", "-highspeed", "-fast", "-latest", ]; -/** - * Dated release tags AIHubMix pins onto a relay ID. Labs name the model itself - * (`google/gemini-2.5-pro`), so the tag has to come off before the ID can match — - * but only when the digits really are a date, or `llama2-70b-4096` would lose its - * context size. Tried after the unstripped ID so a lab that genuinely carries a - * date in its name (`cohere/command-a-03-2025`) still wins. - */ -const DATE_SUFFIXES = [ - /-(?:19|20)\d{2}-(?:0[1-9]|1[0-2])-(?:0[1-9]|[12]\d|3[01])$/, // -2026-01-23 - /-(?:0[1-9]|1[0-2])-(?:19|20)\d{2}$/, // -09-2025 - /-(?:0[1-9]|1[0-2])-(?:0[1-9]|[12]\d|3[01])$/, // -05-20 - /-\d{2}(?:0[1-9]|1[0-2])(?:0[1-9]|[12]\d|3[01])$/, // -260215 -]; /** Catalog effort levels; AIHubMix spells two of them differently. */ const EFFORT_ALIASES: Record = { no_think: "none", instant: "minimal" }; @@ -318,32 +303,17 @@ function resolveBaseModel(model: AihubmixModel, labIDs: LabMetadataIDs | undefin return undefined; } -/** Peel routing prefixes, then routing and date suffixes, to a fixed point. */ +/** Longest match first: strip routing prefixes, then routing suffixes. */ function baseCandidates(modelID: string) { const bare = modelID.split("/").at(-1) ?? modelID; const candidates = new Set([bare]); for (const prefix of ROUTING_PREFIXES) { if (bare.toLowerCase().startsWith(prefix)) candidates.add(bare.slice(prefix.length)); } - // Affixes stack — `gemini-2.5-pro-preview-05-06-search` carries three — and a - // date tag can sit between two of them, so peel to a fixed point rather than - // making one pass per rule. - const add = (candidate: string) => { - const before = candidates.size; - candidates.add(candidate); - return candidates.size > before; - }; - for (let growing = true; growing;) { - growing = false; + for (const suffix of ROUTING_SUFFIXES) { for (const candidate of [...candidates]) { - for (const suffix of ROUTING_SUFFIXES) { - if (candidate.toLowerCase().endsWith(suffix)) { - growing = add(candidate.slice(0, -suffix.length)) || growing; - } - } - for (const pattern of DATE_SUFFIXES) { - const stripped = candidate.replace(pattern, ""); - if (stripped !== candidate) growing = add(stripped) || growing; + if (candidate.toLowerCase().endsWith(suffix)) { + candidates.add(candidate.slice(0, -suffix.length)); } } } diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index b8895b3ffb6..8e84487c066 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5077,29 +5077,6 @@ test("routes AIHubMix prefixes and suffixes back to the upstream lab model", () } }); -test("strips dated release tags from an AIHubMix relay ID", () => { - // Labs name the model, AIHubMix pins the snapshot: three affixes plus a date. - for (const id of [ - "gemini-3.1-flash-lite-preview-05-20", - "gemini-3.1-flash-lite-preview-09-2025", - "gemini-3.1-flash-lite-2026-01-23", - "gemini-3.1-flash-lite-260215", - "coding-gemini-3.1-flash-lite-preview-05-06-search", - ]) { - const model = buildAihubmixModel(aihubmixModel({ model_id: id }), undefined, aihubmixLabIDs); - expect(model).toMatchObject({ base_model: "google/gemini-3.1-flash-lite" }); - } -}); - -test("keeps digits that only look like a date on an AIHubMix relay ID", () => { - // `-4096` is a context size and `-13-45` is not a month and day; stripping - // either would attach the relay to a model it is not a snapshot of. - for (const id of ["gemini-3.1-flash-lite-4096", "gemini-3.1-flash-lite-13-45"]) { - const model = buildAihubmixModel(aihubmixModel({ model_id: id }), undefined, aihubmixLabIDs); - expect(model).toBeUndefined(); - } -}); - test("resolves an AIHubMix relay against a lab that spells its ID differently", () => { // AIHubMix lowercases every relay ID; the lab keeps `minimax/MiniMax-M2`. const model = buildAihubmixModel( diff --git a/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025-nothink.toml b/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025-nothink.toml deleted file mode 100644 index f42f82a1953..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025-nothink.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "google/gemini-2.5-flash-lite" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.4 -cache_read = 0.01 diff --git a/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml b/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml deleted file mode 100644 index f42f82a1953..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "google/gemini-2.5-flash-lite" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.4 -cache_read = 0.01 diff --git a/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-nothink.toml b/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-nothink.toml deleted file mode 100644 index da2afd88e71..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-nothink.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemini-2.5-flash" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.3 -output = 2.499 -cache_read = 0.03 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml b/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml deleted file mode 100644 index da2afd88e71..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemini-2.5-flash" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.3 -output = 2.499 -cache_read = 0.03 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml b/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml deleted file mode 100644 index da2afd88e71..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemini-2.5-flash" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.3 -output = 2.499 -cache_read = 0.03 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-exp-03-25.toml b/providers/aihubmix/models/gemini-2.5-pro-exp-03-25.toml deleted file mode 100644 index 0e70abe3ea1..00000000000 --- a/providers/aihubmix/models/gemini-2.5-pro-exp-03-25.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemini-2.5-pro" -reasoning = false - -[cost] -input = 1.25 -output = 5 -cache_read = 0.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2.5 -output = 15 -cache_read = 0.25 - -[modalities] -input = ["text", "image", "audio", "video"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-03-25-search.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-03-25-search.toml deleted file mode 100644 index 7e32f11ea3d..00000000000 --- a/providers/aihubmix/models/gemini-2.5-pro-preview-03-25-search.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemini-2.5-pro" -reasoning_options = [] - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2.5 -output = 15 -cache_read = 0.25 - -[modalities] -input = ["text", "image", "audio", "video"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-03-25.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-03-25.toml deleted file mode 100644 index 03f51ca6491..00000000000 --- a/providers/aihubmix/models/gemini-2.5-pro-preview-03-25.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemini-2.5-pro" -reasoning_options = [] - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2.5 -output = 15 -cache_read = 0.25 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-05-06-search.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-05-06-search.toml deleted file mode 100644 index 395de9d2fd0..00000000000 --- a/providers/aihubmix/models/gemini-2.5-pro-preview-05-06-search.toml +++ /dev/null @@ -1,17 +0,0 @@ -base_model = "google/gemini-2.5-pro" -tool_call = false -reasoning_options = [] - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2.5 -output = 15 -cache_read = 0.25 - -[modalities] -input = ["text", "image", "audio", "video"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml deleted file mode 100644 index cdc8ebb1103..00000000000 --- a/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml +++ /dev/null @@ -1,20 +0,0 @@ -base_model = "google/gemini-2.5-pro" -tool_call = false - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2.5 -output = 15 -cache_read = 0.25 diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-06-05-search.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-06-05-search.toml deleted file mode 100644 index 7e32f11ea3d..00000000000 --- a/providers/aihubmix/models/gemini-2.5-pro-preview-06-05-search.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemini-2.5-pro" -reasoning_options = [] - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2.5 -output = 15 -cache_read = 0.25 - -[modalities] -input = ["text", "image", "audio", "video"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-06-05.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-06-05.toml deleted file mode 100644 index 7e32f11ea3d..00000000000 --- a/providers/aihubmix/models/gemini-2.5-pro-preview-06-05.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemini-2.5-pro" -reasoning_options = [] - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2.5 -output = 15 -cache_read = 0.25 - -[modalities] -input = ["text", "image", "audio", "video"] diff --git a/providers/aihubmix/models/gpt-4o-mini-2024-07-18.toml b/providers/aihubmix/models/gpt-4o-mini-2024-07-18.toml deleted file mode 100644 index d7bafa9cf42..00000000000 --- a/providers/aihubmix/models/gpt-4o-mini-2024-07-18.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "openai/gpt-4o-mini" -tool_call = false - -[cost] -input = 0.15 -output = 0.6 -cache_read = 0.075 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4o-mini-search-preview.toml b/providers/aihubmix/models/gpt-4o-mini-search-preview.toml deleted file mode 100644 index be2e1c63a52..00000000000 --- a/providers/aihubmix/models/gpt-4o-mini-search-preview.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "openai/gpt-4o-mini" - -[cost] -input = 0.15 -output = 0.6 -cache_read = 0.075 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4o-search-preview.toml b/providers/aihubmix/models/gpt-4o-search-preview.toml deleted file mode 100644 index eb571296634..00000000000 --- a/providers/aihubmix/models/gpt-4o-search-preview.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "openai/gpt-4o" - -[cost] -input = 2.5 -output = 10 -cache_read = 1.25 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/laguna-s-2.1-free.toml b/providers/aihubmix/models/laguna-s-2.1-free.toml deleted file mode 100644 index 98bd8eb9936..00000000000 --- a/providers/aihubmix/models/laguna-s-2.1-free.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "poolside/laguna-s-2.1" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 - -[limit] -output = 262_144 diff --git a/providers/aihubmix/models/laguna-xs-2.1-free.toml b/providers/aihubmix/models/laguna-xs-2.1-free.toml deleted file mode 100644 index 80068092f9a..00000000000 --- a/providers/aihubmix/models/laguna-xs-2.1-free.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "poolside/laguna-xs-2.1" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/muse-spark-1.1.toml b/providers/aihubmix/models/muse-spark-1.1.toml deleted file mode 100644 index 58eb68641cf..00000000000 --- a/providers/aihubmix/models/muse-spark-1.1.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "meta/muse-spark-1.1" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high", "xhigh"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.375 -output = 4.675 - -[modalities] -input = ["text", "image", "video", "audio", "pdf"] diff --git a/providers/aihubmix/models/muse-spark-1.2.toml b/providers/aihubmix/models/muse-spark-1.2.toml deleted file mode 100644 index 86615948cd1..00000000000 --- a/providers/aihubmix/models/muse-spark-1.2.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "meta/muse-spark-1.2" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high", "xhigh"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.375 -output = 4.675 diff --git a/providers/aihubmix/models/muse-spark-1.3.toml b/providers/aihubmix/models/muse-spark-1.3.toml deleted file mode 100644 index a985b6a7e3a..00000000000 --- a/providers/aihubmix/models/muse-spark-1.3.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "meta/muse-spark-1.3" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high", "xhigh"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.375 -output = 4.675 -cache_read = 0.165 - -[limit] -output = 1_000_000 diff --git a/providers/aihubmix/models/o1-2024-12-17.toml b/providers/aihubmix/models/o1-2024-12-17.toml deleted file mode 100644 index df9fac6088e..00000000000 --- a/providers/aihubmix/models/o1-2024-12-17.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "openai/o1" -tool_call = false -reasoning_options = [] - -[cost] -input = 15 -output = 60 -cache_read = 7.5 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen-plus-2025-04-28.toml b/providers/aihubmix/models/qwen-plus-2025-04-28.toml deleted file mode 100644 index 1527008c135..00000000000 --- a/providers/aihubmix/models/qwen-plus-2025-04-28.toml +++ /dev/null @@ -1,23 +0,0 @@ -base_model = "alibaba/qwen-plus" -reasoning = false -tool_call = false - -[cost] -input = 0.1126 -output = 1.126 -cache_read = 0.02252 -cache_write = 0.14075 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.338 -output = 3.38 -cache_read = 0.0676 -cache_write = 0.4225 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 0.676 -output = 9.013311 -cache_read = 0.1352 -cache_write = 0.845 diff --git a/providers/aihubmix/models/qwen-turbo-2024-11-01.toml b/providers/aihubmix/models/qwen-turbo-2024-11-01.toml deleted file mode 100644 index 8a06870fc91..00000000000 --- a/providers/aihubmix/models/qwen-turbo-2024-11-01.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "alibaba/qwen-turbo" -reasoning = false -tool_call = false - -[cost] -input = 0.046 -output = 0.092 diff --git a/providers/aihubmix/models/qwen-turbo-2025-04-28.toml b/providers/aihubmix/models/qwen-turbo-2025-04-28.toml deleted file mode 100644 index 8a06870fc91..00000000000 --- a/providers/aihubmix/models/qwen-turbo-2025-04-28.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "alibaba/qwen-turbo" -reasoning = false -tool_call = false - -[cost] -input = 0.046 -output = 0.092 diff --git a/providers/aihubmix/models/qwen3-coder-plus-2025-07-22.toml b/providers/aihubmix/models/qwen3-coder-plus-2025-07-22.toml deleted file mode 100644 index 5df1810f377..00000000000 --- a/providers/aihubmix/models/qwen3-coder-plus-2025-07-22.toml +++ /dev/null @@ -1,25 +0,0 @@ -base_model = "alibaba/qwen3-coder-plus" -structured_output = true - -[cost] -input = 0.54 -output = 2.16 -cache_read = 0.108 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.821916 -output = 3.287664 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 1.369862 -output = 5.479448 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 2.739726 -output = 27.39726 - -[limit] -context = 128_000 diff --git a/providers/aihubmix/models/qwen3-max-2026-01-23.toml b/providers/aihubmix/models/qwen3-max-2026-01-23.toml deleted file mode 100644 index 50676a9bac9..00000000000 --- a/providers/aihubmix/models/qwen3-max-2026-01-23.toml +++ /dev/null @@ -1,28 +0,0 @@ -base_model = "alibaba/qwen3-max" -reasoning = true -structured_output = true -reasoning_options = [] - -[cost] -input = 0.4508 -output = 1.8032 -cache_read = 0.09016 -cache_write = 0.5635 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.902 -output = 3.608 -cache_read = 0.1804 -cache_write = 1.1275 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 1.3522 -output = 5.4088 -cache_read = 0.27044 -cache_write = 1.69025 - -[limit] -context = 252_000 -output = 32_000 diff --git a/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml b/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml deleted file mode 100644 index 4b9e46342fe..00000000000 --- a/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml +++ /dev/null @@ -1,21 +0,0 @@ -base_model = "alibaba/qwen3.8-max" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.69 -output = 5.07 -cache_read = 0.169 -cache_write = 2.1125 - -[modalities] -input = ["text", "image", "video"] From bb68f937d571140868e9e15df1eee73f14fc9774 Mon Sep 17 00:00:00 2001 From: chenxue Date: Thu, 10 Sep 2026 18:01:21 +0800 Subject: [PATCH 10/20] docs(sync): say why AIHubMix date tags stay on the relay ID Follows the revert: a dated snapshot is its own model, and jiekou, nano-gpt, kilo and openrouter all write those IDs standalone rather than factoring them onto the undated lab entry. Co-Authored-By: Claude Opus 5 --- sync.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sync.md b/sync.md index c8f6c81f9b9..5753d5deccf 100644 --- a/sync.md +++ b/sync.md @@ -255,7 +255,7 @@ xAI is implemented in `packages/core/src/sync/providers/xai.ts`. - No authentication is required; the catalog is public. - The endpoint serves capabilities, limits, modalities, reasoning controls and pricing, so it is authoritative for all of them. Fields it does not serve (`family`, `temperature`, `interleaved`, `knowledge`) keep whatever was authored. - AIHubMix relays upstream models under its own IDs, so a relay is factored onto the lab metadata it serves (`base_model`) whenever that metadata exists, and then records only what it actually changes. `developer_id` maps a relay to its lab. -- Resolving a relay to a lab model peels three kinds of affix to a fixed point, because they stack: routing prefixes that name the upstream compute or an AIHubMix namespace (`alicloud-`, `coding-`, `ahm-`), routing suffixes that select a mode rather than a different model (`-free`, `-nothink`, `-search`), and dated release tags AIHubMix pins onto the ID (`-preview-05-06`, `-2026-01-23`, `-260215`). The date patterns are anchored and validate real month and day ranges, so `llama2-70b-4096` keeps its context size. The unstripped ID is tried first, so a lab that genuinely carries a date in its name (`cohere/command-a-03-2025`) still wins over a stripped candidate. +- Resolving a relay to a lab model strips two kinds of affix: routing prefixes that name the upstream compute or an AIHubMix namespace (`alicloud-`, `coding-`, `ahm-`), and routing suffixes that select a mode rather than a different model (`-free`, `-nothink`, `-search`). Dated release tags are deliberately **not** stripped: `gemini-2.5-pro-preview-06-05` is a pinned snapshot, not the model `google/gemini-2.5-pro`, and every other provider in the catalog writes such IDs as standalone entries. - Lab IDs are matched case-insensitively: AIHubMix spells `minimax-m2` where the lab spells `MiniMax-M2`, and the same model can arrive under several casings, so relays are deduped on the case-folded ID. - A relay with neither resolvable lab metadata nor the `release_date` and `open_weights` a standalone entry requires is skipped rather than written with invented values. The endpoint dates 306 of its 415 models and serves no `open_weights` field at all, which is what the remaining gap is made of. - Two limit sentinels are read as absent rather than as real ceilings: `max_output: 0` (107 of 415 models) and a `max_output` at or above `context_length` (37), which is the context window quoted a second time. From e0ca43893fb290e8ab86c341826b0a380d29f9f4 Mon Sep 17 00:00:00 2001 From: chenxue Date: Fri, 11 Sep 2026 20:19:02 +0800 Subject: [PATCH 11/20] feat(aihubmix): resolve relays from the catalog's own vendor and variant_of MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit AIHubMix now serves `vendor`, `variant_of` and `open_weights`, so nothing about a relay has to be inferred from its ID or mirrored in this repo any more. - `vendor` replaces the hand-maintained `developer_id` table. `VENDOR_LABS` is all that is left of it: the four labs the two registries spell differently. - `variant_of` replaces the routing prefix/suffix lists. A relay is looked up under its own ID first and then under each declared hop, nearest first, so `qwen3.8-max-preview` factors onto the preview rather than its chain root. Following the declared chain also reaches relays no string rule could — `ox-alpha` onto `zhipuai/glm-5.3-flash`, `grok-code-fast-1` onto `xai/grok-build-0.1`, `cohere-command-a` onto `cohere/command-a-03-2025`. - `open_weights` is served for 289 of 408 models, which unblocks standalone creates. A standalone entry also needs limits, so the skip guard now checks them; without it the endpoint's 0-output models fail catalog validation. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 151 +++++++++--------- packages/core/test/sync.test.ts | 89 +++++++++-- .../models/DeepSeek-V3.1-Terminus.toml | 24 +++ .../aihubmix/models/agnes-2.5-pro-alpha.toml | 24 +++ .../aihubmix/models/agnes-3.0-flash.toml | 26 +++ .../aihubmix/models/cohere-command-a.toml | 11 ++ .../aihubmix/models/deepseek-v4.1-flash.toml | 29 ++++ .../aihubmix/models/doubao-seed-1-8.toml | 41 +++++ .../models/doubao-seed-2-0-lite-260215.toml | 41 +++++ providers/aihubmix/models/ernie-5.1.toml | 30 ++++ ...gemini-2.5-flash-lite-preview-09-2025.toml | 29 ++++ ...gemini-2.5-flash-preview-05-20-search.toml | 29 ++++ .../gemini-2.5-flash-preview-09-2025.toml | 29 ++++ .../models/gemini-2.5-pro-preview-05-06.toml | 34 ++++ providers/aihubmix/models/gpt-5.2-high.toml | 10 ++ providers/aihubmix/models/gpt-5.2-low.toml | 10 ++ .../aihubmix/models/gpt-chat-latest.toml | 29 ++++ .../models/grok-4-1-fast-non-reasoning.toml | 13 ++ .../models/grok-4-1-fast-reasoning.toml | 13 ++ .../models/grok-4-fast-non-reasoning.toml | 19 +++ .../models/grok-4-fast-reasoning.toml | 19 +++ providers/aihubmix/models/grok-4.toml | 26 +++ .../aihubmix/models/grok-code-fast-1.toml | 19 +++ .../aihubmix/models/kimi-for-coding-free.toml | 22 +++ providers/aihubmix/models/kimi-k2-0711.toml | 21 +++ .../models/kimi-k2-turbo-preview.toml | 22 +++ providers/aihubmix/models/kimi-k2.5.toml | 3 + .../aihubmix/models/laguna-s-2.1-free.toml | 23 +++ .../aihubmix/models/mercury-2.5-preview.toml | 25 +++ providers/aihubmix/models/mimo-v2-flash.toml | 2 +- .../aihubmix/models/mistral-large-3.toml | 21 +++ providers/aihubmix/models/muse-spark-1.1.toml | 15 ++ providers/aihubmix/models/muse-spark-1.2.toml | 12 ++ providers/aihubmix/models/muse-spark-1.3.toml | 16 ++ .../aihubmix/models/north-mini-code-free.toml | 27 ++++ providers/aihubmix/models/ox-alpha.toml | 18 +++ .../aihubmix/models/qwen3-max-2026-01-23.toml | 43 +++++ providers/aihubmix/models/qwen3-max.toml | 11 +- .../models/qwen3-vl-235b-a22b-instruct.toml | 10 +- .../models/qwen3-vl-235b-a22b-thinking.toml | 10 +- .../models/qwen3-vl-30b-a3b-instruct.toml | 27 ++++ .../models/qwen3-vl-30b-a3b-thinking.toml | 27 ++++ providers/aihubmix/models/qwen3-vl-flash.toml | 40 +++++ providers/aihubmix/models/qwen3-vl-plus.toml | 11 +- .../models/qwen3.6-plus-preview-free.toml | 11 +- .../models/qwen3.8-max-2026-09-02.toml | 33 ++++ sync.md | 10 +- 47 files changed, 1098 insertions(+), 107 deletions(-) create mode 100644 providers/aihubmix/models/DeepSeek-V3.1-Terminus.toml create mode 100644 providers/aihubmix/models/agnes-2.5-pro-alpha.toml create mode 100644 providers/aihubmix/models/agnes-3.0-flash.toml create mode 100644 providers/aihubmix/models/cohere-command-a.toml create mode 100644 providers/aihubmix/models/deepseek-v4.1-flash.toml create mode 100644 providers/aihubmix/models/doubao-seed-1-8.toml create mode 100644 providers/aihubmix/models/doubao-seed-2-0-lite-260215.toml create mode 100644 providers/aihubmix/models/ernie-5.1.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml create mode 100644 providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml create mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml create mode 100644 providers/aihubmix/models/gpt-5.2-high.toml create mode 100644 providers/aihubmix/models/gpt-5.2-low.toml create mode 100644 providers/aihubmix/models/gpt-chat-latest.toml create mode 100644 providers/aihubmix/models/grok-4-1-fast-non-reasoning.toml create mode 100644 providers/aihubmix/models/grok-4-1-fast-reasoning.toml create mode 100644 providers/aihubmix/models/grok-4-fast-non-reasoning.toml create mode 100644 providers/aihubmix/models/grok-4-fast-reasoning.toml create mode 100644 providers/aihubmix/models/grok-4.toml create mode 100644 providers/aihubmix/models/grok-code-fast-1.toml create mode 100644 providers/aihubmix/models/kimi-for-coding-free.toml create mode 100644 providers/aihubmix/models/kimi-k2-0711.toml create mode 100644 providers/aihubmix/models/kimi-k2-turbo-preview.toml create mode 100644 providers/aihubmix/models/laguna-s-2.1-free.toml create mode 100644 providers/aihubmix/models/mercury-2.5-preview.toml create mode 100644 providers/aihubmix/models/mistral-large-3.toml create mode 100644 providers/aihubmix/models/muse-spark-1.1.toml create mode 100644 providers/aihubmix/models/muse-spark-1.2.toml create mode 100644 providers/aihubmix/models/muse-spark-1.3.toml create mode 100644 providers/aihubmix/models/north-mini-code-free.toml create mode 100644 providers/aihubmix/models/ox-alpha.toml create mode 100644 providers/aihubmix/models/qwen3-max-2026-01-23.toml create mode 100644 providers/aihubmix/models/qwen3-vl-30b-a3b-instruct.toml create mode 100644 providers/aihubmix/models/qwen3-vl-30b-a3b-thinking.toml create mode 100644 providers/aihubmix/models/qwen3-vl-flash.toml create mode 100644 providers/aihubmix/models/qwen3.8-max-2026-09-02.toml diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index 2773e188863..0d49d63dfd5 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -49,7 +49,11 @@ export const AihubmixModel = z .object({ model_id: z.string().min(1), model_name: z.string().nullish(), - developer_id: z.number().nullish(), + // The upstream lab that built the model, and the AIHubMix ID this entry is a + // routing variant of. Both are served by the catalog itself, so neither the + // lab nor the variant relationship is inferred from the relay ID here. + vendor: z.string().nullish(), + variant_of: z.string().nullish(), desc: z.string().nullish(), pricing: Pricing.nullish(), features: z.string().nullish(), @@ -62,8 +66,8 @@ export const AihubmixModel = z tool_call: z.boolean().nullish(), release_date: z.string().nullish(), last_updated: z.string().nullish(), - // Not served yet; read opportunistically so creates unblock without a code - // change once AIHubMix adds them. + // `knowledge` is not served yet; read opportunistically so it lands without a + // code change once AIHubMix adds it. knowledge: z.string().nullish(), open_weights: z.boolean().nullish(), retire_stage: z.string().nullish(), @@ -84,48 +88,23 @@ export type AihubmixModel = z.infer; * the lab metadata it serves whenever that metadata exists — the relay then only * records what it actually changes (price, reasoning controls, limits). * - * `developer_id` is AIHubMix's own lab identifier and is the only reliable way - * back to a catalog namespace: relay IDs carry routing prefixes (`coding-`, - * `alicloud-`) and suffixes (`-free`, `-think`, `-nothink`) that are AIHubMix - * routing modes rather than distinct upstream models. + * The catalog answers both halves of that lookup itself: `vendor` names the lab + * that built the model, and `variant_of` names the AIHubMix ID this entry is a + * routing variant of. Relay IDs carry prefixes (`coding-`, `alicloud-`) and + * suffixes (`-free`, `-think`) that are AIHubMix routing modes rather than + * distinct upstream models, and `variant_of` states that relationship instead of + * it being guessed from the string — which also resolves the relays no amount of + * string surgery reaches, such as `ox-alpha` onto `zhipuai/glm-5.3-flash`. */ -const LAB_BY_DEVELOPER: Record = { - 2: "anthropic", - 3: "microsoft", - 4: "bytedance-seed", - 5: "zhipuai", - 6: "cohere", - 7: "deepseek", - 8: "google", - 9: "xai", - 10: "mistral", - 11: "meta", - 12: "openai", - 13: "alibaba", - 15: "moonshotai", - 16: "stepfun", - 17: "nvidia", - 18: "minimax", - 24: "tencent", - 28: "meituan", - 29: "inclusionai", - 31: "xiaomi", - 44: "upstage", +const VENDOR_LABS: Record = { + // The two registries spell four labs differently. This maps namespaces, not + // models: no entry here decides what any model is or which lab built it. + zhipu: "zhipuai", + moonshot: "moonshotai", + bytedance: "bytedance-seed", + "meituan-longcat": "meituan", }; -/** Routing prefixes and suffixes that select a mode, not a different model. */ -const ROUTING_PREFIXES = [ - // Upstream compute the relay routes to. - "alicloud-", "cloudflare-", "deepinfra-", "bai-", "zai-", "anthropic-", "openai-", "nvidia-", - "xiaomi-", "deep-", - // AIHubMix's own routing modes and vanity namespaces. - "coding-", "cc-", "mm-", "aihubmix-", "aihub-", "ahm-", -]; -const ROUTING_SUFFIXES = [ - "-free", "-think", "-nothink", "-search", "-preview", "-disc", "-exp", "-highspeed", "-fast", - "-latest", -]; - /** Catalog effort levels; AIHubMix spells two of them differently. */ const EFFORT_ALIASES: Record = { no_think: "none", instant: "minimal" }; const EFFORT_VALUES = new Set([ @@ -140,8 +119,11 @@ const EFFORT_VALUES = new Set([ ]); type LabMetadataIDs = Map; +/** Every listed relay by lowercased ID, so `variant_of` can be followed. */ +type RelayCatalog = Map; let labMetadataIDs: LabMetadataIDs | undefined; +let relayCatalog: RelayCatalog | undefined; /** * The catalog rejects a `base_model` that resolves to nothing, so relays are @@ -179,7 +161,7 @@ export const aihubmix = { skippedNotice(ids) { return ids.map( (id) => - `AIHubMix lists ${id} but the response carries neither a resolvable base model nor the release_date/open_weights a standalone entry needs.`, + `AIHubMix lists ${id} but the response carries neither a vendor/variant_of that resolves to lab metadata nor the release_date/open_weights/limits a standalone entry needs.`, ); }, async fetchModels() { @@ -195,11 +177,12 @@ export const aihubmix = { // `cc-minimax-m2` and `cc-MiniMax-M2` are the same route under two spellings // and would claim filenames that differ only in case. Keep the last entry // whole rather than mixing two records. - return [...new Map(data.map((model) => [model.model_id.toLowerCase(), model])).values()]; + relayCatalog = new Map(data.map((model) => [model.model_id.toLowerCase(), model])); + return [...relayCatalog.values()]; }, translateModel(model, context) { const existing = context.existing(model.model_id); - const built = buildAihubmixModel(model, existing, labMetadataIDs); + const built = buildAihubmixModel(model, existing, labMetadataIDs, relayCatalog); if (built === undefined) return undefined; return { id: model.model_id, model: built }; }, @@ -209,6 +192,7 @@ export function buildAihubmixModel( model: AihubmixModel, existing: ExistingModel | undefined, labIDs: LabMetadataIDs | undefined = labMetadataIDs, + catalog: RelayCatalog | undefined = relayCatalog, ): SyncedModel | undefined { const input = modalities(model.input_modalities, existing?.modalities?.input ?? ["text"]); const output = modalities(model.output_modalities, existing?.modalities?.output ?? ["text"]); @@ -218,15 +202,11 @@ export function buildAihubmixModel( const structuredOutput = features.has("structured_outputs") || existing?.structured_output; const name = model.model_name ?? existing?.name; const context = tokens(model.context_length); - // AIHubMix backfills an unknown `max_output` from `context_length`, so a value - // equal to the window is read as absent the same way a 0 is — 51 of 415 models - // quote the two as equal, and a model whose output ceiling really is its whole - // context window leaves no room for the prompt. - const quoted = tokens(model.max_output); // Two ways AIHubMix signals an unknown output ceiling: backfilling it from - // `context_length` (51 of 415 models quote the two as equal, which would leave - // no room for the prompt) and quoting a value above the window (6 models, up to - // 10x). Both are read as absent, the same as the 0 the endpoint also uses. + // `context_length` (36 of 408 models quote the two as equal, which would leave + // no room for the prompt) and quoting a value above the window. Both are read + // as absent, the same as the 0 the endpoint also uses. + const quoted = tokens(model.max_output); const maxOutput = context !== undefined && quoted !== undefined && quoted >= context ? undefined : quoted; const limit = { @@ -248,7 +228,7 @@ export function buildAihubmixModel( cost: buildCost(model.pricing, existing?.cost), }; - const base = existing?.base_model ?? resolveBaseModel(model, labIDs); + const base = existing?.base_model ?? resolveBaseModel(model, labIDs, catalog); if (base !== undefined) { return factorBaseModel( base, @@ -258,13 +238,20 @@ export function buildAihubmixModel( ); } - // A standalone entry must carry every required catalog field itself. AIHubMix - // dates only 52 of its 415 models and serves no open_weights flag, so a relay + // A standalone entry must carry every required catalog field itself, and the + // endpoint still leaves gaps: 304 of 408 models are dated, 289 state + // `open_weights`, and the rest quote 0 for a limit they do not know. A relay // with neither metadata to inherit nor those fields is reported rather than // written with invented values. const releaseDate = model.release_date ?? existing?.release_date; const openWeights = model.open_weights ?? existing?.open_weights; - if (name === undefined || releaseDate === undefined || openWeights === undefined) { + if ( + name === undefined || + releaseDate === undefined || + openWeights === undefined || + limit.context === undefined || + limit.output === undefined + ) { return existing === undefined ? undefined : (existing as SyncedModel); } @@ -293,31 +280,43 @@ export function buildAihubmixModel( } as SyncedFullModel; } -function resolveBaseModel(model: AihubmixModel, labIDs: LabMetadataIDs | undefined) { - const lab = LAB_BY_DEVELOPER[model.developer_id ?? -1]; - if (lab === undefined || labIDs === undefined) return undefined; - for (const candidate of baseCandidates(model.model_id)) { +function resolveBaseModel( + model: AihubmixModel, + labIDs: LabMetadataIDs | undefined, + catalog: RelayCatalog | undefined, +) { + const vendor = model.vendor; + if (vendor == null || labIDs === undefined) return undefined; + const lab = VENDOR_LABS[vendor] ?? vendor; + for (const candidate of relayChain(model, catalog)) { const id = labIDs.get(`${lab}/${candidate}`.toLowerCase()); if (id !== undefined) return id; } return undefined; } -/** Longest match first: strip routing prefixes, then routing suffixes. */ -function baseCandidates(modelID: string) { - const bare = modelID.split("/").at(-1) ?? modelID; - const candidates = new Set([bare]); - for (const prefix of ROUTING_PREFIXES) { - if (bare.toLowerCase().startsWith(prefix)) candidates.add(bare.slice(prefix.length)); - } - for (const suffix of ROUTING_SUFFIXES) { - for (const candidate of [...candidates]) { - if (candidate.toLowerCase().endsWith(suffix)) { - candidates.add(candidate.slice(0, -suffix.length)); - } - } +/** + * The relay's own ID first, then one `variant_of` hop at a time toward the + * canonical entry. Nearest first matters: `qwen3.8-max-preview` is a variant of + * `qwen3.8-max` and both are published lab models, so the relay must factor onto + * the preview it actually serves rather than onto the root of its chain. + */ +function relayChain(model: AihubmixModel, catalog: RelayCatalog | undefined) { + const chain = [bareID(model.model_id)]; + const seen = new Set(chain); + let current: AihubmixModel | undefined = model; + while (current?.variant_of != null) { + const parent = bareID(current.variant_of); + if (seen.has(parent)) break; + seen.add(parent); + chain.push(parent); + current = catalog?.get(current.variant_of.toLowerCase()); } - return candidates; + return chain; +} + +function bareID(modelID: string) { + return modelID.split("/").at(-1) ?? modelID; } function reasoningOptions(model: AihubmixModel): SyncedFullModel["reasoning_options"] { @@ -386,7 +385,7 @@ function costTiers(pricing: NonNullable) { /** * AIHubMix sends 0 for a limit it does not know rather than omitting the field — - * 102 of 415 models quote `max_output: 0` — so 0 is read as absent. A model that + * 104 of 408 models quote `max_output: 0` — so 0 is read as absent. A model that * truly emitted no tokens would not be servable. */ function tokens(value: number | null | undefined) { diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 8e84487c066..2faf85e42d2 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5026,7 +5026,7 @@ function aihubmixModel(overrides: Partial = {}): AihubmixModel { return { model_id: "gemini-3.1-flash-lite", model_name: "Gemini 3.1 Flash Lite", - developer_id: 8, + vendor: "google", pricing: { input: 0.25, output: 1.5, cache_read: 0.025 }, ...overrides, }; @@ -5044,22 +5044,38 @@ const aihubmixAuthored: ExistingModel = { }; const aihubmixLabIDs = new Map( - ["google/gemini-3.1-flash-lite", "openai/gpt-5.5", "minimax/MiniMax-M2"].map((id) => [ - id.toLowerCase(), - id, - ]), + [ + "google/gemini-3.1-flash-lite", + "google/gemini-3.1-flash-lite-preview", + "openai/gpt-5.5", + "minimax/MiniMax-M2", + ].map((id) => [id.toLowerCase(), id]), +); + +/** The listing every `variant_of` hop is resolved against. */ +const aihubmixCatalog = new Map( + [ + aihubmixModel(), + aihubmixModel({ + model_id: "gemini-3.1-flash-lite-preview", + variant_of: "gemini-3.1-flash-lite", + }), + aihubmixModel({ model_id: "minimax-m2", vendor: "minimax" }), + ].map((model) => [model.model_id, model]), ); test("factors an AIHubMix relay onto the lab metadata it serves", () => { const model = buildAihubmixModel( aihubmixModel({ model_id: "gemini-3.1-flash-lite-nothink", + variant_of: "gemini-3.1-flash-lite", context_length: 1_048_576, max_output: 65_536, input_modalities: "text,image", }), undefined, aihubmixLabIDs, + aihubmixCatalog, ); expect(model).toMatchObject({ base_model: "google/gemini-3.1-flash-lite", @@ -5070,28 +5086,83 @@ test("factors an AIHubMix relay onto the lab metadata it serves", () => { expect(model).not.toHaveProperty("open_weights"); }); -test("routes AIHubMix prefixes and suffixes back to the upstream lab model", () => { +test("follows the AIHubMix variant chain back to the upstream lab model", () => { + // `coding-` and `-free` are routing modes, and the endpoint says so itself + // rather than the prefix and suffix being stripped from the ID here. for (const id of ["coding-gemini-3.1-flash-lite", "gemini-3.1-flash-lite-free"]) { - const model = buildAihubmixModel(aihubmixModel({ model_id: id }), undefined, aihubmixLabIDs); + const model = buildAihubmixModel( + aihubmixModel({ model_id: id, variant_of: "gemini-3.1-flash-lite" }), + undefined, + aihubmixLabIDs, + aihubmixCatalog, + ); expect(model).toMatchObject({ base_model: "google/gemini-3.1-flash-lite" }); } }); +test("factors an AIHubMix relay onto the nearest published model in its chain", () => { + // `-preview` is a variant of the base model and is itself published, so the + // relay records the preview it actually serves rather than the chain's root. + const model = buildAihubmixModel( + aihubmixModel({ + model_id: "coding-gemini-3.1-flash-lite-preview", + variant_of: "gemini-3.1-flash-lite-preview", + }), + undefined, + aihubmixLabIDs, + aihubmixCatalog, + ); + expect(model).toMatchObject({ base_model: "google/gemini-3.1-flash-lite-preview" }); +}); + test("resolves an AIHubMix relay against a lab that spells its ID differently", () => { // AIHubMix lowercases every relay ID; the lab keeps `minimax/MiniMax-M2`. const model = buildAihubmixModel( - aihubmixModel({ model_id: "coding-minimax-m2-free", developer_id: 18 }), + aihubmixModel({ model_id: "coding-minimax-m2-free", vendor: "minimax", variant_of: "minimax-m2" }), undefined, aihubmixLabIDs, + aihubmixCatalog, ); expect(model).toMatchObject({ base_model: "minimax/MiniMax-M2" }); }); test("skips an AIHubMix relay with neither base metadata nor standalone fields", () => { const model = buildAihubmixModel( - aihubmixModel({ model_id: "house-brand-v1", developer_id: 999 }), + aihubmixModel({ model_id: "house-brand-v1", vendor: null }), + undefined, + aihubmixLabIDs, + aihubmixCatalog, + ); + expect(model).toBeUndefined(); +}); + +test("resolves an AIHubMix lab whose namespace the catalog spells differently", () => { + // AIHubMix says `zhipu` where the catalog namespace is `zhipuai`. + const labIDs = new Map([["zhipuai/glm-5.3", "zhipuai/glm-5.3"]]); + const model = buildAihubmixModel( + aihubmixModel({ model_id: "coding-glm-5.3", vendor: "zhipu", variant_of: "glm-5.3" }), + undefined, + labIDs, + new Map(), + ); + expect(model).toMatchObject({ base_model: "zhipuai/glm-5.3" }); +}); + +test("skips an AIHubMix standalone entry the endpoint quotes no limits for", () => { + // A full catalog entry must carry its own limits; the endpoint sends 0 for a + // ceiling it does not know, which is read as absent rather than written. + const model = buildAihubmixModel( + aihubmixModel({ + model_id: "house-brand-v1", + vendor: null, + release_date: "2026-01-01", + open_weights: false, + context_length: 0, + max_output: 0, + }), undefined, aihubmixLabIDs, + aihubmixCatalog, ); expect(model).toBeUndefined(); }); diff --git a/providers/aihubmix/models/DeepSeek-V3.1-Terminus.toml b/providers/aihubmix/models/DeepSeek-V3.1-Terminus.toml new file mode 100644 index 00000000000..c66bd7430dc --- /dev/null +++ b/providers/aihubmix/models/DeepSeek-V3.1-Terminus.toml @@ -0,0 +1,24 @@ +name = "DeepSeek V3.1 Terminus" +description = "DeepSeek-V3.1 non-thinking mode has now been updated to the DeepSeek-V3.1-Terminus version." +release_date = "2025-09-22" +last_updated = "2025-09-22" +attachment = false +reasoning = true +tool_call = true +structured_output = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.56 +output = 1.68 + +[limit] +context = 160_000 +output = 32_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/agnes-2.5-pro-alpha.toml b/providers/aihubmix/models/agnes-2.5-pro-alpha.toml new file mode 100644 index 00000000000..ce387939e1c --- /dev/null +++ b/providers/aihubmix/models/agnes-2.5-pro-alpha.toml @@ -0,0 +1,24 @@ +name = "Agnes 2.5 Pro Alpha" +description = "Agnes 2.5 Pro Alpha is Agnes AI’s paid inference model, suitable for advanced coding, scientific reasoning, long-context analysis, agent workflows, and multimodal understanding. The model is accessed via an OpenAI-compatible Chat Completions API." +release_date = "2026-07-24" +last_updated = "2026-07-24" +attachment = true +reasoning = true +tool_call = false +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.45 +output = 0.9 +cache_read = 0.00378 + +[limit] +context = 1_000_000 +output = 65_536 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/agnes-3.0-flash.toml b/providers/aihubmix/models/agnes-3.0-flash.toml new file mode 100644 index 00000000000..d1e8248ef27 --- /dev/null +++ b/providers/aihubmix/models/agnes-3.0-flash.toml @@ -0,0 +1,26 @@ +name = "Agnes 3.0 Flash" +description = "Agnes 3.0 Flash is designed for real-world agent tasks and development workflows, covering the full execution chain from task understanding and planning to tool invocation and final delivery. The model focuses on improving stability, instruction-following, factual grounding, and output completeness in complex tasks, helping developers build more reliable agent applications." +release_date = "2026-09-09" +last_updated = "2026-09-09" +attachment = true +reasoning = true +tool_call = false +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.03 +output = 0.15 + +[limit] +context = 512_000 +output = 65_500 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/cohere-command-a.toml b/providers/aihubmix/models/cohere-command-a.toml new file mode 100644 index 00000000000..b00d795de92 --- /dev/null +++ b/providers/aihubmix/models/cohere-command-a.toml @@ -0,0 +1,11 @@ +base_model = "cohere/command-a-03-2025" +reasoning = true +tool_call = false + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 2.5 +output = 10 diff --git a/providers/aihubmix/models/deepseek-v4.1-flash.toml b/providers/aihubmix/models/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..27058ae312d --- /dev/null +++ b/providers/aihubmix/models/deepseek-v4.1-flash.toml @@ -0,0 +1,29 @@ +name = "DeepSeek V4.1 Flash" +description = "DeepSeek-V4.1-Flash model official release. This is the smallest model in DeepSeek’s new model-architecture series, featuring native multimodal visual understanding capabilities. The new architecture is designed to deliver a higher capability ceiling, faster inference speeds, greater throughput, and scalability to larger-parameter models." +release_date = "2026-09-08" +last_updated = "2026-09-08" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.155 +output = 0.62 +cache_read = 0.0031 + +[limit] +context = 1_000_000 +output = 384_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/doubao-seed-1-8.toml b/providers/aihubmix/models/doubao-seed-1-8.toml new file mode 100644 index 00000000000..c456f3f6c12 --- /dev/null +++ b/providers/aihubmix/models/doubao-seed-1-8.toml @@ -0,0 +1,41 @@ +name = "Doubao Seed 1.8" +description = "Doubao's strongest multimodal Agent model Seed1.8 has powerful multimodal capabilities, supports image and text input, and can efficiently and accurately complete tasks in scenarios such as information retrieval, code generation, GUI interaction, and complex workflows, meeting increasingly diverse technical demands." +release_date = "2025-12-28" +last_updated = "2025-12-28" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.10959 +output = 0.273975 +cache_read = 0.021918 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.1644 +output = 2.191995 +cache_read = 0.021915 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.32876 +output = 3.2876 +cache_read = 0.021918 + +[limit] +context = 256_000 +output = 32_768 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/doubao-seed-2-0-lite-260215.toml b/providers/aihubmix/models/doubao-seed-2-0-lite-260215.toml new file mode 100644 index 00000000000..69840b095a2 --- /dev/null +++ b/providers/aihubmix/models/doubao-seed-2-0-lite-260215.toml @@ -0,0 +1,41 @@ +name = "Doubao Seed 2.0 Lite 260215" +description = "Doubao Coding model optimized for real-world programming environments that can reliably invoke tools in common IDEs such as Claude Code. The model is specially optimized for frontend capabilities and performs well with common frontend frameworks. The model supports Skills and can work with various custom skills." +release_date = "2026-02-15" +last_updated = "2026-02-15" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.09041 +output = 0.54246 +cache_read = 0.018082 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.135616 +output = 0.813696 +cache_read = 0.027123 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.271232 +output = 1.627392 +cache_read = 0.054246 + +[limit] +context = 256_000 +output = 128_000 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/ernie-5.1.toml b/providers/aihubmix/models/ernie-5.1.toml new file mode 100644 index 00000000000..9f902de60fb --- /dev/null +++ b/providers/aihubmix/models/ernie-5.1.toml @@ -0,0 +1,30 @@ +name = "ERNIE 5.1" +description = "ERNIE 5.1 is the latest model in the Wenxin series, with comprehensive upgrades to its foundational capabilities and significant improvements in agents, knowledge, reasoning, and deep search. This upgrade uses a decoupled fully-asynchronous reinforcement learning technique to specifically address challenges encountered as large models evolve toward agent-based autonomous decision-making, such as training–inference numerical bias, low utilization of heterogeneous resources, and global issues caused by long-tail effects. It is paired with scaled agent post-training techniques to enhance model capabilities and generalization, enabling a three-step collaboration of environment, expert, and fusion that both ensures training efficiency and significantly improves the model’s stability and performance on complex tasks." +release_date = "2026-05-10" +last_updated = "2026-05-10" +attachment = false +reasoning = true +tool_call = false +open_weights = false + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.5634 +output = 2.5353 +cache_read = 0.5634 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.845 +output = 3.098361 +cache_read = 0.845 + +[limit] +context = 119_000 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml b/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml new file mode 100644 index 00000000000..35d209d0223 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml @@ -0,0 +1,29 @@ +name = "Gemini 2.5 Flash Lite Preview 09 2025" +description = "gemini-2.5-flash-lite latest preview version" +release_date = "2025-09-25" +last_updated = "2025-09-25" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1 +output = 0.4 +cache_read = 0.01 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "video", "audio", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml b/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml new file mode 100644 index 00000000000..e5a39cfc131 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml @@ -0,0 +1,29 @@ +name = "Gemini 2.5 Flash Preview 05-20 Search" +description = "Gemini-2.5 Flash Preview 05-20 Search integrates Google's official search functionality; the search feature will have an additional separate fee log directly integrated into the scoring deduction, with detailed logs not displayed. It will be fixed and displayed later. Only OpenAI-compatible formats are supported for invocation; Gemini SDK is not supported. For Gemini's native SDK, please set parameters directly using the official search parameters." +release_date = "2025-05-20" +last_updated = "2025-05-20" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.3 +output = 2.499 +cache_read = 0.03 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "video", "audio"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml b/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml new file mode 100644 index 00000000000..67d5d794003 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml @@ -0,0 +1,29 @@ +name = "Gemini 2.5 Flash Preview 09 2025" +description = "This latest 2.5 Flash model comes with improvements in two key areas we heard consistent feedback on:\n\nBetter agentic tool use: We've improved how the model uses tools, leading to better performance in more complex, agentic and multi-step applications. This model shows noticeable improvements on key agentic benchmarks, including a 5% gain on SWE-Bench Verified, compared to our last release (48.9% → 54%). More efficient: With thinking on, the model is now significantly more cost-efficient—achieving higher quality outputs while using fewer tokens, reducing latency and cost (see charts above)." +release_date = "2025-09-25" +last_updated = "2025-09-25" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.3 +output = 2.499 +cache_read = 0.03 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "video", "audio"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml new file mode 100644 index 00000000000..6b6a905bb46 --- /dev/null +++ b/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml @@ -0,0 +1,34 @@ +name = "Gemini 2.5 Pro Preview 05-06" +description = "gemini-2.5-pro latest model" +release_date = "2025-05-07" +last_updated = "2025-05-07" +attachment = true +reasoning = true +tool_call = false +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.25 +output = 10 +cache_read = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "video", "audio", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.2-high.toml b/providers/aihubmix/models/gpt-5.2-high.toml new file mode 100644 index 00000000000..d47568fd4e9 --- /dev/null +++ b/providers/aihubmix/models/gpt-5.2-high.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-5.2" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 1.75 +output = 14 +cache_read = 0.175 diff --git a/providers/aihubmix/models/gpt-5.2-low.toml b/providers/aihubmix/models/gpt-5.2-low.toml new file mode 100644 index 00000000000..d47568fd4e9 --- /dev/null +++ b/providers/aihubmix/models/gpt-5.2-low.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-5.2" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 1.75 +output = 14 +cache_read = 0.175 diff --git a/providers/aihubmix/models/gpt-chat-latest.toml b/providers/aihubmix/models/gpt-chat-latest.toml new file mode 100644 index 00000000000..66580e968a4 --- /dev/null +++ b/providers/aihubmix/models/gpt-chat-latest.toml @@ -0,0 +1,29 @@ +name = "GPT Chat" +description = "GPT Chat Latest points to OpenAI's stable API alias chat-latest that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates in the future, they are routed behind this slug automatically." +release_date = "2026-05-05" +last_updated = "2026-05-05" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false +reasoning_options = [] + +[cost] +input = 5 +output = 30 +cache_read = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 10 +output = 45 +cache_read = 1 + +[limit] +context = 400_000 +output = 128_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/grok-4-1-fast-non-reasoning.toml b/providers/aihubmix/models/grok-4-1-fast-non-reasoning.toml new file mode 100644 index 00000000000..3e383f20ae2 --- /dev/null +++ b/providers/aihubmix/models/grok-4-1-fast-non-reasoning.toml @@ -0,0 +1,13 @@ +base_model = "xai/grok-4.3" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.2 +output = 0.5 +cache_read = 0.05 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/grok-4-1-fast-reasoning.toml b/providers/aihubmix/models/grok-4-1-fast-reasoning.toml new file mode 100644 index 00000000000..18531f0d119 --- /dev/null +++ b/providers/aihubmix/models/grok-4-1-fast-reasoning.toml @@ -0,0 +1,13 @@ +base_model = "xai/grok-4.3" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.2 +output = 0.5 +cache_read = 0.05 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/grok-4-fast-non-reasoning.toml b/providers/aihubmix/models/grok-4-fast-non-reasoning.toml new file mode 100644 index 00000000000..f3fddee84df --- /dev/null +++ b/providers/aihubmix/models/grok-4-fast-non-reasoning.toml @@ -0,0 +1,19 @@ +base_model = "xai/grok-4.3" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[cost] +input = 0.2 +output = 0.5 +cache_read = 0.05 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.4 +output = 1 +cache_read = 0.05 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/grok-4-fast-reasoning.toml b/providers/aihubmix/models/grok-4-fast-reasoning.toml new file mode 100644 index 00000000000..81fcf48ef9c --- /dev/null +++ b/providers/aihubmix/models/grok-4-fast-reasoning.toml @@ -0,0 +1,19 @@ +base_model = "xai/grok-4.3" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.2 +output = 0.5 +cache_read = 0.05 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.4 +output = 1 +cache_read = 0.05 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/grok-4.toml b/providers/aihubmix/models/grok-4.toml new file mode 100644 index 00000000000..9bd66831f47 --- /dev/null +++ b/providers/aihubmix/models/grok-4.toml @@ -0,0 +1,26 @@ +name = "Grok 4" +description = "Grok, their latest and greatest flagship model, offers unparalleled performance in natural language, math, and reasoning – the perfect jack of all trades.\nThe current pointing model version is grok-4-0709." +release_date = "2025-07-09" +last_updated = "2025-07-09" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 3.3 +output = 16.5 +cache_read = 0.825 + +[limit] +context = 256_000 +output = 64_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/grok-code-fast-1.toml b/providers/aihubmix/models/grok-code-fast-1.toml new file mode 100644 index 00000000000..bd540da78c4 --- /dev/null +++ b/providers/aihubmix/models/grok-code-fast-1.toml @@ -0,0 +1,19 @@ +base_model = "xai/grok-build-0.1" +reasoning_options = [] + +[cost] +input = 1 +output = 2 +cache_read = 0.2 + +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2 +output = 2 +cache_read = 0.4 + +[limit] +output = 10_000 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/kimi-for-coding-free.toml b/providers/aihubmix/models/kimi-for-coding-free.toml new file mode 100644 index 00000000000..85e6a9bd86e --- /dev/null +++ b/providers/aihubmix/models/kimi-for-coding-free.toml @@ -0,0 +1,22 @@ +name = "Kimi For Coding (free)" +description = "kimi-for-coding-free is a free and open version offered by AIHubMix specifically for Kimi users. To maintain stable service operations, the following usage limits apply: a maximum of 5 requests per minute 500 total requests per day, and a daily quota of 1 million tokens." +release_date = "2026-06-12" +last_updated = "2026-06-12" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = true +reasoning_options = [] + +[cost] +input = 0 +output = 0 + +[limit] +context = 256_000 +output = 32_768 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2-0711.toml b/providers/aihubmix/models/kimi-k2-0711.toml new file mode 100644 index 00000000000..65bb06c8c4f --- /dev/null +++ b/providers/aihubmix/models/kimi-k2-0711.toml @@ -0,0 +1,21 @@ +name = "Kimi K2 0711" +description = "Kimi-K2 is a MoE architecture foundational model with extremely powerful coding and agent capabilities, featuring a total of 1 trillion parameters and activating 32 billion parameters. In benchmark performance tests across major categories such as general knowledge reasoning, programming, mathematics, and agents, the K2 model outperforms other mainstream open-source models.\nThe Kimi-K2 model supports a context length of 128k tokens.\nIt does not support visual capabilities." +release_date = "2025-01-01" +last_updated = "2025-01-01" +attachment = false +reasoning = false +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.54 +output = 2.16 + +[limit] +context = 131_072 +output = 16_384 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2-turbo-preview.toml b/providers/aihubmix/models/kimi-k2-turbo-preview.toml new file mode 100644 index 00000000000..3cadac442fd --- /dev/null +++ b/providers/aihubmix/models/kimi-k2-turbo-preview.toml @@ -0,0 +1,22 @@ +name = "Kimi K2 Turbo Preview" +description = "The kimi-k2-turbo-preview model is a high-speed version of kimi-k2, with the same model parameters as kimi-k2, but the output speed has been increased from 10 tokens per second to 40 tokens per second." +release_date = "2025-07-08" +last_updated = "2025-07-08" +attachment = false +reasoning = false +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 1.2 +output = 4.8 +cache_read = 0.3 + +[limit] +context = 262_144 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2.5.toml b/providers/aihubmix/models/kimi-k2.5.toml index 59191e20864..fbf7c92338a 100644 --- a/providers/aihubmix/models/kimi-k2.5.toml +++ b/providers/aihubmix/models/kimi-k2.5.toml @@ -14,3 +14,6 @@ cache_read = 0.105 [limit] output = 32_768 + +[modalities] +input = ["text", "image"] diff --git a/providers/aihubmix/models/laguna-s-2.1-free.toml b/providers/aihubmix/models/laguna-s-2.1-free.toml new file mode 100644 index 00000000000..e2be262fa86 --- /dev/null +++ b/providers/aihubmix/models/laguna-s-2.1-free.toml @@ -0,0 +1,23 @@ +name = "Laguna S 2.1 (free)" +description = "Laguna S 2.1 is the latest coding agent model from Poolside, featuring an impressive context length of 262,144 tokens. This model is built with 118B total parameters and 8B active parameters, balancing efficiency with high performance. It delivers strong capabilities for developer tasks, scoring 70.2% on the Terminal-Bench 2.1 benchmark." +release_date = "2026-07-21" +last_updated = "2026-07-21" +attachment = false +reasoning = true +tool_call = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_048_576 +output = 262_144 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/mercury-2.5-preview.toml b/providers/aihubmix/models/mercury-2.5-preview.toml new file mode 100644 index 00000000000..2001a76823d --- /dev/null +++ b/providers/aihubmix/models/mercury-2.5-preview.toml @@ -0,0 +1,25 @@ +name = "Mercury 2.5 Preview" +description = "Mercury 2.5 is the latest diffusion-based large language model (dLLM) released by Inception. It is the fastest inference LLM; unlike the sequential token-by-token generation approach, Mercury 2.5 can generate and optimize multiple tokens in parallel, achieving a generation speed of 1,107 tokens per second on standard GPUs. Compared to Mercury 2, its intelligence has increased by more than 10 percentage points, and its quality rivals leading cost-optimized frontier models such as GPT-5.6 Luna (Low), Gemini 3.5 Flash-Lite, and Claude Haiku 4.5." +release_date = "2026-09-02" +last_updated = "2026-09-02" +attachment = false +reasoning = true +tool_call = true +open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0.2 +output = 0.75 +cache_read = 0.02 + +[limit] +context = 260_000 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/mimo-v2-flash.toml b/providers/aihubmix/models/mimo-v2-flash.toml index e367b1430ed..e9a8258948c 100644 --- a/providers/aihubmix/models/mimo-v2-flash.toml +++ b/providers/aihubmix/models/mimo-v2-flash.toml @@ -11,7 +11,7 @@ output = 0.5754 cache_read = 0.03836 [limit] -context = 1_000_000 +context = 1_048_576 output = 131_072 [modalities] diff --git a/providers/aihubmix/models/mistral-large-3.toml b/providers/aihubmix/models/mistral-large-3.toml new file mode 100644 index 00000000000..cfd30382a16 --- /dev/null +++ b/providers/aihubmix/models/mistral-large-3.toml @@ -0,0 +1,21 @@ +name = "Mistral Large 3" +description = "Mistral Large 3 is a MoE model with 67.5B total parameters and 41B active parameters, supporting a 256K-token context window. Trained from scratch on 3,000 NVIDIA H200 GPUs, it is one of the strongest permissively licensed open-weight models available.\n\nDesigned for advanced reasoning and long-context understanding, Mistral Large 3 delivers performance on par with the best instruction-tuned open-weight models for general-purpose tasks, while also offering image understanding capabilities. Its multilingual strengths are particularly notable for non-English/Chinese languages, making it well-suited for global applications.\n\nTypical use cases include enterprise assistants, multilingual customer support, content generation and editing, data analysis over long documents, code assistance, and research workflows that require handling large corpora or complex instructions. With its MoE architecture, Mistral Large 3 balances strong performance with efficient inference, providing a versatile backbone for building reliable, production-grade AI systems." +release_date = "2024-11-01" +last_updated = "2024-11-01" +attachment = true +reasoning = false +tool_call = true +structured_output = true +open_weights = true + +[cost] +input = 0.5 +output = 1.5 + +[limit] +context = 256_000 +output = 131_072 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/muse-spark-1.1.toml b/providers/aihubmix/models/muse-spark-1.1.toml new file mode 100644 index 00000000000..58eb68641cf --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.1.toml @@ -0,0 +1,15 @@ +base_model = "meta/muse-spark-1.1" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.375 +output = 4.675 + +[modalities] +input = ["text", "image", "video", "audio", "pdf"] diff --git a/providers/aihubmix/models/muse-spark-1.2.toml b/providers/aihubmix/models/muse-spark-1.2.toml new file mode 100644 index 00000000000..86615948cd1 --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.2.toml @@ -0,0 +1,12 @@ +base_model = "meta/muse-spark-1.2" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.375 +output = 4.675 diff --git a/providers/aihubmix/models/muse-spark-1.3.toml b/providers/aihubmix/models/muse-spark-1.3.toml new file mode 100644 index 00000000000..a985b6a7e3a --- /dev/null +++ b/providers/aihubmix/models/muse-spark-1.3.toml @@ -0,0 +1,16 @@ +base_model = "meta/muse-spark-1.3" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.375 +output = 4.675 +cache_read = 0.165 + +[limit] +output = 1_000_000 diff --git a/providers/aihubmix/models/north-mini-code-free.toml b/providers/aihubmix/models/north-mini-code-free.toml new file mode 100644 index 00000000000..dee867a0ebe --- /dev/null +++ b/providers/aihubmix/models/north-mini-code-free.toml @@ -0,0 +1,27 @@ +name = "North Mini Code (free)" +description = "Developed by Cohere, north-mini-code-free is the debut model of the North family and Cohere's first agentic coding model. This sparse mixture-of-experts model features 30B total parameters and 3B active parameters, designed and optimized for high performance. With an expansive context length of 256,000 tokens, it is well-suited for handling complex developer workflows and large codebases." +release_date = "2026-06-09" +last_updated = "2026-06-09" +attachment = false +reasoning = true +tool_call = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "high"] + +[cost] +input = 0 +output = 0 + +[limit] +context = 256_000 +output = 64_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/ox-alpha.toml b/providers/aihubmix/models/ox-alpha.toml new file mode 100644 index 00000000000..54c444c537c --- /dev/null +++ b/providers/aihubmix/models/ox-alpha.toml @@ -0,0 +1,18 @@ +base_model = "zhipuai/glm-5.3-flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0 +output = 0 + +[limit] +context = 1_048_576 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3-max-2026-01-23.toml b/providers/aihubmix/models/qwen3-max-2026-01-23.toml new file mode 100644 index 00000000000..5ecc2962be5 --- /dev/null +++ b/providers/aihubmix/models/qwen3-max-2026-01-23.toml @@ -0,0 +1,43 @@ +name = "Qwen3 Max 2026 01-23" +description = "The snapshot version of the Tongyi Qianwen 3 series Max model is from January 23, 2026. By default, it does not require thinking, but thinking mode can be enabled through the enable_thinking parameter, as detailed in the code example. (After enabling thinking by passing parameters, it becomes: Qwen3-Max-Thinking). This model has a total parameter count exceeding one trillion (1T) and a pre-training data volume of up to 36T Tokens, making it the largest and most powerful reasoning model from Alibaba to date." +release_date = "2026-01-23" +last_updated = "2026-01-23" +attachment = false +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.4508 +output = 1.8032 +cache_read = 0.09016 +cache_write = 0.5635 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.902 +output = 3.608 +cache_read = 0.1804 +cache_write = 1.1275 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 1.3522 +output = 5.4088 +cache_read = 0.27044 +cache_write = 1.69025 + +[limit] +context = 262_144 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3-max.toml b/providers/aihubmix/models/qwen3-max.toml index 4e0eae21c31..21d091e1145 100644 --- a/providers/aihubmix/models/qwen3-max.toml +++ b/providers/aihubmix/models/qwen3-max.toml @@ -1,7 +1,13 @@ base_model = "alibaba/qwen3-max" -attachment = true +reasoning = true structured_output = true +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.4508 output = 1.8032 @@ -21,6 +27,3 @@ input = 1.3522 output = 8.1132 cache_read = 0.27044 cache_write = 1.69025 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml b/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml index 740e3a625de..31e731318f6 100644 --- a/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml +++ b/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml @@ -1,11 +1,15 @@ base_model = "alibaba/qwen3-vl-235b-a22b-instruct" +reasoning = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 0.274 output = 1.096 -[limit] -output = 33_000 - [modalities] input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml b/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml index 9152a056352..702c6a92356 100644 --- a/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml +++ b/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml @@ -1,12 +1,14 @@ base_model = "alibaba/qwen3-vl-235b-a22b-thinking" -reasoning_options = [] + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 0.274 output = 2.74 -[limit] -output = 33_000 - [modalities] input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3-vl-30b-a3b-instruct.toml b/providers/aihubmix/models/qwen3-vl-30b-a3b-instruct.toml new file mode 100644 index 00000000000..6be18b91c3c --- /dev/null +++ b/providers/aihubmix/models/qwen3-vl-30b-a3b-instruct.toml @@ -0,0 +1,27 @@ +name = "Qwen3 VL 30B A3B Instruct" +description = "The Qwen3-VL series’ second-largest MoE model Instruct version offers fast response speed and supports ultra-long contexts such as long videos and long documents; it features comprehensive upgrades in image/video understanding, spatial perception, and universal recognition abilities; it also provides visual 2DD/3D localization capabilities, making it capable of handling complex real-world tasks." +release_date = "2025-10-05" +last_updated = "2025-10-05" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1028 +output = 0.4112 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3-vl-30b-a3b-thinking.toml b/providers/aihubmix/models/qwen3-vl-30b-a3b-thinking.toml new file mode 100644 index 00000000000..6760b8f994c --- /dev/null +++ b/providers/aihubmix/models/qwen3-vl-30b-a3b-thinking.toml @@ -0,0 +1,27 @@ +name = "Qwen3 VL 30B A3B Thinking" +description = "The Qwen3-VL series’ second-largest MoE model Thinking version offers fast response speed, stronger multimodal understanding and reasoning, visual agent capabilities, and ultra-long context support for long videos and long documents; it features comprehensive upgrades in image/video understanding, spatial perception, and universal recognition abilities, making it capable of handling complex real-world tasks." +release_date = "2025-10-11" +last_updated = "2025-10-11" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.1028 +output = 1.028 + +[limit] +context = 131_072 +output = 32_768 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3-vl-flash.toml b/providers/aihubmix/models/qwen3-vl-flash.toml new file mode 100644 index 00000000000..9b64faecbc9 --- /dev/null +++ b/providers/aihubmix/models/qwen3-vl-flash.toml @@ -0,0 +1,40 @@ +name = "Qwen3 VL Flash" +description = "The Qwen3 series of compact visual-understanding models achieves an effective fusion of thinking mode and non-thinking mode, outperforming the open-source Qwen3-VL-30B-A3B with faster response speeds. It comprehensively upgrades image and video understanding, supporting ultra-long contexts such as long videos and long documents, spatial awareness, and universal object recognition; it also possesses visual 2D/3D localization capabilities and is capable of handling complex real-world tasks." +release_date = "2025-10-09" +last_updated = "2025-10-09" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.0206 +output = 0.206 +cache_read = 0.00412 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.041 +output = 0.41 +cache_read = 0.0082 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.0822 +output = 0.822 +cache_read = 0.01644 + +[limit] +context = 262_144 +output = 32_768 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3-vl-plus.toml b/providers/aihubmix/models/qwen3-vl-plus.toml index 7a0ac27b37b..4d14d7292b3 100644 --- a/providers/aihubmix/models/qwen3-vl-plus.toml +++ b/providers/aihubmix/models/qwen3-vl-plus.toml @@ -1,8 +1,13 @@ base_model = "alibaba/qwen3-vl-plus" attachment = true -reasoning = false structured_output = true +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.137 output = 1.37 @@ -18,9 +23,5 @@ tier = { type = "context", size = 128_000 } input = 0.411 output = 4.11 -[limit] -context = 256_000 -output = 32_000 - [modalities] input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.6-plus-preview-free.toml b/providers/aihubmix/models/qwen3.6-plus-preview-free.toml index c0771064bae..43fb38a91b3 100644 --- a/providers/aihubmix/models/qwen3.6-plus-preview-free.toml +++ b/providers/aihubmix/models/qwen3.6-plus-preview-free.toml @@ -1,7 +1,16 @@ base_model = "alibaba/qwen3.6-plus" attachment = false tool_call = false -reasoning_options = [] + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 0 diff --git a/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml b/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml new file mode 100644 index 00000000000..56b7d4d0318 --- /dev/null +++ b/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml @@ -0,0 +1,33 @@ +name = "Qwen3.8 Max 2026 09-02" +description = "Qwen3.8-Max-0902 (also known as qwen3.8-max-2026-09-02) is a snapshot version of Alibaba Cloud Tongyi Qianwen's qwen3.8-max. It pushes encoding depth further, enabling it to handle more complex engineering-level projects and long-term autonomous development; collaborative agent capabilities are significantly enhanced, performing more confidently in multi-tool orchestration and end-to-end delivery; visual understanding is comprehensively improved, with more sensitive and accurate chart reasoning, document parsing, and multimodal perception. Continuing the 1-million-context window, reasoning modes, and a complete tool ecosystem, it continues to evolve at a higher level of intelligence." +release_date = "2026-09-02" +last_updated = "2026-09-02" +attachment = true +reasoning = true +tool_call = true +structured_output = true +open_weights = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.69 +output = 5.07 +cache_read = 0.169 +cache_write = 2.1125 + +[limit] +context = 1_000_000 +output = 131_072 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/sync.md b/sync.md index 5753d5deccf..ed6144e9beb 100644 --- a/sync.md +++ b/sync.md @@ -254,11 +254,13 @@ xAI is implemented in `packages/core/src/sync/providers/xai.ts`. - Source endpoint: `https://aihubmix.com/api/v1/models?type=llm`. - No authentication is required; the catalog is public. - The endpoint serves capabilities, limits, modalities, reasoning controls and pricing, so it is authoritative for all of them. Fields it does not serve (`family`, `temperature`, `interleaved`, `knowledge`) keep whatever was authored. -- AIHubMix relays upstream models under its own IDs, so a relay is factored onto the lab metadata it serves (`base_model`) whenever that metadata exists, and then records only what it actually changes. `developer_id` maps a relay to its lab. -- Resolving a relay to a lab model strips two kinds of affix: routing prefixes that name the upstream compute or an AIHubMix namespace (`alicloud-`, `coding-`, `ahm-`), and routing suffixes that select a mode rather than a different model (`-free`, `-nothink`, `-search`). Dated release tags are deliberately **not** stripped: `gemini-2.5-pro-preview-06-05` is a pinned snapshot, not the model `google/gemini-2.5-pro`, and every other provider in the catalog writes such IDs as standalone entries. +- AIHubMix relays upstream models under its own IDs, so a relay is factored onto the lab metadata it serves (`base_model`) whenever that metadata exists, and then records only what it actually changes. +- The endpoint answers both halves of that lookup itself, so nothing about a relay is inferred from its ID here: `vendor` names the lab that built the model, and `variant_of` names the AIHubMix ID the entry is a routing variant of (`variant_kind` labels it a pricing tier, channel tier, mode preset or deprecated alias). A relay is looked up under its own ID first and then under each `variant_of` hop, nearest first — `qwen3.8-max-preview` is a variant of `qwen3.8-max` and both are published, so the relay factors onto the preview it actually serves. Following the declared chain also resolves relays no string rule reaches, such as `ox-alpha` onto `zhipuai/glm-5.3-flash` and `grok-code-fast-1` onto `xai/grok-build-0.1`. +- `VENDOR_LABS` maps the four labs the two registries spell differently (`zhipu`/`zhipuai`, `moonshot`/`moonshotai`, `bytedance`/`bytedance-seed`, `meituan-longcat`/`meituan`). It maps namespaces only; no entry decides what a model is or which lab built it. +- Dated release tags are deliberately not special-cased: `gemini-2.5-pro-preview-06-05` is a pinned snapshot, not the model `google/gemini-2.5-pro`, and the endpoint does not declare it a variant of one. - Lab IDs are matched case-insensitively: AIHubMix spells `minimax-m2` where the lab spells `MiniMax-M2`, and the same model can arrive under several casings, so relays are deduped on the case-folded ID. -- A relay with neither resolvable lab metadata nor the `release_date` and `open_weights` a standalone entry requires is skipped rather than written with invented values. The endpoint dates 306 of its 415 models and serves no `open_weights` field at all, which is what the remaining gap is made of. -- Two limit sentinels are read as absent rather than as real ceilings: `max_output: 0` (107 of 415 models) and a `max_output` at or above `context_length` (37), which is the context window quoted a second time. +- A relay with neither resolvable lab metadata nor the `release_date`, `open_weights` and limits a standalone entry requires is skipped rather than written with invented values. Of 408 listed models the endpoint names a `vendor` for 292, declares `variant_of` for 76, dates 304 and states `open_weights` for 289; the remaining gap is what the skip notices report back upstream. +- Two limit sentinels are read as absent rather than as real ceilings: `max_output: 0` (104 of 408 models) and a `max_output` at or above `context_length` (36), which is the context window quoted a second time. - `reasoning_options[]` entries carry an AIHubMix-only `default` key that the strict `ReasoningOption` schema rejects, and spell two effort levels differently (`no_think`, `instant`), so translation drops the extra key and maps those onto `none` and `minimal`. - A price field is omitted when the model has no such rate, so an omitted `cache_read`/`cache_write` clears an authored one; authored pricing survives only when the endpoint quotes nothing at all for the model. - Routing aliases such as `alicloud-glm-5.1` and `deep-deepseek-v4-pro` are served but not listed by the endpoint, so local files missing from the response are retained (`deleteMissing: false`) and each opens a deduped GitHub issue. From 4bf1dcc4adbfd5af4087069934276860d22070fc Mon Sep 17 00:00:00 2001 From: chenxue Date: Fri, 11 Sep 2026 21:20:30 +0800 Subject: [PATCH 12/20] chore(aihubmix): leave the catalog files to the sync automation The pull request carried 299 generated `providers/aihubmix/models` files alongside the adapter, which pushed the diff past GitHub's 300-file ceiling. `gh pr diff --patch` then answered HTTP 406, and the reviewer workflow died in its context step before the agent ever ran -- so no revision of this branch could earn `reviewer: ready`. Ship the adapter alone. Once it merges, `sync-models.yml` regenerates the catalog on `automation/sync-models-aihubmix`, a branch the reviewer workflow already skips by design. Co-Authored-By: Claude Opus 5 --- .../models/DeepSeek-V3.1-Terminus.toml | 24 ---------- .../aihubmix/models/DeepSeek-V3.1-Think.toml | 13 ------ providers/aihubmix/models/DeepSeek-V3.toml | 9 ---- providers/aihubmix/models/Qwen/QwQ-32B.toml | 7 --- .../aihubmix/models/agnes-2.5-pro-alpha.toml | 24 ---------- .../aihubmix/models/agnes-3.0-flash.toml | 26 ----------- .../models/aihubmix-command-r-08-2024.toml | 6 --- .../aihubmix-command-r-plus-08-2024.toml | 6 --- .../bai-qwen3-vl-235b-a22b-instruct.toml | 10 ---- providers/aihubmix/models/cc-MiniMax-M2.toml | 7 --- .../aihubmix/models/cc-deepseek-v3.1.toml | 7 --- providers/aihubmix/models/cc-glm-5-turbo.toml | 11 ----- providers/aihubmix/models/cc-glm-5.1.toml | 8 ---- providers/aihubmix/models/cc-glm-5.toml | 12 ----- .../aihubmix/models/cc-minimax-m2.1.toml | 13 ------ .../models/cc-minimax-m2.5-highspeed.toml | 13 ------ .../aihubmix/models/cc-minimax-m2.5.toml | 13 ------ .../models/cc-minimax-m2.7-highspeed.toml | 13 ------ .../aihubmix/models/cc-minimax-m2.7.toml | 13 ------ providers/aihubmix/models/cc-minimax-m3.toml | 16 ------- .../models/claude-3-haiku-20240307.toml | 9 ---- .../aihubmix/models/claude-fable-5-1.toml | 15 ------ providers/aihubmix/models/claude-fable-5.toml | 17 +++---- .../aihubmix/models/claude-haiku-4-5.toml | 14 ------ .../aihubmix/models/claude-opus-4-1.toml | 15 ------ .../models/claude-opus-4-5-think.toml | 18 -------- .../aihubmix/models/claude-opus-4-5.toml | 18 -------- .../models/claude-opus-4-6-think.toml | 33 +++++++------ .../aihubmix/models/claude-opus-4-6.toml | 35 ++++++++------ .../models/claude-opus-4-7-think.toml | 30 ++++++++---- .../aihubmix/models/claude-opus-4-7.toml | 32 +++++++++---- .../models/claude-opus-4-8-think.toml | 18 ++++---- .../aihubmix/models/claude-opus-4-8.toml | 18 ++++---- providers/aihubmix/models/claude-opus-5.toml | 13 ++---- .../models/claude-sonnet-4-5-think.toml | 21 --------- .../aihubmix/models/claude-sonnet-4-5.toml | 21 --------- .../models/claude-sonnet-4-6-think.toml | 44 ++++++++++-------- .../aihubmix/models/claude-sonnet-4-6.toml | 46 +++++++++++-------- .../aihubmix/models/claude-sonnet-5.toml | 18 ++++---- .../aihubmix/models/cloudflare-glm-5.2.toml | 14 ------ .../aihubmix/models/coding-glm-4.6-free.toml | 12 ----- providers/aihubmix/models/coding-glm-4.6.toml | 13 ------ .../aihubmix/models/coding-glm-4.7-free.toml | 12 ----- providers/aihubmix/models/coding-glm-4.7.toml | 13 ------ .../aihubmix/models/coding-glm-5-free.toml | 12 ----- .../models/coding-glm-5-turbo-free.toml | 11 ----- .../aihubmix/models/coding-glm-5-turbo.toml | 11 ----- .../aihubmix/models/coding-glm-5.1-free.toml | 22 +++++++-- providers/aihubmix/models/coding-glm-5.1.toml | 23 ++++++++-- .../aihubmix/models/coding-glm-5.2-free.toml | 12 ----- providers/aihubmix/models/coding-glm-5.2.toml | 12 ----- .../aihubmix/models/coding-glm-5.3-free.toml | 15 ------ providers/aihubmix/models/coding-glm-5.3.toml | 16 ------- providers/aihubmix/models/coding-glm-5.toml | 12 ----- .../aihubmix/models/coding-kimi-k3-free.toml | 15 ------ providers/aihubmix/models/coding-kimi-k3.toml | 16 ------- .../models/coding-minimax-m2-free.toml | 13 ------ .../models/coding-minimax-m2.1-free.toml | 13 ------ .../aihubmix/models/coding-minimax-m2.1.toml | 13 ------ .../models/coding-minimax-m2.5-free.toml | 13 ------ .../models/coding-minimax-m2.5-highspeed.toml | 13 ------ .../aihubmix/models/coding-minimax-m2.5.toml | 13 ------ .../models/coding-minimax-m2.7-free.toml | 26 +++++++---- .../models/coding-minimax-m2.7-highspeed.toml | 24 ++++++---- .../aihubmix/models/coding-minimax-m2.7.toml | 24 ++++++---- .../aihubmix/models/coding-minimax-m2.toml | 13 ------ .../models/coding-minimax-m3-free.toml | 16 ------- .../aihubmix/models/coding-minimax-m3.toml | 16 ------- .../models/coding-xiaomi-mimo-v2.5-pro.toml | 11 ++--- .../models/coding-xiaomi-mimo-v2.5.toml | 11 ++--- .../aihubmix/models/cohere-command-a.toml | 11 ----- .../aihubmix/models/command-a-03-2025.toml | 11 ----- .../models/command-a-plus-05-2026.toml | 15 ------ .../aihubmix/models/command-r-08-2024.toml | 6 --- .../models/command-r-plus-08-2024.toml | 6 --- .../models/deepinfra-gemma-4-26b-a4b-it.toml | 16 ------- .../aihubmix/models/deepseek-v3.2-think.toml | 12 ----- providers/aihubmix/models/deepseek-v3.2.toml | 12 ----- .../models/deepseek-v4-flash-0731-fast.toml | 13 ------ .../models/deepseek-v4-flash-vision-exp.toml | 13 ------ .../aihubmix/models/deepseek-v4-flash.toml | 13 ------ .../aihubmix/models/deepseek-v4-pro-0813.toml | 8 +--- .../aihubmix/models/deepseek-v4-pro.toml | 13 ------ .../aihubmix/models/deepseek-v4.1-flash.toml | 29 ------------ .../aihubmix/models/doubao-seed-1-8.toml | 41 ----------------- .../models/doubao-seed-2-0-code-preview.toml | 24 ++++------ .../models/doubao-seed-2-0-lite-260215.toml | 41 ----------------- .../models/doubao-seed-2-0-lite-260428.toml | 31 ++++++------- .../models/doubao-seed-2-0-mini-260428.toml | 29 ++++++------ .../aihubmix/models/doubao-seed-2-0-pro.toml | 24 ++++------ providers/aihubmix/models/ernie-5.1.toml | 30 ------------ .../models/gemini-2.5-flash-image.toml | 10 ---- .../models/gemini-2.5-flash-lite-nothink.toml | 13 ------ ...gemini-2.5-flash-lite-preview-09-2025.toml | 29 ------------ .../models/gemini-2.5-flash-lite.toml | 13 ------ .../models/gemini-2.5-flash-nothink.toml | 13 ------ ...gemini-2.5-flash-preview-05-20-search.toml | 29 ------------ .../gemini-2.5-flash-preview-09-2025.toml | 29 ------------ .../models/gemini-2.5-flash-search.toml | 13 ------ .../aihubmix/models/gemini-2.5-flash.toml | 33 +++++++++---- .../models/gemini-2.5-pro-preview-05-06.toml | 34 -------------- .../models/gemini-2.5-pro-search.toml | 19 -------- providers/aihubmix/models/gemini-2.5-pro.toml | 35 +++++++++----- .../models/gemini-3-flash-preview-free.toml | 12 ----- .../models/gemini-3-flash-preview-search.toml | 13 ------ .../models/gemini-3-flash-preview.toml | 34 +++++++++----- .../models/gemini-3-pro-image-preview.toml | 7 --- .../aihubmix/models/gemini-3-pro-image.toml | 12 ----- .../gemini-3.1-flash-image-preview.toml | 10 ---- .../models/gemini-3.1-flash-image.toml | 12 ----- .../models/gemini-3.1-flash-lite-image.toml | 11 ----- .../models/gemini-3.1-flash-lite-nothink.toml | 13 ------ .../models/gemini-3.1-flash-lite.toml | 29 ++++++++---- .../gemini-3.1-pro-preview-customtools.toml | 30 ++++++++---- .../models/gemini-3.1-pro-preview-search.toml | 19 -------- .../models/gemini-3.1-pro-preview.toml | 36 ++++++++++----- .../models/gemini-3.5-flash-lite-free.toml | 9 ---- .../models/gemini-3.5-flash-lite.toml | 10 ---- .../aihubmix/models/gemini-3.5-flash.toml | 18 ++++---- .../models/gemini-3.6-flash-free.toml | 9 ---- .../aihubmix/models/gemini-3.6-flash.toml | 10 ---- .../models/gemini-3.7-flash-free.toml | 12 ----- .../aihubmix/models/gemini-3.7-flash.toml | 9 ++-- .../models/gemini-3.8-flash-free.toml | 12 ----- .../aihubmix/models/gemini-3.8-flash.toml | 13 ------ .../models/gemma-4-26b-a4b-it-free.toml | 15 ------ .../aihubmix/models/gemma-4-26b-a4b-it.toml | 19 -------- .../aihubmix/models/gemma-4-31b-it-free.toml | 15 ------ providers/aihubmix/models/gemma-4-31b-it.toml | 19 -------- providers/aihubmix/models/glm-4.5v.toml | 21 --------- providers/aihubmix/models/glm-4.6.toml | 19 -------- providers/aihubmix/models/glm-4.6v.toml | 23 ---------- .../aihubmix/models/glm-4.7-flash-free.toml | 9 ---- providers/aihubmix/models/glm-4.7.toml | 19 -------- providers/aihubmix/models/glm-5-turbo.toml | 12 ----- providers/aihubmix/models/glm-5.1.toml | 9 ---- .../aihubmix/models/glm-5.2-fast-preview.toml | 13 ------ providers/aihubmix/models/glm-5.2.toml | 19 +++++--- providers/aihubmix/models/glm-5.3-flash.toml | 11 ++--- providers/aihubmix/models/glm-5.3.toml | 10 +--- providers/aihubmix/models/glm-5.toml | 13 ------ providers/aihubmix/models/glm-5v-turbo.toml | 22 +++++---- providers/aihubmix/models/gpt-4.1-free.toml | 8 ---- .../aihubmix/models/gpt-4.1-mini-free.toml | 8 ---- providers/aihubmix/models/gpt-4.1-mini.toml | 9 ---- .../aihubmix/models/gpt-4.1-nano-free.toml | 5 -- providers/aihubmix/models/gpt-4.1-nano.toml | 6 --- providers/aihubmix/models/gpt-4.1.toml | 9 ---- .../aihubmix/models/gpt-4o-2024-11-20.toml | 7 --- providers/aihubmix/models/gpt-4o-free.toml | 8 ---- providers/aihubmix/models/gpt-4o-mini.toml | 10 ---- providers/aihubmix/models/gpt-4o.toml | 9 ---- .../aihubmix/models/gpt-5-chat-latest.toml | 12 ----- providers/aihubmix/models/gpt-5-codex.toml | 8 ---- providers/aihubmix/models/gpt-5-mini.toml | 7 --- providers/aihubmix/models/gpt-5-nano.toml | 7 --- providers/aihubmix/models/gpt-5-pro.toml | 9 ---- .../aihubmix/models/gpt-5.1-chat-latest.toml | 7 --- .../aihubmix/models/gpt-5.1-codex-max.toml | 7 --- .../aihubmix/models/gpt-5.1-codex-mini.toml | 28 ++++++++--- providers/aihubmix/models/gpt-5.1-codex.toml | 28 ++++++++--- providers/aihubmix/models/gpt-5.1.toml | 29 +++++++++--- .../aihubmix/models/gpt-5.2-chat-latest.toml | 7 --- providers/aihubmix/models/gpt-5.2-codex.toml | 28 ++++++++--- providers/aihubmix/models/gpt-5.2-high.toml | 10 ---- providers/aihubmix/models/gpt-5.2-low.toml | 10 ---- providers/aihubmix/models/gpt-5.2-pro.toml | 11 ----- providers/aihubmix/models/gpt-5.2.toml | 27 ++++++++--- .../aihubmix/models/gpt-5.3-chat-latest.toml | 6 --- providers/aihubmix/models/gpt-5.3-codex.toml | 27 ++++++++--- providers/aihubmix/models/gpt-5.4-mini.toml | 31 ++++++++++--- providers/aihubmix/models/gpt-5.4-nano.toml | 10 ---- providers/aihubmix/models/gpt-5.4-pro.toml | 15 ------ providers/aihubmix/models/gpt-5.4.toml | 41 ++++++++++++----- providers/aihubmix/models/gpt-5.5-free.toml | 9 ---- providers/aihubmix/models/gpt-5.5-pro.toml | 17 ------- providers/aihubmix/models/gpt-5.5.toml | 44 +++++++++++++----- providers/aihubmix/models/gpt-5.6-luna.toml | 24 ++++------ .../aihubmix/models/gpt-5.6-sol-disc.toml | 21 --------- providers/aihubmix/models/gpt-5.6-sol.toml | 24 ++++------ providers/aihubmix/models/gpt-5.6-terra.toml | 24 ++++------ providers/aihubmix/models/gpt-5.toml | 10 ---- providers/aihubmix/models/gpt-6-astra.toml | 21 --------- .../aihubmix/models/gpt-chat-latest.toml | 29 ------------ .../aihubmix/models/gpt-image-2-free.toml | 8 ---- providers/aihubmix/models/gpt-oss-120b.toml | 6 --- .../aihubmix/models/gpt-oss-20b-free.toml | 12 ----- providers/aihubmix/models/gpt-oss-20b.toml | 9 ---- .../models/grok-4-1-fast-non-reasoning.toml | 13 ------ .../models/grok-4-1-fast-reasoning.toml | 13 ------ .../models/grok-4-fast-non-reasoning.toml | 19 -------- .../models/grok-4-fast-reasoning.toml | 19 -------- providers/aihubmix/models/grok-4.3.toml | 25 +++++++--- providers/aihubmix/models/grok-4.5.toml | 17 ++++--- providers/aihubmix/models/grok-4.6.toml | 11 +---- providers/aihubmix/models/grok-4.toml | 26 ----------- providers/aihubmix/models/grok-build-0.1.toml | 9 ++-- .../aihubmix/models/grok-code-fast-1.toml | 19 -------- providers/aihubmix/models/hy3-free.toml | 13 ------ providers/aihubmix/models/hy3-preview.toml | 25 +++------- providers/aihubmix/models/hy3.toml | 14 ------ providers/aihubmix/models/hy4-preview.toml | 20 -------- .../aihubmix/models/kimi-for-coding-free.toml | 22 --------- providers/aihubmix/models/kimi-k2-0711.toml | 21 --------- .../aihubmix/models/kimi-k2-thinking.toml | 8 ---- .../models/kimi-k2-turbo-preview.toml | 22 --------- providers/aihubmix/models/kimi-k2.5.toml | 22 ++++++--- providers/aihubmix/models/kimi-k2.6.toml | 24 +++++++--- .../models/kimi-k2.7-code-highspeed.toml | 9 ++-- providers/aihubmix/models/kimi-k2.7-code.toml | 9 ++-- providers/aihubmix/models/kimi-k3.toml | 13 ++++-- .../aihubmix/models/laguna-s-2.1-free.toml | 23 ---------- .../models/llama-3.3-70b-instruct.toml | 10 ---- providers/aihubmix/models/longcat-2.0.toml | 9 ---- .../aihubmix/models/mercury-2.5-preview.toml | 25 ---------- .../aihubmix/models/mimo-v2-flash-free.toml | 10 ---- providers/aihubmix/models/mimo-v2-flash.toml | 18 -------- providers/aihubmix/models/mimo-v2-omni.toml | 14 ------ providers/aihubmix/models/mimo-v2-pro.toml | 17 ------- providers/aihubmix/models/mimo-v2.5-pro.toml | 10 ---- providers/aihubmix/models/mimo-v2.5.toml | 10 ---- providers/aihubmix/models/minimax-m2.1.toml | 13 ------ .../models/minimax-m2.5-highspeed.toml | 13 ------ providers/aihubmix/models/minimax-m2.5.toml | 13 ------ .../aihubmix/models/minimax-m2.7-free.toml | 13 ------ providers/aihubmix/models/minimax-m2.7.toml | 31 ++++++++----- providers/aihubmix/models/minimax-m2.toml | 13 ------ providers/aihubmix/models/minimax-m3.toml | 16 ------- .../aihubmix/models/mistral-large-3.toml | 21 --------- .../models/mm-minimax-m2.7-highspeed.toml | 13 ------ providers/aihubmix/models/muse-spark-1.1.toml | 15 ------ providers/aihubmix/models/muse-spark-1.2.toml | 12 ----- providers/aihubmix/models/muse-spark-1.3.toml | 16 ------- .../models/nemotron-3-nano-30b-a3b-free.toml | 15 ------ ...on-3-nano-omni-30b-a3b-reasoning-free.toml | 14 ------ .../nemotron-3-super-120b-a12b-free.toml | 11 ----- .../nemotron-3-ultra-550b-a55b-free.toml | 14 ------ .../nemotron-3.5-content-safety-free.toml | 11 ----- .../models/nemotron-3.5-lightning-free.toml | 11 ----- .../models/nemotron-nano-12b-v2-vl-free.toml | 11 ----- .../models/nemotron-nano-9b-v2-free.toml | 11 ----- .../aihubmix/models/north-mini-code-free.toml | 27 ----------- .../nvidia-nemotron-3-super-120b-a12b.toml | 11 ----- providers/aihubmix/models/o1-preview.toml | 11 ----- providers/aihubmix/models/o1-pro.toml | 12 ----- providers/aihubmix/models/o1.toml | 12 ----- providers/aihubmix/models/o3-mini.toml | 12 ----- providers/aihubmix/models/o3-pro.toml | 7 --- providers/aihubmix/models/o3.toml | 13 ------ providers/aihubmix/models/o4-mini.toml | 10 ---- providers/aihubmix/models/ox-alpha.toml | 18 -------- .../aihubmix/models/qwen-plus-latest.toml | 23 ---------- .../aihubmix/models/qwen-turbo-latest.toml | 8 ---- providers/aihubmix/models/qwen-turbo.toml | 8 ---- .../models/qwen3-235b-a22b-instruct-2507.toml | 10 ---- .../aihubmix/models/qwen3-235b-a22b.toml | 11 ----- .../models/qwen3-coder-30b-a3b-instruct.toml | 19 -------- .../qwen3-coder-480b-a35b-instruct.toml | 19 -------- .../aihubmix/models/qwen3-coder-flash.toml | 21 --------- .../aihubmix/models/qwen3-coder-next.toml | 15 ------ .../aihubmix/models/qwen3-coder-plus.toml | 22 --------- .../aihubmix/models/qwen3-max-2026-01-23.toml | 43 ----------------- .../aihubmix/models/qwen3-max-preview.toml | 23 ---------- providers/aihubmix/models/qwen3-max.toml | 29 ------------ .../models/qwen3-next-80b-a3b-instruct.toml | 13 ------ .../models/qwen3-next-80b-a3b-thinking.toml | 14 ------ .../models/qwen3-vl-235b-a22b-instruct.toml | 15 ------ .../models/qwen3-vl-235b-a22b-thinking.toml | 14 ------ .../models/qwen3-vl-30b-a3b-instruct.toml | 27 ----------- .../models/qwen3-vl-30b-a3b-thinking.toml | 27 ----------- providers/aihubmix/models/qwen3-vl-flash.toml | 40 ---------------- providers/aihubmix/models/qwen3-vl-plus.toml | 27 ----------- .../aihubmix/models/qwen3.5-122b-a10b.toml | 23 ---------- providers/aihubmix/models/qwen3.5-27b.toml | 23 ---------- .../aihubmix/models/qwen3.5-35b-a3b.toml | 23 ---------- .../aihubmix/models/qwen3.5-397b-a17b.toml | 23 ---------- providers/aihubmix/models/qwen3.5-flash.toml | 31 ------------- providers/aihubmix/models/qwen3.5-plus.toml | 33 ------------- providers/aihubmix/models/qwen3.6-27b.toml | 11 ----- .../aihubmix/models/qwen3.6-35b-a3b.toml | 18 -------- providers/aihubmix/models/qwen3.6-flash.toml | 41 ++++++++++------- .../aihubmix/models/qwen3.6-max-preview.toml | 36 ++++++++++----- .../models/qwen3.6-plus-preview-free.toml | 23 ---------- providers/aihubmix/models/qwen3.6-plus.toml | 40 +++++++++------- providers/aihubmix/models/qwen3.7-flash.toml | 33 ++++--------- providers/aihubmix/models/qwen3.7-max.toml | 18 +++----- providers/aihubmix/models/qwen3.7-plus.toml | 25 +++------- .../aihubmix/models/qwen3.8-2.4t-a95b.toml | 19 ++++---- providers/aihubmix/models/qwen3.8-flash.toml | 17 ------- .../models/qwen3.8-max-2026-09-02.toml | 33 ------------- .../aihubmix/models/qwen3.8-max-preview.toml | 18 -------- providers/aihubmix/models/qwen3.8-max.toml | 22 ++++----- providers/aihubmix/models/solar-pro4.toml | 8 ---- providers/aihubmix/models/step-3.5-flash.toml | 9 ---- providers/aihubmix/models/step-3.7-flash.toml | 14 ------ .../models/xiaomi-mimo-v2-omni-free.toml | 13 ------ .../models/xiaomi-mimo-v2-pro-free.toml | 10 ---- .../models/xiaomi-mimo-v2.5-free.toml | 7 +-- .../models/xiaomi-mimo-v2.5-pro-free.toml | 7 +-- .../aihubmix/models/zai-glm-5-turbo.toml | 13 ------ 300 files changed, 975 insertions(+), 4127 deletions(-) delete mode 100644 providers/aihubmix/models/DeepSeek-V3.1-Terminus.toml delete mode 100644 providers/aihubmix/models/DeepSeek-V3.1-Think.toml delete mode 100644 providers/aihubmix/models/DeepSeek-V3.toml delete mode 100644 providers/aihubmix/models/Qwen/QwQ-32B.toml delete mode 100644 providers/aihubmix/models/agnes-2.5-pro-alpha.toml delete mode 100644 providers/aihubmix/models/agnes-3.0-flash.toml delete mode 100644 providers/aihubmix/models/aihubmix-command-r-08-2024.toml delete mode 100644 providers/aihubmix/models/aihubmix-command-r-plus-08-2024.toml delete mode 100644 providers/aihubmix/models/bai-qwen3-vl-235b-a22b-instruct.toml delete mode 100644 providers/aihubmix/models/cc-MiniMax-M2.toml delete mode 100644 providers/aihubmix/models/cc-deepseek-v3.1.toml delete mode 100644 providers/aihubmix/models/cc-glm-5-turbo.toml delete mode 100644 providers/aihubmix/models/cc-glm-5.1.toml delete mode 100644 providers/aihubmix/models/cc-glm-5.toml delete mode 100644 providers/aihubmix/models/cc-minimax-m2.1.toml delete mode 100644 providers/aihubmix/models/cc-minimax-m2.5-highspeed.toml delete mode 100644 providers/aihubmix/models/cc-minimax-m2.5.toml delete mode 100644 providers/aihubmix/models/cc-minimax-m2.7-highspeed.toml delete mode 100644 providers/aihubmix/models/cc-minimax-m2.7.toml delete mode 100644 providers/aihubmix/models/cc-minimax-m3.toml delete mode 100644 providers/aihubmix/models/claude-3-haiku-20240307.toml delete mode 100644 providers/aihubmix/models/claude-fable-5-1.toml delete mode 100644 providers/aihubmix/models/claude-haiku-4-5.toml delete mode 100644 providers/aihubmix/models/claude-opus-4-1.toml delete mode 100644 providers/aihubmix/models/claude-opus-4-5-think.toml delete mode 100644 providers/aihubmix/models/claude-opus-4-5.toml delete mode 100644 providers/aihubmix/models/claude-sonnet-4-5-think.toml delete mode 100644 providers/aihubmix/models/claude-sonnet-4-5.toml delete mode 100644 providers/aihubmix/models/cloudflare-glm-5.2.toml delete mode 100644 providers/aihubmix/models/coding-glm-4.6-free.toml delete mode 100644 providers/aihubmix/models/coding-glm-4.6.toml delete mode 100644 providers/aihubmix/models/coding-glm-4.7-free.toml delete mode 100644 providers/aihubmix/models/coding-glm-4.7.toml delete mode 100644 providers/aihubmix/models/coding-glm-5-free.toml delete mode 100644 providers/aihubmix/models/coding-glm-5-turbo-free.toml delete mode 100644 providers/aihubmix/models/coding-glm-5-turbo.toml delete mode 100644 providers/aihubmix/models/coding-glm-5.2-free.toml delete mode 100644 providers/aihubmix/models/coding-glm-5.2.toml delete mode 100644 providers/aihubmix/models/coding-glm-5.3-free.toml delete mode 100644 providers/aihubmix/models/coding-glm-5.3.toml delete mode 100644 providers/aihubmix/models/coding-glm-5.toml delete mode 100644 providers/aihubmix/models/coding-kimi-k3-free.toml delete mode 100644 providers/aihubmix/models/coding-kimi-k3.toml delete mode 100644 providers/aihubmix/models/coding-minimax-m2-free.toml delete mode 100644 providers/aihubmix/models/coding-minimax-m2.1-free.toml delete mode 100644 providers/aihubmix/models/coding-minimax-m2.1.toml delete mode 100644 providers/aihubmix/models/coding-minimax-m2.5-free.toml delete mode 100644 providers/aihubmix/models/coding-minimax-m2.5-highspeed.toml delete mode 100644 providers/aihubmix/models/coding-minimax-m2.5.toml delete mode 100644 providers/aihubmix/models/coding-minimax-m2.toml delete mode 100644 providers/aihubmix/models/coding-minimax-m3-free.toml delete mode 100644 providers/aihubmix/models/coding-minimax-m3.toml delete mode 100644 providers/aihubmix/models/cohere-command-a.toml delete mode 100644 providers/aihubmix/models/command-a-03-2025.toml delete mode 100644 providers/aihubmix/models/command-a-plus-05-2026.toml delete mode 100644 providers/aihubmix/models/command-r-08-2024.toml delete mode 100644 providers/aihubmix/models/command-r-plus-08-2024.toml delete mode 100644 providers/aihubmix/models/deepinfra-gemma-4-26b-a4b-it.toml delete mode 100644 providers/aihubmix/models/deepseek-v3.2-think.toml delete mode 100644 providers/aihubmix/models/deepseek-v3.2.toml delete mode 100644 providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml delete mode 100644 providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml delete mode 100644 providers/aihubmix/models/deepseek-v4-flash.toml delete mode 100644 providers/aihubmix/models/deepseek-v4-pro.toml delete mode 100644 providers/aihubmix/models/deepseek-v4.1-flash.toml delete mode 100644 providers/aihubmix/models/doubao-seed-1-8.toml delete mode 100644 providers/aihubmix/models/doubao-seed-2-0-lite-260215.toml delete mode 100644 providers/aihubmix/models/ernie-5.1.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-image.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-lite-nothink.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-lite.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-nothink.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-flash-search.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml delete mode 100644 providers/aihubmix/models/gemini-2.5-pro-search.toml delete mode 100644 providers/aihubmix/models/gemini-3-flash-preview-free.toml delete mode 100644 providers/aihubmix/models/gemini-3-flash-preview-search.toml delete mode 100644 providers/aihubmix/models/gemini-3-pro-image-preview.toml delete mode 100644 providers/aihubmix/models/gemini-3-pro-image.toml delete mode 100644 providers/aihubmix/models/gemini-3.1-flash-image-preview.toml delete mode 100644 providers/aihubmix/models/gemini-3.1-flash-image.toml delete mode 100644 providers/aihubmix/models/gemini-3.1-flash-lite-image.toml delete mode 100644 providers/aihubmix/models/gemini-3.1-flash-lite-nothink.toml delete mode 100644 providers/aihubmix/models/gemini-3.1-pro-preview-search.toml delete mode 100644 providers/aihubmix/models/gemini-3.5-flash-lite-free.toml delete mode 100644 providers/aihubmix/models/gemini-3.5-flash-lite.toml delete mode 100644 providers/aihubmix/models/gemini-3.6-flash-free.toml delete mode 100644 providers/aihubmix/models/gemini-3.6-flash.toml delete mode 100644 providers/aihubmix/models/gemini-3.7-flash-free.toml delete mode 100644 providers/aihubmix/models/gemini-3.8-flash-free.toml delete mode 100644 providers/aihubmix/models/gemini-3.8-flash.toml delete mode 100644 providers/aihubmix/models/gemma-4-26b-a4b-it-free.toml delete mode 100644 providers/aihubmix/models/gemma-4-26b-a4b-it.toml delete mode 100644 providers/aihubmix/models/gemma-4-31b-it-free.toml delete mode 100644 providers/aihubmix/models/gemma-4-31b-it.toml delete mode 100644 providers/aihubmix/models/glm-4.5v.toml delete mode 100644 providers/aihubmix/models/glm-4.6.toml delete mode 100644 providers/aihubmix/models/glm-4.6v.toml delete mode 100644 providers/aihubmix/models/glm-4.7-flash-free.toml delete mode 100644 providers/aihubmix/models/glm-4.7.toml delete mode 100644 providers/aihubmix/models/glm-5-turbo.toml delete mode 100644 providers/aihubmix/models/glm-5.1.toml delete mode 100644 providers/aihubmix/models/glm-5.2-fast-preview.toml delete mode 100644 providers/aihubmix/models/glm-5.toml delete mode 100644 providers/aihubmix/models/gpt-4.1-free.toml delete mode 100644 providers/aihubmix/models/gpt-4.1-mini-free.toml delete mode 100644 providers/aihubmix/models/gpt-4.1-mini.toml delete mode 100644 providers/aihubmix/models/gpt-4.1-nano-free.toml delete mode 100644 providers/aihubmix/models/gpt-4.1-nano.toml delete mode 100644 providers/aihubmix/models/gpt-4.1.toml delete mode 100644 providers/aihubmix/models/gpt-4o-2024-11-20.toml delete mode 100644 providers/aihubmix/models/gpt-4o-free.toml delete mode 100644 providers/aihubmix/models/gpt-4o-mini.toml delete mode 100644 providers/aihubmix/models/gpt-4o.toml delete mode 100644 providers/aihubmix/models/gpt-5-chat-latest.toml delete mode 100644 providers/aihubmix/models/gpt-5-codex.toml delete mode 100644 providers/aihubmix/models/gpt-5-mini.toml delete mode 100644 providers/aihubmix/models/gpt-5-nano.toml delete mode 100644 providers/aihubmix/models/gpt-5-pro.toml delete mode 100644 providers/aihubmix/models/gpt-5.1-chat-latest.toml delete mode 100644 providers/aihubmix/models/gpt-5.1-codex-max.toml delete mode 100644 providers/aihubmix/models/gpt-5.2-chat-latest.toml delete mode 100644 providers/aihubmix/models/gpt-5.2-high.toml delete mode 100644 providers/aihubmix/models/gpt-5.2-low.toml delete mode 100644 providers/aihubmix/models/gpt-5.2-pro.toml delete mode 100644 providers/aihubmix/models/gpt-5.3-chat-latest.toml delete mode 100644 providers/aihubmix/models/gpt-5.4-nano.toml delete mode 100644 providers/aihubmix/models/gpt-5.4-pro.toml delete mode 100644 providers/aihubmix/models/gpt-5.5-free.toml delete mode 100644 providers/aihubmix/models/gpt-5.5-pro.toml delete mode 100644 providers/aihubmix/models/gpt-5.6-sol-disc.toml delete mode 100644 providers/aihubmix/models/gpt-5.toml delete mode 100644 providers/aihubmix/models/gpt-6-astra.toml delete mode 100644 providers/aihubmix/models/gpt-chat-latest.toml delete mode 100644 providers/aihubmix/models/gpt-image-2-free.toml delete mode 100644 providers/aihubmix/models/gpt-oss-120b.toml delete mode 100644 providers/aihubmix/models/gpt-oss-20b-free.toml delete mode 100644 providers/aihubmix/models/gpt-oss-20b.toml delete mode 100644 providers/aihubmix/models/grok-4-1-fast-non-reasoning.toml delete mode 100644 providers/aihubmix/models/grok-4-1-fast-reasoning.toml delete mode 100644 providers/aihubmix/models/grok-4-fast-non-reasoning.toml delete mode 100644 providers/aihubmix/models/grok-4-fast-reasoning.toml delete mode 100644 providers/aihubmix/models/grok-4.toml delete mode 100644 providers/aihubmix/models/grok-code-fast-1.toml delete mode 100644 providers/aihubmix/models/hy3-free.toml delete mode 100644 providers/aihubmix/models/hy3.toml delete mode 100644 providers/aihubmix/models/hy4-preview.toml delete mode 100644 providers/aihubmix/models/kimi-for-coding-free.toml delete mode 100644 providers/aihubmix/models/kimi-k2-0711.toml delete mode 100644 providers/aihubmix/models/kimi-k2-thinking.toml delete mode 100644 providers/aihubmix/models/kimi-k2-turbo-preview.toml delete mode 100644 providers/aihubmix/models/laguna-s-2.1-free.toml delete mode 100644 providers/aihubmix/models/llama-3.3-70b-instruct.toml delete mode 100644 providers/aihubmix/models/longcat-2.0.toml delete mode 100644 providers/aihubmix/models/mercury-2.5-preview.toml delete mode 100644 providers/aihubmix/models/mimo-v2-flash-free.toml delete mode 100644 providers/aihubmix/models/mimo-v2-flash.toml delete mode 100644 providers/aihubmix/models/mimo-v2-omni.toml delete mode 100644 providers/aihubmix/models/mimo-v2-pro.toml delete mode 100644 providers/aihubmix/models/mimo-v2.5-pro.toml delete mode 100644 providers/aihubmix/models/mimo-v2.5.toml delete mode 100644 providers/aihubmix/models/minimax-m2.1.toml delete mode 100644 providers/aihubmix/models/minimax-m2.5-highspeed.toml delete mode 100644 providers/aihubmix/models/minimax-m2.5.toml delete mode 100644 providers/aihubmix/models/minimax-m2.7-free.toml delete mode 100644 providers/aihubmix/models/minimax-m2.toml delete mode 100644 providers/aihubmix/models/minimax-m3.toml delete mode 100644 providers/aihubmix/models/mistral-large-3.toml delete mode 100644 providers/aihubmix/models/mm-minimax-m2.7-highspeed.toml delete mode 100644 providers/aihubmix/models/muse-spark-1.1.toml delete mode 100644 providers/aihubmix/models/muse-spark-1.2.toml delete mode 100644 providers/aihubmix/models/muse-spark-1.3.toml delete mode 100644 providers/aihubmix/models/nemotron-3-nano-30b-a3b-free.toml delete mode 100644 providers/aihubmix/models/nemotron-3-nano-omni-30b-a3b-reasoning-free.toml delete mode 100644 providers/aihubmix/models/nemotron-3-super-120b-a12b-free.toml delete mode 100644 providers/aihubmix/models/nemotron-3-ultra-550b-a55b-free.toml delete mode 100644 providers/aihubmix/models/nemotron-3.5-content-safety-free.toml delete mode 100644 providers/aihubmix/models/nemotron-3.5-lightning-free.toml delete mode 100644 providers/aihubmix/models/nemotron-nano-12b-v2-vl-free.toml delete mode 100644 providers/aihubmix/models/nemotron-nano-9b-v2-free.toml delete mode 100644 providers/aihubmix/models/north-mini-code-free.toml delete mode 100644 providers/aihubmix/models/nvidia-nemotron-3-super-120b-a12b.toml delete mode 100644 providers/aihubmix/models/o1-preview.toml delete mode 100644 providers/aihubmix/models/o1-pro.toml delete mode 100644 providers/aihubmix/models/o1.toml delete mode 100644 providers/aihubmix/models/o3-mini.toml delete mode 100644 providers/aihubmix/models/o3-pro.toml delete mode 100644 providers/aihubmix/models/o3.toml delete mode 100644 providers/aihubmix/models/o4-mini.toml delete mode 100644 providers/aihubmix/models/ox-alpha.toml delete mode 100644 providers/aihubmix/models/qwen-plus-latest.toml delete mode 100644 providers/aihubmix/models/qwen-turbo-latest.toml delete mode 100644 providers/aihubmix/models/qwen-turbo.toml delete mode 100644 providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml delete mode 100644 providers/aihubmix/models/qwen3-235b-a22b.toml delete mode 100644 providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml delete mode 100644 providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml delete mode 100644 providers/aihubmix/models/qwen3-coder-flash.toml delete mode 100644 providers/aihubmix/models/qwen3-coder-next.toml delete mode 100644 providers/aihubmix/models/qwen3-coder-plus.toml delete mode 100644 providers/aihubmix/models/qwen3-max-2026-01-23.toml delete mode 100644 providers/aihubmix/models/qwen3-max-preview.toml delete mode 100644 providers/aihubmix/models/qwen3-max.toml delete mode 100644 providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml delete mode 100644 providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml delete mode 100644 providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml delete mode 100644 providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml delete mode 100644 providers/aihubmix/models/qwen3-vl-30b-a3b-instruct.toml delete mode 100644 providers/aihubmix/models/qwen3-vl-30b-a3b-thinking.toml delete mode 100644 providers/aihubmix/models/qwen3-vl-flash.toml delete mode 100644 providers/aihubmix/models/qwen3-vl-plus.toml delete mode 100644 providers/aihubmix/models/qwen3.5-122b-a10b.toml delete mode 100644 providers/aihubmix/models/qwen3.5-27b.toml delete mode 100644 providers/aihubmix/models/qwen3.5-35b-a3b.toml delete mode 100644 providers/aihubmix/models/qwen3.5-397b-a17b.toml delete mode 100644 providers/aihubmix/models/qwen3.5-flash.toml delete mode 100644 providers/aihubmix/models/qwen3.5-plus.toml delete mode 100644 providers/aihubmix/models/qwen3.6-27b.toml delete mode 100644 providers/aihubmix/models/qwen3.6-35b-a3b.toml delete mode 100644 providers/aihubmix/models/qwen3.6-plus-preview-free.toml delete mode 100644 providers/aihubmix/models/qwen3.8-flash.toml delete mode 100644 providers/aihubmix/models/qwen3.8-max-2026-09-02.toml delete mode 100644 providers/aihubmix/models/qwen3.8-max-preview.toml delete mode 100644 providers/aihubmix/models/solar-pro4.toml delete mode 100644 providers/aihubmix/models/step-3.5-flash.toml delete mode 100644 providers/aihubmix/models/step-3.7-flash.toml delete mode 100644 providers/aihubmix/models/xiaomi-mimo-v2-omni-free.toml delete mode 100644 providers/aihubmix/models/xiaomi-mimo-v2-pro-free.toml delete mode 100644 providers/aihubmix/models/zai-glm-5-turbo.toml diff --git a/providers/aihubmix/models/DeepSeek-V3.1-Terminus.toml b/providers/aihubmix/models/DeepSeek-V3.1-Terminus.toml deleted file mode 100644 index c66bd7430dc..00000000000 --- a/providers/aihubmix/models/DeepSeek-V3.1-Terminus.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "DeepSeek V3.1 Terminus" -description = "DeepSeek-V3.1 non-thinking mode has now been updated to the DeepSeek-V3.1-Terminus version." -release_date = "2025-09-22" -last_updated = "2025-09-22" -attachment = false -reasoning = true -tool_call = true -structured_output = true -open_weights = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.56 -output = 1.68 - -[limit] -context = 160_000 -output = 32_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/DeepSeek-V3.1-Think.toml b/providers/aihubmix/models/DeepSeek-V3.1-Think.toml deleted file mode 100644 index 2f19ee69fbc..00000000000 --- a/providers/aihubmix/models/DeepSeek-V3.1-Think.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "deepseek/deepseek-v3.1" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.56 -output = 1.68 - -[limit] -context = 128_000 -output = 32_000 diff --git a/providers/aihubmix/models/DeepSeek-V3.toml b/providers/aihubmix/models/DeepSeek-V3.toml deleted file mode 100644 index 6531beb0ddb..00000000000 --- a/providers/aihubmix/models/DeepSeek-V3.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "deepseek/deepseek-v3" -structured_output = true - -[cost] -input = 0.272 -output = 1.088 - -[limit] -context = 163_840 diff --git a/providers/aihubmix/models/Qwen/QwQ-32B.toml b/providers/aihubmix/models/Qwen/QwQ-32B.toml deleted file mode 100644 index 7cf8bfe98c3..00000000000 --- a/providers/aihubmix/models/Qwen/QwQ-32B.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "alibaba/qwq-32b" -reasoning = false -structured_output = true - -[cost] -input = 0.14 -output = 0.56 diff --git a/providers/aihubmix/models/agnes-2.5-pro-alpha.toml b/providers/aihubmix/models/agnes-2.5-pro-alpha.toml deleted file mode 100644 index ce387939e1c..00000000000 --- a/providers/aihubmix/models/agnes-2.5-pro-alpha.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Agnes 2.5 Pro Alpha" -description = "Agnes 2.5 Pro Alpha is Agnes AI’s paid inference model, suitable for advanced coding, scientific reasoning, long-context analysis, agent workflows, and multimodal understanding. The model is accessed via an OpenAI-compatible Chat Completions API." -release_date = "2026-07-24" -last_updated = "2026-07-24" -attachment = true -reasoning = true -tool_call = false -open_weights = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.45 -output = 0.9 -cache_read = 0.00378 - -[limit] -context = 1_000_000 -output = 65_536 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/agnes-3.0-flash.toml b/providers/aihubmix/models/agnes-3.0-flash.toml deleted file mode 100644 index d1e8248ef27..00000000000 --- a/providers/aihubmix/models/agnes-3.0-flash.toml +++ /dev/null @@ -1,26 +0,0 @@ -name = "Agnes 3.0 Flash" -description = "Agnes 3.0 Flash is designed for real-world agent tasks and development workflows, covering the full execution chain from task understanding and planning to tool invocation and final delivery. The model focuses on improving stability, instruction-following, factual grounding, and output completeness in complex tasks, helping developers build more reliable agent applications." -release_date = "2026-09-09" -last_updated = "2026-09-09" -attachment = true -reasoning = true -tool_call = false -open_weights = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.03 -output = 0.15 - -[limit] -context = 512_000 -output = 65_500 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/aihubmix-command-r-08-2024.toml b/providers/aihubmix/models/aihubmix-command-r-08-2024.toml deleted file mode 100644 index 3eb86afe515..00000000000 --- a/providers/aihubmix/models/aihubmix-command-r-08-2024.toml +++ /dev/null @@ -1,6 +0,0 @@ -base_model = "cohere/command-r-08-2024" -tool_call = false - -[cost] -input = 0.2 -output = 0.8 diff --git a/providers/aihubmix/models/aihubmix-command-r-plus-08-2024.toml b/providers/aihubmix/models/aihubmix-command-r-plus-08-2024.toml deleted file mode 100644 index e0e3ee2ad05..00000000000 --- a/providers/aihubmix/models/aihubmix-command-r-plus-08-2024.toml +++ /dev/null @@ -1,6 +0,0 @@ -base_model = "cohere/command-r-plus-08-2024" -tool_call = false - -[cost] -input = 2.8 -output = 11.2 diff --git a/providers/aihubmix/models/bai-qwen3-vl-235b-a22b-instruct.toml b/providers/aihubmix/models/bai-qwen3-vl-235b-a22b-instruct.toml deleted file mode 100644 index 776ec2d52ef..00000000000 --- a/providers/aihubmix/models/bai-qwen3-vl-235b-a22b-instruct.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "alibaba/qwen3-vl-235b-a22b-instruct" -attachment = false -tool_call = false - -[cost] -input = 0.274 -output = 1.096 - -[modalities] -input = ["text"] diff --git a/providers/aihubmix/models/cc-MiniMax-M2.toml b/providers/aihubmix/models/cc-MiniMax-M2.toml deleted file mode 100644 index a4debfe801b..00000000000 --- a/providers/aihubmix/models/cc-MiniMax-M2.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "minimax/MiniMax-M2" -reasoning = false -structured_output = true - -[cost] -input = 0.1 -output = 0.1 diff --git a/providers/aihubmix/models/cc-deepseek-v3.1.toml b/providers/aihubmix/models/cc-deepseek-v3.1.toml deleted file mode 100644 index 488e4952a8a..00000000000 --- a/providers/aihubmix/models/cc-deepseek-v3.1.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "deepseek/deepseek-v3.1" -reasoning = false -structured_output = true - -[cost] -input = 0.56 -output = 1.68 diff --git a/providers/aihubmix/models/cc-glm-5-turbo.toml b/providers/aihubmix/models/cc-glm-5-turbo.toml deleted file mode 100644 index 50c916e5495..00000000000 --- a/providers/aihubmix/models/cc-glm-5-turbo.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "zhipuai/glm-5-turbo" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.06 -output = 0.22 - -[limit] -context = 204_800 diff --git a/providers/aihubmix/models/cc-glm-5.1.toml b/providers/aihubmix/models/cc-glm-5.1.toml deleted file mode 100644 index ac1e901ddf4..00000000000 --- a/providers/aihubmix/models/cc-glm-5.1.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "zhipuai/glm-5.1" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.06 -output = 0.22 diff --git a/providers/aihubmix/models/cc-glm-5.toml b/providers/aihubmix/models/cc-glm-5.toml deleted file mode 100644 index 43a367a1581..00000000000 --- a/providers/aihubmix/models/cc-glm-5.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "zhipuai/glm-5" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.06 -output = 0.22 - -[limit] -context = 200_000 diff --git a/providers/aihubmix/models/cc-minimax-m2.1.toml b/providers/aihubmix/models/cc-minimax-m2.1.toml deleted file mode 100644 index 8314c6dfb6a..00000000000 --- a/providers/aihubmix/models/cc-minimax-m2.1.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.1" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.1 diff --git a/providers/aihubmix/models/cc-minimax-m2.5-highspeed.toml b/providers/aihubmix/models/cc-minimax-m2.5-highspeed.toml deleted file mode 100644 index 84c99647716..00000000000 --- a/providers/aihubmix/models/cc-minimax-m2.5-highspeed.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.5-highspeed" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.1 diff --git a/providers/aihubmix/models/cc-minimax-m2.5.toml b/providers/aihubmix/models/cc-minimax-m2.5.toml deleted file mode 100644 index 139bb3236d5..00000000000 --- a/providers/aihubmix/models/cc-minimax-m2.5.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.5" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.1 diff --git a/providers/aihubmix/models/cc-minimax-m2.7-highspeed.toml b/providers/aihubmix/models/cc-minimax-m2.7-highspeed.toml deleted file mode 100644 index 1e66879bd49..00000000000 --- a/providers/aihubmix/models/cc-minimax-m2.7-highspeed.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.7-highspeed" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.1 diff --git a/providers/aihubmix/models/cc-minimax-m2.7.toml b/providers/aihubmix/models/cc-minimax-m2.7.toml deleted file mode 100644 index 79e4e4b2888..00000000000 --- a/providers/aihubmix/models/cc-minimax-m2.7.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.7" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.1 diff --git a/providers/aihubmix/models/cc-minimax-m3.toml b/providers/aihubmix/models/cc-minimax-m3.toml deleted file mode 100644 index d090ab46f95..00000000000 --- a/providers/aihubmix/models/cc-minimax-m3.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "minimax/MiniMax-M3" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.1 - -[limit] -context = 1_000_000 -output = 524_288 diff --git a/providers/aihubmix/models/claude-3-haiku-20240307.toml b/providers/aihubmix/models/claude-3-haiku-20240307.toml deleted file mode 100644 index d404c3f9957..00000000000 --- a/providers/aihubmix/models/claude-3-haiku-20240307.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "anthropic/claude-3-haiku-20240307" -tool_call = false - -[cost] -input = 0.275 -output = 1.375 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/claude-fable-5-1.toml b/providers/aihubmix/models/claude-fable-5-1.toml deleted file mode 100644 index 4896a255a1f..00000000000 --- a/providers/aihubmix/models/claude-fable-5-1.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "anthropic/claude-fable-5-1" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - -[cost] -input = 11 -output = 55 -cache_read = 0.275 -cache_write = 13.75 diff --git a/providers/aihubmix/models/claude-fable-5.toml b/providers/aihubmix/models/claude-fable-5.toml index c7e1152cec4..afa2221178b 100644 --- a/providers/aihubmix/models/claude-fable-5.toml +++ b/providers/aihubmix/models/claude-fable-5.toml @@ -1,17 +1,18 @@ base_model = "anthropic/claude-fable-5" structured_output = true - +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] interleaved = true -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - [cost] input = 11 output = 55 cache_read = 1.1 cache_write = 13.75 + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/claude-haiku-4-5.toml b/providers/aihubmix/models/claude-haiku-4-5.toml deleted file mode 100644 index 6e5b90ce32e..00000000000 --- a/providers/aihubmix/models/claude-haiku-4-5.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "anthropic/claude-haiku-4-5" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.1 -output = 5.5 -cache_read = 0.11 -cache_write = 1.375 diff --git a/providers/aihubmix/models/claude-opus-4-1.toml b/providers/aihubmix/models/claude-opus-4-1.toml deleted file mode 100644 index 8b33d3e0342..00000000000 --- a/providers/aihubmix/models/claude-opus-4-1.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "anthropic/claude-opus-4-1" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 16.5 -output = 82.5 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/claude-opus-4-5-think.toml b/providers/aihubmix/models/claude-opus-4-5-think.toml deleted file mode 100644 index c9dac00196a..00000000000 --- a/providers/aihubmix/models/claude-opus-4-5-think.toml +++ /dev/null @@ -1,18 +0,0 @@ -base_model = "anthropic/claude-opus-4-5" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 5 -output = 25 -cache_read = 0.5 -cache_write = 6.25 diff --git a/providers/aihubmix/models/claude-opus-4-5.toml b/providers/aihubmix/models/claude-opus-4-5.toml deleted file mode 100644 index c9dac00196a..00000000000 --- a/providers/aihubmix/models/claude-opus-4-5.toml +++ /dev/null @@ -1,18 +0,0 @@ -base_model = "anthropic/claude-opus-4-5" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 5 -output = 25 -cache_read = 0.5 -cache_write = 6.25 diff --git a/providers/aihubmix/models/claude-opus-4-6-think.toml b/providers/aihubmix/models/claude-opus-4-6-think.toml index 5226b66dec7..58007ca6a42 100644 --- a/providers/aihubmix/models/claude-opus-4-6-think.toml +++ b/providers/aihubmix/models/claude-opus-4-6-think.toml @@ -1,21 +1,20 @@ -base_model = "anthropic/claude-opus-4-6" name = "Claude Opus 4.6 Thinking" description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" +family = "claude-opus" +release_date = "2026-02-05" +last_updated = "2026-03-13" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] +temperature = true +tool_call = true structured_output = true +knowledge = "2025-05-31" +open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "max"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 5 output = 25 @@ -23,8 +22,16 @@ cache_read = 0.5 cache_write = 6.25 [[cost.tiers]] -tier = { type = "context", size = 200_000 } +tier = { size = 200_000 } input = 10 output = 37.5 -cache_read = 1 +cache_read = 1.0 cache_write = 12.5 + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-4-6.toml b/providers/aihubmix/models/claude-opus-4-6.toml index 44952c1e2bc..57b40db1786 100644 --- a/providers/aihubmix/models/claude-opus-4-6.toml +++ b/providers/aihubmix/models/claude-opus-4-6.toml @@ -1,19 +1,20 @@ -base_model = "anthropic/claude-opus-4-6" +name = "Claude Opus 4.6" description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" +family = "claude-opus" +release_date = "2026-02-05" +last_updated = "2026-03-13" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] +# Native Messages prefers $.thinking.type = "adaptive" with $.output_config.effort = "low"|"medium"|"high"|"max"; enabled budget_tokens >= 1024 is deprecated and must be < $.max_tokens. https://docs.aihubmix.com/cn/api/Claude-Native (accessed 2026-06-25) +temperature = true +tool_call = true structured_output = true +knowledge = "2025-05-31" +open_weights = false interleaved = true -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "max"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 5 output = 25 @@ -21,8 +22,16 @@ cache_read = 0.5 cache_write = 6.25 [[cost.tiers]] -tier = { type = "context", size = 200_000 } +tier = { size = 200_000 } input = 10 output = 37.5 -cache_read = 1 +cache_read = 1.0 cache_write = 12.5 + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-4-7-think.toml b/providers/aihubmix/models/claude-opus-4-7-think.toml index d14e3b269be..e9da40d2fae 100644 --- a/providers/aihubmix/models/claude-opus-4-7-think.toml +++ b/providers/aihubmix/models/claude-opus-4-7-think.toml @@ -1,18 +1,20 @@ -base_model = "anthropic/claude-opus-4-7" name = "Claude Opus 4.7 Thinking" description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" +family = "claude-opus" +release_date = "2026-04-16" +last_updated = "2026-04-16" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +temperature = false +tool_call = true structured_output = true +knowledge = "2026-01-31" +open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - [cost] input = 5 output = 25 @@ -20,8 +22,16 @@ cache_read = 0.5 cache_write = 6.25 [[cost.tiers]] -tier = { type = "context", size = 200_000 } +tier = { size = 200_000 } input = 10 output = 37.5 -cache_read = 1 +cache_read = 1.0 cache_write = 12.5 + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-4-7.toml b/providers/aihubmix/models/claude-opus-4-7.toml index f3267b25d12..0fb0b1c89f3 100644 --- a/providers/aihubmix/models/claude-opus-4-7.toml +++ b/providers/aihubmix/models/claude-opus-4-7.toml @@ -1,16 +1,20 @@ -base_model = "anthropic/claude-opus-4-7" +name = "Claude Opus 4.7" description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" +family = "claude-opus" +release_date = "2026-04-16" +last_updated = "2026-04-16" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +# Native Messages uses $.thinking.type = "adaptive" and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected. https://docs.aihubmix.com/cn/blogs/Claude-Opus4.7 (accessed 2026-06-25) +temperature = false +tool_call = true structured_output = true +knowledge = "2026-01-31" +open_weights = false interleaved = true -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - [cost] input = 5 output = 25 @@ -18,8 +22,16 @@ cache_read = 0.5 cache_write = 6.25 [[cost.tiers]] -tier = { type = "context", size = 200_000 } +tier = { size = 200_000 } input = 10 output = 37.5 -cache_read = 1 +cache_read = 1.0 cache_write = 12.5 + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-4-8-think.toml b/providers/aihubmix/models/claude-opus-4-8-think.toml index af514c8d231..52f226453d2 100644 --- a/providers/aihubmix/models/claude-opus-4-8-think.toml +++ b/providers/aihubmix/models/claude-opus-4-8-think.toml @@ -1,18 +1,20 @@ base_model = "anthropic/claude-opus-4-8" -structured_output = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +# Anthropic-compatible /v1/messages: $.thinking.type = "enabled"|"disabled"|"adaptive" (disabled turns reasoning off = toggle) and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected on this Opus tier. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message (accessed 2026-07-02) [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - [cost] input = 5 output = 25 cache_read = 0.5 cache_write = 6.25 + +[limit] +context = 200_000 +output = 32_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-4-8.toml b/providers/aihubmix/models/claude-opus-4-8.toml index 7605942029d..e8d612cf002 100644 --- a/providers/aihubmix/models/claude-opus-4-8.toml +++ b/providers/aihubmix/models/claude-opus-4-8.toml @@ -1,17 +1,19 @@ base_model = "anthropic/claude-opus-4-8" -structured_output = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] +# Anthropic-compatible /v1/messages: $.thinking.type = "enabled"|"disabled"|"adaptive" (disabled turns reasoning off = toggle) and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected on this Opus tier. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message (accessed 2026-07-02) interleaved = true -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - [cost] input = 5 output = 25 cache_read = 0.5 cache_write = 6.25 + +[limit] +context = 200_000 +output = 32_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/claude-opus-5.toml b/providers/aihubmix/models/claude-opus-5.toml index 70accd1aad8..6566c208cd6 100644 --- a/providers/aihubmix/models/claude-opus-5.toml +++ b/providers/aihubmix/models/claude-opus-5.toml @@ -1,18 +1,15 @@ # AIHubMix Anthropic-compatible /v1/messages: $.thinking.type = "disabled"|"adaptive" (toggle) and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; verified live 2026-08-11. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message base_model = "anthropic/claude-opus-5" structured_output = true - +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] interleaved = true -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - [cost] input = 5 output = 25 cache_read = 0.5 cache_write = 6.25 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/claude-sonnet-4-5-think.toml b/providers/aihubmix/models/claude-sonnet-4-5-think.toml deleted file mode 100644 index c0d6579cc19..00000000000 --- a/providers/aihubmix/models/claude-sonnet-4-5-think.toml +++ /dev/null @@ -1,21 +0,0 @@ -base_model = "anthropic/claude-sonnet-4-5" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 3.3 -output = 16.5 -cache_read = 0.33 -cache_write = 4.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 6.6 -output = 24.75 -cache_read = 0.66 -cache_write = 8.25 diff --git a/providers/aihubmix/models/claude-sonnet-4-5.toml b/providers/aihubmix/models/claude-sonnet-4-5.toml deleted file mode 100644 index c0d6579cc19..00000000000 --- a/providers/aihubmix/models/claude-sonnet-4-5.toml +++ /dev/null @@ -1,21 +0,0 @@ -base_model = "anthropic/claude-sonnet-4-5" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 3.3 -output = 16.5 -cache_read = 0.33 -cache_write = 4.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 6.6 -output = 24.75 -cache_read = 0.66 -cache_write = 8.25 diff --git a/providers/aihubmix/models/claude-sonnet-4-6-think.toml b/providers/aihubmix/models/claude-sonnet-4-6-think.toml index 4e000db48bd..9e3fc461f4e 100644 --- a/providers/aihubmix/models/claude-sonnet-4-6-think.toml +++ b/providers/aihubmix/models/claude-sonnet-4-6-think.toml @@ -1,33 +1,37 @@ -base_model = "anthropic/claude-sonnet-4-6" name = "Claude Sonnet 4.6 Thinking" description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" +family = "claude-sonnet" +release_date = "2026-02-17" +last_updated = "2026-03-13" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] +temperature = true +tool_call = true structured_output = true +knowledge = "2025-08-31" +open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "max"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] -input = 3 -output = 15 -cache_read = 0.3 +input = 3.00 +output = 15.00 +cache_read = 0.30 cache_write = 3.75 [[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 6 -output = 22.5 -cache_read = 0.6 -cache_write = 7.5 +tier = { size = 200_000 } +input = 6.00 +output = 22.50 +cache_read = 0.60 +cache_write = 7.50 [limit] -output = 128_000 +context = 1_000_000 +output = 64_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/claude-sonnet-4-6.toml b/providers/aihubmix/models/claude-sonnet-4-6.toml index df8ab08fbfa..b252b0f52b3 100644 --- a/providers/aihubmix/models/claude-sonnet-4-6.toml +++ b/providers/aihubmix/models/claude-sonnet-4-6.toml @@ -1,31 +1,37 @@ -base_model = "anthropic/claude-sonnet-4-6" +name = "Claude Sonnet 4.6" description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" +family = "claude-sonnet" +release_date = "2026-02-17" +last_updated = "2026-03-13" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] +# Native Messages prefers $.thinking.type = "adaptive" with $.output_config.effort = "low"|"medium"|"high"; enabled budget_tokens >= 1024 is deprecated and must be < $.max_tokens. Chat effort "max" maps to native "high". https://docs.aihubmix.com/cn/api/Claude-Native (accessed 2026-06-25) +temperature = true +tool_call = true structured_output = true +knowledge = "2025-08-31" +open_weights = false interleaved = true -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "max"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] -input = 3 -output = 15 -cache_read = 0.3 +input = 3.00 +output = 15.00 +cache_read = 0.30 cache_write = 3.75 [[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 6 -output = 22.5 -cache_read = 0.6 -cache_write = 7.5 +tier = { size = 200_000 } +input = 6.00 +output = 22.50 +cache_read = 0.60 +cache_write = 7.50 [limit] -output = 128_000 +context = 1_000_000 +output = 64_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/claude-sonnet-5.toml b/providers/aihubmix/models/claude-sonnet-5.toml index 6ff86b6a08f..880ddcd2f41 100644 --- a/providers/aihubmix/models/claude-sonnet-5.toml +++ b/providers/aihubmix/models/claude-sonnet-5.toml @@ -1,17 +1,19 @@ base_model = "anthropic/claude-sonnet-5" structured_output = true - +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] interleaved = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] +temperature = false [cost] input = 2 output = 10 cache_read = 0.2 cache_write = 2.5 + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/cloudflare-glm-5.2.toml b/providers/aihubmix/models/cloudflare-glm-5.2.toml deleted file mode 100644 index 1ab27569fed..00000000000 --- a/providers/aihubmix/models/cloudflare-glm-5.2.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "zhipuai/glm-5.2" -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 1.4 -output = 4.4002 -cache_read = 0.2604 diff --git a/providers/aihubmix/models/coding-glm-4.6-free.toml b/providers/aihubmix/models/coding-glm-4.6-free.toml deleted file mode 100644 index 6d567a1b123..00000000000 --- a/providers/aihubmix/models/coding-glm-4.6-free.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "zhipuai/glm-4.6" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 - -[limit] -context = 200_000 diff --git a/providers/aihubmix/models/coding-glm-4.6.toml b/providers/aihubmix/models/coding-glm-4.6.toml deleted file mode 100644 index 26f0de57cbe..00000000000 --- a/providers/aihubmix/models/coding-glm-4.6.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "zhipuai/glm-4.6" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.06 -output = 0.22 -cache_read = 0.010998 - -[limit] -context = 200_000 diff --git a/providers/aihubmix/models/coding-glm-4.7-free.toml b/providers/aihubmix/models/coding-glm-4.7-free.toml deleted file mode 100644 index e6f1d52ca64..00000000000 --- a/providers/aihubmix/models/coding-glm-4.7-free.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "zhipuai/glm-4.7" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 - -[limit] -context = 200_000 diff --git a/providers/aihubmix/models/coding-glm-4.7.toml b/providers/aihubmix/models/coding-glm-4.7.toml deleted file mode 100644 index 679d37ca186..00000000000 --- a/providers/aihubmix/models/coding-glm-4.7.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "zhipuai/glm-4.7" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.06 -output = 0.22 -cache_read = 0.010998 - -[limit] -context = 200_000 diff --git a/providers/aihubmix/models/coding-glm-5-free.toml b/providers/aihubmix/models/coding-glm-5-free.toml deleted file mode 100644 index c955d2f4274..00000000000 --- a/providers/aihubmix/models/coding-glm-5-free.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "zhipuai/glm-5" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 - -[limit] -context = 200_000 diff --git a/providers/aihubmix/models/coding-glm-5-turbo-free.toml b/providers/aihubmix/models/coding-glm-5-turbo-free.toml deleted file mode 100644 index 29147dbed38..00000000000 --- a/providers/aihubmix/models/coding-glm-5-turbo-free.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "zhipuai/glm-5-turbo" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 - -[limit] -context = 204_800 diff --git a/providers/aihubmix/models/coding-glm-5-turbo.toml b/providers/aihubmix/models/coding-glm-5-turbo.toml deleted file mode 100644 index 50c916e5495..00000000000 --- a/providers/aihubmix/models/coding-glm-5-turbo.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "zhipuai/glm-5-turbo" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.06 -output = 0.22 - -[limit] -context = 204_800 diff --git a/providers/aihubmix/models/coding-glm-5.1-free.toml b/providers/aihubmix/models/coding-glm-5.1-free.toml index b3fa0b48e57..c44e059e3ad 100644 --- a/providers/aihubmix/models/coding-glm-5.1-free.toml +++ b/providers/aihubmix/models/coding-glm-5.1-free.toml @@ -1,13 +1,27 @@ -base_model = "zhipuai/glm-5.1" name = "Coding GLM 5.1 (free)" description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" +family = "glm-free" +release_date = "2026-04-11" +last_updated = "2026-04-11" +attachment = false +reasoning = true +reasoning_options = [{ type = "toggle" }] +temperature = true +tool_call = true +structured_output = true +open_weights = true [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 0 output = 0 + +[limit] +context = 200_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/coding-glm-5.1.toml b/providers/aihubmix/models/coding-glm-5.1.toml index c6a5d070c82..ad67b0aeab4 100644 --- a/providers/aihubmix/models/coding-glm-5.1.toml +++ b/providers/aihubmix/models/coding-glm-5.1.toml @@ -1,13 +1,28 @@ -base_model = "zhipuai/glm-5.1" name = "Coding GLM 5.1" description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" +family = "glm" +release_date = "2026-04-11" +last_updated = "2026-04-11" +attachment = false +reasoning = true +reasoning_options = [{ type = "toggle" }] +temperature = true +tool_call = true +structured_output = true +open_weights = true [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 0.06 output = 0.22 +cache_read = 0.013 + +[limit] +context = 200_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/coding-glm-5.2-free.toml b/providers/aihubmix/models/coding-glm-5.2-free.toml deleted file mode 100644 index 9e3d289059d..00000000000 --- a/providers/aihubmix/models/coding-glm-5.2-free.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "zhipuai/glm-5.2" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/coding-glm-5.2.toml b/providers/aihubmix/models/coding-glm-5.2.toml deleted file mode 100644 index e5ae0dc0bb7..00000000000 --- a/providers/aihubmix/models/coding-glm-5.2.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "zhipuai/glm-5.2" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.06 -output = 0.22 diff --git a/providers/aihubmix/models/coding-glm-5.3-free.toml b/providers/aihubmix/models/coding-glm-5.3-free.toml deleted file mode 100644 index 05636bbb0b7..00000000000 --- a/providers/aihubmix/models/coding-glm-5.3-free.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "zhipuai/glm-5.3" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 diff --git a/providers/aihubmix/models/coding-glm-5.3.toml b/providers/aihubmix/models/coding-glm-5.3.toml deleted file mode 100644 index 2803d9d5996..00000000000 --- a/providers/aihubmix/models/coding-glm-5.3.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "zhipuai/glm-5.3" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0.06 -output = 0.22 -cache_read = 0.015 - -[limit] -context = 1_048_576 diff --git a/providers/aihubmix/models/coding-glm-5.toml b/providers/aihubmix/models/coding-glm-5.toml deleted file mode 100644 index 43a367a1581..00000000000 --- a/providers/aihubmix/models/coding-glm-5.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "zhipuai/glm-5" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.06 -output = 0.22 - -[limit] -context = 200_000 diff --git a/providers/aihubmix/models/coding-kimi-k3-free.toml b/providers/aihubmix/models/coding-kimi-k3-free.toml deleted file mode 100644 index 0a47dcba8ec..00000000000 --- a/providers/aihubmix/models/coding-kimi-k3-free.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "moonshotai/kimi-k3" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0 -output = 0 - -[limit] -output = 1_048_576 diff --git a/providers/aihubmix/models/coding-kimi-k3.toml b/providers/aihubmix/models/coding-kimi-k3.toml deleted file mode 100644 index 2001fbbeb87..00000000000 --- a/providers/aihubmix/models/coding-kimi-k3.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "moonshotai/kimi-k3" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0.44 -output = 1.61333 -cache_read = 0.066 - -[limit] -output = 1_048_576 diff --git a/providers/aihubmix/models/coding-minimax-m2-free.toml b/providers/aihubmix/models/coding-minimax-m2-free.toml deleted file mode 100644 index b87ca9d6a6b..00000000000 --- a/providers/aihubmix/models/coding-minimax-m2-free.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/coding-minimax-m2.1-free.toml b/providers/aihubmix/models/coding-minimax-m2.1-free.toml deleted file mode 100644 index 36a757f4ab4..00000000000 --- a/providers/aihubmix/models/coding-minimax-m2.1-free.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.1" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/coding-minimax-m2.1.toml b/providers/aihubmix/models/coding-minimax-m2.1.toml deleted file mode 100644 index a853c211e95..00000000000 --- a/providers/aihubmix/models/coding-minimax-m2.1.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.1" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.2 -output = 0.2 diff --git a/providers/aihubmix/models/coding-minimax-m2.5-free.toml b/providers/aihubmix/models/coding-minimax-m2.5-free.toml deleted file mode 100644 index 3a00a80be66..00000000000 --- a/providers/aihubmix/models/coding-minimax-m2.5-free.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.5" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/coding-minimax-m2.5-highspeed.toml b/providers/aihubmix/models/coding-minimax-m2.5-highspeed.toml deleted file mode 100644 index e35e6b64892..00000000000 --- a/providers/aihubmix/models/coding-minimax-m2.5-highspeed.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.5-highspeed" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.2 -output = 0.2 diff --git a/providers/aihubmix/models/coding-minimax-m2.5.toml b/providers/aihubmix/models/coding-minimax-m2.5.toml deleted file mode 100644 index b9b6b77d2e0..00000000000 --- a/providers/aihubmix/models/coding-minimax-m2.5.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.5" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.2 -output = 0.2 diff --git a/providers/aihubmix/models/coding-minimax-m2.7-free.toml b/providers/aihubmix/models/coding-minimax-m2.7-free.toml index e62a1236d97..de8f6b2b766 100644 --- a/providers/aihubmix/models/coding-minimax-m2.7-free.toml +++ b/providers/aihubmix/models/coding-minimax-m2.7-free.toml @@ -1,21 +1,27 @@ -base_model = "minimax/MiniMax-M2.7" -name = "Coding MiniMax M2.7 (free)" +name = "Coding MiniMax M2.7 (Free)" description = "MiniMax model for chat, coding, office work, and agentic tasks" +family = "minimax-free" +release_date = "2026-03-18" +last_updated = "2026-03-18" +attachment = false +reasoning = true +reasoning_options = [] +temperature = true +tool_call = true structured_output = true +open_weights = true [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 0 output = 0 [limit] -output = 204_800 +context = 204_800 +output = 128_100 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml b/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml index 5defdbe12d7..24b69d7fa7b 100644 --- a/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml +++ b/providers/aihubmix/models/coding-minimax-m2.7-highspeed.toml @@ -1,21 +1,27 @@ -base_model = "minimax/MiniMax-M2.7-highspeed" name = "Coding MiniMax M2.7 Highspeed" description = "High-speed MiniMax model for low-latency coding and agent workflows" +family = "minimax" +release_date = "2026-03-18" +last_updated = "2026-03-18" +attachment = false +reasoning = true +reasoning_options = [] +temperature = true +tool_call = true structured_output = true +open_weights = true [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 0.2 output = 0.2 [limit] -output = 204_800 +context = 204_800 +output = 128_100 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/coding-minimax-m2.7.toml b/providers/aihubmix/models/coding-minimax-m2.7.toml index 35ab1809da0..c17cc1f4e4c 100644 --- a/providers/aihubmix/models/coding-minimax-m2.7.toml +++ b/providers/aihubmix/models/coding-minimax-m2.7.toml @@ -1,21 +1,27 @@ -base_model = "minimax/MiniMax-M2.7" name = "Coding MiniMax M2.7" description = "MiniMax model for chat, coding, office work, and agentic tasks" +family = "minimax" +release_date = "2026-03-18" +last_updated = "2026-03-18" +attachment = false +reasoning = true +reasoning_options = [] +temperature = true +tool_call = true structured_output = true +open_weights = true [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 0.2 output = 0.2 [limit] -output = 204_800 +context = 204_800 +output = 128_100 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/coding-minimax-m2.toml b/providers/aihubmix/models/coding-minimax-m2.toml deleted file mode 100644 index b2d95706530..00000000000 --- a/providers/aihubmix/models/coding-minimax-m2.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.2 -output = 0.2 diff --git a/providers/aihubmix/models/coding-minimax-m3-free.toml b/providers/aihubmix/models/coding-minimax-m3-free.toml deleted file mode 100644 index 259a40c1fda..00000000000 --- a/providers/aihubmix/models/coding-minimax-m3-free.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "minimax/MiniMax-M3" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_000_000 -output = 524_288 diff --git a/providers/aihubmix/models/coding-minimax-m3.toml b/providers/aihubmix/models/coding-minimax-m3.toml deleted file mode 100644 index 4ae57e13033..00000000000 --- a/providers/aihubmix/models/coding-minimax-m3.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "minimax/MiniMax-M3" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.2 -output = 0.2 - -[limit] -context = 1_000_000 -output = 524_288 diff --git a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml index ffbb47916c3..a106666530c 100644 --- a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml +++ b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5-pro.toml @@ -1,17 +1,16 @@ base_model = "xiaomi/mimo-v2.5-pro" +reasoning_options = [{ type = "toggle" }] name = "Coding Xiaomi MiMo-V2.5-Pro" -structured_output = true +family = "mimo-v2.5-pro" +last_updated = "2026-05-13" [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 0.2 -output = 0.4 -cache_read = 0.0016 +output = 0.6 +cache_read = 0.04 [[cost.tiers]] tier = { type = "context", size = 256_000 } diff --git a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml index bf5cb9d4e99..fa49278c22d 100644 --- a/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml +++ b/providers/aihubmix/models/coding-xiaomi-mimo-v2.5.toml @@ -1,17 +1,16 @@ base_model = "xiaomi/mimo-v2.5" +reasoning_options = [{ type = "toggle" }] name = "Coding Xiaomi MiMo-V2.5" -structured_output = true +family = "mimo-v2.5" +last_updated = "2026-05-13" [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 0.08 -output = 0.16 -cache_read = 0.0016 +output = 0.4 +cache_read = 0.016 [[cost.tiers]] tier = { type = "context", size = 256_000 } diff --git a/providers/aihubmix/models/cohere-command-a.toml b/providers/aihubmix/models/cohere-command-a.toml deleted file mode 100644 index b00d795de92..00000000000 --- a/providers/aihubmix/models/cohere-command-a.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "cohere/command-a-03-2025" -reasoning = true -tool_call = false - -[[reasoning_options]] -type = "effort" -values = ["none", "high"] - -[cost] -input = 2.5 -output = 10 diff --git a/providers/aihubmix/models/command-a-03-2025.toml b/providers/aihubmix/models/command-a-03-2025.toml deleted file mode 100644 index b00d795de92..00000000000 --- a/providers/aihubmix/models/command-a-03-2025.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "cohere/command-a-03-2025" -reasoning = true -tool_call = false - -[[reasoning_options]] -type = "effort" -values = ["none", "high"] - -[cost] -input = 2.5 -output = 10 diff --git a/providers/aihubmix/models/command-a-plus-05-2026.toml b/providers/aihubmix/models/command-a-plus-05-2026.toml deleted file mode 100644 index 562165dc4dc..00000000000 --- a/providers/aihubmix/models/command-a-plus-05-2026.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "cohere/command-a-plus-05-2026" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 2.5 -output = 10 diff --git a/providers/aihubmix/models/command-r-08-2024.toml b/providers/aihubmix/models/command-r-08-2024.toml deleted file mode 100644 index 3eb86afe515..00000000000 --- a/providers/aihubmix/models/command-r-08-2024.toml +++ /dev/null @@ -1,6 +0,0 @@ -base_model = "cohere/command-r-08-2024" -tool_call = false - -[cost] -input = 0.2 -output = 0.8 diff --git a/providers/aihubmix/models/command-r-plus-08-2024.toml b/providers/aihubmix/models/command-r-plus-08-2024.toml deleted file mode 100644 index e0e3ee2ad05..00000000000 --- a/providers/aihubmix/models/command-r-plus-08-2024.toml +++ /dev/null @@ -1,6 +0,0 @@ -base_model = "cohere/command-r-plus-08-2024" -tool_call = false - -[cost] -input = 2.8 -output = 11.2 diff --git a/providers/aihubmix/models/deepinfra-gemma-4-26b-a4b-it.toml b/providers/aihubmix/models/deepinfra-gemma-4-26b-a4b-it.toml deleted file mode 100644 index bd9adb9f07c..00000000000 --- a/providers/aihubmix/models/deepinfra-gemma-4-26b-a4b-it.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "google/gemma-4-26b-a4b-it" -attachment = false -reasoning = false -tool_call = false - -[cost] -input = 0.088 -output = 0.385 -cache_read = 0.011 - -[limit] -context = 262_100 -output = 131_100 - -[modalities] -input = ["text"] diff --git a/providers/aihubmix/models/deepseek-v3.2-think.toml b/providers/aihubmix/models/deepseek-v3.2-think.toml deleted file mode 100644 index b0d1587bebd..00000000000 --- a/providers/aihubmix/models/deepseek-v3.2-think.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "deepseek/deepseek-v3.2" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.302 -output = 0.453 -cache_read = 0.0302 - -[limit] -context = 163_840 diff --git a/providers/aihubmix/models/deepseek-v3.2.toml b/providers/aihubmix/models/deepseek-v3.2.toml deleted file mode 100644 index b0d1587bebd..00000000000 --- a/providers/aihubmix/models/deepseek-v3.2.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "deepseek/deepseek-v3.2" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.302 -output = 0.453 -cache_read = 0.0302 - -[limit] -context = 163_840 diff --git a/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml b/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml deleted file mode 100644 index 8a68fecf328..00000000000 --- a/providers/aihubmix/models/deepseek-v4-flash-0731-fast.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "deepseek/deepseek-v4-flash-0731" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0.28 -output = 1.4 -cache_read = 0.07 diff --git a/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml b/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml deleted file mode 100644 index 53b6813a437..00000000000 --- a/providers/aihubmix/models/deepseek-v4-flash-vision-exp.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "deepseek/deepseek-v4-flash-vision-exp" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0.155 -output = 0.62 -cache_read = 0.0031 diff --git a/providers/aihubmix/models/deepseek-v4-flash.toml b/providers/aihubmix/models/deepseek-v4-flash.toml deleted file mode 100644 index fd5c8bd99ac..00000000000 --- a/providers/aihubmix/models/deepseek-v4-flash.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "deepseek/deepseek-v4-flash" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0.142 -output = 0.284 -cache_read = 0.0284 diff --git a/providers/aihubmix/models/deepseek-v4-pro-0813.toml b/providers/aihubmix/models/deepseek-v4-pro-0813.toml index dab59a28608..fbe29803f4c 100644 --- a/providers/aihubmix/models/deepseek-v4-pro-0813.toml +++ b/providers/aihubmix/models/deepseek-v4-pro-0813.toml @@ -2,17 +2,11 @@ # Effort: reasoning_effort = high|max # AIHubMix effort levels returned HTTP 200 (validated 2026-08-31T04:17:24Z). base_model = "deepseek/deepseek-v4-pro-0813" +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["high", "max"] }] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - [cost] input = 0.6918 output = 2.0754 diff --git a/providers/aihubmix/models/deepseek-v4-pro.toml b/providers/aihubmix/models/deepseek-v4-pro.toml deleted file mode 100644 index 097f0ad22a7..00000000000 --- a/providers/aihubmix/models/deepseek-v4-pro.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "deepseek/deepseek-v4-pro" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 1.69 -output = 3.38 -cache_read = 0.14027 diff --git a/providers/aihubmix/models/deepseek-v4.1-flash.toml b/providers/aihubmix/models/deepseek-v4.1-flash.toml deleted file mode 100644 index 27058ae312d..00000000000 --- a/providers/aihubmix/models/deepseek-v4.1-flash.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "DeepSeek V4.1 Flash" -description = "DeepSeek-V4.1-Flash model official release. This is the smallest model in DeepSeek’s new model-architecture series, featuring native multimodal visual understanding capabilities. The new architecture is designed to deliver a higher capability ceiling, faster inference speeds, greater throughput, and scalability to larger-parameter models." -release_date = "2026-09-08" -last_updated = "2026-09-08" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0.155 -output = 0.62 -cache_read = 0.0031 - -[limit] -context = 1_000_000 -output = 384_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/doubao-seed-1-8.toml b/providers/aihubmix/models/doubao-seed-1-8.toml deleted file mode 100644 index c456f3f6c12..00000000000 --- a/providers/aihubmix/models/doubao-seed-1-8.toml +++ /dev/null @@ -1,41 +0,0 @@ -name = "Doubao Seed 1.8" -description = "Doubao's strongest multimodal Agent model Seed1.8 has powerful multimodal capabilities, supports image and text input, and can efficiently and accurately complete tasks in scenarios such as information retrieval, code generation, GUI interaction, and complex workflows, meeting increasingly diverse technical demands." -release_date = "2025-12-28" -last_updated = "2025-12-28" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.10959 -output = 0.273975 -cache_read = 0.021918 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.1644 -output = 2.191995 -cache_read = 0.021915 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.32876 -output = 3.2876 -cache_read = 0.021918 - -[limit] -context = 256_000 -output = 32_768 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml b/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml index b37e7c9b461..4eb6666263d 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-code-preview.toml @@ -5,6 +5,7 @@ release_date = "2026-02-14" last_updated = "2026-02-14" attachment = true reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] temperature = true tool_call = true structured_output = true @@ -13,28 +14,21 @@ open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - [cost] -input = 0.4822 -output = 2.411 +input = 0.48 +output = 2.41 cache_read = 0.09644 [[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.72328 -output = 3.6164 +tier = { size = 32_000 } +input = 0.72 +output = 3.62 cache_read = 0.144656 [[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 1.4466 -output = 7.233 +tier = { size = 128_000 } +input = 1.45 +output = 7.23 cache_read = 0.28932 [limit] diff --git a/providers/aihubmix/models/doubao-seed-2-0-lite-260215.toml b/providers/aihubmix/models/doubao-seed-2-0-lite-260215.toml deleted file mode 100644 index 69840b095a2..00000000000 --- a/providers/aihubmix/models/doubao-seed-2-0-lite-260215.toml +++ /dev/null @@ -1,41 +0,0 @@ -name = "Doubao Seed 2.0 Lite 260215" -description = "Doubao Coding model optimized for real-world programming environments that can reliably invoke tools in common IDEs such as Claude Code. The model is specially optimized for frontend capabilities and performs well with common frontend frameworks. The model supports Skills and can work with various custom skills." -release_date = "2026-02-15" -last_updated = "2026-02-15" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.09041 -output = 0.54246 -cache_read = 0.018082 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.135616 -output = 0.813696 -cache_read = 0.027123 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.271232 -output = 1.627392 -cache_read = 0.054246 - -[limit] -context = 256_000 -output = 128_000 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml b/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml index bd50446b1af..dc89d4bccc6 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-lite-260428.toml @@ -5,6 +5,7 @@ release_date = "2026-04-28" last_updated = "2026-04-28" attachment = true reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] temperature = true tool_call = true structured_output = true @@ -13,34 +14,30 @@ open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - [cost] -input = 0.09041 -output = 0.54246 -cache_read = 0.018082 +input = 0.08 +output = 0.51 +cache_read = 0.01692 +input_audio = 1.269 [[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.1268 -output = 0.7608 +tier = { size = 32_000 } +input = 0.13 +output = 0.76 cache_read = 0.02536 +input_audio = 1.902 [[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.2536 -output = 1.5216 +tier = { size = 128_000 } +input = 0.25 +output = 1.52 cache_read = 0.05072 +input_audio = 3.804 [limit] context = 256_000 output = 128_000 [modalities] -input = ["text", "image", "video", "audio"] +input = ["text", "image", "video"] output = ["text"] diff --git a/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml b/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml index 8f929910c37..2f14132485f 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-mini-260428.toml @@ -5,6 +5,7 @@ release_date = "2026-04-28" last_updated = "2026-04-28" attachment = true reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] temperature = true tool_call = true structured_output = true @@ -13,34 +14,30 @@ open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - [cost] -input = 0.0282 -output = 0.282 +input = 0.03 +output = 0.28 cache_read = 0.00564 +input_audio = 0.423 [[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.0564 -output = 0.564 +tier = { size = 32_000 } +input = 0.06 +output = 0.56 cache_read = 0.01128 +input_audio = 0.846 [[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.1128 -output = 1.128 +tier = { size = 128_000 } +input = 0.11 +output = 1.13 cache_read = 0.02256 +input_audio = 1.692 [limit] context = 256_000 output = 128_000 [modalities] -input = ["text", "image", "video", "audio"] +input = ["text", "image", "video"] output = ["text"] diff --git a/providers/aihubmix/models/doubao-seed-2-0-pro.toml b/providers/aihubmix/models/doubao-seed-2-0-pro.toml index dbe58911d3d..a3f4582f729 100644 --- a/providers/aihubmix/models/doubao-seed-2-0-pro.toml +++ b/providers/aihubmix/models/doubao-seed-2-0-pro.toml @@ -5,6 +5,7 @@ release_date = "2026-02-14" last_updated = "2026-02-14" attachment = true reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["minimal", "low", "medium", "high"] }] temperature = true tool_call = true structured_output = true @@ -13,28 +14,21 @@ open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - [cost] -input = 0.4822 -output = 2.411 +input = 0.48 +output = 2.41 cache_read = 0.09644 [[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.72328 -output = 3.6164 +tier = { size = 32_000 } +input = 0.72 +output = 3.62 cache_read = 0.144656 [[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 1.4466 -output = 7.233 +tier = { size = 128_000 } +input = 1.45 +output = 7.23 cache_read = 0.28932 [limit] diff --git a/providers/aihubmix/models/ernie-5.1.toml b/providers/aihubmix/models/ernie-5.1.toml deleted file mode 100644 index 9f902de60fb..00000000000 --- a/providers/aihubmix/models/ernie-5.1.toml +++ /dev/null @@ -1,30 +0,0 @@ -name = "ERNIE 5.1" -description = "ERNIE 5.1 is the latest model in the Wenxin series, with comprehensive upgrades to its foundational capabilities and significant improvements in agents, knowledge, reasoning, and deep search. This upgrade uses a decoupled fully-asynchronous reinforcement learning technique to specifically address challenges encountered as large models evolve toward agent-based autonomous decision-making, such as training–inference numerical bias, low utilization of heterogeneous resources, and global issues caused by long-tail effects. It is paired with scaled agent post-training techniques to enhance model capabilities and generalization, enabling a three-step collaboration of environment, expert, and fusion that both ensures training efficiency and significantly improves the model’s stability and performance on complex tasks." -release_date = "2026-05-10" -last_updated = "2026-05-10" -attachment = false -reasoning = true -tool_call = false -open_weights = false - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.5634 -output = 2.5353 -cache_read = 0.5634 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.845 -output = 3.098361 -cache_read = 0.845 - -[limit] -context = 119_000 -output = 65_536 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-image.toml b/providers/aihubmix/models/gemini-2.5-flash-image.toml deleted file mode 100644 index c46f3c2e4b7..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-image.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-2.5-flash-image" -reasoning = false - -[cost] -input = 0.3 -output = 2.499 -cache_read = 0.03 - -[limit] -context = 65_536 diff --git a/providers/aihubmix/models/gemini-2.5-flash-lite-nothink.toml b/providers/aihubmix/models/gemini-2.5-flash-lite-nothink.toml deleted file mode 100644 index f42f82a1953..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-lite-nothink.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "google/gemini-2.5-flash-lite" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.4 -cache_read = 0.01 diff --git a/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml b/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml deleted file mode 100644 index 35d209d0223..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-lite-preview-09-2025.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Gemini 2.5 Flash Lite Preview 09 2025" -description = "gemini-2.5-flash-lite latest preview version" -release_date = "2025-09-25" -last_updated = "2025-09-25" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = false - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.4 -cache_read = 0.01 - -[limit] -context = 1_048_576 -output = 65_536 - -[modalities] -input = ["text", "image", "video", "audio", "pdf"] -output = ["text"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-lite.toml b/providers/aihubmix/models/gemini-2.5-flash-lite.toml deleted file mode 100644 index f42f82a1953..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-lite.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "google/gemini-2.5-flash-lite" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.4 -cache_read = 0.01 diff --git a/providers/aihubmix/models/gemini-2.5-flash-nothink.toml b/providers/aihubmix/models/gemini-2.5-flash-nothink.toml deleted file mode 100644 index 5f59e3eef8f..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-nothink.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "google/gemini-2.5-flash" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.3 -output = 2.499 -cache_read = 0.03 diff --git a/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml b/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml deleted file mode 100644 index e5a39cfc131..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-preview-05-20-search.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Gemini 2.5 Flash Preview 05-20 Search" -description = "Gemini-2.5 Flash Preview 05-20 Search integrates Google's official search functionality; the search feature will have an additional separate fee log directly integrated into the scoring deduction, with detailed logs not displayed. It will be fixed and displayed later. Only OpenAI-compatible formats are supported for invocation; Gemini SDK is not supported. For Gemini's native SDK, please set parameters directly using the official search parameters." -release_date = "2025-05-20" -last_updated = "2025-05-20" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = false - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.3 -output = 2.499 -cache_read = 0.03 - -[limit] -context = 1_048_576 -output = 65_536 - -[modalities] -input = ["text", "image", "video", "audio"] -output = ["text"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml b/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml deleted file mode 100644 index 67d5d794003..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-preview-09-2025.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Gemini 2.5 Flash Preview 09 2025" -description = "This latest 2.5 Flash model comes with improvements in two key areas we heard consistent feedback on:\n\nBetter agentic tool use: We've improved how the model uses tools, leading to better performance in more complex, agentic and multi-step applications. This model shows noticeable improvements on key agentic benchmarks, including a 5% gain on SWE-Bench Verified, compared to our last release (48.9% → 54%). More efficient: With thinking on, the model is now significantly more cost-efficient—achieving higher quality outputs while using fewer tokens, reducing latency and cost (see charts above)." -release_date = "2025-09-25" -last_updated = "2025-09-25" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = false - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.3 -output = 2.499 -cache_read = 0.03 - -[limit] -context = 1_048_576 -output = 65_536 - -[modalities] -input = ["text", "image", "video", "audio"] -output = ["text"] diff --git a/providers/aihubmix/models/gemini-2.5-flash-search.toml b/providers/aihubmix/models/gemini-2.5-flash-search.toml deleted file mode 100644 index 5f59e3eef8f..00000000000 --- a/providers/aihubmix/models/gemini-2.5-flash-search.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "google/gemini-2.5-flash" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.3 -output = 2.499 -cache_read = 0.03 diff --git a/providers/aihubmix/models/gemini-2.5-flash.toml b/providers/aihubmix/models/gemini-2.5-flash.toml index 8462b4a9581..23f238b714b 100644 --- a/providers/aihubmix/models/gemini-2.5-flash.toml +++ b/providers/aihubmix/models/gemini-2.5-flash.toml @@ -1,15 +1,28 @@ -# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: 0 disables, -1 is dynamic, and manual budgets are 1..24576. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) -base_model = "google/gemini-2.5-flash" +name = "Gemini 2.5 Flash" description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" +family = "gemini-flash" +release_date = "2025-03-20" +last_updated = "2025-06-05" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 0, max = 24_576 }] +# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: 0 disables, -1 is dynamic, and manual budgets are 1..24576. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-01" +open_weights = false [cost] input = 0.3 -output = 2.499 +output = 2.50 cache_read = 0.03 +input_audio = 1.00 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "audio", "video", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml b/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml deleted file mode 100644 index 6b6a905bb46..00000000000 --- a/providers/aihubmix/models/gemini-2.5-pro-preview-05-06.toml +++ /dev/null @@ -1,34 +0,0 @@ -name = "Gemini 2.5 Pro Preview 05-06" -description = "gemini-2.5-pro latest model" -release_date = "2025-05-07" -last_updated = "2025-05-07" -attachment = true -reasoning = true -tool_call = false -open_weights = false - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2.5 -output = 15 -cache_read = 0.25 - -[limit] -context = 1_048_576 -output = 65_536 - -[modalities] -input = ["text", "image", "video", "audio", "pdf"] -output = ["text"] diff --git a/providers/aihubmix/models/gemini-2.5-pro-search.toml b/providers/aihubmix/models/gemini-2.5-pro-search.toml deleted file mode 100644 index a97c9c65bb6..00000000000 --- a/providers/aihubmix/models/gemini-2.5-pro-search.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "google/gemini-2.5-pro" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2.5 -output = 15 -cache_read = 0.25 diff --git a/providers/aihubmix/models/gemini-2.5-pro.toml b/providers/aihubmix/models/gemini-2.5-pro.toml index 75750cf69ba..8ef984bdbf3 100644 --- a/providers/aihubmix/models/gemini-2.5-pro.toml +++ b/providers/aihubmix/models/gemini-2.5-pro.toml @@ -1,12 +1,17 @@ -base_model = "google/gemini-2.5-pro" +name = "Gemini 2.5 Pro" description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" +family = "gemini-pro" +release_date = "2025-03-20" +last_updated = "2025-06-05" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] +# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: -1 is dynamic and manual budgets are 128..32768; 0/off is unsupported. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-01" +open_weights = false [cost] input = 1.25 @@ -14,7 +19,15 @@ output = 10 cache_read = 0.125 [[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2.5 -output = 15 +tier = { size = 200_000 } +input = 2.50 +output = 15.00 cache_read = 0.25 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "audio", "video", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-3-flash-preview-free.toml b/providers/aihubmix/models/gemini-3-flash-preview-free.toml deleted file mode 100644 index 274c6a78644..00000000000 --- a/providers/aihubmix/models/gemini-3-flash-preview-free.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "google/gemini-3-flash-preview" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/gemini-3-flash-preview-search.toml b/providers/aihubmix/models/gemini-3-flash-preview-search.toml deleted file mode 100644 index 522c79650d8..00000000000 --- a/providers/aihubmix/models/gemini-3-flash-preview-search.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "google/gemini-3-flash-preview" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.5 -output = 3 -cache_read = 0.05 diff --git a/providers/aihubmix/models/gemini-3-flash-preview.toml b/providers/aihubmix/models/gemini-3-flash-preview.toml index 991ea923538..379e0cf258d 100644 --- a/providers/aihubmix/models/gemini-3-flash-preview.toml +++ b/providers/aihubmix/models/gemini-3-flash-preview.toml @@ -1,12 +1,16 @@ -base_model = "google/gemini-3-flash-preview" +name = "Gemini 3 Flash Preview" description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" +family = "gemini-flash" +release_date = "2025-12-17" +last_updated = "2025-12-17" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-01" +open_weights = false [cost] input = 0.5 @@ -14,7 +18,15 @@ output = 3 cache_read = 0.05 [[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 0.5 -output = 3 +tier = { size = 200_000 } +input = 0.50 +output = 3.00 cache_read = 0.05 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "audio", "video", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-3-pro-image-preview.toml b/providers/aihubmix/models/gemini-3-pro-image-preview.toml deleted file mode 100644 index 0bf04c931f5..00000000000 --- a/providers/aihubmix/models/gemini-3-pro-image-preview.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "google/gemini-3-pro-image-preview" -status = "deprecated" -reasoning_options = [] - -[cost] -input = 2 -output = 12 diff --git a/providers/aihubmix/models/gemini-3-pro-image.toml b/providers/aihubmix/models/gemini-3-pro-image.toml deleted file mode 100644 index bd71f0099f2..00000000000 --- a/providers/aihubmix/models/gemini-3-pro-image.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "google/gemini-3-pro-image" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 2 -output = 12 diff --git a/providers/aihubmix/models/gemini-3.1-flash-image-preview.toml b/providers/aihubmix/models/gemini-3.1-flash-image-preview.toml deleted file mode 100644 index 2409dc19d4e..00000000000 --- a/providers/aihubmix/models/gemini-3.1-flash-image-preview.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3.1-flash-image-preview" -status = "deprecated" -reasoning_options = [] - -[cost] -input = 0.5 -output = 3 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gemini-3.1-flash-image.toml b/providers/aihubmix/models/gemini-3.1-flash-image.toml deleted file mode 100644 index bf9b9becb85..00000000000 --- a/providers/aihubmix/models/gemini-3.1-flash-image.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "google/gemini-3.1-flash-image" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.5 -output = 3 diff --git a/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml b/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml deleted file mode 100644 index 91587bf22be..00000000000 --- a/providers/aihubmix/models/gemini-3.1-flash-lite-image.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "google/gemini-3.1-flash-lite-image" -tool_call = false - -[[reasoning_options]] -type = "effort" -values = ["minimal", "high"] - -[cost] -input = 0.25 -output = 1.5 -cache_read = 0.025 diff --git a/providers/aihubmix/models/gemini-3.1-flash-lite-nothink.toml b/providers/aihubmix/models/gemini-3.1-flash-lite-nothink.toml deleted file mode 100644 index 51db76c6d9d..00000000000 --- a/providers/aihubmix/models/gemini-3.1-flash-lite-nothink.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "google/gemini-3.1-flash-lite" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.25 -output = 1.5 -cache_read = 0.025 diff --git a/providers/aihubmix/models/gemini-3.1-flash-lite.toml b/providers/aihubmix/models/gemini-3.1-flash-lite.toml index 4adbf5f45d6..1454eedebea 100644 --- a/providers/aihubmix/models/gemini-3.1-flash-lite.toml +++ b/providers/aihubmix/models/gemini-3.1-flash-lite.toml @@ -1,14 +1,27 @@ -base_model = "google/gemini-3.1-flash-lite" +name = "Gemini 3.1 Flash Lite" description = "Low-latency Gemini model for high-volume multimodal and agent workloads" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" +family = "gemini-flash-lite" +release_date = "2026-05-07" +last_updated = "2026-05-07" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-01" +open_weights = false [cost] input = 0.25 output = 1.5 cache_read = 0.025 +cache_write = 1.00 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "audio", "video", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-3.1-pro-preview-customtools.toml b/providers/aihubmix/models/gemini-3.1-pro-preview-customtools.toml index c0d271ad953..7a53e5d5345 100644 --- a/providers/aihubmix/models/gemini-3.1-pro-preview-customtools.toml +++ b/providers/aihubmix/models/gemini-3.1-pro-preview-customtools.toml @@ -1,12 +1,16 @@ -base_model = "google/gemini-3.1-pro-preview-customtools" +name = "Gemini 3.1 Pro Preview Custom Tools" description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" +family = "gemini-pro" +release_date = "2026-02-19" +last_updated = "2026-02-19" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-01" +open_weights = false [cost] input = 2 @@ -14,7 +18,15 @@ output = 12 cache_read = 0.2 [[cost.tiers]] -tier = { type = "context", size = 200_000 } +tier = { size = 200_000 } input = 4 output = 18 cache_read = 0.4 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "audio", "video", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-3.1-pro-preview-search.toml b/providers/aihubmix/models/gemini-3.1-pro-preview-search.toml deleted file mode 100644 index 6fdb9a70c1b..00000000000 --- a/providers/aihubmix/models/gemini-3.1-pro-preview-search.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "google/gemini-3.1-pro-preview" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 2 -output = 12 -cache_read = 0.2 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 4 -output = 18 -cache_read = 0.4 diff --git a/providers/aihubmix/models/gemini-3.1-pro-preview.toml b/providers/aihubmix/models/gemini-3.1-pro-preview.toml index 559bc725893..c7dc961e75d 100644 --- a/providers/aihubmix/models/gemini-3.1-pro-preview.toml +++ b/providers/aihubmix/models/gemini-3.1-pro-preview.toml @@ -1,12 +1,16 @@ -base_model = "google/gemini-3.1-pro-preview" +name = "Gemini 3.1 Pro Preview" description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" +family = "gemini-pro" +release_date = "2026-02-19" +last_updated = "2026-02-19" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-01" +open_weights = false [cost] input = 2 @@ -14,7 +18,15 @@ output = 12 cache_read = 0.2 [[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 4 -output = 18 -cache_read = 0.4 +tier = { size = 200_000 } +input = 4.00 +output = 18.00 +cache_read = 0.40 + +[limit] +context = 1_048_576 +output = 65_536 + +[modalities] +input = ["text", "image", "audio", "video", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-3.5-flash-lite-free.toml b/providers/aihubmix/models/gemini-3.5-flash-lite-free.toml deleted file mode 100644 index 8ba9484d2bf..00000000000 --- a/providers/aihubmix/models/gemini-3.5-flash-lite-free.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "google/gemini-3.5-flash-lite" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/gemini-3.5-flash-lite.toml b/providers/aihubmix/models/gemini-3.5-flash-lite.toml deleted file mode 100644 index 980fef0620f..00000000000 --- a/providers/aihubmix/models/gemini-3.5-flash-lite.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3.5-flash-lite" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 0.3 -output = 2.499999 -cache_read = 0.03 diff --git a/providers/aihubmix/models/gemini-3.5-flash.toml b/providers/aihubmix/models/gemini-3.5-flash.toml index 1a5b79c9241..fb3e8ecd08a 100644 --- a/providers/aihubmix/models/gemini-3.5-flash.toml +++ b/providers/aihubmix/models/gemini-3.5-flash.toml @@ -1,13 +1,15 @@ base_model = "google/gemini-3.5-flash" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" +reasoning_options = [{ type = "effort", values = ["minimal", "low", "medium", "high"] }] [cost] input = 1.5 output = 9 -cache_read = 0.15 +cache_read = 1.5 + +[limit] +context = 1_000_000 +output = 64_000 + +[modalities] +input = ["text", "image", "audio", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/gemini-3.6-flash-free.toml b/providers/aihubmix/models/gemini-3.6-flash-free.toml deleted file mode 100644 index 76b7a0c3dfa..00000000000 --- a/providers/aihubmix/models/gemini-3.6-flash-free.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "google/gemini-3.6-flash" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/gemini-3.6-flash.toml b/providers/aihubmix/models/gemini-3.6-flash.toml deleted file mode 100644 index 9307549671d..00000000000 --- a/providers/aihubmix/models/gemini-3.6-flash.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "google/gemini-3.6-flash" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 1.5 -output = 7.5 -cache_read = 0.15 diff --git a/providers/aihubmix/models/gemini-3.7-flash-free.toml b/providers/aihubmix/models/gemini-3.7-flash-free.toml deleted file mode 100644 index 7a29ae2e6da..00000000000 --- a/providers/aihubmix/models/gemini-3.7-flash-free.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "google/gemini-3.7-flash" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/gemini-3.7-flash.toml b/providers/aihubmix/models/gemini-3.7-flash.toml index 1e23441b9c9..31d0f319a9a 100644 --- a/providers/aihubmix/models/gemini-3.7-flash.toml +++ b/providers/aihubmix/models/gemini-3.7-flash.toml @@ -1,11 +1,8 @@ base_model = "google/gemini-3.7-flash" -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" +# AIHubMix unified Chat: $.reasoning_effort = low|medium|high +# https://docs.aihubmix.com/cn/api/unified-inference +reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] input = 0.75 diff --git a/providers/aihubmix/models/gemini-3.8-flash-free.toml b/providers/aihubmix/models/gemini-3.8-flash-free.toml deleted file mode 100644 index f4f34ce371c..00000000000 --- a/providers/aihubmix/models/gemini-3.8-flash-free.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "google/gemini-3.8-flash" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/gemini-3.8-flash.toml b/providers/aihubmix/models/gemini-3.8-flash.toml deleted file mode 100644 index 1252ad652d8..00000000000 --- a/providers/aihubmix/models/gemini-3.8-flash.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "google/gemini-3.8-flash" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.75 -output = 3.75 -cache_read = 0.075 diff --git a/providers/aihubmix/models/gemma-4-26b-a4b-it-free.toml b/providers/aihubmix/models/gemma-4-26b-a4b-it-free.toml deleted file mode 100644 index fdbe0240df6..00000000000 --- a/providers/aihubmix/models/gemma-4-26b-a4b-it-free.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "google/gemma-4-26b-a4b-it" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "high"] - -[cost] -input = 0 -output = 0 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/gemma-4-26b-a4b-it.toml b/providers/aihubmix/models/gemma-4-26b-a4b-it.toml deleted file mode 100644 index f8ea6d86182..00000000000 --- a/providers/aihubmix/models/gemma-4-26b-a4b-it.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "google/gemma-4-26b-a4b-it" -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "high"] - -[cost] -input = 0.14 -output = 0.39998 - -[limit] -output = 131_100 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/gemma-4-31b-it-free.toml b/providers/aihubmix/models/gemma-4-31b-it-free.toml deleted file mode 100644 index 4d1d6d38a83..00000000000 --- a/providers/aihubmix/models/gemma-4-31b-it-free.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "google/gemma-4-31b-it" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "high"] - -[cost] -input = 0 -output = 0 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/gemma-4-31b-it.toml b/providers/aihubmix/models/gemma-4-31b-it.toml deleted file mode 100644 index adcf4ac6145..00000000000 --- a/providers/aihubmix/models/gemma-4-31b-it.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "google/gemma-4-31b-it" -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "high"] - -[cost] -input = 0.14 -output = 0.39998 - -[limit] -output = 131_100 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/glm-4.5v.toml b/providers/aihubmix/models/glm-4.5v.toml deleted file mode 100644 index 93d89be4fbd..00000000000 --- a/providers/aihubmix/models/glm-4.5v.toml +++ /dev/null @@ -1,21 +0,0 @@ -base_model = "zhipuai/glm-4.5v" -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0.274 -output = 0.822 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.548 -output = 1.644 - -[limit] -context = 65_536 diff --git a/providers/aihubmix/models/glm-4.6.toml b/providers/aihubmix/models/glm-4.6.toml deleted file mode 100644 index 17741f5557e..00000000000 --- a/providers/aihubmix/models/glm-4.6.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "zhipuai/glm-4.6" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.273974 -output = 1.095896 -cache_read = 0.054795 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.547946 -output = 2.191784 -cache_read = 0.109589 - -[limit] -context = 200_000 diff --git a/providers/aihubmix/models/glm-4.6v.toml b/providers/aihubmix/models/glm-4.6v.toml deleted file mode 100644 index f3deb7ff53f..00000000000 --- a/providers/aihubmix/models/glm-4.6v.toml +++ /dev/null @@ -1,23 +0,0 @@ -base_model = "zhipuai/glm-4.6v" -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0.137 -output = 0.411 -cache_read = 0.0274 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.274 -output = 0.822 -cache_read = 0.0548 - -[limit] -context = 131_072 diff --git a/providers/aihubmix/models/glm-4.7-flash-free.toml b/providers/aihubmix/models/glm-4.7-flash-free.toml deleted file mode 100644 index c7baa849a9b..00000000000 --- a/providers/aihubmix/models/glm-4.7-flash-free.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "zhipuai/glm-4.7-flash" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/glm-4.7.toml b/providers/aihubmix/models/glm-4.7.toml deleted file mode 100644 index 302a2bd98b4..00000000000 --- a/providers/aihubmix/models/glm-4.7.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "zhipuai/glm-4.7" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.273974 -output = 1.095896 -cache_read = 0.054795 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.547946 -output = 2.191784 -cache_read = 0.109589 - -[limit] -context = 200_000 diff --git a/providers/aihubmix/models/glm-5-turbo.toml b/providers/aihubmix/models/glm-5-turbo.toml deleted file mode 100644 index aa300d4a5bd..00000000000 --- a/providers/aihubmix/models/glm-5-turbo.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "zhipuai/glm-5-turbo" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 1.2 -output = 3.9996 -cache_read = 0.24 - -[limit] -context = 204_800 diff --git a/providers/aihubmix/models/glm-5.1.toml b/providers/aihubmix/models/glm-5.1.toml deleted file mode 100644 index c357a3506f4..00000000000 --- a/providers/aihubmix/models/glm-5.1.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "zhipuai/glm-5.1" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.845 -output = 3.38 -cache_read = 0.183112 diff --git a/providers/aihubmix/models/glm-5.2-fast-preview.toml b/providers/aihubmix/models/glm-5.2-fast-preview.toml deleted file mode 100644 index eb6b0815ff3..00000000000 --- a/providers/aihubmix/models/glm-5.2-fast-preview.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "zhipuai/glm-5.2" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 2.254 -output = 7.889 -cache_read = 0.5635 diff --git a/providers/aihubmix/models/glm-5.2.toml b/providers/aihubmix/models/glm-5.2.toml index 7a62d7aa858..6fb0761ec4c 100644 --- a/providers/aihubmix/models/glm-5.2.toml +++ b/providers/aihubmix/models/glm-5.2.toml @@ -1,16 +1,21 @@ base_model = "zhipuai/glm-5.2" -[interleaved] -field = "reasoning_content" - -[[reasoning_options]] -type = "toggle" - [[reasoning_options]] type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] +values = ["high", "max"] + +[interleaved] +field = "reasoning_content" [cost] input = 1.1268 output = 3.9438 cache_read = 0.2817 + +[limit] +context = 1_000_000 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/glm-5.3-flash.toml b/providers/aihubmix/models/glm-5.3-flash.toml index 11124a3e9e9..58250ec9ab4 100644 --- a/providers/aihubmix/models/glm-5.3-flash.toml +++ b/providers/aihubmix/models/glm-5.3-flash.toml @@ -1,22 +1,17 @@ base_model = "zhipuai/glm-5.3-flash" +reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - [cost] input = 0.11268 output = 0.39438 cache_read = 0.02817 [limit] -context = 1_048_576 +output = 128_000 [modalities] input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/glm-5.3.toml b/providers/aihubmix/models/glm-5.3.toml index 5d67e72462b..f9de29a855e 100644 --- a/providers/aihubmix/models/glm-5.3.toml +++ b/providers/aihubmix/models/glm-5.3.toml @@ -1,19 +1,13 @@ base_model = "zhipuai/glm-5.3" +reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - [cost] input = 1.1268 output = 3.9438 cache_read = 0.2817 [limit] -context = 1_048_576 +output = 128_000 diff --git a/providers/aihubmix/models/glm-5.toml b/providers/aihubmix/models/glm-5.toml deleted file mode 100644 index 015f06590c4..00000000000 --- a/providers/aihubmix/models/glm-5.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "zhipuai/glm-5" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.88 -output = 2.816 -cache_read = 0.176 - -[limit] -context = 200_000 diff --git a/providers/aihubmix/models/glm-5v-turbo.toml b/providers/aihubmix/models/glm-5v-turbo.toml index e7f491d8196..32398231e9d 100644 --- a/providers/aihubmix/models/glm-5v-turbo.toml +++ b/providers/aihubmix/models/glm-5v-turbo.toml @@ -1,24 +1,28 @@ -base_model = "zhipuai/glm-5v-turbo" name = "GLM 5 Vision Turbo" description = "GLM vision model for visual reasoning, documents, and multimodal agents" +family = "glmv" +release_date = "2026-05-09" +last_updated = "2026-05-09" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }] +temperature = true +tool_call = true structured_output = true +open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 0.7042 output = 3.09848 cache_read = 0.169008 -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.986 -output = 3.662004 -cache_read = 0.253402 +[limit] +context = 200_000 +output = 128_000 [modalities] input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-4.1-free.toml b/providers/aihubmix/models/gpt-4.1-free.toml deleted file mode 100644 index 7fa52106de5..00000000000 --- a/providers/aihubmix/models/gpt-4.1-free.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-4.1" - -[cost] -input = 0 -output = 0 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4.1-mini-free.toml b/providers/aihubmix/models/gpt-4.1-mini-free.toml deleted file mode 100644 index bba3f71945f..00000000000 --- a/providers/aihubmix/models/gpt-4.1-mini-free.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-4.1-mini" - -[cost] -input = 0 -output = 0 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4.1-mini.toml b/providers/aihubmix/models/gpt-4.1-mini.toml deleted file mode 100644 index 27124c07c4d..00000000000 --- a/providers/aihubmix/models/gpt-4.1-mini.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "openai/gpt-4.1-mini" - -[cost] -input = 0.4 -output = 1.6 -cache_read = 0.1 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4.1-nano-free.toml b/providers/aihubmix/models/gpt-4.1-nano-free.toml deleted file mode 100644 index 75bcde2c355..00000000000 --- a/providers/aihubmix/models/gpt-4.1-nano-free.toml +++ /dev/null @@ -1,5 +0,0 @@ -base_model = "openai/gpt-4.1-nano" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/gpt-4.1-nano.toml b/providers/aihubmix/models/gpt-4.1-nano.toml deleted file mode 100644 index f85c0947701..00000000000 --- a/providers/aihubmix/models/gpt-4.1-nano.toml +++ /dev/null @@ -1,6 +0,0 @@ -base_model = "openai/gpt-4.1-nano" - -[cost] -input = 0.1 -output = 0.4 -cache_read = 0.025 diff --git a/providers/aihubmix/models/gpt-4.1.toml b/providers/aihubmix/models/gpt-4.1.toml deleted file mode 100644 index f618ee47235..00000000000 --- a/providers/aihubmix/models/gpt-4.1.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "openai/gpt-4.1" - -[cost] -input = 2 -output = 8 -cache_read = 0.5 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4o-2024-11-20.toml b/providers/aihubmix/models/gpt-4o-2024-11-20.toml deleted file mode 100644 index db72e0c8807..00000000000 --- a/providers/aihubmix/models/gpt-4o-2024-11-20.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "openai/gpt-4o-2024-11-20" -tool_call = false - -[cost] -input = 2.5 -output = 10 -cache_read = 1.25 diff --git a/providers/aihubmix/models/gpt-4o-free.toml b/providers/aihubmix/models/gpt-4o-free.toml deleted file mode 100644 index f87a9a8d6d3..00000000000 --- a/providers/aihubmix/models/gpt-4o-free.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-4o" - -[cost] -input = 0 -output = 0 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4o-mini.toml b/providers/aihubmix/models/gpt-4o-mini.toml deleted file mode 100644 index d7bafa9cf42..00000000000 --- a/providers/aihubmix/models/gpt-4o-mini.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "openai/gpt-4o-mini" -tool_call = false - -[cost] -input = 0.15 -output = 0.6 -cache_read = 0.075 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-4o.toml b/providers/aihubmix/models/gpt-4o.toml deleted file mode 100644 index eb571296634..00000000000 --- a/providers/aihubmix/models/gpt-4o.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "openai/gpt-4o" - -[cost] -input = 2.5 -output = 10 -cache_read = 1.25 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5-chat-latest.toml b/providers/aihubmix/models/gpt-5-chat-latest.toml deleted file mode 100644 index e3a239e5180..00000000000 --- a/providers/aihubmix/models/gpt-5-chat-latest.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/gpt-5-chat-latest" -base_model_omit = ["limit.input"] -reasoning = false - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 - -[limit] -context = 128_000 -output = 16_384 diff --git a/providers/aihubmix/models/gpt-5-codex.toml b/providers/aihubmix/models/gpt-5-codex.toml deleted file mode 100644 index 2b9e666df4d..00000000000 --- a/providers/aihubmix/models/gpt-5-codex.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-5-codex" -attachment = true -reasoning_options = [] - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 diff --git a/providers/aihubmix/models/gpt-5-mini.toml b/providers/aihubmix/models/gpt-5-mini.toml deleted file mode 100644 index ec2e0406779..00000000000 --- a/providers/aihubmix/models/gpt-5-mini.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "openai/gpt-5-mini" -reasoning_options = [] - -[cost] -input = 0.25 -output = 2 -cache_read = 0.025 diff --git a/providers/aihubmix/models/gpt-5-nano.toml b/providers/aihubmix/models/gpt-5-nano.toml deleted file mode 100644 index 35e4f5d9e48..00000000000 --- a/providers/aihubmix/models/gpt-5-nano.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "openai/gpt-5-nano" -reasoning_options = [] - -[cost] -input = 0.05 -output = 0.4 -cache_read = 0.005 diff --git a/providers/aihubmix/models/gpt-5-pro.toml b/providers/aihubmix/models/gpt-5-pro.toml deleted file mode 100644 index 3d221044701..00000000000 --- a/providers/aihubmix/models/gpt-5-pro.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "openai/gpt-5-pro" - -[[reasoning_options]] -type = "effort" -values = ["high"] - -[cost] -input = 15 -output = 120 diff --git a/providers/aihubmix/models/gpt-5.1-chat-latest.toml b/providers/aihubmix/models/gpt-5.1-chat-latest.toml deleted file mode 100644 index d422acc79fc..00000000000 --- a/providers/aihubmix/models/gpt-5.1-chat-latest.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "openai/gpt-5.1-chat-latest" -reasoning = false - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 diff --git a/providers/aihubmix/models/gpt-5.1-codex-max.toml b/providers/aihubmix/models/gpt-5.1-codex-max.toml deleted file mode 100644 index a137a421adb..00000000000 --- a/providers/aihubmix/models/gpt-5.1-codex-max.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "openai/gpt-5.1-codex-max" -reasoning_options = [] - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 diff --git a/providers/aihubmix/models/gpt-5.1-codex-mini.toml b/providers/aihubmix/models/gpt-5.1-codex-mini.toml index abc2b111223..dea686856b0 100644 --- a/providers/aihubmix/models/gpt-5.1-codex-mini.toml +++ b/providers/aihubmix/models/gpt-5.1-codex-mini.toml @@ -1,11 +1,27 @@ -base_model = "openai/gpt-5.1-codex-mini" +name = "GPT-5.1 Codex mini" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] +family = "gpt-codex" +release_date = "2025-11-13" +last_updated = "2025-11-13" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] +temperature = false +knowledge = "2024-09-30" +tool_call = true +structured_output = true +open_weights = false [cost] input = 0.25 -output = 2 +output = 2.00 cache_read = 0.025 + +[limit] +context = 400_000 +input = 272_000 +output = 128_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.1-codex.toml b/providers/aihubmix/models/gpt-5.1-codex.toml index 63ed5678cb6..be453c3b691 100644 --- a/providers/aihubmix/models/gpt-5.1-codex.toml +++ b/providers/aihubmix/models/gpt-5.1-codex.toml @@ -1,11 +1,27 @@ -base_model = "openai/gpt-5.1-codex" +name = "GPT-5.1 Codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] +family = "gpt-codex" +release_date = "2025-11-13" +last_updated = "2025-11-13" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] +temperature = false +knowledge = "2024-09-30" +tool_call = true +structured_output = true +open_weights = false [cost] input = 1.25 -output = 10 +output = 10.00 cache_read = 0.125 + +[limit] +context = 400_000 +input = 272_000 +output = 128_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.1.toml b/providers/aihubmix/models/gpt-5.1.toml index 68263b27fb6..d707b927063 100644 --- a/providers/aihubmix/models/gpt-5.1.toml +++ b/providers/aihubmix/models/gpt-5.1.toml @@ -1,12 +1,27 @@ -base_model = "openai/gpt-5.1" +name = "GPT-5.1" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" +family = "gpt" +release_date = "2025-11-13" +last_updated = "2025-11-13" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] temperature = false - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high"] +knowledge = "2024-09-30" +tool_call = true +structured_output = true +open_weights = false [cost] input = 1.25 -output = 10 -cache_read = 0.125 +output = 10.00 +cache_read = 0.13 + +[limit] +context = 400_000 +input = 272_000 +output = 128_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.2-chat-latest.toml b/providers/aihubmix/models/gpt-5.2-chat-latest.toml deleted file mode 100644 index 6218a841ec7..00000000000 --- a/providers/aihubmix/models/gpt-5.2-chat-latest.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "openai/gpt-5.2-chat-latest" -reasoning = false - -[cost] -input = 1.75 -output = 14 -cache_read = 0.175 diff --git a/providers/aihubmix/models/gpt-5.2-codex.toml b/providers/aihubmix/models/gpt-5.2-codex.toml index a8725acddb2..3e67be5f408 100644 --- a/providers/aihubmix/models/gpt-5.2-codex.toml +++ b/providers/aihubmix/models/gpt-5.2-codex.toml @@ -1,11 +1,27 @@ -base_model = "openai/gpt-5.2-codex" +name = "GPT-5.2 Codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh"] +family = "gpt-codex" +release_date = "2025-12-11" +last_updated = "2025-12-11" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] +temperature = false +knowledge = "2025-08-31" +tool_call = true +structured_output = true +open_weights = false [cost] input = 1.75 -output = 14 +output = 14.00 cache_read = 0.175 + +[limit] +context = 400_000 +input = 272_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.2-high.toml b/providers/aihubmix/models/gpt-5.2-high.toml deleted file mode 100644 index d47568fd4e9..00000000000 --- a/providers/aihubmix/models/gpt-5.2-high.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "openai/gpt-5.2" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 1.75 -output = 14 -cache_read = 0.175 diff --git a/providers/aihubmix/models/gpt-5.2-low.toml b/providers/aihubmix/models/gpt-5.2-low.toml deleted file mode 100644 index d47568fd4e9..00000000000 --- a/providers/aihubmix/models/gpt-5.2-low.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "openai/gpt-5.2" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 1.75 -output = 14 -cache_read = 0.175 diff --git a/providers/aihubmix/models/gpt-5.2-pro.toml b/providers/aihubmix/models/gpt-5.2-pro.toml deleted file mode 100644 index ea3515497a5..00000000000 --- a/providers/aihubmix/models/gpt-5.2-pro.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "openai/gpt-5.2-pro" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["medium", "high", "xhigh"] - -[cost] -input = 21 -output = 168 -cache_read = 2.1 diff --git a/providers/aihubmix/models/gpt-5.2.toml b/providers/aihubmix/models/gpt-5.2.toml index 5f6de738b50..ee922648878 100644 --- a/providers/aihubmix/models/gpt-5.2.toml +++ b/providers/aihubmix/models/gpt-5.2.toml @@ -1,12 +1,27 @@ -base_model = "openai/gpt-5.2" +name = "GPT-5.2" description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" +family = "gpt" +release_date = "2025-12-11" +last_updated = "2025-12-11" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] temperature = false - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh"] +knowledge = "2025-08-31" +tool_call = true +structured_output = true +open_weights = false [cost] input = 1.75 -output = 14 +output = 14.00 cache_read = 0.175 + +[limit] +context = 400_000 +input = 272_000 +output = 128_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.3-chat-latest.toml b/providers/aihubmix/models/gpt-5.3-chat-latest.toml deleted file mode 100644 index 197284ef5f0..00000000000 --- a/providers/aihubmix/models/gpt-5.3-chat-latest.toml +++ /dev/null @@ -1,6 +0,0 @@ -base_model = "openai/gpt-5.3-chat-latest" - -[cost] -input = 1.75 -output = 14 -cache_read = 0.175 diff --git a/providers/aihubmix/models/gpt-5.3-codex.toml b/providers/aihubmix/models/gpt-5.3-codex.toml index d58667f384b..82ef5c400e3 100644 --- a/providers/aihubmix/models/gpt-5.3-codex.toml +++ b/providers/aihubmix/models/gpt-5.3-codex.toml @@ -1,12 +1,27 @@ -base_model = "openai/gpt-5.3-codex" +name = "GPT-5.3 Codex" description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" +family = "gpt-codex" +release_date = "2026-02-05" +last_updated = "2026-02-05" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] temperature = false - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh"] +knowledge = "2025-08-31" +tool_call = true +structured_output = true +open_weights = false [cost] input = 1.75 -output = 14 +output = 14.00 cache_read = 0.175 + +[limit] +context = 400_000 +input = 272_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.4-mini.toml b/providers/aihubmix/models/gpt-5.4-mini.toml index afd01cdd821..5697dfff19f 100644 --- a/providers/aihubmix/models/gpt-5.4-mini.toml +++ b/providers/aihubmix/models/gpt-5.4-mini.toml @@ -1,12 +1,31 @@ -base_model = "openai/gpt-5.4-mini" +name = "GPT-5.4 mini" description = "Compact GPT model for low-latency assistance and high-volume workloads" +family = "gpt-mini" +release_date = "2026-03-17" +last_updated = "2026-03-17" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] temperature = false - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh"] +knowledge = "2025-08-31" +tool_call = true +structured_output = true +open_weights = false [cost] input = 0.75 -output = 4.5 +output = 4.50 cache_read = 0.075 + +[limit] +context = 400_000 +input = 272_000 +output = 128_000 + +[modalities] +input = ["text", "image"] +output = ["text"] + +[experimental.modes.fast] +cost = { input = 1.50, output = 9.00, cache_read = 0.15 } +provider = { body = { service_tier = "priority" } } diff --git a/providers/aihubmix/models/gpt-5.4-nano.toml b/providers/aihubmix/models/gpt-5.4-nano.toml deleted file mode 100644 index 8d6d5218295..00000000000 --- a/providers/aihubmix/models/gpt-5.4-nano.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "openai/gpt-5.4-nano" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh"] - -[cost] -input = 0.2 -output = 1.25 -cache_read = 0.02 diff --git a/providers/aihubmix/models/gpt-5.4-pro.toml b/providers/aihubmix/models/gpt-5.4-pro.toml deleted file mode 100644 index 6765bbb4917..00000000000 --- a/providers/aihubmix/models/gpt-5.4-pro.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "openai/gpt-5.4-pro" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["medium", "high", "xhigh"] - -[cost] -input = 30 -output = 180 - -[[cost.tiers]] -tier = { type = "context", size = 272_000 } -input = 60 -output = 270 diff --git a/providers/aihubmix/models/gpt-5.4.toml b/providers/aihubmix/models/gpt-5.4.toml index 52b1382fb53..1ee0c18fe50 100644 --- a/providers/aihubmix/models/gpt-5.4.toml +++ b/providers/aihubmix/models/gpt-5.4.toml @@ -1,18 +1,37 @@ -base_model = "openai/gpt-5.4" +name = "GPT-5.4" description = "Frontier GPT model for professional reasoning, coding, and multimodal work" +family = "gpt" +release_date = "2026-03-05" +last_updated = "2026-03-05" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] temperature = false - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh"] +knowledge = "2025-08-31" +tool_call = true +structured_output = true +open_weights = false [cost] -input = 2.5 -output = 15 +input = 2.50 +output = 15.00 cache_read = 0.25 [[cost.tiers]] -tier = { type = "context", size = 272_000 } -input = 5 -output = 22.5 -cache_read = 0.5 +tier = { size = 272_000 } +input = 5.00 +output = 22.50 +cache_read = 0.50 + +[limit] +context = 1_050_000 +input = 922_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] + +[experimental.modes.fast] +cost = { input = 5.00, output = 30.00, cache_read = 0.50 } +provider = { body = { service_tier = "priority" } } diff --git a/providers/aihubmix/models/gpt-5.5-free.toml b/providers/aihubmix/models/gpt-5.5-free.toml deleted file mode 100644 index dc9fead01bb..00000000000 --- a/providers/aihubmix/models/gpt-5.5-free.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "openai/gpt-5.5" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh"] - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/gpt-5.5-pro.toml b/providers/aihubmix/models/gpt-5.5-pro.toml deleted file mode 100644 index fbf1ce5c232..00000000000 --- a/providers/aihubmix/models/gpt-5.5-pro.toml +++ /dev/null @@ -1,17 +0,0 @@ -base_model = "openai/gpt-5.5-pro" - -[[reasoning_options]] -type = "effort" -values = ["medium", "high", "xhigh"] - -[cost] -input = 30 -output = 180 - -[[cost.tiers]] -tier = { type = "context", size = 272_000 } -input = 60 -output = 270 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.5.toml b/providers/aihubmix/models/gpt-5.5.toml index 965823dab6d..ad2fe4a56d7 100644 --- a/providers/aihubmix/models/gpt-5.5.toml +++ b/providers/aihubmix/models/gpt-5.5.toml @@ -1,17 +1,37 @@ -base_model = "openai/gpt-5.5" +name = "GPT-5.5" description = "Frontier GPT model for professional reasoning, coding, and multimodal work" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh"] +family = "gpt" +release_date = "2026-04-23" +last_updated = "2026-04-23" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh"] }] +temperature = false +knowledge = "2025-12-01" +tool_call = true +structured_output = true +open_weights = false [cost] -input = 5 -output = 30 -cache_read = 0.5 +input = 5.00 +output = 30.00 +cache_read = 0.50 [[cost.tiers]] -tier = { type = "context", size = 272_000 } -input = 10 -output = 45 -cache_read = 1 +tier = { size = 272_000 } +input = 10.00 +output = 45.00 +cache_read = 1.00 + +[limit] +context = 1_050_000 +input = 922_000 +output = 128_000 + +[modalities] +input = ["text", "image", "pdf"] +output = ["text"] + +[experimental.modes.fast] +cost = { input = 12.50, output = 75.00, cache_read = 1.25 } +provider = { body = { service_tier = "priority" } } diff --git a/providers/aihubmix/models/gpt-5.6-luna.toml b/providers/aihubmix/models/gpt-5.6-luna.toml index 6271edcdca5..7213c25165e 100644 --- a/providers/aihubmix/models/gpt-5.6-luna.toml +++ b/providers/aihubmix/models/gpt-5.6-luna.toml @@ -1,21 +1,17 @@ base_model = "openai/gpt-5.6-luna" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh", "max"] +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +temperature = false [cost] -input = 0.2 -output = 1.2 -cache_read = 0.02 -cache_write = 0.25 +input = 1 +output = 6 +cache_read = 0.1 +cache_write = 1.25 -[[cost.tiers]] -tier = { type = "context", size = 272_000 } -input = 0.4 -output = 1.8 -cache_read = 0.04 -cache_write = 0.5 +[limit] +context = 1_050_000 +output = 128_000 [modalities] input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.6-sol-disc.toml b/providers/aihubmix/models/gpt-5.6-sol-disc.toml deleted file mode 100644 index ceb6dc11d8c..00000000000 --- a/providers/aihubmix/models/gpt-5.6-sol-disc.toml +++ /dev/null @@ -1,21 +0,0 @@ -base_model = "openai/gpt-5.6-sol" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 4 -output = 20 -cache_read = 0.4 -cache_write = 5 - -[[cost.tiers]] -tier = { type = "context", size = 272_000 } -input = 8 -output = 30 -cache_read = 0.8 -cache_write = 10 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-5.6-sol.toml b/providers/aihubmix/models/gpt-5.6-sol.toml index ceb6dc11d8c..2d0fb532e7e 100644 --- a/providers/aihubmix/models/gpt-5.6-sol.toml +++ b/providers/aihubmix/models/gpt-5.6-sol.toml @@ -1,21 +1,17 @@ base_model = "openai/gpt-5.6-sol" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh", "max"] +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +temperature = false [cost] -input = 4 -output = 20 -cache_read = 0.4 -cache_write = 5 - -[[cost.tiers]] -tier = { type = "context", size = 272_000 } -input = 8 +input = 5 output = 30 -cache_read = 0.8 -cache_write = 10 +cache_read = 0.5 +cache_write = 6.25 + +[limit] +context = 1_050_000 +output = 128_000 [modalities] input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.6-terra.toml b/providers/aihubmix/models/gpt-5.6-terra.toml index 2236c6b23c2..1c68d95d455 100644 --- a/providers/aihubmix/models/gpt-5.6-terra.toml +++ b/providers/aihubmix/models/gpt-5.6-terra.toml @@ -1,21 +1,17 @@ base_model = "openai/gpt-5.6-terra" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high", "xhigh", "max"] +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] +temperature = false [cost] -input = 2 -output = 12 -cache_read = 0.2 -cache_write = 2.5 +input = 2.5 +output = 15 +cache_read = 0.25 +cache_write = 3.125 -[[cost.tiers]] -tier = { type = "context", size = 272_000 } -input = 4 -output = 18 -cache_read = 0.4 -cache_write = 5 +[limit] +context = 1_050_000 +output = 128_000 [modalities] input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/gpt-5.toml b/providers/aihubmix/models/gpt-5.toml deleted file mode 100644 index f1d834f171a..00000000000 --- a/providers/aihubmix/models/gpt-5.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "openai/gpt-5" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 1.25 -output = 10 -cache_read = 0.125 diff --git a/providers/aihubmix/models/gpt-6-astra.toml b/providers/aihubmix/models/gpt-6-astra.toml deleted file mode 100644 index 3f1fcc6d753..00000000000 --- a/providers/aihubmix/models/gpt-6-astra.toml +++ /dev/null @@ -1,21 +0,0 @@ -base_model = "openai/gpt-6-astra" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh", "max"] - -[cost] -input = 10 -output = 50 -cache_read = 1 -cache_write = 12.5 - -[[cost.tiers]] -tier = { type = "context", size = 272_000 } -input = 20 -output = 75 -cache_read = 2 -cache_write = 25 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/gpt-chat-latest.toml b/providers/aihubmix/models/gpt-chat-latest.toml deleted file mode 100644 index 66580e968a4..00000000000 --- a/providers/aihubmix/models/gpt-chat-latest.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "GPT Chat" -description = "GPT Chat Latest points to OpenAI's stable API alias chat-latest that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates in the future, they are routed behind this slug automatically." -release_date = "2026-05-05" -last_updated = "2026-05-05" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = false -reasoning_options = [] - -[cost] -input = 5 -output = 30 -cache_read = 0.5 - -[[cost.tiers]] -tier = { type = "context", size = 272_000 } -input = 10 -output = 45 -cache_read = 1 - -[limit] -context = 400_000 -output = 128_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/gpt-image-2-free.toml b/providers/aihubmix/models/gpt-image-2-free.toml deleted file mode 100644 index 63bece6ae44..00000000000 --- a/providers/aihubmix/models/gpt-image-2-free.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "openai/gpt-image-2" - -[cost] -input = 0 -output = 0 - -[modalities] -output = ["image", "text"] diff --git a/providers/aihubmix/models/gpt-oss-120b.toml b/providers/aihubmix/models/gpt-oss-120b.toml deleted file mode 100644 index afef8b74204..00000000000 --- a/providers/aihubmix/models/gpt-oss-120b.toml +++ /dev/null @@ -1,6 +0,0 @@ -base_model = "openai/gpt-oss-120b" -reasoning_options = [] - -[cost] -input = 0.18 -output = 0.9 diff --git a/providers/aihubmix/models/gpt-oss-20b-free.toml b/providers/aihubmix/models/gpt-oss-20b-free.toml deleted file mode 100644 index 9ddc252b4d6..00000000000 --- a/providers/aihubmix/models/gpt-oss-20b-free.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/gpt-oss-20b" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 0 -output = 0 - -[limit] -output = 131_072 diff --git a/providers/aihubmix/models/gpt-oss-20b.toml b/providers/aihubmix/models/gpt-oss-20b.toml deleted file mode 100644 index e878de3f58e..00000000000 --- a/providers/aihubmix/models/gpt-oss-20b.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "openai/gpt-oss-20b" -reasoning_options = [] - -[cost] -input = 0.11 -output = 0.55 - -[limit] -context = 128_000 diff --git a/providers/aihubmix/models/grok-4-1-fast-non-reasoning.toml b/providers/aihubmix/models/grok-4-1-fast-non-reasoning.toml deleted file mode 100644 index 3e383f20ae2..00000000000 --- a/providers/aihubmix/models/grok-4-1-fast-non-reasoning.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "xai/grok-4.3" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.2 -output = 0.5 -cache_read = 0.05 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/grok-4-1-fast-reasoning.toml b/providers/aihubmix/models/grok-4-1-fast-reasoning.toml deleted file mode 100644 index 18531f0d119..00000000000 --- a/providers/aihubmix/models/grok-4-1-fast-reasoning.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "xai/grok-4.3" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high"] - -[cost] -input = 0.2 -output = 0.5 -cache_read = 0.05 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/grok-4-fast-non-reasoning.toml b/providers/aihubmix/models/grok-4-fast-non-reasoning.toml deleted file mode 100644 index f3fddee84df..00000000000 --- a/providers/aihubmix/models/grok-4-fast-non-reasoning.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "xai/grok-4.3" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[cost] -input = 0.2 -output = 0.5 -cache_read = 0.05 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.4 -output = 1 -cache_read = 0.05 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/grok-4-fast-reasoning.toml b/providers/aihubmix/models/grok-4-fast-reasoning.toml deleted file mode 100644 index 81fcf48ef9c..00000000000 --- a/providers/aihubmix/models/grok-4-fast-reasoning.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "xai/grok-4.3" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high"] - -[cost] -input = 0.2 -output = 0.5 -cache_read = 0.05 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.4 -output = 1 -cache_read = 0.05 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/grok-4.3.toml b/providers/aihubmix/models/grok-4.3.toml index d40e8180f6c..82cee7b576b 100644 --- a/providers/aihubmix/models/grok-4.3.toml +++ b/providers/aihubmix/models/grok-4.3.toml @@ -1,9 +1,15 @@ -base_model = "xai/grok-4.3" +name = "Grok 4.3" description = "Grok model for agentic tool use, reasoning, coding, and live assistance" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high"] +family = "grok" +release_date = "2026-05-01" +last_updated = "2026-05-01" +attachment = true +reasoning = true +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] +temperature = true +tool_call = true +structured_output = true +open_weights = false [cost] input = 1.25 @@ -11,10 +17,15 @@ output = 2.5 cache_read = 0.2 [[cost.tiers]] -tier = { type = "context", size = 200_000 } +tier = { size = 200_000 } input = 2.5 -output = 5 +output = 5.0 cache_read = 0.4 +[limit] +context = 1_000_000 +output = 1_000_000 + [modalities] input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/grok-4.5.toml b/providers/aihubmix/models/grok-4.5.toml index 1fac2993623..9545ba4a2d5 100644 --- a/providers/aihubmix/models/grok-4.5.toml +++ b/providers/aihubmix/models/grok-4.5.toml @@ -1,16 +1,15 @@ base_model = "xai/grok-4.5" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] +reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }] [cost] input = 2 output = 6 cache_read = 0.5 -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 4.4 -output = 13.2 -cache_read = 1.1 +[limit] +context = 1_000_000 +output = 1_000_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/grok-4.6.toml b/providers/aihubmix/models/grok-4.6.toml index fd607334f23..5264268a96f 100644 --- a/providers/aihubmix/models/grok-4.6.toml +++ b/providers/aihubmix/models/grok-4.6.toml @@ -1,16 +1,7 @@ base_model = "xai/grok-4.6" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high", "xhigh"] +reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }] [cost] input = 2 output = 6 cache_read = 0.5 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 4.4 -output = 13.2 -cache_read = 1.1 diff --git a/providers/aihubmix/models/grok-4.toml b/providers/aihubmix/models/grok-4.toml deleted file mode 100644 index 9bd66831f47..00000000000 --- a/providers/aihubmix/models/grok-4.toml +++ /dev/null @@ -1,26 +0,0 @@ -name = "Grok 4" -description = "Grok, their latest and greatest flagship model, offers unparalleled performance in natural language, math, and reasoning – the perfect jack of all trades.\nThe current pointing model version is grok-4-0709." -release_date = "2025-07-09" -last_updated = "2025-07-09" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = false - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "medium", "high"] - -[cost] -input = 3.3 -output = 16.5 -cache_read = 0.825 - -[limit] -context = 256_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/grok-build-0.1.toml b/providers/aihubmix/models/grok-build-0.1.toml index b12f06edb15..ff7db61874c 100644 --- a/providers/aihubmix/models/grok-build-0.1.toml +++ b/providers/aihubmix/models/grok-build-0.1.toml @@ -6,11 +6,10 @@ input = 1 output = 2 cache_read = 0.2 -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2 -output = 2 -cache_read = 0.4 +[limit] +context = 256_000 +output = 256_000 [modalities] input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/grok-code-fast-1.toml b/providers/aihubmix/models/grok-code-fast-1.toml deleted file mode 100644 index bd540da78c4..00000000000 --- a/providers/aihubmix/models/grok-code-fast-1.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "xai/grok-build-0.1" -reasoning_options = [] - -[cost] -input = 1 -output = 2 -cache_read = 0.2 - -[[cost.tiers]] -tier = { type = "context", size = 200_000 } -input = 2 -output = 2 -cache_read = 0.4 - -[limit] -output = 10_000 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/hy3-free.toml b/providers/aihubmix/models/hy3-free.toml deleted file mode 100644 index 0fcbaf489e2..00000000000 --- a/providers/aihubmix/models/hy3-free.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "tencent/hy3" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "high"] - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/hy3-preview.toml b/providers/aihubmix/models/hy3-preview.toml index 91e8ba00aea..0ea62d7c422 100644 --- a/providers/aihubmix/models/hy3-preview.toml +++ b/providers/aihubmix/models/hy3-preview.toml @@ -1,30 +1,17 @@ base_model = "tencent/hy3-preview" name = "Hy3 Preview" structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "high"] +reasoning_options = [{ type = "effort", values = ["none", "low", "high"] }] [cost] input = 0.17 output = 0.566661 cache_read = 0.051 -[[cost.tiers]] -tier = { type = "context", size = 16_000 } -input = 0.2254 -output = 0.9016 -cache_read = 0.084525 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.2818 -output = 1.1272 -cache_read = 0.11272 - [limit] +context = 256_000 output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/hy3.toml b/providers/aihubmix/models/hy3.toml deleted file mode 100644 index 38502dbd27a..00000000000 --- a/providers/aihubmix/models/hy3.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "tencent/hy3" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "low", "high"] - -[cost] -input = 0.1562 -output = 0.6248 -cache_read = 0.03905 diff --git a/providers/aihubmix/models/hy4-preview.toml b/providers/aihubmix/models/hy4-preview.toml deleted file mode 100644 index ab83f489ac5..00000000000 --- a/providers/aihubmix/models/hy4-preview.toml +++ /dev/null @@ -1,20 +0,0 @@ -base_model = "tencent/hy4-preview" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.845 -output = 2.535 -cache_read = 0.04225 - -[limit] -context = 1_048_576 diff --git a/providers/aihubmix/models/kimi-for-coding-free.toml b/providers/aihubmix/models/kimi-for-coding-free.toml deleted file mode 100644 index 85e6a9bd86e..00000000000 --- a/providers/aihubmix/models/kimi-for-coding-free.toml +++ /dev/null @@ -1,22 +0,0 @@ -name = "Kimi For Coding (free)" -description = "kimi-for-coding-free is a free and open version offered by AIHubMix specifically for Kimi users. To maintain stable service operations, the following usage limits apply: a maximum of 5 requests per minute 500 total requests per day, and a daily quota of 1 million tokens." -release_date = "2026-06-12" -last_updated = "2026-06-12" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = true -reasoning_options = [] - -[cost] -input = 0 -output = 0 - -[limit] -context = 256_000 -output = 32_768 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2-0711.toml b/providers/aihubmix/models/kimi-k2-0711.toml deleted file mode 100644 index 65bb06c8c4f..00000000000 --- a/providers/aihubmix/models/kimi-k2-0711.toml +++ /dev/null @@ -1,21 +0,0 @@ -name = "Kimi K2 0711" -description = "Kimi-K2 is a MoE architecture foundational model with extremely powerful coding and agent capabilities, featuring a total of 1 trillion parameters and activating 32 billion parameters. In benchmark performance tests across major categories such as general knowledge reasoning, programming, mathematics, and agents, the K2 model outperforms other mainstream open-source models.\nThe Kimi-K2 model supports a context length of 128k tokens.\nIt does not support visual capabilities." -release_date = "2025-01-01" -last_updated = "2025-01-01" -attachment = false -reasoning = false -tool_call = true -structured_output = true -open_weights = true - -[cost] -input = 0.54 -output = 2.16 - -[limit] -context = 131_072 -output = 16_384 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2-thinking.toml b/providers/aihubmix/models/kimi-k2-thinking.toml deleted file mode 100644 index 9eb147d8db3..00000000000 --- a/providers/aihubmix/models/kimi-k2-thinking.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "moonshotai/kimi-k2-thinking" -structured_output = true -reasoning_options = [] - -[cost] -input = 0.548 -output = 2.192 -cache_read = 0.137 diff --git a/providers/aihubmix/models/kimi-k2-turbo-preview.toml b/providers/aihubmix/models/kimi-k2-turbo-preview.toml deleted file mode 100644 index 3cadac442fd..00000000000 --- a/providers/aihubmix/models/kimi-k2-turbo-preview.toml +++ /dev/null @@ -1,22 +0,0 @@ -name = "Kimi K2 Turbo Preview" -description = "The kimi-k2-turbo-preview model is a high-speed version of kimi-k2, with the same model parameters as kimi-k2, but the output speed has been increased from 10 tokens per second to 40 tokens per second." -release_date = "2025-07-08" -last_updated = "2025-07-08" -attachment = false -reasoning = false -tool_call = true -structured_output = true -open_weights = true - -[cost] -input = 1.2 -output = 4.8 -cache_read = 0.3 - -[limit] -context = 262_144 -output = 8_192 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2.5.toml b/providers/aihubmix/models/kimi-k2.5.toml index fbf7c92338a..ce3eca895df 100644 --- a/providers/aihubmix/models/kimi-k2.5.toml +++ b/providers/aihubmix/models/kimi-k2.5.toml @@ -1,19 +1,29 @@ -base_model = "moonshotai/kimi-k2.5" +name = "Kimi K2.5" description = "Kimi multimodal agent model for visual understanding, coding, and planning" +family = "kimi-k2" +release_date = "2026-01" +last_updated = "2026-01" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }] +temperature = false +tool_call = true +structured_output = true +knowledge = "2025-01" +open_weights = true [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 0.6 output = 3 -cache_read = 0.105 +cache_read = 0.10 [limit] +context = 262_144 output = 32_768 [modalities] -input = ["text", "image"] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2.6.toml b/providers/aihubmix/models/kimi-k2.6.toml index b352a4b2103..ddb87134c56 100644 --- a/providers/aihubmix/models/kimi-k2.6.toml +++ b/providers/aihubmix/models/kimi-k2.6.toml @@ -1,17 +1,29 @@ -base_model = "moonshotai/kimi-k2.6" +name = "Kimi K2.6" description = "Kimi multimodal agent model for visual understanding, coding, and planning" +family = "kimi-k2" +release_date = "2026-04-21" +last_updated = "2026-04-21" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }] temperature = false +tool_call = true +structured_output = true +knowledge = "2025-01" +open_weights = true [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 0.95 -output = 3.9995 -cache_read = 0.160835 +output = 4 +cache_read = 0.16 [limit] +context = 262_144 output = 32_768 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2.7-code-highspeed.toml b/providers/aihubmix/models/kimi-k2.7-code-highspeed.toml index eb3684dc6f4..a64a84d4cd5 100644 --- a/providers/aihubmix/models/kimi-k2.7-code-highspeed.toml +++ b/providers/aihubmix/models/kimi-k2.7-code-highspeed.toml @@ -1,15 +1,18 @@ base_model = "moonshotai/kimi-k2.7-code-highspeed" +reasoning_options = [{ type = "toggle" }] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 1.9 output = 7.999 cache_read = 0.32167 [limit] +context = 262_144 output = 32_768 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/kimi-k2.7-code.toml b/providers/aihubmix/models/kimi-k2.7-code.toml index edaca21f35b..529b52a2d8b 100644 --- a/providers/aihubmix/models/kimi-k2.7-code.toml +++ b/providers/aihubmix/models/kimi-k2.7-code.toml @@ -1,15 +1,18 @@ base_model = "moonshotai/kimi-k2.7-code" +reasoning_options = [{ type = "toggle" }] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 0.95 output = 3.9995 cache_read = 0.160835 [limit] +context = 262_144 output = 32_768 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/kimi-k3.toml b/providers/aihubmix/models/kimi-k3.toml index 21e178e7339..cc93d685469 100644 --- a/providers/aihubmix/models/kimi-k3.toml +++ b/providers/aihubmix/models/kimi-k3.toml @@ -2,7 +2,9 @@ # (text, vision, video), and a 1M-token context window. # Source accessed 2026-07-27: # https://aihubmix.com/model/kimi-k3 + base_model = "moonshotai/kimi-k3" +last_updated = "2026-07-27" [interleaved] field = "reasoning_content" @@ -15,9 +17,10 @@ type = "effort" values = ["low", "high", "max"] [cost] -input = 3 -output = 15 -cache_read = 0.3 +input = 3.00 +output = 15.00 +cache_read = 0.30 -[limit] -output = 1_048_576 +[modalities] +input = ["text", "image", "video"] +output = ["text"] \ No newline at end of file diff --git a/providers/aihubmix/models/laguna-s-2.1-free.toml b/providers/aihubmix/models/laguna-s-2.1-free.toml deleted file mode 100644 index e2be262fa86..00000000000 --- a/providers/aihubmix/models/laguna-s-2.1-free.toml +++ /dev/null @@ -1,23 +0,0 @@ -name = "Laguna S 2.1 (free)" -description = "Laguna S 2.1 is the latest coding agent model from Poolside, featuring an impressive context length of 262,144 tokens. This model is built with 118B total parameters and 8B active parameters, balancing efficiency with high performance. It delivers strong capabilities for developer tasks, scoring 70.2% on the Terminal-Bench 2.1 benchmark." -release_date = "2026-07-21" -last_updated = "2026-07-21" -attachment = false -reasoning = true -tool_call = true -open_weights = true - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 -output = 262_144 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/llama-3.3-70b-instruct.toml b/providers/aihubmix/models/llama-3.3-70b-instruct.toml deleted file mode 100644 index 88ae0b0bfd2..00000000000 --- a/providers/aihubmix/models/llama-3.3-70b-instruct.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "meta/llama-3.3-70b-instruct" -attachment = false -tool_call = false - -[cost] -input = 0.6 -output = 1.2 - -[limit] -context = 131_072 diff --git a/providers/aihubmix/models/longcat-2.0.toml b/providers/aihubmix/models/longcat-2.0.toml deleted file mode 100644 index 62f5ab0db00..00000000000 --- a/providers/aihubmix/models/longcat-2.0.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "meituan/longcat-2.0" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.7746 -output = 3.0984 -cache_read = 0.015492 diff --git a/providers/aihubmix/models/mercury-2.5-preview.toml b/providers/aihubmix/models/mercury-2.5-preview.toml deleted file mode 100644 index 2001a76823d..00000000000 --- a/providers/aihubmix/models/mercury-2.5-preview.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Mercury 2.5 Preview" -description = "Mercury 2.5 is the latest diffusion-based large language model (dLLM) released by Inception. It is the fastest inference LLM; unlike the sequential token-by-token generation approach, Mercury 2.5 can generate and optimize multiple tokens in parallel, achieving a generation speed of 1,107 tokens per second on standard GPUs. Compared to Mercury 2, its intelligence has increased by more than 10 percentage points, and its quality rivals leading cost-optimized frontier models such as GPT-5.6 Luna (Low), Gemini 3.5 Flash-Lite, and Claude Haiku 4.5." -release_date = "2026-09-02" -last_updated = "2026-09-02" -attachment = false -reasoning = true -tool_call = true -open_weights = false - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high"] - -[cost] -input = 0.2 -output = 0.75 -cache_read = 0.02 - -[limit] -context = 260_000 -output = 65_536 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/mimo-v2-flash-free.toml b/providers/aihubmix/models/mimo-v2-flash-free.toml deleted file mode 100644 index 6dda24ef887..00000000000 --- a/providers/aihubmix/models/mimo-v2-flash-free.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "xiaomi/mimo-v2-flash" -reasoning = false -tool_call = false - -[cost] -input = 0 -output = 0 - -[limit] -context = 256_000 diff --git a/providers/aihubmix/models/mimo-v2-flash.toml b/providers/aihubmix/models/mimo-v2-flash.toml deleted file mode 100644 index e9a8258948c..00000000000 --- a/providers/aihubmix/models/mimo-v2-flash.toml +++ /dev/null @@ -1,18 +0,0 @@ -base_model = "xiaomi/mimo-v2-flash" -attachment = true -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.1918 -output = 0.5754 -cache_read = 0.03836 - -[limit] -context = 1_048_576 -output = 131_072 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/mimo-v2-omni.toml b/providers/aihubmix/models/mimo-v2-omni.toml deleted file mode 100644 index a873d045063..00000000000 --- a/providers/aihubmix/models/mimo-v2-omni.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "xiaomi/mimo-v2-omni" -reasoning = false -tool_call = false - -[cost] -input = 0.44 -output = 2.2 -cache_read = 0.088 - -[limit] -context = 256_000 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/mimo-v2-pro.toml b/providers/aihubmix/models/mimo-v2-pro.toml deleted file mode 100644 index 71418db87b0..00000000000 --- a/providers/aihubmix/models/mimo-v2-pro.toml +++ /dev/null @@ -1,17 +0,0 @@ -base_model = "xiaomi/mimo-v2-pro" -reasoning = false -tool_call = false - -[cost] -input = 1.1 -output = 3.3 -cache_read = 0.22 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 2.2 -output = 6.6 -cache_read = 0.44 - -[limit] -context = 1_000_000 diff --git a/providers/aihubmix/models/mimo-v2.5-pro.toml b/providers/aihubmix/models/mimo-v2.5-pro.toml deleted file mode 100644 index 5075fe759a9..00000000000 --- a/providers/aihubmix/models/mimo-v2.5-pro.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "xiaomi/mimo-v2.5-pro" -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.48 -output = 0.96 -cache_read = 0.00384 diff --git a/providers/aihubmix/models/mimo-v2.5.toml b/providers/aihubmix/models/mimo-v2.5.toml deleted file mode 100644 index 86372dd670f..00000000000 --- a/providers/aihubmix/models/mimo-v2.5.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "xiaomi/mimo-v2.5" -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.155 -output = 0.31 -cache_read = 0.0031 diff --git a/providers/aihubmix/models/minimax-m2.1.toml b/providers/aihubmix/models/minimax-m2.1.toml deleted file mode 100644 index 96e0cd924d1..00000000000 --- a/providers/aihubmix/models/minimax-m2.1.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.1" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.288 -output = 1.152 diff --git a/providers/aihubmix/models/minimax-m2.5-highspeed.toml b/providers/aihubmix/models/minimax-m2.5-highspeed.toml deleted file mode 100644 index 8ee59cf05a4..00000000000 --- a/providers/aihubmix/models/minimax-m2.5-highspeed.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.5-highspeed" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.288 -output = 1.152 diff --git a/providers/aihubmix/models/minimax-m2.5.toml b/providers/aihubmix/models/minimax-m2.5.toml deleted file mode 100644 index e4e96f80dc6..00000000000 --- a/providers/aihubmix/models/minimax-m2.5.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.5" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.288 -output = 1.152 diff --git a/providers/aihubmix/models/minimax-m2.7-free.toml b/providers/aihubmix/models/minimax-m2.7-free.toml deleted file mode 100644 index a9b38c992fc..00000000000 --- a/providers/aihubmix/models/minimax-m2.7-free.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.7" -tool_call = false - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/minimax-m2.7.toml b/providers/aihubmix/models/minimax-m2.7.toml index 097633f4c4d..f00a935f81f 100644 --- a/providers/aihubmix/models/minimax-m2.7.toml +++ b/providers/aihubmix/models/minimax-m2.7.toml @@ -1,22 +1,29 @@ -base_model = "minimax/MiniMax-M2.7" name = "MiniMax M2.7" description = "MiniMax model for chat, coding, office work, and agentic tasks" +family = "minimax" +release_date = "2026-03-18" +last_updated = "2026-03-18" +attachment = false +reasoning = true +reasoning_options = [] +temperature = true +tool_call = true structured_output = true +open_weights = true [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] -input = 0.2958 -output = 1.1832 -cache_read = 0.05916 +input = 0.3 +output = 1.2 +cache_read = 0.06 +cache_write = 0.375 [limit] -output = 204_800 +context = 204_800 +output = 128_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/minimax-m2.toml b/providers/aihubmix/models/minimax-m2.toml deleted file mode 100644 index 91837b885d0..00000000000 --- a/providers/aihubmix/models/minimax-m2.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.288 -output = 1.152 diff --git a/providers/aihubmix/models/minimax-m3.toml b/providers/aihubmix/models/minimax-m3.toml deleted file mode 100644 index ee1919ca4ef..00000000000 --- a/providers/aihubmix/models/minimax-m3.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "minimax/MiniMax-M3" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.288 -output = 1.152 - -[limit] -context = 1_000_000 -output = 524_288 diff --git a/providers/aihubmix/models/mistral-large-3.toml b/providers/aihubmix/models/mistral-large-3.toml deleted file mode 100644 index cfd30382a16..00000000000 --- a/providers/aihubmix/models/mistral-large-3.toml +++ /dev/null @@ -1,21 +0,0 @@ -name = "Mistral Large 3" -description = "Mistral Large 3 is a MoE model with 67.5B total parameters and 41B active parameters, supporting a 256K-token context window. Trained from scratch on 3,000 NVIDIA H200 GPUs, it is one of the strongest permissively licensed open-weight models available.\n\nDesigned for advanced reasoning and long-context understanding, Mistral Large 3 delivers performance on par with the best instruction-tuned open-weight models for general-purpose tasks, while also offering image understanding capabilities. Its multilingual strengths are particularly notable for non-English/Chinese languages, making it well-suited for global applications.\n\nTypical use cases include enterprise assistants, multilingual customer support, content generation and editing, data analysis over long documents, code assistance, and research workflows that require handling large corpora or complex instructions. With its MoE architecture, Mistral Large 3 balances strong performance with efficient inference, providing a versatile backbone for building reliable, production-grade AI systems." -release_date = "2024-11-01" -last_updated = "2024-11-01" -attachment = true -reasoning = false -tool_call = true -structured_output = true -open_weights = true - -[cost] -input = 0.5 -output = 1.5 - -[limit] -context = 256_000 -output = 131_072 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/aihubmix/models/mm-minimax-m2.7-highspeed.toml b/providers/aihubmix/models/mm-minimax-m2.7-highspeed.toml deleted file mode 100644 index 1e66879bd49..00000000000 --- a/providers/aihubmix/models/mm-minimax-m2.7-highspeed.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "minimax/MiniMax-M2.7-highspeed" -structured_output = true - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1 -output = 0.1 diff --git a/providers/aihubmix/models/muse-spark-1.1.toml b/providers/aihubmix/models/muse-spark-1.1.toml deleted file mode 100644 index 58eb68641cf..00000000000 --- a/providers/aihubmix/models/muse-spark-1.1.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "meta/muse-spark-1.1" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high", "xhigh"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.375 -output = 4.675 - -[modalities] -input = ["text", "image", "video", "audio", "pdf"] diff --git a/providers/aihubmix/models/muse-spark-1.2.toml b/providers/aihubmix/models/muse-spark-1.2.toml deleted file mode 100644 index 86615948cd1..00000000000 --- a/providers/aihubmix/models/muse-spark-1.2.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "meta/muse-spark-1.2" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high", "xhigh"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.375 -output = 4.675 diff --git a/providers/aihubmix/models/muse-spark-1.3.toml b/providers/aihubmix/models/muse-spark-1.3.toml deleted file mode 100644 index a985b6a7e3a..00000000000 --- a/providers/aihubmix/models/muse-spark-1.3.toml +++ /dev/null @@ -1,16 +0,0 @@ -base_model = "meta/muse-spark-1.3" - -[[reasoning_options]] -type = "effort" -values = ["minimal", "low", "medium", "high", "xhigh"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.375 -output = 4.675 -cache_read = 0.165 - -[limit] -output = 1_000_000 diff --git a/providers/aihubmix/models/nemotron-3-nano-30b-a3b-free.toml b/providers/aihubmix/models/nemotron-3-nano-30b-a3b-free.toml deleted file mode 100644 index d00601f3ca4..00000000000 --- a/providers/aihubmix/models/nemotron-3-nano-30b-a3b-free.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "nvidia/nemotron-3-nano-30b-a3b" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 -output = 1_048_576 diff --git a/providers/aihubmix/models/nemotron-3-nano-omni-30b-a3b-reasoning-free.toml b/providers/aihubmix/models/nemotron-3-nano-omni-30b-a3b-reasoning-free.toml deleted file mode 100644 index 1cfeb8d900c..00000000000 --- a/providers/aihubmix/models/nemotron-3-nano-omni-30b-a3b-reasoning-free.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 - -[limit] -context = 262_144 diff --git a/providers/aihubmix/models/nemotron-3-super-120b-a12b-free.toml b/providers/aihubmix/models/nemotron-3-super-120b-a12b-free.toml deleted file mode 100644 index bbb11b6100c..00000000000 --- a/providers/aihubmix/models/nemotron-3-super-120b-a12b-free.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "nvidia/nemotron-3-super-120b-a12b" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 diff --git a/providers/aihubmix/models/nemotron-3-ultra-550b-a55b-free.toml b/providers/aihubmix/models/nemotron-3-ultra-550b-a55b-free.toml deleted file mode 100644 index 9be7df33a6e..00000000000 --- a/providers/aihubmix/models/nemotron-3-ultra-550b-a55b-free.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "nvidia/nemotron-3-ultra-550b-a55b" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 diff --git a/providers/aihubmix/models/nemotron-3.5-content-safety-free.toml b/providers/aihubmix/models/nemotron-3.5-content-safety-free.toml deleted file mode 100644 index da1f6cacd6b..00000000000 --- a/providers/aihubmix/models/nemotron-3.5-content-safety-free.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "nvidia/nemotron-3.5-content-safety" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 - -[limit] -context = 131_072 diff --git a/providers/aihubmix/models/nemotron-3.5-lightning-free.toml b/providers/aihubmix/models/nemotron-3.5-lightning-free.toml deleted file mode 100644 index 030e52ea0f4..00000000000 --- a/providers/aihubmix/models/nemotron-3.5-lightning-free.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "nvidia/nemotron-3.5-lightning" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 diff --git a/providers/aihubmix/models/nemotron-nano-12b-v2-vl-free.toml b/providers/aihubmix/models/nemotron-nano-12b-v2-vl-free.toml deleted file mode 100644 index 947e56253a9..00000000000 --- a/providers/aihubmix/models/nemotron-nano-12b-v2-vl-free.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "nvidia/nemotron-nano-12b-v2-vl" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0 -output = 0 - -[limit] -context = 131_072 diff --git a/providers/aihubmix/models/nemotron-nano-9b-v2-free.toml b/providers/aihubmix/models/nemotron-nano-9b-v2-free.toml deleted file mode 100644 index bc0512d2657..00000000000 --- a/providers/aihubmix/models/nemotron-nano-9b-v2-free.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "nvidia/nemotron-nano-9b-v2" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 diff --git a/providers/aihubmix/models/north-mini-code-free.toml b/providers/aihubmix/models/north-mini-code-free.toml deleted file mode 100644 index dee867a0ebe..00000000000 --- a/providers/aihubmix/models/north-mini-code-free.toml +++ /dev/null @@ -1,27 +0,0 @@ -name = "North Mini Code (free)" -description = "Developed by Cohere, north-mini-code-free is the debut model of the North family and Cohere's first agentic coding model. This sparse mixture-of-experts model features 30B total parameters and 3B active parameters, designed and optimized for high performance. With an expansive context length of 256,000 tokens, it is well-suited for handling complex developer workflows and large codebases." -release_date = "2026-06-09" -last_updated = "2026-06-09" -attachment = false -reasoning = true -tool_call = true -open_weights = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "high"] - -[cost] -input = 0 -output = 0 - -[limit] -context = 256_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/nvidia-nemotron-3-super-120b-a12b.toml b/providers/aihubmix/models/nvidia-nemotron-3-super-120b-a12b.toml deleted file mode 100644 index 20b31b0b7bf..00000000000 --- a/providers/aihubmix/models/nvidia-nemotron-3-super-120b-a12b.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "nvidia/nemotron-3-super-120b-a12b" -structured_output = true -reasoning_options = [] - -[cost] -input = 0.11 -output = 0.55 -cache_read = 0.0275 - -[limit] -context = 1_000_000 diff --git a/providers/aihubmix/models/o1-preview.toml b/providers/aihubmix/models/o1-preview.toml deleted file mode 100644 index df9fac6088e..00000000000 --- a/providers/aihubmix/models/o1-preview.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "openai/o1" -tool_call = false -reasoning_options = [] - -[cost] -input = 15 -output = 60 -cache_read = 7.5 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/o1-pro.toml b/providers/aihubmix/models/o1-pro.toml deleted file mode 100644 index 960adb68778..00000000000 --- a/providers/aihubmix/models/o1-pro.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/o1-pro" -attachment = false -tool_call = false -reasoning_options = [] - -[cost] -input = 170 -output = 680 -cache_read = 170 - -[modalities] -input = ["text"] diff --git a/providers/aihubmix/models/o1.toml b/providers/aihubmix/models/o1.toml deleted file mode 100644 index e560a52586b..00000000000 --- a/providers/aihubmix/models/o1.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/o1" -attachment = false -tool_call = false -reasoning_options = [] - -[cost] -input = 15 -output = 60 -cache_read = 7.5 - -[modalities] -input = ["text"] diff --git a/providers/aihubmix/models/o3-mini.toml b/providers/aihubmix/models/o3-mini.toml deleted file mode 100644 index 1a0d9b8945a..00000000000 --- a/providers/aihubmix/models/o3-mini.toml +++ /dev/null @@ -1,12 +0,0 @@ -base_model = "openai/o3-mini" -attachment = true -tool_call = false -reasoning_options = [] - -[cost] -input = 1.1 -output = 4.4 -cache_read = 0.55 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/o3-pro.toml b/providers/aihubmix/models/o3-pro.toml deleted file mode 100644 index 75813e2bcd7..00000000000 --- a/providers/aihubmix/models/o3-pro.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "openai/o3-pro" -reasoning_options = [] - -[cost] -input = 20 -output = 80 -cache_read = 20 diff --git a/providers/aihubmix/models/o3.toml b/providers/aihubmix/models/o3.toml deleted file mode 100644 index e8f09e33452..00000000000 --- a/providers/aihubmix/models/o3.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "openai/o3" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 2 -output = 8 -cache_read = 0.5 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/o4-mini.toml b/providers/aihubmix/models/o4-mini.toml deleted file mode 100644 index 75807e417c1..00000000000 --- a/providers/aihubmix/models/o4-mini.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "openai/o4-mini" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 1.1 -output = 4.4 -cache_read = 0.275 diff --git a/providers/aihubmix/models/ox-alpha.toml b/providers/aihubmix/models/ox-alpha.toml deleted file mode 100644 index 54c444c537c..00000000000 --- a/providers/aihubmix/models/ox-alpha.toml +++ /dev/null @@ -1,18 +0,0 @@ -base_model = "zhipuai/glm-5.3-flash" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "high", "max"] - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_048_576 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen-plus-latest.toml b/providers/aihubmix/models/qwen-plus-latest.toml deleted file mode 100644 index 1527008c135..00000000000 --- a/providers/aihubmix/models/qwen-plus-latest.toml +++ /dev/null @@ -1,23 +0,0 @@ -base_model = "alibaba/qwen-plus" -reasoning = false -tool_call = false - -[cost] -input = 0.1126 -output = 1.126 -cache_read = 0.02252 -cache_write = 0.14075 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.338 -output = 3.38 -cache_read = 0.0676 -cache_write = 0.4225 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 0.676 -output = 9.013311 -cache_read = 0.1352 -cache_write = 0.845 diff --git a/providers/aihubmix/models/qwen-turbo-latest.toml b/providers/aihubmix/models/qwen-turbo-latest.toml deleted file mode 100644 index 3b191626bcc..00000000000 --- a/providers/aihubmix/models/qwen-turbo-latest.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "alibaba/qwen-turbo" -reasoning = false -tool_call = false - -[cost] -input = 0.046 -output = 0.092 -cache_read = 0.0092 diff --git a/providers/aihubmix/models/qwen-turbo.toml b/providers/aihubmix/models/qwen-turbo.toml deleted file mode 100644 index 3b191626bcc..00000000000 --- a/providers/aihubmix/models/qwen-turbo.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "alibaba/qwen-turbo" -reasoning = false -tool_call = false - -[cost] -input = 0.046 -output = 0.092 -cache_read = 0.0092 diff --git a/providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml b/providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml deleted file mode 100644 index 94dc9e653ce..00000000000 --- a/providers/aihubmix/models/qwen3-235b-a22b-instruct-2507.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "alibaba/qwen3-235b-a22b-instruct-2507" -attachment = true -structured_output = true - -[cost] -input = 0.28 -output = 1.12 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-235b-a22b.toml b/providers/aihubmix/models/qwen3-235b-a22b.toml deleted file mode 100644 index f4b5bbbc406..00000000000 --- a/providers/aihubmix/models/qwen3-235b-a22b.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "alibaba/qwen3-235b-a22b" -reasoning = false -structured_output = true - -[cost] -input = 0.28 -output = 1.12 - -[limit] -context = 131_100 -output = 128_000 diff --git a/providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml b/providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml deleted file mode 100644 index 751e0fcd3f3..00000000000 --- a/providers/aihubmix/models/qwen3-coder-30b-a3b-instruct.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "alibaba/qwen3-coder-30b-a3b-instruct" -structured_output = true - -[cost] -input = 0.2 -output = 0.8 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.308218 -output = 1.232872 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 1.027396 -output = 5.13698 - -[limit] -output = 262_000 diff --git a/providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml b/providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml deleted file mode 100644 index 7b61078de8e..00000000000 --- a/providers/aihubmix/models/qwen3-coder-480b-a35b-instruct.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "alibaba/qwen3-coder-480b-a35b-instruct" -structured_output = true - -[cost] -input = 0.82 -output = 3.28 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 1.232876 -output = 4.931504 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 2.054794 -output = 8.219176 - -[limit] -context = 262_000 diff --git a/providers/aihubmix/models/qwen3-coder-flash.toml b/providers/aihubmix/models/qwen3-coder-flash.toml deleted file mode 100644 index 3ed4e460af4..00000000000 --- a/providers/aihubmix/models/qwen3-coder-flash.toml +++ /dev/null @@ -1,21 +0,0 @@ -base_model = "alibaba/qwen3-coder-flash" -structured_output = true - -[cost] -input = 0.136 -output = 0.544 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.205478 -output = 0.821912 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.342464 -output = 1.369856 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 0.68493 -output = 3.42465 diff --git a/providers/aihubmix/models/qwen3-coder-next.toml b/providers/aihubmix/models/qwen3-coder-next.toml deleted file mode 100644 index 7c3121eb26d..00000000000 --- a/providers/aihubmix/models/qwen3-coder-next.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "alibaba/qwen3-coder-next" - -[cost] -input = 0.137 -output = 0.548 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.2054 -output = 0.8216 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.3424 -output = 1.3696 diff --git a/providers/aihubmix/models/qwen3-coder-plus.toml b/providers/aihubmix/models/qwen3-coder-plus.toml deleted file mode 100644 index de385311d0a..00000000000 --- a/providers/aihubmix/models/qwen3-coder-plus.toml +++ /dev/null @@ -1,22 +0,0 @@ -base_model = "alibaba/qwen3-coder-plus" -structured_output = true - -[cost] -input = 0.54 -output = 2.16 -cache_read = 0.108 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.821916 -output = 3.287664 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 1.369862 -output = 5.479448 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 2.739726 -output = 27.39726 diff --git a/providers/aihubmix/models/qwen3-max-2026-01-23.toml b/providers/aihubmix/models/qwen3-max-2026-01-23.toml deleted file mode 100644 index 5ecc2962be5..00000000000 --- a/providers/aihubmix/models/qwen3-max-2026-01-23.toml +++ /dev/null @@ -1,43 +0,0 @@ -name = "Qwen3 Max 2026 01-23" -description = "The snapshot version of the Tongyi Qianwen 3 series Max model is from January 23, 2026. By default, it does not require thinking, but thinking mode can be enabled through the enable_thinking parameter, as detailed in the code example. (After enabling thinking by passing parameters, it becomes: Qwen3-Max-Thinking). This model has a total parameter count exceeding one trillion (1T) and a pre-training data volume of up to 36T Tokens, making it the largest and most powerful reasoning model from Alibaba to date." -release_date = "2026-01-23" -last_updated = "2026-01-23" -attachment = false -reasoning = true -tool_call = true -structured_output = true -open_weights = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.4508 -output = 1.8032 -cache_read = 0.09016 -cache_write = 0.5635 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.902 -output = 3.608 -cache_read = 0.1804 -cache_write = 1.1275 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 1.3522 -output = 5.4088 -cache_read = 0.27044 -cache_write = 1.69025 - -[limit] -context = 262_144 -output = 65_536 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/qwen3-max-preview.toml b/providers/aihubmix/models/qwen3-max-preview.toml deleted file mode 100644 index bec3493a846..00000000000 --- a/providers/aihubmix/models/qwen3-max-preview.toml +++ /dev/null @@ -1,23 +0,0 @@ -base_model = "alibaba/qwen3-max" -attachment = true -structured_output = true - -[cost] -input = 0.846 -output = 3.384 -cache_read = 0.1692 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 1.408 -output = 5.632 -cache_read = 0.2816 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 2.1128 -output = 8.4512 -cache_read = 0.42256 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-max.toml b/providers/aihubmix/models/qwen3-max.toml deleted file mode 100644 index 21d091e1145..00000000000 --- a/providers/aihubmix/models/qwen3-max.toml +++ /dev/null @@ -1,29 +0,0 @@ -base_model = "alibaba/qwen3-max" -reasoning = true -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.4508 -output = 1.8032 -cache_read = 0.09016 -cache_write = 0.5635 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.902 -output = 5.412 -cache_read = 0.1804 -cache_write = 1.1275 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 1.3522 -output = 8.1132 -cache_read = 0.27044 -cache_write = 1.69025 diff --git a/providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml b/providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml deleted file mode 100644 index 7a21444834b..00000000000 --- a/providers/aihubmix/models/qwen3-next-80b-a3b-instruct.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "alibaba/qwen3-next-80b-a3b-instruct" -attachment = true -structured_output = true - -[cost] -input = 0.138 -output = 0.552 - -[limit] -context = 256_000 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml b/providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml deleted file mode 100644 index fa6363729f4..00000000000 --- a/providers/aihubmix/models/qwen3-next-80b-a3b-thinking.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "alibaba/qwen3-next-80b-a3b-thinking" -attachment = true -structured_output = true -reasoning_options = [] - -[cost] -input = 0.142 -output = 1.42 - -[limit] -context = 256_000 - -[modalities] -input = ["text", "image"] diff --git a/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml b/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml deleted file mode 100644 index 31e731318f6..00000000000 --- a/providers/aihubmix/models/qwen3-vl-235b-a22b-instruct.toml +++ /dev/null @@ -1,15 +0,0 @@ -base_model = "alibaba/qwen3-vl-235b-a22b-instruct" -reasoning = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.274 -output = 1.096 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml b/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml deleted file mode 100644 index 702c6a92356..00000000000 --- a/providers/aihubmix/models/qwen3-vl-235b-a22b-thinking.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "alibaba/qwen3-vl-235b-a22b-thinking" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.274 -output = 2.74 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3-vl-30b-a3b-instruct.toml b/providers/aihubmix/models/qwen3-vl-30b-a3b-instruct.toml deleted file mode 100644 index 6be18b91c3c..00000000000 --- a/providers/aihubmix/models/qwen3-vl-30b-a3b-instruct.toml +++ /dev/null @@ -1,27 +0,0 @@ -name = "Qwen3 VL 30B A3B Instruct" -description = "The Qwen3-VL series’ second-largest MoE model Instruct version offers fast response speed and supports ultra-long contexts such as long videos and long documents; it features comprehensive upgrades in image/video understanding, spatial perception, and universal recognition abilities; it also provides visual 2DD/3D localization capabilities, making it capable of handling complex real-world tasks." -release_date = "2025-10-05" -last_updated = "2025-10-05" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1028 -output = 0.4112 - -[limit] -context = 131_072 -output = 32_768 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/qwen3-vl-30b-a3b-thinking.toml b/providers/aihubmix/models/qwen3-vl-30b-a3b-thinking.toml deleted file mode 100644 index 6760b8f994c..00000000000 --- a/providers/aihubmix/models/qwen3-vl-30b-a3b-thinking.toml +++ /dev/null @@ -1,27 +0,0 @@ -name = "Qwen3 VL 30B A3B Thinking" -description = "The Qwen3-VL series’ second-largest MoE model Thinking version offers fast response speed, stronger multimodal understanding and reasoning, visual agent capabilities, and ultra-long context support for long videos and long documents; it features comprehensive upgrades in image/video understanding, spatial perception, and universal recognition abilities, making it capable of handling complex real-world tasks." -release_date = "2025-10-11" -last_updated = "2025-10-11" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1028 -output = 1.028 - -[limit] -context = 131_072 -output = 32_768 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/qwen3-vl-flash.toml b/providers/aihubmix/models/qwen3-vl-flash.toml deleted file mode 100644 index 9b64faecbc9..00000000000 --- a/providers/aihubmix/models/qwen3-vl-flash.toml +++ /dev/null @@ -1,40 +0,0 @@ -name = "Qwen3 VL Flash" -description = "The Qwen3 series of compact visual-understanding models achieves an effective fusion of thinking mode and non-thinking mode, outperforming the open-source Qwen3-VL-30B-A3B with faster response speeds. It comprehensively upgrades image and video understanding, supporting ultra-long contexts such as long videos and long documents, spatial awareness, and universal object recognition; it also possesses visual 2D/3D localization capabilities and is capable of handling complex real-world tasks." -release_date = "2025-10-09" -last_updated = "2025-10-09" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.0206 -output = 0.206 -cache_read = 0.00412 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.041 -output = 0.41 -cache_read = 0.0082 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.0822 -output = 0.822 -cache_read = 0.01644 - -[limit] -context = 262_144 -output = 32_768 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/qwen3-vl-plus.toml b/providers/aihubmix/models/qwen3-vl-plus.toml deleted file mode 100644 index 4d14d7292b3..00000000000 --- a/providers/aihubmix/models/qwen3-vl-plus.toml +++ /dev/null @@ -1,27 +0,0 @@ -base_model = "alibaba/qwen3-vl-plus" -attachment = true -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.137 -output = 1.37 -cache_read = 0.0274 - -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.2054 -output = 2.054 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.411 -output = 4.11 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.5-122b-a10b.toml b/providers/aihubmix/models/qwen3.5-122b-a10b.toml deleted file mode 100644 index aa89331fc32..00000000000 --- a/providers/aihubmix/models/qwen3.5-122b-a10b.toml +++ /dev/null @@ -1,23 +0,0 @@ -base_model = "alibaba/qwen3.5-122b-a10b" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1126 -output = 0.9008 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.2818 -output = 2.2544 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.5-27b.toml b/providers/aihubmix/models/qwen3.5-27b.toml deleted file mode 100644 index afe6ada6bd0..00000000000 --- a/providers/aihubmix/models/qwen3.5-27b.toml +++ /dev/null @@ -1,23 +0,0 @@ -base_model = "alibaba/qwen3.5-27b" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.0846 -output = 0.6768 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.2536 -output = 2.0288 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.5-35b-a3b.toml b/providers/aihubmix/models/qwen3.5-35b-a3b.toml deleted file mode 100644 index cbef8c4b6a0..00000000000 --- a/providers/aihubmix/models/qwen3.5-35b-a3b.toml +++ /dev/null @@ -1,23 +0,0 @@ -base_model = "alibaba/qwen3.5-35b-a3b" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.0564 -output = 0.4512 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.2254 -output = 1.8032 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.5-397b-a17b.toml b/providers/aihubmix/models/qwen3.5-397b-a17b.toml deleted file mode 100644 index d57dea0957d..00000000000 --- a/providers/aihubmix/models/qwen3.5-397b-a17b.toml +++ /dev/null @@ -1,23 +0,0 @@ -base_model = "alibaba/qwen3.5-397b-a17b" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1644 -output = 0.9864 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.411 -output = 2.466 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.5-flash.toml b/providers/aihubmix/models/qwen3.5-flash.toml deleted file mode 100644 index f7a2e526d20..00000000000 --- a/providers/aihubmix/models/qwen3.5-flash.toml +++ /dev/null @@ -1,31 +0,0 @@ -base_model = "alibaba/qwen3.5-flash" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.0282 -output = 0.282 -cache_read = 0.00282 -cache_write = 0.03525 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.1126 -output = 1.126 -cache_read = 0.01126 -cache_write = 0.14075 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 0.169 -output = 1.69 -cache_read = 0.0169 -cache_write = 0.21125 diff --git a/providers/aihubmix/models/qwen3.5-plus.toml b/providers/aihubmix/models/qwen3.5-plus.toml deleted file mode 100644 index a6279667d80..00000000000 --- a/providers/aihubmix/models/qwen3.5-plus.toml +++ /dev/null @@ -1,33 +0,0 @@ -base_model = "alibaba/qwen3.5-plus" -attachment = true -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1096 -output = 0.6576 -cache_read = 0.01096 -cache_write = 0.137 - -[[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 0.274 -output = 1.644 -cache_read = 0.0274 -cache_write = 0.3425 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 0.548 -output = 3.288 -cache_read = 0.0548 -cache_write = 0.685 diff --git a/providers/aihubmix/models/qwen3.6-27b.toml b/providers/aihubmix/models/qwen3.6-27b.toml deleted file mode 100644 index b7bcc334bc0..00000000000 --- a/providers/aihubmix/models/qwen3.6-27b.toml +++ /dev/null @@ -1,11 +0,0 @@ -base_model = "alibaba/qwen3.6-27b" - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.422 -output = 2.532 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.6-35b-a3b.toml b/providers/aihubmix/models/qwen3.6-35b-a3b.toml deleted file mode 100644 index 5516b701f13..00000000000 --- a/providers/aihubmix/models/qwen3.6-35b-a3b.toml +++ /dev/null @@ -1,18 +0,0 @@ -base_model = "alibaba/qwen3.6-35b-a3b" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.254 -output = 1.524 - -[modalities] -input = ["text", "image", "video"] diff --git a/providers/aihubmix/models/qwen3.6-flash.toml b/providers/aihubmix/models/qwen3.6-flash.toml index bf52da4cbcf..2371f400435 100644 --- a/providers/aihubmix/models/qwen3.6-flash.toml +++ b/providers/aihubmix/models/qwen3.6-flash.toml @@ -1,28 +1,37 @@ -base_model = "alibaba/qwen3.6-flash" +name = "Qwen3.6 Flash" description = "Multimodal reasoning model for visual analysis, planning, and tool use" +family = "qwen3.6" +release_date = "2026-04-02" +last_updated = "2026-04-02" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }] +temperature = true +tool_call = true +structured_output = true +knowledge = "2025-04" +open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] -input = 0.169 -output = 1.014 +input = 0.17 +output = 1.01 cache_read = 0.0169 cache_write = 0.21125 [[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 0.676 -output = 4.056 +tier = { size = 256_000 } +input = 0.68 +output = 4.06 cache_read = 0.0676 cache_write = 0.845 + +[limit] +context = 991_000 +output = 64_000 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3.6-max-preview.toml b/providers/aihubmix/models/qwen3.6-max-preview.toml index bab3477e100..32c2d9f4bfe 100644 --- a/providers/aihubmix/models/qwen3.6-max-preview.toml +++ b/providers/aihubmix/models/qwen3.6-max-preview.toml @@ -1,25 +1,37 @@ -base_model = "alibaba/qwen3.6-max-preview" +name = "Qwen3.6 Max Preview" description = "Flagship model for demanding analysis, coding, and production agent workflows" +family = "qwen3.6" +release_date = "2026-05-09" +last_updated = "2026-05-09" +attachment = false +reasoning = true +reasoning_options = [{ type = "toggle" }] +temperature = true +tool_call = true structured_output = true +knowledge = "2025-04" +open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "budget_tokens" - [cost] -input = 1.268 -output = 7.608 +input = 1.27 +output = 7.61 cache_read = 0.1268 cache_write = 1.585 [[cost.tiers]] -tier = { type = "context", size = 128_000 } -input = 2.112 -output = 12.672 +tier = { size = 128_000 } +input = 2.11 +output = 12.67 cache_read = 0.2112 cache_write = 2.64 + +[limit] +context = 240_000 +output = 64_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3.6-plus-preview-free.toml b/providers/aihubmix/models/qwen3.6-plus-preview-free.toml deleted file mode 100644 index 43fb38a91b3..00000000000 --- a/providers/aihubmix/models/qwen3.6-plus-preview-free.toml +++ /dev/null @@ -1,23 +0,0 @@ -base_model = "alibaba/qwen3.6-plus" -attachment = false -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0 -output = 0 - -[limit] -output = 65_535 - -[modalities] -input = ["text"] diff --git a/providers/aihubmix/models/qwen3.6-plus.toml b/providers/aihubmix/models/qwen3.6-plus.toml index 800cb342527..722d03774fc 100644 --- a/providers/aihubmix/models/qwen3.6-plus.toml +++ b/providers/aihubmix/models/qwen3.6-plus.toml @@ -1,29 +1,37 @@ -base_model = "alibaba/qwen3.6-plus" +name = "Qwen3.6 Plus" description = "Multimodal reasoning model for visual analysis, planning, and tool use" +family = "qwen3.6" +release_date = "2026-05-09" +last_updated = "2026-05-09" +attachment = true +reasoning = true +reasoning_options = [{ type = "toggle" }] +temperature = true +tool_call = true structured_output = true +knowledge = "2025-04" +open_weights = false [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] -input = 0.282 -output = 1.692 +input = 0.28 +output = 1.69 cache_read = 0.0282 cache_write = 0.3525 [[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 1.128 -output = 6.768 +tier = { size = 256_000 } +input = 1.13 +output = 6.77 cache_read = 0.1128 cache_write = 1.41 + +[limit] +context = 991_000 +output = 64_000 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3.7-flash.toml b/providers/aihubmix/models/qwen3.7-flash.toml index a97f1ac9ee4..3f34f7fd74f 100644 --- a/providers/aihubmix/models/qwen3.7-flash.toml +++ b/providers/aihubmix/models/qwen3.7-flash.toml @@ -1,39 +1,22 @@ # Toggle: enable_thinking = true|false # Budget: thinking_budget = integer reasoning tokens base_model = "alibaba/qwen3.7-flash" +attachment = false +reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", max = 262_144 }] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 0.0282 output = 0.1128 cache_read = 0.00564 cache_write = 0.03525 -[[cost.tiers]] -tier = { type = "context", size = 32_000 } -input = 0.0845 -output = 0.338 -cache_read = 0.0169 -cache_write = 0.105625 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 0.169 -output = 0.676 -cache_read = 0.0338 -cache_write = 0.21125 - [limit] -output = 131_072 +context = 991_000 +output = 64_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3.7-max.toml b/providers/aihubmix/models/qwen3.7-max.toml index 6aa08a6f013..3370cc1f374 100644 --- a/providers/aihubmix/models/qwen3.7-max.toml +++ b/providers/aihubmix/models/qwen3.7-max.toml @@ -1,19 +1,10 @@ base_model = "alibaba/qwen3.7-max" structured_output = true +reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", max = 262_144 }] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 1.69 output = 5.07 @@ -21,4 +12,9 @@ cache_read = 0.169 cache_write = 2.1125 [limit] -output = 131_072 +context = 991_000 +output = 64_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3.7-plus.toml b/providers/aihubmix/models/qwen3.7-plus.toml index c92a5926949..d797f8535c7 100644 --- a/providers/aihubmix/models/qwen3.7-plus.toml +++ b/providers/aihubmix/models/qwen3.7-plus.toml @@ -1,31 +1,20 @@ base_model = "alibaba/qwen3.7-plus" structured_output = true +reasoning_options = [{ type = "toggle" }, { type = "budget_tokens", max = 262_144 }] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 0.282 output = 1.128 cache_read = 0.0564 cache_write = 0.3525 -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 0.845 -output = 3.38 -cache_read = 0.169 -cache_write = 1.05625 - [limit] -output = 131_072 +context = 991_000 +output = 64_000 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3.8-2.4t-a95b.toml b/providers/aihubmix/models/qwen3.8-2.4t-a95b.toml index 2d7c0a8782a..717b12b3b19 100644 --- a/providers/aihubmix/models/qwen3.8-2.4t-a95b.toml +++ b/providers/aihubmix/models/qwen3.8-2.4t-a95b.toml @@ -1,24 +1,21 @@ # AIHubMix Models API reports text,image input (queried 2026-08-31T04:06:29Z). # OpenAI-compatible HTTPS image input returned HTTP 200 (validated 2026-08-31T04:08:19Z). base_model = "alibaba/qwen3.8-2.4t-a95b" +attachment = true +reasoning_options = [{ type = "effort", values = ["low", "medium", "xhigh"] }] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "xhigh"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 2 output = 6 cache_read = 0.5 [limit] -context = 1_000_000 +context = 262_000 +output = 262_000 + +[modalities] +input = ["text", "image"] +output = ["text"] diff --git a/providers/aihubmix/models/qwen3.8-flash.toml b/providers/aihubmix/models/qwen3.8-flash.toml deleted file mode 100644 index c4be8d4c10b..00000000000 --- a/providers/aihubmix/models/qwen3.8-flash.toml +++ /dev/null @@ -1,17 +0,0 @@ -base_model = "alibaba/qwen3.8-flash" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "xhigh"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.1126 -output = 0.380025 -cache_read = 0.014075 -cache_write = 0.175937 diff --git a/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml b/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml deleted file mode 100644 index 56b7d4d0318..00000000000 --- a/providers/aihubmix/models/qwen3.8-max-2026-09-02.toml +++ /dev/null @@ -1,33 +0,0 @@ -name = "Qwen3.8 Max 2026 09-02" -description = "Qwen3.8-Max-0902 (also known as qwen3.8-max-2026-09-02) is a snapshot version of Alibaba Cloud Tongyi Qianwen's qwen3.8-max. It pushes encoding depth further, enabling it to handle more complex engineering-level projects and long-term autonomous development; collaborative agent capabilities are significantly enhanced, performing more confidently in multi-tool orchestration and end-to-end delivery; visual understanding is comprehensively improved, with more sensitive and accurate chart reasoning, document parsing, and multimodal perception. Continuing the 1-million-context window, reasoning modes, and a complete tool ecosystem, it continues to evolve at a higher level of intelligence." -release_date = "2026-09-02" -last_updated = "2026-09-02" -attachment = true -reasoning = true -tool_call = true -structured_output = true -open_weights = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 1.69 -output = 5.07 -cache_read = 0.169 -cache_write = 2.1125 - -[limit] -context = 1_000_000 -output = 131_072 - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/aihubmix/models/qwen3.8-max-preview.toml b/providers/aihubmix/models/qwen3.8-max-preview.toml deleted file mode 100644 index e160a294c1a..00000000000 --- a/providers/aihubmix/models/qwen3.8-max-preview.toml +++ /dev/null @@ -1,18 +0,0 @@ -base_model = "alibaba/qwen3.8-max-preview" -structured_output = true - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - -[cost] -input = 0.338 -output = 1.014 -cache_read = 0.0676 -cache_write = 0.4225 diff --git a/providers/aihubmix/models/qwen3.8-max.toml b/providers/aihubmix/models/qwen3.8-max.toml index 202ce3e10cc..b9bd3e40b33 100644 --- a/providers/aihubmix/models/qwen3.8-max.toml +++ b/providers/aihubmix/models/qwen3.8-max.toml @@ -1,26 +1,26 @@ # AIHubMix OpenAI-compatible /v1/chat/completions: $.enable_thinking = true|false (toggle) and $.reasoning_effort = "low"|"medium"|"xhigh"; verified live 2026-08-11. https://docs.aihubmix.com/cn/api/unified-inference # AIHubMix Models API (queried 2026-08-11T09:45:11Z): input 1.69, output 5.07, cache_read 0.169, cache_write 2.1125 USD/MTok. https://aihubmix.com/api/v1/models?model=qwen3.8-max base_model = "alibaba/qwen3.8-max" +attachment = false structured_output = true +reasoning_options = [ + { type = "toggle" }, + { type = "effort", values = ["low", "medium", "xhigh"] }, +] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] - -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 1.69 output = 5.07 cache_read = 0.169 cache_write = 2.1125 +[limit] +context = 991_000 +output = 128_000 + [modalities] -input = ["text", "image", "video"] +input = ["text"] +output = ["text"] diff --git a/providers/aihubmix/models/solar-pro4.toml b/providers/aihubmix/models/solar-pro4.toml deleted file mode 100644 index 7ea7766cc3e..00000000000 --- a/providers/aihubmix/models/solar-pro4.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model = "upstage/solar-pro4" -reasoning = false -tool_call = false - -[cost] -input = 0.3 -output = 1.2 -cache_read = 0.06 diff --git a/providers/aihubmix/models/step-3.5-flash.toml b/providers/aihubmix/models/step-3.5-flash.toml deleted file mode 100644 index 3ecb188643b..00000000000 --- a/providers/aihubmix/models/step-3.5-flash.toml +++ /dev/null @@ -1,9 +0,0 @@ -base_model = "stepfun/step-3.5-flash" -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 0.11 -output = 0.33 diff --git a/providers/aihubmix/models/step-3.7-flash.toml b/providers/aihubmix/models/step-3.7-flash.toml deleted file mode 100644 index 4cae5014cae..00000000000 --- a/providers/aihubmix/models/step-3.7-flash.toml +++ /dev/null @@ -1,14 +0,0 @@ -base_model = "stepfun/step-3.7-flash" -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["low", "medium", "high"] - -[cost] -input = 0.22 -output = 1.32 -cache_read = 0.044 diff --git a/providers/aihubmix/models/xiaomi-mimo-v2-omni-free.toml b/providers/aihubmix/models/xiaomi-mimo-v2-omni-free.toml deleted file mode 100644 index c2f827b8f3e..00000000000 --- a/providers/aihubmix/models/xiaomi-mimo-v2-omni-free.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "xiaomi/mimo-v2-omni" -reasoning = false -tool_call = false - -[cost] -input = 0 -output = 0 - -[limit] -context = 256_000 - -[modalities] -input = ["text", "image", "video", "audio"] diff --git a/providers/aihubmix/models/xiaomi-mimo-v2-pro-free.toml b/providers/aihubmix/models/xiaomi-mimo-v2-pro-free.toml deleted file mode 100644 index 823d3fb6c43..00000000000 --- a/providers/aihubmix/models/xiaomi-mimo-v2-pro-free.toml +++ /dev/null @@ -1,10 +0,0 @@ -base_model = "xiaomi/mimo-v2-pro" -reasoning = false -tool_call = false - -[cost] -input = 0 -output = 0 - -[limit] -context = 1_000_000 diff --git a/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml b/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml index 8c02f9c85b9..5d9852914e9 100644 --- a/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml +++ b/providers/aihubmix/models/xiaomi-mimo-v2.5-free.toml @@ -1,12 +1,13 @@ base_model = "xiaomi/mimo-v2.5" +reasoning_options = [{ type = "toggle" }] name = "Xiaomi MiMo-V2.5 (free)" +family = "mimo-v2.5" +last_updated = "2026-05-13" [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 0 output = 0 +cache_read = 0 diff --git a/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml b/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml index 12a97b3e0e7..a6277d9bb9c 100644 --- a/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml +++ b/providers/aihubmix/models/xiaomi-mimo-v2.5-pro-free.toml @@ -1,12 +1,13 @@ base_model = "xiaomi/mimo-v2.5-pro" +reasoning_options = [{ type = "toggle" }] name = "Xiaomi MiMo-V2.5-Pro (free)" +family = "mimo-v2.5-pro" +last_updated = "2026-05-13" [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "toggle" - [cost] input = 0 output = 0 +cache_read = 0 diff --git a/providers/aihubmix/models/zai-glm-5-turbo.toml b/providers/aihubmix/models/zai-glm-5-turbo.toml deleted file mode 100644 index 486cfe4ab8c..00000000000 --- a/providers/aihubmix/models/zai-glm-5-turbo.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "zhipuai/glm-5-turbo" -tool_call = false - -[[reasoning_options]] -type = "toggle" - -[cost] -input = 1.2 -output = 3.9996 -cache_read = 0.24 - -[limit] -context = 204_800 From fe493be601190f3a2eb615de2184ed5ea110adf3 Mon Sep 17 00:00:00 2001 From: chenxue Date: Fri, 11 Sep 2026 21:36:33 +0800 Subject: [PATCH 13/20] feat(aihubmix): re-author the toggle wire path on every sync A sync rewrites the model file whole, so any header a human wrote on it is lost the first time the model changes. AIHubMix reaches the same thinking toggle from four dialects -- `enable_thinking` on the OpenAI-compatible path, `thinking.type` on `/v1/messages`, `generationConfig.thinkingConfig` on the Gemini path -- so `toggle` alone does not tell a caller which field to send. Emit the header from `translateModel` whenever the model carries a toggle, the way the OpenRouter adapter already does, so the wire path survives the rewrite. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 15 ++++++++- packages/core/test/sync.test.ts | 32 ++++++++++++++++++++ 2 files changed, 46 insertions(+), 1 deletion(-) diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index 0d49d63dfd5..ada37d7d600 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -141,6 +141,13 @@ async function readLabMetadataIDs(modelsDir: string) { return ids; } +// The same off state is reachable from whichever dialect the caller speaks, so +// the toggle has no single wire path. Name one per protocol. +const TOGGLE_HEADER = + '# Toggle: $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11);\n' + + '# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path.\n' + + "# https://docs.aihubmix.com/cn/api/unified-inference\n"; + export const aihubmix = { id: "aihubmix", name: "AIHubMix", @@ -184,7 +191,13 @@ export const aihubmix = { const existing = context.existing(model.model_id); const built = buildAihubmixModel(model, existing, labMetadataIDs, relayCatalog); if (built === undefined) return undefined; - return { id: model.model_id, model: built }; + return { + id: model.model_id, + model: built, + // A rewrite drops whatever header the file carried, so re-author it here + // or the wire path is lost on the first sync that touches the model. + header: built.reasoning_options?.some((option) => option.type === "toggle") ? TOGGLE_HEADER : undefined, + }; }, } satisfies SyncProvider; diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 2faf85e42d2..60307001997 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5186,6 +5186,38 @@ test("normalizes AIHubMix reasoning options to the catalog vocabulary", () => { ]); }); +test("authors the AIHubMix toggle wire-path header so a rewrite cannot drop it", () => { + // Standalone relays, so the toggle stays on the written file instead of being + // factored onto a lab base model. + const standalone = { + vendor: "somelab", + release_date: "2026-05-01", + open_weights: false, + context_length: 262_144, + max_output: 65_536, + } satisfies Partial; + const toggled = aihubmixModel({ + ...standalone, + model_id: "somelab-thinker", + model_name: "SomeLab Thinker", + reasoning: true, + reasoning_options: [{ type: "toggle" }] as AihubmixModel["reasoning_options"], + }); + const plain = aihubmixModel({ + ...standalone, + model_id: "somelab-plain", + model_name: "SomeLab Plain", + }); + aihubmix.parseModels({ data: [toggled, plain] }); + + const context = { existing: () => undefined, authored: () => undefined }; + const translated = aihubmix.translateModel(toggled, context); + expect(translated?.model.reasoning_options).toEqual([{ type: "toggle" }]); + expect(translated?.header).toStartWith("# Toggle: $.enable_thinking = true|false"); + // Only a toggle needs the wire path spelled out; everything else stays bare. + expect(aihubmix.translateModel(plain, context)?.header).toBeUndefined(); +}); + test("reads a zero AIHubMix limit as absent rather than a real ceiling", () => { // 102 of 415 models quote `max_output: 0` for a limit the endpoint does not know. const zeroed = buildAihubmixModel( From 7556964f66757507dbeac7fc61df1c48dc9a5e22 Mon Sep 17 00:00:00 2001 From: chenxue Date: Fri, 11 Sep 2026 21:53:41 +0800 Subject: [PATCH 14/20] fix(aihubmix): keep prices the endpoint does not quote The catalog endpoint carries text and cache rates only -- no audio and no reasoning rate, at the top level or inside a tier. `buildCost` rebuilt the cost object from that answer alone, so the next sync would have wiped the `input_audio` already authored on `doubao-seed-2-0-lite`, `doubao-seed-2-0-mini` and `gemini-2.5-flash`. Carry the authored audio and reasoning rates through, matching tiers by context size, the way the other gateway adapters do. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 17 ++++++++-- packages/core/test/sync.test.ts | 34 ++++++++++++++++++++ 2 files changed, 49 insertions(+), 2 deletions(-) diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index ada37d7d600..c9de36ac3f4 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -374,15 +374,25 @@ function buildCost( output, cache_read: price(pricing.cache_read), cache_write: price(pricing.cache_write), - tiers: costTiers(pricing) ?? authored?.tiers, + // AIHubMix quotes only text and cache rates, so an audio or reasoning price + // exists on the file and nowhere else. A rewrite would drop it. + input_audio: authored?.input_audio, + output_audio: authored?.output_audio, + reasoning: authored?.reasoning, + tiers: costTiers(pricing, authored?.tiers) ?? authored?.tiers, }; } -function costTiers(pricing: NonNullable) { +function costTiers( + pricing: NonNullable, + authored: NonNullable["tiers"], +) { const tiers = (pricing.tiers ?? []).flatMap((tier) => { const input = price(tier.input); const output = price(tier.output); if (input === undefined || output === undefined) return []; + // Audio rates are per tier too, and the endpoint quotes none of them. + const priced = authored?.find((entry) => entry.tier.size === tier.tier.size); return [ { tier: { type: tier.tier.type ?? "context", size: tier.tier.size }, @@ -390,6 +400,9 @@ function costTiers(pricing: NonNullable) { output, cache_read: price(tier.cache_read), cache_write: price(tier.cache_write), + input_audio: priced?.input_audio, + output_audio: priced?.output_audio, + reasoning: priced?.reasoning, }, ]; }); diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 60307001997..45ead6142ba 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5218,6 +5218,40 @@ test("authors the AIHubMix toggle wire-path header so a rewrite cannot drop it", expect(aihubmix.translateModel(plain, context)?.header).toBeUndefined(); }); +test("keeps AIHubMix audio and reasoning prices the endpoint never quotes", () => { + // The endpoint models only text and cache rates, so an audio rate lives on the + // file and nowhere else -- at the top level and inside each context tier. + const authored: ExistingModel = { + ...aihubmixAuthored, + cost: { + input: 0.25, + output: 1.5, + input_audio: 1, + output_audio: 2, + reasoning: 3, + tiers: [ + { tier: { type: "context", size: 32_000 }, input: 0.5, output: 3, input_audio: 1.9 }, + ], + }, + }; + const model = buildAihubmixModel( + aihubmixModel({ + pricing: { + input: 0.25, + output: 1.5, + tiers: [{ tier: { type: "context", size: 32_000 }, input: 0.6, output: 3.2 }], + }, + }), + authored, + aihubmixLabIDs, + ); + expect(model?.cost?.input_audio).toBe(1); + expect(model?.cost?.output_audio).toBe(2); + expect(model?.cost?.reasoning).toBe(3); + // The endpoint still owns the text rates it does quote. + expect(model?.cost?.tiers?.[0]).toMatchObject({ input: 0.6, output: 3.2, input_audio: 1.9 }); +}); + test("reads a zero AIHubMix limit as absent rather than a real ceiling", () => { // 102 of 415 models quote `max_output: 0` for a limit the endpoint does not know. const zeroed = buildAihubmixModel( From 15d1e913e7f18ad173d5f778e5aa5c89c082094d Mon Sep 17 00:00:00 2001 From: chenxue Date: Sat, 12 Sep 2026 00:19:40 +0800 Subject: [PATCH 15/20] fix(aihubmix): fold the toggle into effort=none per AGENTS.md AIHubMix accepts whichever off switch the caller's SDK speaks -- `reasoning_effort: "none"` on the Chat path, `enable_thinking: false`, `thinking.type: "disabled"` on `/v1/messages`, `thinkingBudget: 0` on the Gemini path -- and maps each onto the vendor's real control instead of rejecting it. 34 of 408 routes therefore publish both a toggle and a graded effort list carrying `none`. The catalog spells that one way: `AGENTS.md` says graded effort that already includes `none` stands alone, with no toggle. Fold it, and name the dialects that reach the same off state in the file header, which is where that belongs. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 37 ++++++++++++++--- packages/core/test/sync.test.ts | 42 +++++++++++++++++--- 2 files changed, 68 insertions(+), 11 deletions(-) diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index c9de36ac3f4..16389dad2f0 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -142,11 +142,15 @@ async function readLabMetadataIDs(modelsDir: string) { } // The same off state is reachable from whichever dialect the caller speaks, so -// the toggle has no single wire path. Name one per protocol. -const TOGGLE_HEADER = - '# Toggle: $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11);\n' + +// an off switch has no single wire path. Name one per protocol. +const DIALECTS = + '# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11);\n' + '# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path.\n' + "# https://docs.aihubmix.com/cn/api/unified-inference\n"; +const TOGGLE_HEADER = "# Toggle:\n" + DIALECTS; +// Where the catalog spells the off state as `effort = none`, the other dialects +// still reach it, and the folded toggle is the only place that was recorded. +const FOLDED_HEADER = "# Off is effort=none; graded levels — no toggle. The same off elsewhere:\n" + DIALECTS; export const aihubmix = { id: "aihubmix", @@ -196,7 +200,7 @@ export const aihubmix = { model: built, // A rewrite drops whatever header the file carried, so re-author it here // or the wire path is lost on the first sync that touches the model. - header: built.reasoning_options?.some((option) => option.type === "toggle") ? TOGGLE_HEADER : undefined, + header: reasoningHeader(model, built), }; }, } satisfies SyncProvider; @@ -344,7 +348,30 @@ function reasoningOptions(model: AihubmixModel): SyncedFullModel["reasoning_opti .filter((value) => EFFORT_VALUES.has(value)); return values.length > 0 ? [{ type: "effort" as const, values }] : []; }); - return options.length > 0 ? (options as SyncedFullModel["reasoning_options"]) : undefined; + // AIHubMix accepts whichever off switch the caller's SDK speaks and maps it, + // so a model can publish both a toggle and `effort = none`. The catalog spells + // that one way: graded effort carrying `none` stands alone, and the dialects + // that reach the same off state are named in the file header instead. + const folded = foldsToggle(options) ? options.filter((option) => option.type !== "toggle") : options; + return folded.length > 0 ? (folded as SyncedFullModel["reasoning_options"]) : undefined; +} + +function foldsToggle(options: { type: string; values?: string[] }[]) { + return ( + options.some((option) => option.type === "toggle") && + options.some((option) => option.type === "effort" && (option.values ?? []).includes("none")) + ); +} + +function reasoningHeader(model: AihubmixModel, built: SyncedModel) { + const options = built.reasoning_options; + if (options === undefined) return undefined; + if (options.some((option) => option.type === "toggle")) return TOGGLE_HEADER; + // Only say where the off state moved to on a file that actually spells it out. + return options.some((option) => option.type === "effort" && option.values?.includes("none")) && + (model.reasoning_options ?? []).some((option) => option.type === "toggle") + ? FOLDED_HEADER + : undefined; } function modalities(value: string | null | undefined, fallback: string[]) { diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 45ead6142ba..c173adc6894 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5180,10 +5180,24 @@ test("normalizes AIHubMix reasoning options to the catalog vocabulary", () => { undefined, aihubmixLabIDs, ); - expect(model?.reasoning_options).toEqual([ - { type: "effort", values: ["none", "minimal", "high"] }, - { type: "toggle" }, - ]); + // `AGENTS.md`: graded effort that already carries `none` stands alone. AIHubMix + // publishes both because it accepts either dialect's off switch and maps it. + expect(model?.reasoning_options).toEqual([{ type: "effort", values: ["none", "minimal", "high"] }]); +}); + +test("keeps the AIHubMix toggle when its effort list has no off value", () => { + const model = buildAihubmixModel( + aihubmixModel({ + reasoning: true, + reasoning_options: [ + { type: "effort", values: ["high", "max"] }, + { type: "toggle" }, + ] as AihubmixModel["reasoning_options"], + }), + undefined, + aihubmixLabIDs, + ); + expect(model?.reasoning_options).toEqual([{ type: "effort", values: ["high", "max"] }, { type: "toggle" }]); }); test("authors the AIHubMix toggle wire-path header so a rewrite cannot drop it", () => { @@ -5213,9 +5227,25 @@ test("authors the AIHubMix toggle wire-path header so a rewrite cannot drop it", const context = { existing: () => undefined, authored: () => undefined }; const translated = aihubmix.translateModel(toggled, context); expect(translated?.model.reasoning_options).toEqual([{ type: "toggle" }]); - expect(translated?.header).toStartWith("# Toggle: $.enable_thinking = true|false"); - // Only a toggle needs the wire path spelled out; everything else stays bare. + expect(translated?.header).toStartWith("# Toggle:\n# $.enable_thinking = true|false"); + // Only a reasoning control needs the wire path spelled out; everything else stays bare. expect(aihubmix.translateModel(plain, context)?.header).toBeUndefined(); + + // A toggle folded into `effort = none` still records where the off state lives. + const folded = aihubmixModel({ + ...standalone, + model_id: "somelab-folded", + model_name: "SomeLab Folded", + reasoning: true, + reasoning_options: [ + { type: "toggle" }, + { type: "effort", values: ["no_think", "high"] }, + ] as AihubmixModel["reasoning_options"], + }); + aihubmix.parseModels({ data: [folded] }); + const dropped = aihubmix.translateModel(folded, context); + expect(dropped?.model.reasoning_options).toEqual([{ type: "effort", values: ["none", "high"] }]); + expect(dropped?.header).toStartWith("# Off is effort=none"); }); test("keeps AIHubMix audio and reasoning prices the endpoint never quotes", () => { From 38492050e60dbcfaf4a766c6a7c2edc90c9a15b0 Mon Sep 17 00:00:00 2001 From: chenxue Date: Sat, 12 Sep 2026 00:37:20 +0800 Subject: [PATCH 16/20] fix(aihubmix): stop a sync from wiping what only the file records The endpoint has no surface for several things a provider file carries, and the adapter rebuilt each object from the endpoint answer alone, so the first automation run would have dropped them: - `experimental` and `provider` -- the `[experimental.modes.fast]` block and its nested request body on `gpt-5.4`, `gpt-5.4-mini`, `gpt-5.5` - `limit.input` -- the 922k input cap the catalog models and AIHubMix does not - reasoning budget bounds -- the endpoint states that a budget exists but never its range, so a bare `{ type = "budget_tokens" }` written onto a `base_model` file would override the lab's real `min`/`max` with an unbounded control Carry all four through from the authored file, the way the Anthropic, OpenRouter and Merge Gateway adapters do. Six models and four context tiers also repeat the input price in `cache_read`, which is how the endpoint spells "no cache discount" rather than a real rate; an omitted field already means "no such rate" here, so an echoed one is now read the same way instead of publishing a full-price read as a 10x discount. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 42 +++++++++++--- packages/core/test/sync.test.ts | 59 ++++++++++++++++++++ 2 files changed, 94 insertions(+), 7 deletions(-) diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index 16389dad2f0..c8bea20eb01 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -42,6 +42,8 @@ const ReasoningOption = z .object({ type: z.string(), values: z.array(z.string()).nullish(), + min: z.number().nullish(), + max: z.number().nullish(), }) .passthrough(); @@ -228,17 +230,22 @@ export function buildAihubmixModel( context !== undefined && quoted !== undefined && quoted >= context ? undefined : quoted; const limit = { context: context ?? existing?.limit?.context, + // The endpoint models no input cap, so an authored one is the only record of it. + input: existing?.limit?.input, output: maxOutput ?? existing?.limit?.output, }; const shared = { attachment: input.some((value) => value !== "text"), reasoning, - reasoning_options: reasoningOptions(model) ?? existing?.reasoning_options, + reasoning_options: reasoningOptions(model, existing) ?? existing?.reasoning_options, tool_call: toolCall, structured_output: structuredOutput, - // AIHubMix serves no temperature or interleaved flags; keep what was authored. + // AIHubMix serves no temperature, interleaved, fast-mode or request-shape + // surface; all four exist on the file and nowhere else, so keep them. temperature: existing?.temperature, interleaved: existing?.interleaved, + experimental: existing?.experimental, + provider: existing?.provider, status: model.retire_stage === "deprecated" ? ("deprecated" as const) : existing?.status, modalities: { input, output }, limit, @@ -336,11 +343,21 @@ function bareID(modelID: string) { return modelID.split("/").at(-1) ?? modelID; } -function reasoningOptions(model: AihubmixModel): SyncedFullModel["reasoning_options"] { +function reasoningOptions( + model: AihubmixModel, + existing?: ExistingModel, +): SyncedFullModel["reasoning_options"] { if (model.reasoning_options == null) return undefined; const options = model.reasoning_options.flatMap((option) => { - if (option.type === "toggle" || option.type === "budget_tokens") { - return [{ type: option.type }]; + if (option.type === "toggle") return [{ type: option.type }]; + // The endpoint states that a budget exists but not its bounds. A bare option + // written onto a `base_model` file would override the lab's real range with + // an unbounded one, so the authored bounds are carried through. + if (option.type === "budget_tokens") { + const authored = existing?.reasoning_options?.find((entry) => entry.type === "budget_tokens"); + const min = option.min ?? (authored?.type === "budget_tokens" ? authored.min : undefined); + const max = option.max ?? (authored?.type === "budget_tokens" ? authored.max : undefined); + return [{ type: "budget_tokens" as const, min: min ?? undefined, max: max ?? undefined }]; } if (option.type !== "effort") return []; const values = (option.values ?? []) @@ -399,7 +416,7 @@ function buildCost( return { input, output, - cache_read: price(pricing.cache_read), + cache_read: cacheRead(pricing.cache_read, input), cache_write: price(pricing.cache_write), // AIHubMix quotes only text and cache rates, so an audio or reasoning price // exists on the file and nowhere else. A rewrite would drop it. @@ -425,7 +442,7 @@ function costTiers( tier: { type: tier.tier.type ?? "context", size: tier.tier.size }, input, output, - cache_read: price(tier.cache_read), + cache_read: cacheRead(tier.cache_read, input), cache_write: price(tier.cache_write), input_audio: priced?.input_audio, output_audio: priced?.output_audio, @@ -446,6 +463,17 @@ function tokens(value: number | null | undefined) { return value; } +/** + * Six models and four context tiers repeat the input price in `cache_read`, + * which is how the endpoint spells "no cache discount" rather than a real rate. + * Publishing it would understate a cached read by up to 10x. An omitted field + * already means "no such rate" here, so an echoed one is read the same way. + */ +function cacheRead(value: number | null | undefined, input: number) { + const parsed = price(value); + return parsed !== undefined && parsed >= input ? undefined : parsed; +} + function price(value: number | null | undefined) { if (value == null || !Number.isFinite(value) || value < 0) return undefined; return Math.round(value * 1_000_000) / 1_000_000; diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index c173adc6894..33a581e17f9 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5282,6 +5282,65 @@ test("keeps AIHubMix audio and reasoning prices the endpoint never quotes", () = expect(model?.cost?.tiers?.[0]).toMatchObject({ input: 0.6, output: 3.2, input_audio: 1.9 }); }); +test("keeps AIHubMix fields the endpoint has no surface for", () => { + // Fast mode, request-shape overrides and the input cap live on the file only. + const authored: ExistingModel = { + ...aihubmixAuthored, + limit: { context: 1_050_000, input: 922_000, output: 128_000 }, + experimental: { modes: { fast: { cost: { input: 5, output: 30 }, provider: { body: { service_tier: "priority" } } } } }, + provider: { body: { service_tier: "flex" } }, + } as ExistingModel; + const model = buildAihubmixModel( + aihubmixModel({ context_length: 1_050_000, max_output: 128_000 }), + authored, + aihubmixLabIDs, + ); + expect(model?.limit?.input).toBe(922_000); + expect(model?.experimental).toEqual(authored.experimental); + expect(model?.provider).toEqual(authored.provider); +}); + +test("reads an AIHubMix cache rate that just repeats input as no discount", () => { + // 6 models and 4 tiers echo `input` in `cache_read`; publishing it would + // understate a cached read by up to 10x. + const model = buildAihubmixModel( + aihubmixModel({ + pricing: { + input: 2, + output: 8, + cache_read: 2, + tiers: [{ tier: { type: "context", size: 200_000 }, input: 4, output: 16, cache_read: 4 }], + }, + }), + aihubmixAuthored, + aihubmixLabIDs, + ); + expect(model?.cost?.cache_read).toBeUndefined(); + expect(model?.cost?.tiers?.[0]?.cache_read).toBeUndefined(); + // A genuine discount is still published. + const discounted = buildAihubmixModel( + aihubmixModel({ pricing: { input: 2, output: 8, cache_read: 0.2 } }), + aihubmixAuthored, + aihubmixLabIDs, + ); + expect(discounted?.cost?.cache_read).toBe(0.2); +}); + +test("keeps authored reasoning budget bounds the AIHubMix endpoint omits", () => { + // The endpoint states that a budget exists but never its range, and a bare + // option on a `base_model` file would override the lab's real bounds. + const authored: ExistingModel = { + ...aihubmixAuthored, + reasoning_options: [{ type: "budget_tokens", min: 1_024, max: 32_000 }], + }; + const model = buildAihubmixModel( + aihubmixModel({ reasoning: true, reasoning_options: [{ type: "budget_tokens" }] as AihubmixModel["reasoning_options"] }), + authored, + aihubmixLabIDs, + ); + expect(model?.reasoning_options).toEqual([{ type: "budget_tokens", min: 1_024, max: 32_000 }]); +}); + test("reads a zero AIHubMix limit as absent rather than a real ceiling", () => { // 102 of 415 models quote `max_output: 0` for a limit the endpoint does not know. const zeroed = buildAihubmixModel( From f62d6a62304b691e874fc25c837e5749f421b280 Mon Sep 17 00:00:00 2001 From: chenxue Date: Sat, 12 Sep 2026 00:39:09 +0800 Subject: [PATCH 17/20] fix(aihubmix): let the endpoint widen modalities, never narrow them The catalog under-reports what a route accepts. It lists `text,image` for `kimi-k2.5`, whose lab entry and this repo both record video, and `text` for `qwen3.8-2.4t-a95b`, whose own file carries a note that live image input returned 200 on 2026-08-31. Treating the endpoint as authoritative would delete both on the first sync. Union the endpoint list with what the file recorded instead. A modality the endpoint adds still lands; one it never listed is removed by editing the file, which is where it came from. Both gaps are reported upstream. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 13 +++++++++++- packages/core/test/sync.test.ts | 21 ++++++++++++++++++++ 2 files changed, 33 insertions(+), 1 deletion(-) diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index c8bea20eb01..374a3d146e2 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -391,12 +391,23 @@ function reasoningHeader(model: AihubmixModel, built: SyncedModel) { : undefined; } +/** + * The endpoint under-reports what a route accepts: it lists `text,image` for + * `kimi-k2.5`, whose lab entry and this repo both record video, and `text` for + * `qwen3.8-2.4t-a95b`, whose own file notes a live 200 on image input. Both are + * reported upstream, but a sync must not delete an accepted modality in the + * meantime, so the endpoint adds to what the file recorded rather than replacing + * it. A modality the endpoint never listed can still be removed by editing the + * file, which is where it came from. + */ function modalities(value: string | null | undefined, fallback: string[]) { const parsed = (value ?? "") .split(",") .map((entry) => entry.trim()) .filter((entry) => ["text", "audio", "image", "video", "pdf"].includes(entry)); - return (parsed.length > 0 ? parsed : fallback) as SyncedFullModel["modalities"]["input"]; + if (parsed.length === 0) return fallback as SyncedFullModel["modalities"]["input"]; + // Endpoint order first, so a file only changes when its content changes. + return [...new Set([...parsed, ...fallback])] as SyncedFullModel["modalities"]["input"]; } /** diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 33a581e17f9..d936dc9cc9b 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5341,6 +5341,27 @@ test("keeps authored reasoning budget bounds the AIHubMix endpoint omits", () => expect(model?.reasoning_options).toEqual([{ type: "budget_tokens", min: 1_024, max: 32_000 }]); }); +test("does not let a narrower AIHubMix modality list delete an accepted one", () => { + // The endpoint lists `text,image` for kimi-k2.5, whose file records video. + const authored: ExistingModel = { + ...aihubmixAuthored, + modalities: { input: ["text", "image", "video"], output: ["text"] }, + }; + const model = buildAihubmixModel( + aihubmixModel({ input_modalities: "text,image" }), + authored, + aihubmixLabIDs, + ); + expect(model?.modalities?.input).toEqual(["text", "image", "video"]); + // A modality the endpoint adds still lands. + const widened = buildAihubmixModel( + aihubmixModel({ input_modalities: "text,image,pdf" }), + { ...aihubmixAuthored, modalities: { input: ["text"], output: ["text"] } }, + aihubmixLabIDs, + ); + expect(widened?.modalities?.input).toEqual(["text", "image", "pdf"]); +}); + test("reads a zero AIHubMix limit as absent rather than a real ceiling", () => { // 102 of 415 models quote `max_output: 0` for a limit the endpoint does not know. const zeroed = buildAihubmixModel( From 7f76e79c2c42269a3135cc6e7b1528e1be3f4493 Mon Sep 17 00:00:00 2001 From: chenxue Date: Sat, 12 Sep 2026 01:15:04 +0800 Subject: [PATCH 18/20] fix(aihubmix): stop a create from overriding lab metadata off MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The endpoint never sends `false` for a capability it does not know: 107 of 408 routes omit `reasoning` and 100 omit `tool_call`, and no route sends `false` at all. Reading a missing flag as `false` wrote an override that disabled a reasoner the lab entry declares. Modalities had the same shape of bug one level down. The union added in the previous commit merged the endpoint's list with the existing file, but `dev` carries only 77 aihubmix files, so most of the catalog arrives as a create with no file to merge against — 14 creates in the current listing would have written a narrowing override (`gpt-4o` losing pdf, `qwen3.5-27b` losing audio). The union now also includes the lab entry the relay factors onto. Headers were retained rather than refreshed, so a wire path could outlive the options it documents and a folded toggle kept advertising a toggle. `authoritativeHeaders` fixes that but would have deleted the price citations and live-test records humans wrote in the same block, so translateModel now reads the existing header and supersedes only the wire-path lines it authors. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/index.ts | 9 +++ packages/core/src/sync/providers/aihubmix.ts | 78 +++++++++++++++++--- packages/core/test/sync.test.ts | 76 +++++++++++++++++-- sync.md | 10 ++- 4 files changed, 153 insertions(+), 20 deletions(-) diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index 1f2596dd057..c2dee34d039 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -104,6 +104,12 @@ export interface SyncProvider { context: { existing(id: string): ExistingModel | undefined; authored(id: string): ExistingModel | undefined; + /** + * The leading comment block already on the file, so a provider that owns + * its header (authoritativeHeaders) can refresh the part it generates + * without discarding notes a human wrote around it. + */ + header?(id: string): string | undefined; }, ): { id: string; @@ -268,6 +274,9 @@ export async function syncProvider( authored(id) { return existing.get(`${id}.toml`)?.authored; }, + header(id) { + return existing.get(`${id}.toml`)?.header || undefined; + }, }); } catch (error) { if (!(error instanceof MissingReasoningOptionsError)) throw error; diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index 374a3d146e2..4d873f601fe 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -4,7 +4,7 @@ import { z } from "zod"; import { describeModel } from "../../describe.js"; import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js"; -import { factorBaseModel } from "./openrouter.js"; +import { factorBaseModel, modelMetadata } from "./openrouter.js"; const API_ENDPOINT = "https://aihubmix.com/api/v1/models?type=llm"; @@ -159,6 +159,11 @@ export const aihubmix = { name: "AIHubMix", modelsDir: "providers/aihubmix/models", trackMissingModels: true, + // A rewrite keeps whatever leading comment the file already had, so a stale + // wire path would outlive the options it documents — and a model whose toggle + // folds into `effort = none` would keep advertising a toggle. translateModel + // re-derives the header from the response, so let it own the block. + authoritativeHeaders: true, // Routing aliases such as `alicloud-glm-5.1` are served but unlisted, so a // local file absent from the response is retained rather than deleted. deleteMissing: false, @@ -202,7 +207,7 @@ export const aihubmix = { model: built, // A rewrite drops whatever header the file carried, so re-author it here // or the wire path is lost on the first sync that touches the model. - header: reasoningHeader(model, built), + header: composeHeader(context.header?.(model.model_id), reasoningHeader(model, built)), }; }, } satisfies SyncProvider; @@ -213,11 +218,28 @@ export function buildAihubmixModel( labIDs: LabMetadataIDs | undefined = labMetadataIDs, catalog: RelayCatalog | undefined = relayCatalog, ): SyncedModel | undefined { - const input = modalities(model.input_modalities, existing?.modalities?.input ?? ["text"]); - const output = modalities(model.output_modalities, existing?.modalities?.output ?? ["text"]); + const base = existing?.base_model ?? resolveBaseModel(model, labIDs, catalog); + // `dev` carries 77 aihubmix files, so most of the catalog arrives as a create + // with no file to union against. The lab entry the relay factors onto is the + // only baseline those have, and 14 creates in the current listing would + // otherwise write a narrowing override onto it (`gpt-4o` losing pdf, + // `qwen3.5-27b` losing audio). + const baseModalities = labModalities(base); + const input = modalities(model.input_modalities, [ + ...(existing?.modalities?.input ?? []), + ...(baseModalities?.input ?? []), + ]); + const output = modalities(model.output_modalities, [ + ...(existing?.modalities?.output ?? []), + ...(baseModalities?.output ?? []), + ]); const features = new Set((model.features ?? "").split(",").map((value) => value.trim())); - const reasoning = model.reasoning ?? existing?.reasoning ?? false; - const toolCall = model.tool_call ?? existing?.tool_call ?? false; + // The endpoint never sends `false`. 107 of 408 routes omit `reasoning` and 100 + // omit `tool_call` rather than denying them, and no route sends `false` at + // all, so a missing flag means unknown. Reading it as `false` would write an + // override that turns off a reasoner or tool use the lab declares. + const reasoning = model.reasoning ?? existing?.reasoning; + const toolCall = model.tool_call ?? existing?.tool_call; const structuredOutput = features.has("structured_outputs") || existing?.structured_output; const name = model.model_name ?? existing?.name; const context = tokens(model.context_length); @@ -252,7 +274,6 @@ export function buildAihubmixModel( cost: buildCost(model.pricing, existing?.cost), }; - const base = existing?.base_model ?? resolveBaseModel(model, labIDs, catalog); if (base !== undefined) { return factorBaseModel( base, @@ -279,8 +300,15 @@ export function buildAihubmixModel( return existing === undefined ? undefined : (existing as SyncedModel); } + // A standalone entry has no lab entry to inherit from, so the two flags have + // to resolve to a boolean here. Published reasoning options are the model's + // own statement that it reasons; absent both, the route is recorded as not. + const standaloneReasoning = reasoning ?? shared.reasoning_options !== undefined; + const standaloneToolCall = toolCall ?? false; return { ...shared, + reasoning: standaloneReasoning, + tool_call: standaloneToolCall, name, description: existing?.description ?? @@ -289,8 +317,8 @@ export function buildAihubmixModel( id: model.model_id, providerId: "aihubmix", name, - reasoning, - tool_call: toolCall, + reasoning: standaloneReasoning, + tool_call: standaloneToolCall, structured_output: structuredOutput, open_weights: openWeights, limit, @@ -380,6 +408,21 @@ function foldsToggle(options: { type: string; values?: string[] }[]) { ); } +/** + * Lines this adapter authors, plus the hand-written wire paths it supersedes. + * Anything else in the header is a human note — a price citation, a live-test + * record — that the response cannot reproduce, so it is carried through. + */ +const WIRE_PATH_LINE = /^#\s*(Toggle|Effort|Budget|Off is effort)\b|\$\.|thinkingConfig|docs\.aihubmix\.com\/cn\/api/; + +function composeHeader(existingHeader: string | undefined, derived: string | undefined) { + const notes = (existingHeader ?? "") + .split("\n") + .filter((line) => line.trim() !== "" && !WIRE_PATH_LINE.test(line)); + const header = (derived ?? "") + (notes.length > 0 ? `${notes.join("\n")}\n` : ""); + return header === "" ? undefined : header; +} + function reasoningHeader(model: AihubmixModel, built: SyncedModel) { const options = built.reasoning_options; if (options === undefined) return undefined; @@ -400,14 +443,27 @@ function reasoningHeader(model: AihubmixModel, built: SyncedModel) { * it. A modality the endpoint never listed can still be removed by editing the * file, which is where it came from. */ +/** + * The lab entry a relay factors onto, read for the one thing the endpoint can + * under-report. Returns nothing when the relay is standalone. + */ +function labModalities(base: string | undefined) { + if (base === undefined) return undefined; + const metadata = modelMetadata(base) as { + modalities?: { input?: string[]; output?: string[] }; + }; + return metadata.modalities; +} + function modalities(value: string | null | undefined, fallback: string[]) { const parsed = (value ?? "") .split(",") .map((entry) => entry.trim()) .filter((entry) => ["text", "audio", "image", "video", "pdf"].includes(entry)); - if (parsed.length === 0) return fallback as SyncedFullModel["modalities"]["input"]; + const known = fallback.length > 0 ? fallback : ["text"]; + if (parsed.length === 0) return known as SyncedFullModel["modalities"]["input"]; // Endpoint order first, so a file only changes when its content changes. - return [...new Set([...parsed, ...fallback])] as SyncedFullModel["modalities"]["input"]; + return [...new Set([...parsed, ...known])] as SyncedFullModel["modalities"]["input"]; } /** diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index d936dc9cc9b..887c0dc3bc9 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5342,26 +5342,88 @@ test("keeps authored reasoning budget bounds the AIHubMix endpoint omits", () => }); test("does not let a narrower AIHubMix modality list delete an accepted one", () => { - // The endpoint lists `text,image` for kimi-k2.5, whose file records video. + // The endpoint lists `text,image` for kimi-k2.5, whose lab entry records video. + // Most of the catalog arrives as a create with no file to union against, so + // the lab entry is the only baseline — and the narrowing must not be written. + const created = buildAihubmixModel( + aihubmixModel({ input_modalities: "text,image" }), + undefined, + aihubmixLabIDs, + ); + expect(created?.modalities).toBeUndefined(); + + // Where the file is the wider record, its modality survives an update too. const authored: ExistingModel = { ...aihubmixAuthored, - modalities: { input: ["text", "image", "video"], output: ["text"] }, + id: "minimax-m2", + modalities: { input: ["text", "image"], output: ["text"] }, }; - const model = buildAihubmixModel( - aihubmixModel({ input_modalities: "text,image" }), + const updated = buildAihubmixModel( + aihubmixModel({ model_id: "minimax-m2", vendor: "minimax", input_modalities: "text" }), authored, aihubmixLabIDs, ); - expect(model?.modalities?.input).toEqual(["text", "image", "video"]); + expect(updated?.modalities?.input).toEqual(["text", "image"]); + // A modality the endpoint adds still lands. const widened = buildAihubmixModel( - aihubmixModel({ input_modalities: "text,image,pdf" }), - { ...aihubmixAuthored, modalities: { input: ["text"], output: ["text"] } }, + aihubmixModel({ model_id: "minimax-m2", vendor: "minimax", input_modalities: "text,image,pdf" }), + undefined, aihubmixLabIDs, ); expect(widened?.modalities?.input).toEqual(["text", "image", "pdf"]); }); +test("refreshes the AIHubMix wire path without discarding a human note", () => { + // The header is authoritative so a stale wire path cannot outlive the options + // it documents, but a price citation or live-test record is not reproducible + // from the response and has to survive the rewrite. + const toggled = aihubmixModel({ + vendor: "somelab", + model_id: "somelab-noted", + model_name: "SomeLab Noted", + release_date: "2026-05-01", + open_weights: false, + context_length: 262_144, + max_output: 65_536, + reasoning: true, + reasoning_options: [{ type: "toggle" }] as AihubmixModel["reasoning_options"], + }); + aihubmix.parseModels({ data: [toggled] }); + const translated = aihubmix.translateModel(toggled, { + existing: () => undefined, + authored: () => undefined, + header: () => + "# Toggle: enable_thinking = true|false\n" + + "# AIHubMix Models API (queried 2026-08-11T09:45:11Z): input 1.69, output 5.07.\n", + }); + expect(translated?.header).toStartWith("# Toggle:\n# $.enable_thinking = true|false"); + // The hand-written wire path it supersedes is gone; the citation is not. + expect(translated?.header).not.toContain("# Toggle: enable_thinking"); + expect(translated?.header).toContain("queried 2026-08-11T09:45:11Z"); +}); + +test("reads a missing AIHubMix reasoning or tool flag as unknown, not as false", () => { + // 107 of 408 routes omit `reasoning` and 100 omit `tool_call`; none send + // `false`. A create must not write the omission as an override that turns off + // what the lab entry declares. + const created = buildAihubmixModel( + aihubmixModel({ input_modalities: "text,image,video,audio,pdf" }), + undefined, + aihubmixLabIDs, + ); + expect(created?.reasoning).toBeUndefined(); + expect(created?.tool_call).toBeUndefined(); + + // An explicit boolean is still honoured. + const denied = buildAihubmixModel( + aihubmixModel({ tool_call: false, input_modalities: "text,image,video,audio,pdf" }), + undefined, + aihubmixLabIDs, + ); + expect(denied?.tool_call).toBe(false); +}); + test("reads a zero AIHubMix limit as absent rather than a real ceiling", () => { // 102 of 415 models quote `max_output: 0` for a limit the endpoint does not know. const zeroed = buildAihubmixModel( diff --git a/sync.md b/sync.md index ed6144e9beb..ce651872830 100644 --- a/sync.md +++ b/sync.md @@ -253,7 +253,13 @@ xAI is implemented in `packages/core/src/sync/providers/xai.ts`. - AIHubMix is implemented in `packages/core/src/sync/providers/aihubmix.ts`. - Source endpoint: `https://aihubmix.com/api/v1/models?type=llm`. - No authentication is required; the catalog is public. -- The endpoint serves capabilities, limits, modalities, reasoning controls and pricing, so it is authoritative for all of them. Fields it does not serve (`family`, `temperature`, `interleaved`, `knowledge`) keep whatever was authored. +- The endpoint owns capabilities, limits, reasoning controls and text/cache pricing. Fields it has no surface for keep whatever was authored: `family`, `temperature`, `interleaved`, `knowledge`, `experimental`, `provider`, `limit.input`, and the `input_audio`/`output_audio`/`reasoning` rates (its `pricing` object carries only `input`, `output`, `cache_read`, `cache_write` and `tiers`). +- Booleans it omits mean unknown, not denied: 107 of 408 routes send no `reasoning` and 100 send no `tool_call`, and none sends `false`. A missing flag is left undefined so the lab entry's value is inherited rather than overridden off. +- Modalities can only widen. The endpoint under-reports some routes (it lists `text,image` for `kimi-k2.5`, whose lab entry records video), so its list is unioned with the lab entry the relay factors onto and with the existing file, never used to replace either. A modality the endpoint never listed is still removable by editing the file. +- `cache_read` that merely repeats `input` is how the endpoint spells "no cache discount" (6 models and 4 context tiers do this); it is read as absent rather than published as a rate, which would understate a cached read by up to 10x. +- `budget_tokens` arrives bare — 103 live entries state that a budget exists but never its bounds — so authored `min`/`max` are carried through instead of being replaced with an unbounded range. +- A `toggle` published alongside an effort list containing `none` is folded to the effort list alone, per the `AGENTS.md` reasoning-options table. AIHubMix accepts whichever off switch the caller's SDK speaks and maps it, so the other dialects that reach the same off state are named in the file header instead. +- `authoritativeHeaders` is on: the adapter re-derives the toggle/folded wire-path comment from each response, so a stale header cannot outlive the options it documents. It supersedes hand-written wire paths only — price citations, source links and live-test records in the same header are carried through, since the response cannot reproduce them. - AIHubMix relays upstream models under its own IDs, so a relay is factored onto the lab metadata it serves (`base_model`) whenever that metadata exists, and then records only what it actually changes. - The endpoint answers both halves of that lookup itself, so nothing about a relay is inferred from its ID here: `vendor` names the lab that built the model, and `variant_of` names the AIHubMix ID the entry is a routing variant of (`variant_kind` labels it a pricing tier, channel tier, mode preset or deprecated alias). A relay is looked up under its own ID first and then under each `variant_of` hop, nearest first — `qwen3.8-max-preview` is a variant of `qwen3.8-max` and both are published, so the relay factors onto the preview it actually serves. Following the declared chain also resolves relays no string rule reaches, such as `ox-alpha` onto `zhipuai/glm-5.3-flash` and `grok-code-fast-1` onto `xai/grok-build-0.1`. - `VENDOR_LABS` maps the four labs the two registries spell differently (`zhipu`/`zhipuai`, `moonshot`/`moonshotai`, `bytedance`/`bytedance-seed`, `meituan-longcat`/`meituan`). It maps namespaces only; no entry decides what a model is or which lab built it. @@ -262,7 +268,7 @@ xAI is implemented in `packages/core/src/sync/providers/xai.ts`. - A relay with neither resolvable lab metadata nor the `release_date`, `open_weights` and limits a standalone entry requires is skipped rather than written with invented values. Of 408 listed models the endpoint names a `vendor` for 292, declares `variant_of` for 76, dates 304 and states `open_weights` for 289; the remaining gap is what the skip notices report back upstream. - Two limit sentinels are read as absent rather than as real ceilings: `max_output: 0` (104 of 408 models) and a `max_output` at or above `context_length` (36), which is the context window quoted a second time. - `reasoning_options[]` entries carry an AIHubMix-only `default` key that the strict `ReasoningOption` schema rejects, and spell two effort levels differently (`no_think`, `instant`), so translation drops the extra key and maps those onto `none` and `minimal`. -- A price field is omitted when the model has no such rate, so an omitted `cache_read`/`cache_write` clears an authored one; authored pricing survives only when the endpoint quotes nothing at all for the model. +- A price field is omitted when the model has no such rate, so an omitted `cache_read`/`cache_write` clears an authored one; authored pricing survives only when the endpoint quotes nothing at all for the model, or for the rates it never quotes at all (see above). - Routing aliases such as `alicloud-glm-5.1` and `deep-deepseek-v4-pro` are served but not listed by the endpoint, so local files missing from the response are retained (`deleteMissing: false`) and each opens a deduped GitHub issue. ## Tinfoil Notes From 1ca206675e643f27cddd3a48bdc602ae8c17c6da Mon Sep 17 00:00:00 2001 From: chenxue Date: Mon, 14 Sep 2026 11:46:44 +0800 Subject: [PATCH 19/20] =?UTF-8?q?fix(aihubmix):=20answer=20review=20round?= =?UTF-8?q?=205=20=E2=80=94=20limits,=20standalone,=20skip=20tracking?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four review findings, all fixed at the adapter's shape rules rather than with per-model tables. - Limits no longer publish a decimal restatement of a binary window as a narrowing override. A stated limit below an accepted one but at or above 1000³/1024³ resolves to the accepted value; genuine host caps still land. 25 narrowing overrides become 7, and three MiniMax restatements an earlier sync wrote into files are retired. - A full standalone entry is only authored where the response names no vendor. A named lab means the relay belongs on base_model, so 38 would-be standalone creates for lab models become skips that name the file a human must add. - SyncProvider gains trackMissingModels, so a provider that creates models but still skips the ones it cannot write opens deduped [missing-model] issues instead of notices nobody acts on. - budget_tokens bounds cannot be copied from a lab entry: ModelMetadata has no reasoning_options field, so there is no such baseline to shadow. The comment records why rather than adding a fallback that could never fire. Also drops eight channel-alias files (alicloud-glm-5.1, zai-glm-5.1, the four deepseek-v4 channel routes, two xiaomi-mimo-v2.5 routes). Each relays to a model already in the catalog and echoes that model's ID back; the endpoint's main model list, not callability, is the catalog boundary. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/index.ts | 9 +- packages/core/src/sync/missing-issues.ts | 2 +- packages/core/src/sync/providers/aihubmix.ts | 88 +++++++++++--- packages/core/test/sync.test.ts | 110 +++++++++++++++++- .../models/alicloud-deepseek-v4-flash.toml | 29 ----- .../models/alicloud-deepseek-v4-pro.toml | 29 ----- .../aihubmix/models/alicloud-glm-5.1.toml | 29 ----- .../models/deep-deepseek-v4-flash.toml | 29 ----- .../aihubmix/models/deep-deepseek-v4-pro.toml | 29 ----- .../aihubmix/models/xiaomi-mimo-v2.5-pro.toml | 19 --- .../aihubmix/models/xiaomi-mimo-v2.5.toml | 19 --- providers/aihubmix/models/zai-glm-5.1.toml | 28 ----- sync.md | 11 +- 13 files changed, 196 insertions(+), 235 deletions(-) delete mode 100644 providers/aihubmix/models/alicloud-deepseek-v4-flash.toml delete mode 100644 providers/aihubmix/models/alicloud-deepseek-v4-pro.toml delete mode 100644 providers/aihubmix/models/alicloud-glm-5.1.toml delete mode 100644 providers/aihubmix/models/deep-deepseek-v4-flash.toml delete mode 100644 providers/aihubmix/models/deep-deepseek-v4-pro.toml delete mode 100644 providers/aihubmix/models/xiaomi-mimo-v2.5-pro.toml delete mode 100644 providers/aihubmix/models/xiaomi-mimo-v2.5.toml delete mode 100644 providers/aihubmix/models/zai-glm-5.1.toml diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index c2dee34d039..b6755b85b07 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -81,7 +81,12 @@ export interface SyncProvider { * deduped GitHub issue per missing model ID. */ skipCreates?: boolean; - /** Report remote-only models skipped by skipCreates as GitHub issues. */ + /** + * Open one deduped GitHub issue per model the provider skipped. Implied by + * skipCreates, and settable on its own by a provider that creates models but + * still skips the ones it cannot write — without it those skips produce a + * notice nobody acts on. + */ trackMissingModels?: boolean; deleteMissing?: boolean; preserveSymlinks?: boolean; @@ -498,7 +503,7 @@ export async function syncProvider( ]; const issueModels = [ - ...(provider.skipCreates === true ? skippedRemote : []), + ...(provider.skipCreates === true || provider.trackMissingModels === true ? skippedRemote : []), ...missingReasoning.keys(), ]; if ( diff --git a/packages/core/src/sync/missing-issues.ts b/packages/core/src/sync/missing-issues.ts index 81a01a3742e..ebb88f4a791 100644 --- a/packages/core/src/sync/missing-issues.ts +++ b/packages/core/src/sync/missing-issues.ts @@ -26,7 +26,7 @@ function issueBody(provider: MissingModelIssueTarget, modelId: string, reason?: `| Expected path | \`${provider.modelsDir}/${modelId}.toml\` |`, "", reason === undefined - ? "This provider uses `skipCreates` because the remote source is not enough to auto-author a full TOML." + ? "The sync did not author this model: the remote source is not enough to write a full TOML, or the model belongs on `base_model` and its `models/` metadata is missing." : `Sync diagnostic: ${reason}`, "Add the model manually (prefer `base_model` when matching `models/` metadata exists).", ...(reason === undefined ? [] : [ diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index 4d873f601fe..0127848f397 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -164,8 +164,11 @@ export const aihubmix = { // folds into `effort = none` would keep advertising a toggle. translateModel // re-derives the header from the response, so let it own the block. authoritativeHeaders: true, - // Routing aliases such as `alicloud-glm-5.1` are served but unlisted, so a - // local file absent from the response is retained rather than deleted. + // The listing is AIHubMix's main model list, and a route rotates out of it for a + // spell without being retired, so a local file absent from one response is + // retained rather than deleted. It is not a licence to keep anything: a hidden + // channel alias (`zai-glm-5.1`, which the gateway answers by routing to the + // listed `glm-5.1`) is deliberately outside that list and does not belong here. deleteMissing: false, sourceID(model) { return model.retire_stage === "deprecated" ? undefined : model.model_id; @@ -173,13 +176,13 @@ export const aihubmix = { missingNotice(paths) { return paths.map( (file) => - `AIHubMix no longer lists ${file}; confirm it is still a served routing alias or deprecate it.`, + `AIHubMix does not list ${file} in its main model list; confirm the route rotated out for a spell, or drop the file if it is a hidden channel alias of a model already in the catalog.`, ); }, skippedNotice(ids) { return ids.map( (id) => - `AIHubMix lists ${id} but the response carries neither a vendor/variant_of that resolves to lab metadata nor the release_date/open_weights/limits a standalone entry needs.`, + `AIHubMix lists ${id} but it cannot be written yet. If it names a vendor, add the lab model under \`models/${"/"}.toml\` and the relay factors onto it automatically; if it names none, the response is missing the release_date/open_weights/limits a standalone entry has to carry.`, ); }, async fetchModels() { @@ -224,7 +227,8 @@ export function buildAihubmixModel( // only baseline those have, and 14 creates in the current listing would // otherwise write a narrowing override onto it (`gpt-4o` losing pdf, // `qwen3.5-27b` losing audio). - const baseModalities = labModalities(base); + const lab = labMetadata(base); + const baseModalities = lab?.modalities; const input = modalities(model.input_modalities, [ ...(existing?.modalities?.input ?? []), ...(baseModalities?.input ?? []), @@ -250,11 +254,16 @@ export function buildAihubmixModel( const quoted = tokens(model.max_output); const maxOutput = context !== undefined && quoted !== undefined && quoted >= context ? undefined : quoted; + // Same class of bug as the modalities: the endpoint restates windows in decimal + // (8 glm routes quote 204800 as 200000) and quotes conservative output ceilings, + // and treating those as owned fields writes a narrowing override onto a limit the + // lab entry already states correctly. A restatement resolves to the accepted + // value; a genuine cap the host imposes still lands. const limit = { - context: context ?? existing?.limit?.context, + context: resolveLimit(context, existing?.limit?.context, lab?.limit?.context), // The endpoint models no input cap, so an authored one is the only record of it. input: existing?.limit?.input, - output: maxOutput ?? existing?.limit?.output, + output: resolveLimit(maxOutput, existing?.limit?.output, lab?.limit?.output), }; const shared = { attachment: input.some((value) => value !== "text"), @@ -283,6 +292,16 @@ export function buildAihubmixModel( ); } + // `vendor` names the lab that built the model, so this relay hosts someone + // else's model and belongs on `base_model` — AGENTS.md treats a full standalone + // definition for a nameable lab model as a blocker. Reaching here means the lab + // entry does not exist yet (81 of 407 routes, 38 of them complete enough that the + // endpoint answer alone would have satisfied the standalone guard), so the relay is + // reported for a human to add `models//.toml`, after which it factors + // with no change here. A file already in the repo keeps being updated: what it + // should have been is upstream's call, and freezing it would only stall its prices. + if (existing === undefined && model.vendor != null) return undefined; + // A standalone entry must carry every required catalog field itself, and the // endpoint still leaves gaps: 304 of 408 models are dated, 289 state // `open_weights`, and the rest quote 0 for a limit they do not know. A relay @@ -378,9 +397,12 @@ function reasoningOptions( if (model.reasoning_options == null) return undefined; const options = model.reasoning_options.flatMap((option) => { if (option.type === "toggle") return [{ type: option.type }]; - // The endpoint states that a budget exists but not its bounds. A bare option - // written onto a `base_model` file would override the lab's real range with - // an unbounded one, so the authored bounds are carried through. + // The endpoint states that a budget exists but not its bounds, so the bounds a + // file already carries are the only record of them and are carried through. + // There is no second baseline to fall back on: `ModelMetadata` has no + // `reasoning_options` field, so a bare budget written here cannot be shadowing + // a range stated on the lab entry — that range can only live on a provider file, + // and a budget range is a property of the host's API, not of the model. if (option.type === "budget_tokens") { const authored = existing?.reasoning_options?.find((entry) => entry.type === "budget_tokens"); const min = option.min ?? (authored?.type === "budget_tokens" ? authored.min : undefined); @@ -444,15 +466,51 @@ function reasoningHeader(model: AihubmixModel, built: SyncedModel) { * file, which is where it came from. */ /** - * The lab entry a relay factors onto, read for the one thing the endpoint can - * under-report. Returns nothing when the relay is standalone. + * The lab entry a relay factors onto, read for what the endpoint can under-report: + * modalities it omits and windows it restates in decimal. Returns nothing when the + * relay is standalone. Reasoning controls are not here to be read: `ModelMetadata` + * has no `reasoning_options` field, so a lab entry cannot state them and only a + * provider file ever does. */ -function labModalities(base: string | undefined) { +function labMetadata(base: string | undefined) { if (base === undefined) return undefined; - const metadata = modelMetadata(base) as { + return modelMetadata(base) as { modalities?: { input?: string[]; output?: string[] }; + limit?: { context?: number; output?: number }; }; - return metadata.modalities; +} + +/** + * A decimal restatement of a binary window can only lose `1000/1024` per K unit, + * so three nested unit swaps — 1024³ tokens quoted as 1000³ — is the floor of what + * a restatement can explain. The endpoint quotes 204800 as 200000 (0.977), 1048576 + * as 1000000 (0.954) and 65536 as 65535; none of that is the host narrowing the + * window, and writing it as an override invents a difference that is not there. + * A real restriction sits far below: grok-code-fast-1 caps output at 10000 of + * 256000 (0.039) and gpt-5-chat-latest serves 128000 of a 400000 window (0.320). + */ +const UNIT_RESTATEMENT_FLOOR = 1000 ** 3 / 1024 ** 3; + +/** + * The limit to record. The endpoint speaks first and the file stands in when it + * says nothing — the authored value is not a worse version of the lab's but a + * narrower one on purpose, the host's own cap (`kimi-k2.5` serves 32768 of a + * 262144 window), so it is kept rather than widened away. + * + * Whichever of the two states the limit, it is only recorded if it is a limit: a + * value that merely restates an accepted window in decimal resolves to that + * window and no override is written. Applying the test to the stated value rather + * than to the endpoint's quote also retires the restatements an earlier sync + * already wrote onto three MiniMax files (128000 and 128100 of 131072). + */ +function resolveLimit(quoted?: number, authored?: number, lab?: number) { + const stated = quoted ?? authored; + if (stated === undefined) return undefined; + for (const accepted of [authored, lab]) { + if (accepted === undefined || stated >= accepted) continue; + if (stated / accepted >= UNIT_RESTATEMENT_FLOOR) return accepted; + } + return stated; } function modalities(value: string | null | undefined, fallback: string[]) { diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index 887c0dc3bc9..c9d8d00f026 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5202,9 +5202,10 @@ test("keeps the AIHubMix toggle when its effort list has no off value", () => { test("authors the AIHubMix toggle wire-path header so a rewrite cannot drop it", () => { // Standalone relays, so the toggle stays on the written file instead of being - // factored onto a lab base model. + // factored onto a lab base model. A standalone entry is only allowed where the + // response names no vendor — a named lab belongs on `base_model`. const standalone = { - vendor: "somelab", + vendor: null, release_date: "2026-05-01", open_weights: false, context_length: 262_144, @@ -5379,7 +5380,7 @@ test("refreshes the AIHubMix wire path without discarding a human note", () => { // it documents, but a price citation or live-test record is not reproducible // from the response and has to survive the rewrite. const toggled = aihubmixModel({ - vendor: "somelab", + vendor: null, model_id: "somelab-noted", model_name: "SomeLab Noted", release_date: "2026-05-01", @@ -5474,6 +5475,109 @@ test("keeps authored AIHubMix pricing when the endpoint quotes no rate at all", expect(model?.cost).toEqual(aihubmixAuthored.cost); }); +test("inherits a limit the AIHubMix endpoint only restates in decimal", () => { + // The lab window is 1_048_576 and the endpoint quotes 1_000_000 for it — the same + // window in decimal, not a cap — so the relay must inherit rather than write an + // override claiming it lost 48_576 tokens. + const restated = buildAihubmixModel( + aihubmixModel({ context_length: 1_000_000, max_output: 65_536 }), + undefined, + aihubmixLabIDs, + ); + expect(restated?.limit?.context).toBeUndefined(); + + // A window the host genuinely restricts is far below any restatement and lands. + const capped = buildAihubmixModel( + aihubmixModel({ context_length: 128_000, max_output: 65_536 }), + undefined, + aihubmixLabIDs, + ); + expect(capped?.limit?.context).toBe(128_000); + + // And an authored cap survives an endpoint that quotes nothing for the ceiling. + const authored: ExistingModel = { + ...aihubmixAuthored, + limit: { context: 1_048_576, output: 32_768 }, + }; + const unknown = buildAihubmixModel( + aihubmixModel({ max_output: 0 }), + authored, + aihubmixLabIDs, + ); + expect(unknown?.limit?.output).toBe(32_768); +}); + +test("does not author a standalone AIHubMix entry for a model a lab made", () => { + // Every field a standalone entry needs is present, but `vendor` names the lab + // that built the model and no `models/somelab/…` entry exists to factor onto. + // AGENTS.md makes that a blocker, so the relay is reported, not written. + const lab = { + vendor: "somelab", + model_id: "somelab-unmapped", + model_name: "SomeLab Unmapped", + release_date: "2026-05-01", + open_weights: false, + context_length: 262_144, + max_output: 65_536, + } satisfies Partial; + expect(buildAihubmixModel(aihubmixModel(lab), undefined, aihubmixLabIDs)).toBeUndefined(); + expect(aihubmix.sourceID?.(aihubmixModel(lab))).toBe("somelab-unmapped"); + + // A relay the response names no vendor for is the host's own alias, and still writes. + const hostOwn = buildAihubmixModel( + aihubmixModel({ ...lab, vendor: null }), + undefined, + aihubmixLabIDs, + ); + expect(hostOwn?.name).toBe("SomeLab Unmapped"); + + // A standalone file already in the repo keeps being updated rather than freezing: + // what it should have been is upstream's call, and stalling its prices helps no one. + const authored: ExistingModel = { + id: "somelab-unmapped", + name: "SomeLab Unmapped", + release_date: "2026-05-01", + open_weights: false, + cost: { input: 1, output: 2 }, + limit: { context: 262_144, output: 65_536 }, + modalities: { input: ["text"], output: ["text"] }, + }; + const updated = buildAihubmixModel(aihubmixModel(lab), authored, aihubmixLabIDs); + expect(updated?.cost).toEqual({ input: 0.25, output: 1.5, cache_read: 0.025 }); +}); + +test("opens missing-model issues for a provider that creates but still skips", async () => { + // aihubmix does not set skipCreates — it creates what it can — so gating the + // issue path on skipCreates left every relay it cannot write as a notice nobody + // acts on. An explicit trackMissingModels is the opt-in for exactly that case. + const dir = await mkdtemp(path.join(tmpdir(), "sync-track-")); + const modelsDir = path.join(dir, "providers", "tracked", "models"); + await mkdir(modelsDir, { recursive: true }); + try { + const result = await syncProvider({ + id: "aihubmix", + name: "Tracked", + modelsDir, + trackMissingModels: true, + async fetchModels() { + return { data: [] }; + }, + parseModels() { + return [{ model_id: "unmapped-relay" }]; + }, + translateModel() { + return undefined; + }, + sourceID(model: { model_id: string }) { + return model.model_id; + }, + } as never, { dryRun: true, openIssues: true }); + expect(result.notices.some((notice) => notice.includes("unmapped-relay"))).toBe(true); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + test("marks retired AIHubMix relays deprecated and stops tracking them", () => { const retired = aihubmixModel({ retire_stage: "deprecated" }); expect(buildAihubmixModel(retired, aihubmixAuthored, aihubmixLabIDs)?.status).toBe("deprecated"); diff --git a/providers/aihubmix/models/alicloud-deepseek-v4-flash.toml b/providers/aihubmix/models/alicloud-deepseek-v4-flash.toml deleted file mode 100644 index 36c21d7cfb1..00000000000 --- a/providers/aihubmix/models/alicloud-deepseek-v4-flash.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "DeepSeek V4 Flash (Alibaba Cloud)" -description = "Fast DeepSeek model for efficient chat, coding help, and agent loops" -family = "deepseek-flash" -release_date = "2026-04-24" -last_updated = "2026-04-24" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-05" -open_weights = true - -[interleaved] -field = "reasoning_content" - -[cost] -input = 0.14 -output = 0.28 -cache_read = 0.028 - -[limit] -context = 1_000_000 -output = 384_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/alicloud-deepseek-v4-pro.toml b/providers/aihubmix/models/alicloud-deepseek-v4-pro.toml deleted file mode 100644 index 57e26625400..00000000000 --- a/providers/aihubmix/models/alicloud-deepseek-v4-pro.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "DeepSeek V4 Pro (Alibaba Cloud)" -description = "Flagship DeepSeek model for coding, reasoning, and agentic work" -family = "deepseek-thinking" -release_date = "2026-04-24" -last_updated = "2026-04-24" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-05" -open_weights = true - -[interleaved] -field = "reasoning_content" - -[cost] -input = 1.69 -output = 3.38 -cache_read = 0.13 - -[limit] -context = 1_000_000 -output = 384_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/alicloud-glm-5.1.toml b/providers/aihubmix/models/alicloud-glm-5.1.toml deleted file mode 100644 index 877d9e29a4b..00000000000 --- a/providers/aihubmix/models/alicloud-glm-5.1.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "GLM-5.1 (Alibaba Cloud)" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -family = "glm" -release_date = "2026-03-27" -last_updated = "2026-03-27" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -open_weights = true - -[interleaved] -field = "reasoning_content" - -[cost] -input = 0.84 -output = 3.38 -cache_read = 0.169 -cache_write = 1.05625 - -[limit] -context = 200_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/deep-deepseek-v4-flash.toml b/providers/aihubmix/models/deep-deepseek-v4-flash.toml deleted file mode 100644 index 89f36a266b7..00000000000 --- a/providers/aihubmix/models/deep-deepseek-v4-flash.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "DeepSeek V4 Flash (DeepSeek)" -description = "Fast DeepSeek model for efficient chat, coding help, and agent loops" -family = "deepseek-flash" -release_date = "2026-04-24" -last_updated = "2026-04-24" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-05" -open_weights = true - -[interleaved] -field = "reasoning_content" - -[cost] -input = 0.154 -output = 0.308 -cache_read = 0.0308 - -[limit] -context = 1_000_000 -output = 384_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/deep-deepseek-v4-pro.toml b/providers/aihubmix/models/deep-deepseek-v4-pro.toml deleted file mode 100644 index eb762b884f9..00000000000 --- a/providers/aihubmix/models/deep-deepseek-v4-pro.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "DeepSeek V4 Pro (DeepSeek)" -description = "Flagship DeepSeek model for coding, reasoning, and agentic work" -family = "deepseek-thinking" -release_date = "2026-04-24" -last_updated = "2026-04-24" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -knowledge = "2025-05" -open_weights = true - -[interleaved] -field = "reasoning_content" - -[cost] -input = 0.478 -output = 0.956 -cache_read = 0.004302 - -[limit] -context = 1_000_000 -output = 384_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/aihubmix/models/xiaomi-mimo-v2.5-pro.toml b/providers/aihubmix/models/xiaomi-mimo-v2.5-pro.toml deleted file mode 100644 index 2d493daadb9..00000000000 --- a/providers/aihubmix/models/xiaomi-mimo-v2.5-pro.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "xiaomi/mimo-v2.5-pro" -reasoning_options = [{ type = "toggle" }] -name = "Xiaomi MiMo-V2.5-Pro" -family = "mimo-v2.5-pro" -last_updated = "2026-05-13" - -[interleaved] -field = "reasoning_content" - -[cost] -input = 1.1 -output = 3.3 -cache_read = 0.22 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 2.2 -output = 6.6 -cache_read = 0.44 diff --git a/providers/aihubmix/models/xiaomi-mimo-v2.5.toml b/providers/aihubmix/models/xiaomi-mimo-v2.5.toml deleted file mode 100644 index 04bbbaaaf19..00000000000 --- a/providers/aihubmix/models/xiaomi-mimo-v2.5.toml +++ /dev/null @@ -1,19 +0,0 @@ -base_model = "xiaomi/mimo-v2.5" -reasoning_options = [{ type = "toggle" }] -name = "Xiaomi MiMo-V2.5" -family = "mimo-v2.5" -last_updated = "2026-05-13" - -[interleaved] -field = "reasoning_content" - -[cost] -input = 0.44 -output = 2.2 -cache_read = 0.088 - -[[cost.tiers]] -tier = { type = "context", size = 256_000 } -input = 0.88 -output = 4.4 -cache_read = 0.176 diff --git a/providers/aihubmix/models/zai-glm-5.1.toml b/providers/aihubmix/models/zai-glm-5.1.toml deleted file mode 100644 index 9f09d1e0b0f..00000000000 --- a/providers/aihubmix/models/zai-glm-5.1.toml +++ /dev/null @@ -1,28 +0,0 @@ -name = "GLM-5.1 (Z.ai)" -description = "Flagship GLM model for hybrid reasoning, coding, and agentic engineering" -family = "glm" -release_date = "2026-03-27" -last_updated = "2026-03-27" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -structured_output = true -open_weights = true - -[interleaved] -field = "reasoning_content" - -[cost] -input = 0.845 -output = 3.38 -cache_read = 0.183112 - -[limit] -context = 200_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/sync.md b/sync.md index ce651872830..70b284d35db 100644 --- a/sync.md +++ b/sync.md @@ -256,8 +256,10 @@ xAI is implemented in `packages/core/src/sync/providers/xai.ts`. - The endpoint owns capabilities, limits, reasoning controls and text/cache pricing. Fields it has no surface for keep whatever was authored: `family`, `temperature`, `interleaved`, `knowledge`, `experimental`, `provider`, `limit.input`, and the `input_audio`/`output_audio`/`reasoning` rates (its `pricing` object carries only `input`, `output`, `cache_read`, `cache_write` and `tiers`). - Booleans it omits mean unknown, not denied: 107 of 408 routes send no `reasoning` and 100 send no `tool_call`, and none sends `false`. A missing flag is left undefined so the lab entry's value is inherited rather than overridden off. - Modalities can only widen. The endpoint under-reports some routes (it lists `text,image` for `kimi-k2.5`, whose lab entry records video), so its list is unioned with the lab entry the relay factors onto and with the existing file, never used to replace either. A modality the endpoint never listed is still removable by editing the file. +- Limits are read the same way, and for the same reason: the endpoint restates windows in decimal (8 `glm` routes quote 204800 as 200000, four quote 1048576 as 1000000) and a restatement is not the host narrowing the window. A decimal restatement of a binary window loses at most `1000/1024` per K unit, so `1000³/1024³` — three nested unit swaps — is the floor of what a restatement can explain; a stated limit below an accepted one but at or above that ratio resolves to the accepted value and writes no override. The test is applied to whichever side states the limit, so it also retires restatements an earlier sync already wrote into files (three MiniMax entries quoted 131072 as 128000/128100). +- A limit the endpoint does not quote falls back to the authored one rather than to the lab's, because an authored value is not a worse copy of the lab's but a narrower one on purpose: `kimi-k2.5` serves 32768 of a 262144 window. Genuine host caps survive the ratio test and are still written — 7 of the current 408 routes, from `grok-code-fast-1` at 10000 of 256000 to `gpt-5-chat-latest` serving 128000 of a 400000 window. - `cache_read` that merely repeats `input` is how the endpoint spells "no cache discount" (6 models and 4 context tiers do this); it is read as absent rather than published as a rate, which would understate a cached read by up to 10x. -- `budget_tokens` arrives bare — 103 live entries state that a budget exists but never its bounds — so authored `min`/`max` are carried through instead of being replaced with an unbounded range. +- `budget_tokens` arrives bare — 103 live entries state that a budget exists but never its bounds — so authored `min`/`max` are carried through instead of being replaced with an unbounded range. There is no second baseline behind the file: `ModelMetadata` has no `reasoning_options` field, so a lab entry cannot state a budget range and a bare budget written here is not shadowing one. A budget range is a property of the host's API, not of the model. - A `toggle` published alongside an effort list containing `none` is folded to the effort list alone, per the `AGENTS.md` reasoning-options table. AIHubMix accepts whichever off switch the caller's SDK speaks and maps it, so the other dialects that reach the same off state are named in the file header instead. - `authoritativeHeaders` is on: the adapter re-derives the toggle/folded wire-path comment from each response, so a stale header cannot outlive the options it documents. It supersedes hand-written wire paths only — price citations, source links and live-test records in the same header are carried through, since the response cannot reproduce them. - AIHubMix relays upstream models under its own IDs, so a relay is factored onto the lab metadata it serves (`base_model`) whenever that metadata exists, and then records only what it actually changes. @@ -265,11 +267,14 @@ xAI is implemented in `packages/core/src/sync/providers/xai.ts`. - `VENDOR_LABS` maps the four labs the two registries spell differently (`zhipu`/`zhipuai`, `moonshot`/`moonshotai`, `bytedance`/`bytedance-seed`, `meituan-longcat`/`meituan`). It maps namespaces only; no entry decides what a model is or which lab built it. - Dated release tags are deliberately not special-cased: `gemini-2.5-pro-preview-06-05` is a pinned snapshot, not the model `google/gemini-2.5-pro`, and the endpoint does not declare it a variant of one. - Lab IDs are matched case-insensitively: AIHubMix spells `minimax-m2` where the lab spells `MiniMax-M2`, and the same model can arrive under several casings, so relays are deduped on the case-folded ID. -- A relay with neither resolvable lab metadata nor the `release_date`, `open_weights` and limits a standalone entry requires is skipped rather than written with invented values. Of 408 listed models the endpoint names a `vendor` for 292, declares `variant_of` for 76, dates 304 and states `open_weights` for 289; the remaining gap is what the skip notices report back upstream. +- A standalone entry is only authored where the response names no `vendor`. A named lab built the model, so the relay belongs on `base_model`, and AGENTS.md treats a full standalone definition for a nameable lab model as a blocker — so a relay whose lab entry does not exist yet (81 of the current 407 routes, 38 of them described completely enough that the adapter would otherwise have written a full standalone definition) is skipped and reported for a human to add `models//.toml`, after which it factors with no change to the adapter. A standalone file already in the repo keeps being updated: what it should have been is upstream's call, and freezing it would only stall its prices. +- Where no vendor is named, the entry still has to carry the `release_date`, `open_weights` and limits a full definition requires, and is skipped rather than written with invented values when it does not. Of 408 listed models the endpoint names a `vendor` for 292, declares `variant_of` for 76, dates 304 and states `open_weights` for 289; the remaining gap is what the skip notices report back upstream. - Two limit sentinels are read as absent rather than as real ceilings: `max_output: 0` (104 of 408 models) and a `max_output` at or above `context_length` (36), which is the context window quoted a second time. - `reasoning_options[]` entries carry an AIHubMix-only `default` key that the strict `ReasoningOption` schema rejects, and spell two effort levels differently (`no_think`, `instant`), so translation drops the extra key and maps those onto `none` and `minimal`. - A price field is omitted when the model has no such rate, so an omitted `cache_read`/`cache_write` clears an authored one; authored pricing survives only when the endpoint quotes nothing at all for the model, or for the rates it never quotes at all (see above). -- Routing aliases such as `alicloud-glm-5.1` and `deep-deepseek-v4-pro` are served but not listed by the endpoint, so local files missing from the response are retained (`deleteMissing: false`) and each opens a deduped GitHub issue. +- The catalog boundary is AIHubMix's main model list, which is what the endpoint returns. Callable is not the boundary: hidden channel aliases such as `alicloud-glm-5.1` and `deep-deepseek-v4-pro` answer HTTP 200 by routing to a listed model and echo that model's ID back, and several hundred further routes are callable without being listed. Eight such alias files were dropped from this provider; they duplicated entries already in the catalog under the listed ID. +- A route can still rotate out of the list for a spell without being retired, so a local file absent from one response is retained (`deleteMissing: false`) and opens a deduped GitHub issue naming both readings — confirm the rotation, or drop the file if it is a hidden channel alias. +- `trackMissingModels` is set, so relays the adapter skips open deduped `[missing-model]` issues even though creates are enabled. Without it a provider that creates most models but cannot write some of them would emit notices nobody acts on; the flag is implied by `skipCreates` and settable on its own for exactly this case. ## Tinfoil Notes From 8d02b14f3ca241f730f9310fe2c482d669b58f53 Mon Sep 17 00:00:00 2001 From: chenxue Date: Mon, 14 Sep 2026 13:42:18 +0800 Subject: [PATCH 20/20] aihubmix: resolve limit restatements against the lab window, keep human notes Limits - The unit-restatement test now compares a ratio instead of a direction. The endpoint restates 204800 as 200000 and 1000000 as 1048576, and neither is the host stating a different window; checking only the narrowing side left 20 routes writing an override that states no difference at all. - The accepted value is looked for in the lab entry first and only then in the provider file, so a restatement resolves to the spelling that makes the override disappear. Resolving file-first pinned 10 imprecise numbers forever (qwen3.7-flash's 991000 for the lab's 1000000). - Whatever the restatement resolves to is clamped to the lab's window, after the resolution rather than instead of it: a relay cannot serve a wider window than the model it relays, and an endpoint quoting back the file's own stale ceiling is only caught by a later clamp (grok-4.5 held 1000000 against a lab 500000). Header - The two wire-path lines this adapter owns are dropped whether or not a derived block replaces them. Keeping them when nothing is derived left a route advertising a toggle it no longer had, and no later sync could tell. - A superseded line is recognised by its opening on the trimmed line, so an indented ` # Toggle:` no longer outlives its block. - Eight provider files carried human notes mid-body, which a sync drops; moved above the first key as AGENTS.md requires. Names - The endpoint label is compared on the bare ID, which is what resolved the base model, so 10 namespaced routes stop taking a redundant storefront override. - A blank label is skipped rather than written: ModelBase.name is min(1), and writing one through aborted the whole provider's sync at validation. normalizeModelSlug is exported from openrouter.ts, which already serves as the shared helper module for the other provider adapters. Co-Authored-By: Claude Opus 5 --- packages/core/src/sync/providers/aihubmix.ts | 122 ++++++++++-- .../core/src/sync/providers/openrouter.ts | 2 +- packages/core/test/sync.test.ts | 179 +++++++++++++++++- .../aihubmix/models/claude-opus-4-6.toml | 2 +- .../aihubmix/models/claude-opus-4-7.toml | 2 +- .../models/claude-opus-4-8-think.toml | 2 +- .../aihubmix/models/claude-opus-4-8.toml | 2 +- .../aihubmix/models/claude-sonnet-4-6.toml | 2 +- .../aihubmix/models/gemini-2.5-flash.toml | 2 +- providers/aihubmix/models/gemini-2.5-pro.toml | 2 +- .../aihubmix/models/gemini-3.7-flash.toml | 4 +- sync.md | 8 +- 12 files changed, 303 insertions(+), 26 deletions(-) diff --git a/packages/core/src/sync/providers/aihubmix.ts b/packages/core/src/sync/providers/aihubmix.ts index 0127848f397..d7a30c3352d 100644 --- a/packages/core/src/sync/providers/aihubmix.ts +++ b/packages/core/src/sync/providers/aihubmix.ts @@ -4,7 +4,7 @@ import { z } from "zod"; import { describeModel } from "../../describe.js"; import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js"; -import { factorBaseModel, modelMetadata } from "./openrouter.js"; +import { factorBaseModel, modelMetadata, normalizeModelSlug } from "./openrouter.js"; const API_ENDPOINT = "https://aihubmix.com/api/v1/models?type=llm"; @@ -145,10 +145,15 @@ async function readLabMetadataIDs(modelsDir: string) { // The same off state is reachable from whichever dialect the caller speaks, so // an off switch has no single wire path. Name one per protocol. -const DIALECTS = +const DIALECT_PATHS = '# $.enable_thinking = true|false on the OpenAI-compatible /v1/chat/completions path (verified live 2026-09-11);\n' + - '# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path.\n' + - "# https://docs.aihubmix.com/cn/api/unified-inference\n"; + '# $.thinking.type = "enabled"|"disabled"|"adaptive" on /v1/messages; $.generationConfig.thinkingConfig on the Gemini path.\n'; +// Cited on its own line, because a human wrote this exact line by hand in +// `gemini-3.7-flash.toml` — it is a source for the whole gateway, not a claim +// about one model's options, and so it is carried through as a note rather than +// being owned by the block. The two lines above are only ever this adapter's own. +const DIALECT_SOURCE = "# https://docs.aihubmix.com/cn/api/unified-inference\n"; +const DIALECTS = DIALECT_PATHS + DIALECT_SOURCE; const TOGGLE_HEADER = "# Toggle:\n" + DIALECTS; // Where the catalog spells the off state as `effort = none`, the other dialects // still reach it, and the folded toggle is the only place that was recorded. @@ -286,7 +291,7 @@ export function buildAihubmixModel( if (base !== undefined) { return factorBaseModel( base, - { name: existing?.name, description: existing?.description, ...shared }, + { name: factoredName(model, base, existing), description: existing?.description, ...shared }, limit, existing?.base_model === base ? existing.base_model_omit : undefined, ); @@ -431,16 +436,80 @@ function foldsToggle(options: { type: string; values?: string[] }[]) { } /** - * Lines this adapter authors, plus the hand-written wire paths it supersedes. - * Anything else in the header is a human note — a price citation, a live-test - * record — that the response cannot reproduce, so it is carried through. + * The display name to record on a factored entry. `inheritedOverride` already + * drops a name the lab entry states identically, but the two registries punctuate + * the same name differently — the endpoint writes `GLM 5.3` where the lab writes + * `GLM-5.3` — and taking the endpoint's spelling as an override on 78 entries + * would fight the lab's own naming across the catalog for no gain. + * + * So the endpoint's label is recorded only where the relay is not simply that lab + * model under another punctuation: its ID, normalised, differs from the base + * model's slug. That is the same test `shouldPreserveFactoredName` applies for + * OpenRouter, and it is what keeps `coding-glm-4.6-free` reading "Coding GLM 4.6 + * (free)" instead of inheriting a bare "GLM-4.6" it shares with two other routes. + * A relay that *is* the lab model keeps deferring to the lab's spelling, including + * where the lab renamed it (`gemini-3-pro-image` shows as "Nano Banana Pro"). And + * the endpoint's label only ever fills a create: an update keeps the name the file + * states, so this cannot rewrite a spelling a human chose. + */ +function factoredName(model: AihubmixModel, base: string, existing: ExistingModel | undefined) { + // A name already on the file is a human's call and outranks the endpoint's label, + // which is a storefront string: 4 files spell their model the way its lab does + // (`MiMo-V2.5`, `MiniMax-M2.7`) where the endpoint sends `Mimo V2.5`. Handing it + // straight through stays correct anyway — `inheritedOverride` drops a name the lab + // states identically, which is what retires the 27 redundant ones `dev` carries. + if (existing?.name !== undefined) return existing.name; + // A blank label is not a name. `ModelBase.name` is `min(1)`, so writing one + // through would abort the whole provider's sync at validation rather than skip + // the field, and the standalone path never had to care because it only ever + // passed a name that had already been validated. + if (model.model_name == null || model.model_name.trim() === "") return undefined; + // Compared on the bare ID, because that is what resolved the base model: + // `relayChain` walks `bareID(model_id)`, so `Qwen/QwQ-32B` reaches + // `qwen/qwq-32b`. Normalising the namespaced form instead would never match its + // own slug, and each of the 10 namespaced routes would take a redundant + // storefront override the moment its lab file lands. + const slug = base.split("/").slice(1).join("/"); + return normalizeModelSlug(bareID(model.model_id)) === normalizeModelSlug(slug) ? undefined : model.model_name; +} + +/** + * A bare wire-path line, which the derived block restates in full. These four + * openings introduce nothing but the field to send, so replacing one loses + * nothing — `# Effort: reasoning_effort = low|high|max` says less than the block + * that supersedes it. + * + * Matching an opening rather than a substring is the point. Keying on `$.` or on + * the docs host would also delete lines that merely mention one: seven files on + * `dev` carry a header, and `claude-opus-5` and `qwen3.8-max` each state a wire + * path together with a dated live test the response cannot reproduce + * ("verified live 2026-08-11"). Those are notes, and notes are carried through + * even where they overlap the block — a second statement of the same wire path + * costs nothing, a deleted verification date cannot be recovered. */ -const WIRE_PATH_LINE = /^#\s*(Toggle|Effort|Budget|Off is effort)\b|\$\.|thinkingConfig|docs\.aihubmix\.com\/cn\/api/; +const AUTHORED_OPENING = /^#\s*(Toggle|Effort|Budget|Off is effort)\b/; function composeHeader(existingHeader: string | undefined, derived: string | undefined) { + // The two wire-path lines are this adapter's own restatement of the block, so + // they go whether or not a block replaces them. Keeping them when nothing is + // derived is what left a route advertising a toggle it no longer has: the block + // vanished, its tail survived as a "note", and no later sync could tell the + // difference — the file never self-corrected. + // + // The source line is kept unless a derived block restates it, which is only to + // avoid stating it twice. It is a citation for the gateway rather than a claim + // about this model, and a human wrote this exact line in `gemini-3.7-flash`. + const authored = new Set( + (derived === undefined ? DIALECT_PATHS : DIALECTS) + .split("\n") + .map((line) => line.trim()) + .filter((line) => line !== ""), + ); const notes = (existingHeader ?? "") .split("\n") - .filter((line) => line.trim() !== "" && !WIRE_PATH_LINE.test(line)); + .filter((line) => + line.trim() !== "" && !AUTHORED_OPENING.test(line.trim()) && !authored.has(line.trim()), + ); const header = (derived ?? "") + (notes.length > 0 ? `${notes.join("\n")}\n` : ""); return header === "" ? undefined : header; } @@ -506,11 +575,36 @@ const UNIT_RESTATEMENT_FLOOR = 1000 ** 3 / 1024 ** 3; function resolveLimit(quoted?: number, authored?: number, lab?: number) { const stated = quoted ?? authored; if (stated === undefined) return undefined; - for (const accepted of [authored, lab]) { - if (accepted === undefined || stated >= accepted) continue; - if (stated / accepted >= UNIT_RESTATEMENT_FLOOR) return accepted; + // The restatement reads the same from either side, so the comparison is a ratio + // rather than a direction: the endpoint quotes an accepted 204800 as 200000 and + // an accepted 1000000 as 1048576, and neither is the host stating a different + // window. Checking only the narrowing side left 20 routes writing an override + // that states no difference at all (`glm-5.3` recording 1048576 against a lab + // window of 1000000). + // + // The lab entry is tried first, because a restatement should resolve to the + // spelling that makes the override disappear: matching the lab means + // `inheritedOverride` drops the key entirely, while resolving to the value the + // provider file happens to hold would pin that spelling forever — `991000` is + // itself just an imprecise way of writing the lab's 1000000. The file's own + // value still decides where the lab states no such key, and a genuine host + // restriction falls below the floor and is written as the delta it is. + let resolved = stated; + for (const accepted of [lab, authored]) { + if (accepted === undefined) continue; + const ratio = Math.min(stated, accepted) / Math.max(stated, accepted); + if (ratio < UNIT_RESTATEMENT_FLOOR) continue; + resolved = accepted; + break; } - return stated; + // A relay cannot serve a wider window than the model it relays: the window is the + // model's property and a host can only restrict it. Applied to whatever the + // restatement resolved, not in place of it — an endpoint quoting the same stale + // ceiling the file already holds resolves to that number, and clamping only + // afterwards is what catches it (`grok-4.5` quoting the file's own 1000000 output + // against a lab window of 500000). Where the lab entry is the stale side, + // `models/` is where that gets corrected. + return lab !== undefined && resolved > lab ? lab : resolved; } function modalities(value: string | null | undefined, fallback: string[]) { diff --git a/packages/core/src/sync/providers/openrouter.ts b/packages/core/src/sync/providers/openrouter.ts index 77164f7caa6..2121c93bf8d 100644 --- a/packages/core/src/sync/providers/openrouter.ts +++ b/packages/core/src/sync/providers/openrouter.ts @@ -440,7 +440,7 @@ function shouldPreserveFactoredName( return normalizeModelSlug(modelSlug) !== normalizeModelSlug(canonicalSlug); } -function normalizeModelSlug(value: string) { +export function normalizeModelSlug(value: string) { return value.toLowerCase().replaceAll(/[^a-z0-9]/g, ""); } diff --git a/packages/core/test/sync.test.ts b/packages/core/test/sync.test.ts index c9d8d00f026..08d28fcba12 100644 --- a/packages/core/test/sync.test.ts +++ b/packages/core/test/sync.test.ts @@ -5375,6 +5375,77 @@ test("does not let a narrower AIHubMix modality list delete an accepted one", () expect(widened?.modalities?.input).toEqual(["text", "image", "pdf"]); }); +test("records an AIHubMix route's own name only where its ID is not the lab slug", () => { + // A factored create used to pass the file's name, so a route with no file yet + // recorded none at all and rendered as the lab model: `coding-glm-4.6-free`, + // `coding-glm-4.6` and `glm-4.6` all read "GLM-4.6". + const catalog = new Map(aihubmixCatalog); + const free = aihubmixModel({ + model_id: "coding-gemini-3.1-flash-lite-free", + model_name: "Coding Gemini 3.1 Flash Lite (free)", + variant_of: "gemini-3.1-flash-lite", + }); + catalog.set(free.model_id, free); + const created = buildAihubmixModel(free, undefined, aihubmixLabIDs, catalog); + expect(created).toMatchObject({ + base_model: "google/gemini-3.1-flash-lite", + name: "Coding Gemini 3.1 Flash Lite (free)", + }); + + // An update keeps the name the file states. Four files spell their model the way + // its lab does (`MiMo-V2.5`) where the endpoint sends a storefront `Mimo V2.5`, + // and the endpoint's label is not a reason to rewrite a human's spelling. + const authored = buildAihubmixModel( + free, + { id: free.model_id, name: "Coding Gemini 3.1 Flash-Lite (free)" }, + aihubmixLabIDs, + catalog, + ); + expect(authored).toMatchObject({ name: "Coding Gemini 3.1 Flash-Lite (free)" }); + + // A relay that *is* that lab model keeps deferring to the lab's spelling, so a + // registry punctuating the same name differently does not become an override + // on every entry in the catalog. + // A blank label is not a name, and it is not merely ignored: `ModelBase.name` is + // `min(1)`, so a `""` handed through aborts the whole provider's sync at + // validation and writes no file at all. The endpoint types the field + // `nullish()`, so both shapes have to resolve to "no name". + for (const blank of [null, "", " "]) { + const bare = aihubmixModel({ + model_id: "coding-gemini-3.1-flash-lite-free", + model_name: blank as never, + variant_of: "gemini-3.1-flash-lite", + }); + const built = buildAihubmixModel(bare, undefined, aihubmixLabIDs, catalog); + expect(built).toMatchObject({ base_model: "google/gemini-3.1-flash-lite" }); + expect(built).not.toHaveProperty("name"); + } + + // A namespaced route is compared on the bare ID, because that is what resolved + // its base model: `relayChain` walks `bareID(model_id)`. Normalising the + // namespaced form would never equal its own slug, and the route would take a + // storefront override restating the lab's own name. + const namespaced = buildAihubmixModel( + // The label deliberately differs from the lab's, so an override would actually + // be recorded if the comparison used the namespaced form. + aihubmixModel({ model_id: "Google/gemini-3.1-flash-lite", model_name: "Gemini 3.1 Flash Lite Turbo" }), + undefined, + aihubmixLabIDs, + aihubmixCatalog, + ); + expect(namespaced).toMatchObject({ base_model: "google/gemini-3.1-flash-lite" }); + expect(namespaced).not.toHaveProperty("name"); + + const punctuated = buildAihubmixModel( + aihubmixModel({ model_name: "Gemini 3.1 Flash-Lite" }), + undefined, + aihubmixLabIDs, + aihubmixCatalog, + ); + expect(punctuated).toMatchObject({ base_model: "google/gemini-3.1-flash-lite" }); + expect(punctuated).not.toHaveProperty("name"); +}); + test("refreshes the AIHubMix wire path without discarding a human note", () => { // The header is authoritative so a stale wire path cannot outlive the options // it documents, but a price citation or live-test record is not reproducible @@ -5390,7 +5461,16 @@ test("refreshes the AIHubMix wire path without discarding a human note", () => { reasoning: true, reasoning_options: [{ type: "toggle" }] as AihubmixModel["reasoning_options"], }); - aihubmix.parseModels({ data: [toggled] }); + const plain = aihubmixModel({ + vendor: null, + model_id: "somelab-bare", + model_name: "SomeLab Bare", + release_date: "2026-05-01", + open_weights: false, + context_length: 262_144, + max_output: 65_536, + }); + aihubmix.parseModels({ data: [toggled, plain] }); const translated = aihubmix.translateModel(toggled, { existing: () => undefined, authored: () => undefined, @@ -5402,6 +5482,51 @@ test("refreshes the AIHubMix wire path without discarding a human note", () => { // The hand-written wire path it supersedes is gone; the citation is not. expect(translated?.header).not.toContain("# Toggle: enable_thinking"); expect(translated?.header).toContain("queried 2026-08-11T09:45:11Z"); + + // A note is recognised by what it opens with, not by mentioning a wire path or + // the docs host: two files on `dev` state a wire path together with a live test + // the response cannot reproduce, and a second statement of the same path costs + // nothing next to a deleted verification date. + const noted = aihubmix.translateModel(toggled, { + existing: () => undefined, + authored: () => undefined, + header: () => + '# Native Messages prefers $.thinking.type = "adaptive" (verified live 2026-08-11).\n' + + "# https://docs.aihubmix.com/cn/api/Claude-Native\n", + }); + expect(noted?.header).toContain("verified live 2026-08-11"); + expect(noted?.header).toContain("docs.aihubmix.com/cn/api/Claude-Native"); + + // A dialect line is only dropped when a derived block restates it. With nothing + // derived there is nothing to restate, and one file's whole citation is that + // line byte for byte. + const citation = "# https://docs.aihubmix.com/cn/api/unified-inference\n"; + expect( + aihubmix.translateModel(plain, { + existing: () => undefined, + authored: () => undefined, + header: () => citation, + })?.header, + ).toBe(citation); + // An opening is recognised after trimming, because `leadingComments` matches on + // the trimmed line but keeps the raw one — so an indented `# Toggle:` arrives + // here still indented, and would otherwise survive as a "note" restating the + // block written directly above it. + const indented = aihubmix.translateModel(toggled, { + existing: () => undefined, + authored: () => undefined, + header: () => " # Toggle: enable_thinking = true|false\n", + })?.header; + expect(indented).not.toContain("# Toggle: enable_thinking"); + + // Written alongside a block that does restate it, it is not doubled. + const doubled = + aihubmix.translateModel(toggled, { + existing: () => undefined, + authored: () => undefined, + header: () => citation, + })?.header ?? ""; + expect(doubled.split(citation.trim()).length - 1).toBe(1); }); test("reads a missing AIHubMix reasoning or tool flag as unknown, not as false", () => { @@ -5494,6 +5619,58 @@ test("inherits a limit the AIHubMix endpoint only restates in decimal", () => { ); expect(capped?.limit?.context).toBe(128_000); + // The restatement reads the same from the other side: the lab window is + // 1_048_576 and the endpoint quotes 1_048_576 against a provider file holding the + // decimal 1_000_000. Resolving to the lab is what lets `inheritedOverride` drop + // the key — resolving to the file's own spelling would pin 1_000_000 forever even + // though it is only an imprecise way of writing the same window. + const reverse = buildAihubmixModel( + aihubmixModel({ context_length: 1_048_576, max_output: 65_536 }), + { ...aihubmixAuthored, limit: { context: 1_000_000, output: 65_536 } }, + aihubmixLabIDs, + ); + expect(reverse?.limit?.context).toBeUndefined(); + + // With no lab entry to defer to, the file's own value is the accepted one, and the + // restatement still reads from either side: a host route quoting the binary + // 1_048_576 for the 1_000_000 already on the file keeps the file's number instead + // of rewriting it to say the same window differently. + const hostRestated = buildAihubmixModel( + aihubmixModel({ + vendor: null, + model_id: "somelab-restated", + model_name: "SomeLab Restated", + release_date: "2026-05-01", + open_weights: false, + context_length: 1_048_576, + max_output: 65_536, + }), + { ...aihubmixAuthored, id: "somelab-restated", limit: { context: 1_000_000, output: 65_536 } }, + aihubmixLabIDs, + ); + expect(hostRestated?.limit?.context).toBe(1_000_000); + + // A relay cannot serve a wider window than the model it relays, so a ceiling above + // the lab's own resolves to the lab's rather than advertising tokens no request can + // reach. Doubling the output is far past any restatement. + const overreach = buildAihubmixModel( + aihubmixModel({ context_length: 1_048_576, max_output: 131_072 }), + undefined, + aihubmixLabIDs, + ); + expect(overreach?.limit?.output).toBeUndefined(); + + // The clamp applies to whatever the restatement resolved, not instead of it. An + // endpoint quoting back the same stale ceiling the file already holds resolves to + // that number — so clamping only afterwards is what catches it. `grok-4.5` did + // exactly this, recording a 1_000_000 output against a 500_000 lab window. + const stale = buildAihubmixModel( + aihubmixModel({ context_length: 1_048_576, max_output: 1_000_000 }), + { ...aihubmixAuthored, limit: { context: 1_048_576, output: 1_000_000 } }, + aihubmixLabIDs, + ); + expect(stale?.limit?.output).toBeUndefined(); + // And an authored cap survives an endpoint that quotes nothing for the ceiling. const authored: ExistingModel = { ...aihubmixAuthored, diff --git a/providers/aihubmix/models/claude-opus-4-6.toml b/providers/aihubmix/models/claude-opus-4-6.toml index 57b40db1786..8f0d7241820 100644 --- a/providers/aihubmix/models/claude-opus-4-6.toml +++ b/providers/aihubmix/models/claude-opus-4-6.toml @@ -1,3 +1,4 @@ +# Native Messages prefers $.thinking.type = "adaptive" with $.output_config.effort = "low"|"medium"|"high"|"max"; enabled budget_tokens >= 1024 is deprecated and must be < $.max_tokens. https://docs.aihubmix.com/cn/api/Claude-Native (accessed 2026-06-25) name = "Claude Opus 4.6" description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" family = "claude-opus" @@ -6,7 +7,6 @@ last_updated = "2026-03-13" attachment = true reasoning = true reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] -# Native Messages prefers $.thinking.type = "adaptive" with $.output_config.effort = "low"|"medium"|"high"|"max"; enabled budget_tokens >= 1024 is deprecated and must be < $.max_tokens. https://docs.aihubmix.com/cn/api/Claude-Native (accessed 2026-06-25) temperature = true tool_call = true structured_output = true diff --git a/providers/aihubmix/models/claude-opus-4-7.toml b/providers/aihubmix/models/claude-opus-4-7.toml index 0fb0b1c89f3..b085c8e69a2 100644 --- a/providers/aihubmix/models/claude-opus-4-7.toml +++ b/providers/aihubmix/models/claude-opus-4-7.toml @@ -1,3 +1,4 @@ +# Native Messages uses $.thinking.type = "adaptive" and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected. https://docs.aihubmix.com/cn/blogs/Claude-Opus4.7 (accessed 2026-06-25) name = "Claude Opus 4.7" description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" family = "claude-opus" @@ -6,7 +7,6 @@ last_updated = "2026-04-16" attachment = true reasoning = true reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] -# Native Messages uses $.thinking.type = "adaptive" and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected. https://docs.aihubmix.com/cn/blogs/Claude-Opus4.7 (accessed 2026-06-25) temperature = false tool_call = true structured_output = true diff --git a/providers/aihubmix/models/claude-opus-4-8-think.toml b/providers/aihubmix/models/claude-opus-4-8-think.toml index 52f226453d2..a829e13abba 100644 --- a/providers/aihubmix/models/claude-opus-4-8-think.toml +++ b/providers/aihubmix/models/claude-opus-4-8-think.toml @@ -1,6 +1,6 @@ +# Anthropic-compatible /v1/messages: $.thinking.type = "enabled"|"disabled"|"adaptive" (disabled turns reasoning off = toggle) and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected on this Opus tier. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message (accessed 2026-07-02) base_model = "anthropic/claude-opus-4-8" reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] -# Anthropic-compatible /v1/messages: $.thinking.type = "enabled"|"disabled"|"adaptive" (disabled turns reasoning off = toggle) and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected on this Opus tier. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message (accessed 2026-07-02) [interleaved] field = "reasoning_content" diff --git a/providers/aihubmix/models/claude-opus-4-8.toml b/providers/aihubmix/models/claude-opus-4-8.toml index e8d612cf002..3dd610d9365 100644 --- a/providers/aihubmix/models/claude-opus-4-8.toml +++ b/providers/aihubmix/models/claude-opus-4-8.toml @@ -1,6 +1,6 @@ +# Anthropic-compatible /v1/messages: $.thinking.type = "enabled"|"disabled"|"adaptive" (disabled turns reasoning off = toggle) and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected on this Opus tier. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message (accessed 2026-07-02) base_model = "anthropic/claude-opus-4-8" reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "xhigh", "max"] }] -# Anthropic-compatible /v1/messages: $.thinking.type = "enabled"|"disabled"|"adaptive" (disabled turns reasoning off = toggle) and $.output_config.effort = "low"|"medium"|"high"|"xhigh"|"max"; manual budget_tokens is rejected on this Opus tier. https://docs.aihubmix.com/cn/api-reference/anthropic-compatible/create-a-message (accessed 2026-07-02) interleaved = true diff --git a/providers/aihubmix/models/claude-sonnet-4-6.toml b/providers/aihubmix/models/claude-sonnet-4-6.toml index b252b0f52b3..be1a367a194 100644 --- a/providers/aihubmix/models/claude-sonnet-4-6.toml +++ b/providers/aihubmix/models/claude-sonnet-4-6.toml @@ -1,3 +1,4 @@ +# Native Messages prefers $.thinking.type = "adaptive" with $.output_config.effort = "low"|"medium"|"high"; enabled budget_tokens >= 1024 is deprecated and must be < $.max_tokens. Chat effort "max" maps to native "high". https://docs.aihubmix.com/cn/api/Claude-Native (accessed 2026-06-25) name = "Claude Sonnet 4.6" description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" family = "claude-sonnet" @@ -6,7 +7,6 @@ last_updated = "2026-03-13" attachment = true reasoning = true reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high", "max"] }, { type = "budget_tokens", min = 1_024 }] -# Native Messages prefers $.thinking.type = "adaptive" with $.output_config.effort = "low"|"medium"|"high"; enabled budget_tokens >= 1024 is deprecated and must be < $.max_tokens. Chat effort "max" maps to native "high". https://docs.aihubmix.com/cn/api/Claude-Native (accessed 2026-06-25) temperature = true tool_call = true structured_output = true diff --git a/providers/aihubmix/models/gemini-2.5-flash.toml b/providers/aihubmix/models/gemini-2.5-flash.toml index 23f238b714b..22234024838 100644 --- a/providers/aihubmix/models/gemini-2.5-flash.toml +++ b/providers/aihubmix/models/gemini-2.5-flash.toml @@ -1,3 +1,4 @@ +# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: 0 disables, -1 is dynamic, and manual budgets are 1..24576. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) name = "Gemini 2.5 Flash" description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost" family = "gemini-flash" @@ -6,7 +7,6 @@ last_updated = "2025-06-05" attachment = true reasoning = true reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }, { type = "budget_tokens", min = 0, max = 24_576 }] -# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: 0 disables, -1 is dynamic, and manual budgets are 1..24576. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) temperature = true tool_call = true structured_output = true diff --git a/providers/aihubmix/models/gemini-2.5-pro.toml b/providers/aihubmix/models/gemini-2.5-pro.toml index 8ef984bdbf3..e77eedd0faf 100644 --- a/providers/aihubmix/models/gemini-2.5-pro.toml +++ b/providers/aihubmix/models/gemini-2.5-pro.toml @@ -1,3 +1,4 @@ +# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: -1 is dynamic and manual budgets are 128..32768; 0/off is unsupported. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) name = "Gemini 2.5 Pro" description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" family = "gemini-pro" @@ -6,7 +7,6 @@ last_updated = "2025-06-05" attachment = true reasoning = true reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -# Native Gemini uses $.generationConfig.thinkingConfig.thinkingBudget: -1 is dynamic and manual budgets are 128..32768; 0/off is unsupported. https://cloud.google.com/vertex-ai/generative-ai/docs/thinking (accessed 2026-06-25) temperature = true tool_call = true structured_output = true diff --git a/providers/aihubmix/models/gemini-3.7-flash.toml b/providers/aihubmix/models/gemini-3.7-flash.toml index 31d0f319a9a..63ad68c0ba2 100644 --- a/providers/aihubmix/models/gemini-3.7-flash.toml +++ b/providers/aihubmix/models/gemini-3.7-flash.toml @@ -1,7 +1,7 @@ -base_model = "google/gemini-3.7-flash" - # AIHubMix unified Chat: $.reasoning_effort = low|medium|high # https://docs.aihubmix.com/cn/api/unified-inference +base_model = "google/gemini-3.7-flash" + reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] [cost] diff --git a/sync.md b/sync.md index 70b284d35db..20dccc55516 100644 --- a/sync.md +++ b/sync.md @@ -256,14 +256,20 @@ xAI is implemented in `packages/core/src/sync/providers/xai.ts`. - The endpoint owns capabilities, limits, reasoning controls and text/cache pricing. Fields it has no surface for keep whatever was authored: `family`, `temperature`, `interleaved`, `knowledge`, `experimental`, `provider`, `limit.input`, and the `input_audio`/`output_audio`/`reasoning` rates (its `pricing` object carries only `input`, `output`, `cache_read`, `cache_write` and `tiers`). - Booleans it omits mean unknown, not denied: 107 of 408 routes send no `reasoning` and 100 send no `tool_call`, and none sends `false`. A missing flag is left undefined so the lab entry's value is inherited rather than overridden off. - Modalities can only widen. The endpoint under-reports some routes (it lists `text,image` for `kimi-k2.5`, whose lab entry records video), so its list is unioned with the lab entry the relay factors onto and with the existing file, never used to replace either. A modality the endpoint never listed is still removable by editing the file. -- Limits are read the same way, and for the same reason: the endpoint restates windows in decimal (8 `glm` routes quote 204800 as 200000, four quote 1048576 as 1000000) and a restatement is not the host narrowing the window. A decimal restatement of a binary window loses at most `1000/1024` per K unit, so `1000³/1024³` — three nested unit swaps — is the floor of what a restatement can explain; a stated limit below an accepted one but at or above that ratio resolves to the accepted value and writes no override. The test is applied to whichever side states the limit, so it also retires restatements an earlier sync already wrote into files (three MiniMax entries quoted 131072 as 128000/128100). +- Limits are read the same way, and for the same reason: the endpoint restates windows in decimal (8 `glm` routes quote 204800 as 200000) and in binary (four quote 1000000 as 1048576), and a restatement in either direction is not the host stating a different window. A decimal restatement of a binary window loses at most `1000/1024` per K unit, so `1000³/1024³` — three nested unit swaps — is the floor of what a restatement can explain. The comparison is therefore a ratio and not a direction: two limits within that floor of each other are one window spelled twice, and the stated one resolves to the accepted value and writes no override. Checking only the narrowing side left 20 routes recording an override that states no difference at all, `glm-5.3` among them writing 1048576 against a lab window of 1000000. Applying it to whichever side states the limit is also what retires restatements an earlier sync already wrote into files (three MiniMax entries quoted 131072 as 128000/128100). +- The accepted value tried first is the lab entry's, and the file's own only where the lab states no such key, because a restatement should resolve to the spelling that makes the override disappear: matching the lab lets the factoring step drop the key entirely, while resolving to whatever the provider file happens to hold would pin that spelling forever — `qwen3.7-flash` carries 991000, which is itself only an imprecise way of writing the lab's 1000000. Resolving file-first preserved 10 such overrides, including a `claude-opus-4-8` still recording 200000/32000 against a 1000000/128000 lab window. +- Whatever the restatement resolves to is then clamped to the lab's window, since a relay cannot serve a wider one than the model it relays — the window is the model's property and a host can only restrict it. The clamp runs after the resolution rather than in place of it, because an endpoint quoting back the same stale ceiling the file already holds resolves to that number and only a later clamp catches it: `grok-4.5` carried 1000000 for both limits against a 500000 lab window, and the endpoint quotes that same 1000000. Past the two sentinels below, 14 routes quote a window wider than their lab entry's (`qwen3.8-2.4t-a95b` at 1000000 of 262144, `gemma-4-31b-it` at 131100 of 32768), all reported upstream. Where the lab entry is the stale side, `models/` is where that gets corrected. - A limit the endpoint does not quote falls back to the authored one rather than to the lab's, because an authored value is not a worse copy of the lab's but a narrower one on purpose: `kimi-k2.5` serves 32768 of a 262144 window. Genuine host caps survive the ratio test and are still written — 7 of the current 408 routes, from `grok-code-fast-1` at 10000 of 256000 to `gpt-5-chat-latest` serving 128000 of a 400000 window. - `cache_read` that merely repeats `input` is how the endpoint spells "no cache discount" (6 models and 4 context tiers do this); it is read as absent rather than published as a rate, which would understate a cached read by up to 10x. - `budget_tokens` arrives bare — 103 live entries state that a budget exists but never its bounds — so authored `min`/`max` are carried through instead of being replaced with an unbounded range. There is no second baseline behind the file: `ModelMetadata` has no `reasoning_options` field, so a lab entry cannot state a budget range and a bare budget written here is not shadowing one. A budget range is a property of the host's API, not of the model. - A `toggle` published alongside an effort list containing `none` is folded to the effort list alone, per the `AGENTS.md` reasoning-options table. AIHubMix accepts whichever off switch the caller's SDK speaks and maps it, so the other dialects that reach the same off state are named in the file header instead. - `authoritativeHeaders` is on: the adapter re-derives the toggle/folded wire-path comment from each response, so a stale header cannot outlive the options it documents. It supersedes hand-written wire paths only — price citations, source links and live-test records in the same header are carried through, since the response cannot reproduce them. +- A superseded wire path is told from a note to keep by what the line opens with (`# Toggle:`, `# Effort:`, `# Budget:`, `# Off is effort`) — the openings `AGENTS.md` itself prescribes — and not by whether the line mentions a field path or the docs host, since keying on the substring would also delete lines that merely contain one: two files state a wire path together with a dated live test the response cannot reproduce. A second statement of the same path costs nothing; a deleted verification date cannot be recovered. The opening is matched on the trimmed line, or an indented ` # Toggle:` outlives the block it documented. +- The two lines naming this gateway's wire paths are the adapter's own restatement of the block, so they go whether or not a block replaces them; only the docs link survives as a note, and only where no derived block already states it. Keeping the wire paths when nothing is derived is what once left a route advertising a toggle it no longer had — the block vanished, its tail survived as a "note", and no later sync could tell the difference, so the file never self-corrected. +- `AGENTS.md` also requires these notes above the first key, since a sync keeps only the leading comment block and drops every comment below it. Eight aihubmix files carried theirs mid-body and are moved up here; without the move the next sync deletes them silently. - AIHubMix relays upstream models under its own IDs, so a relay is factored onto the lab metadata it serves (`base_model`) whenever that metadata exists, and then records only what it actually changes. - The endpoint answers both halves of that lookup itself, so nothing about a relay is inferred from its ID here: `vendor` names the lab that built the model, and `variant_of` names the AIHubMix ID the entry is a routing variant of (`variant_kind` labels it a pricing tier, channel tier, mode preset or deprecated alias). A relay is looked up under its own ID first and then under each `variant_of` hop, nearest first — `qwen3.8-max-preview` is a variant of `qwen3.8-max` and both are published, so the relay factors onto the preview it actually serves. Following the declared chain also resolves relays no string rule reaches, such as `ox-alpha` onto `zhipuai/glm-5.3-flash` and `grok-code-fast-1` onto `xai/grok-build-0.1`. +- The endpoint's label is recorded only where the relay is not that lab model under other punctuation — its bare ID, normalised, differs from the base model's slug. It is compared on the bare ID because that is what resolved the base model, so `Qwen/QwQ-32B` reaches `qwen/qwq-32b`; normalising the namespaced form instead matches nothing and each of the 10 namespaced routes would take a redundant storefront override. The rule keeps `coding-glm-4.6-free` reading "Coding GLM 4.6 (free)" instead of the bare "GLM-4.6" it would share with two other routes, while entries whose endpoint spelling differs only in punctuation (`GLM 5.3` against the lab's `GLM-5.3`) keep deferring to the lab and write nothing. A name already on the file outranks both, since four files spell their model the way its lab does (`MiMo-V2.5`) where the endpoint sends a storefront `Mimo V2.5`; handing it through stays correct because the factoring step drops a name the lab states identically, which is what retires the redundant ones. A blank label is not a name: `ModelBase.name` is `min(1)`, so writing one through aborts the whole provider's sync at validation rather than skipping the field. - `VENDOR_LABS` maps the four labs the two registries spell differently (`zhipu`/`zhipuai`, `moonshot`/`moonshotai`, `bytedance`/`bytedance-seed`, `meituan-longcat`/`meituan`). It maps namespaces only; no entry decides what a model is or which lab built it. - Dated release tags are deliberately not special-cased: `gemini-2.5-pro-preview-06-05` is a pinned snapshot, not the model `google/gemini-2.5-pro`, and the endpoint does not declare it a variant of one. - Lab IDs are matched case-insensitively: AIHubMix spells `minimax-m2` where the lab spells `MiniMax-M2`, and the same model can arrive under several casings, so relays are deduped on the case-folded ID.