From ac76dfb44dba635f27350262437dd7229930fe4e Mon Sep 17 00:00:00 2001 From: siyoon Date: Sun, 13 Sep 2026 20:47:45 +0900 Subject: [PATCH 1/6] feat(sync): replace Friendli generator with SyncProvider --- packages/core/script/generate-friendli.ts | 505 ---------------- packages/core/src/sync/index.ts | 5 +- packages/core/src/sync/providers/friendli.ts | 543 ++++++++++++++++++ .../models/MiniMaxAI/MiniMax-M2.5.toml | 10 +- .../models/deepseek-ai/DeepSeek-V3.2.toml | 28 +- .../models/google/gemma-4-31B-it.toml | 14 +- .../friendli/models/zai-org/GLM-5.1.toml | 16 +- .../friendli/models/zai-org/GLM-5.2.toml | 25 +- .../models/zai-org/GLM-5.3-Flash.toml | 26 + .../friendli/models/zai-org/GLM-5.3.toml | 35 +- 10 files changed, 638 insertions(+), 569 deletions(-) delete mode 100644 packages/core/script/generate-friendli.ts create mode 100644 packages/core/src/sync/providers/friendli.ts create mode 100644 providers/friendli/models/zai-org/GLM-5.3-Flash.toml diff --git a/packages/core/script/generate-friendli.ts b/packages/core/script/generate-friendli.ts deleted file mode 100644 index d048f3d8369..00000000000 --- a/packages/core/script/generate-friendli.ts +++ /dev/null @@ -1,505 +0,0 @@ -#!/usr/bin/env bun - -import { mkdir } from "node:fs/promises"; -import path from "node:path"; -import { z } from "zod"; - -import { inferKimiFamily } from "../src/family.js"; - -// Friendli API endpoint -const API_ENDPOINT = "https://api.friendli.ai/serverless/v1/models"; - -// Zod schemas for API response validation -const Functionality = z.object({ - tool_call: z.boolean(), - parallel_tool_call: z.boolean(), - structured_output: z.boolean(), -}); - -const Pricing = z.object({ - input: z.number(), - output: z.number(), - response_time: z.number(), - unit_type: z.enum(["TOKEN", "SECOND"]), -}); - -const FriendliModel = z - .object({ - id: z.string(), - name: z.string(), - max_completion_tokens: z.number(), - context_length: z.number(), - functionality: Functionality, - pricing: Pricing, - hugging_face_url: z.string().optional(), - description: z.string().optional(), - license: z.string().optional(), - policy: z.string().optional().nullable(), - created: z.number(), // Unix timestamp - }) - .passthrough(); - -const FriendliResponse = z.object({ - data: z.array(FriendliModel), -}); - -// Family inference patterns -const familyPatterns: [RegExp, string][] = [ - [/qwen3/i, "qwen3"], - [/deepseek-r1/i, "deepseek-r1"], - [/glm-4/i, "glm-4"], - [/glm-5/i, "glm"], -]; - -function inferFamily(modelId: string, modelName: string): string | undefined { - const kimiFamily = inferKimiFamily(modelId, modelName); - if (kimiFamily !== undefined) return kimiFamily; - - for (const [pattern, family] of familyPatterns) { - if (pattern.test(modelId) || pattern.test(modelName)) { - return family; - } - } - return undefined; -} - -function extractModelName(fullName: string): string { - // "meta-llama/Llama-3.3-70B-Instruct" -> "Llama 3.3 70B Instruct" - const parts = fullName.split("/"); - const modelName = parts.at(-1) ?? fullName; - return modelName - .replace(/-/g, " ") - .replace(/\b\w/g, (l) => l.toUpperCase()); -} - -// TODO: Replace with functionality.parse_reasoning from API when available -function isReasoningModel(modelId: string): boolean { - const nonReasoningPatterns = [ - /qwen3.*instruct/i, - ]; - - for (const pattern of nonReasoningPatterns) { - if (pattern.test(modelId)) { - return false; - } - } - - // Everything else is reasoning or hybrid reasoning - return true; -} - -function formatNumber(n: number): string { - if (n >= 1000) { - // Format with underscores for readability (e.g., 131_072) - return n.toString().replace(/\B(?=(\d{3})+(?!\d))/g, "_"); - } - return n.toString(); -} - -function timestampToDate(timestamp: number): string { - const date = new Date(timestamp * 1000); - return date.toISOString().slice(0, 10); -} - -function getTodayDate(): string { - return new Date().toISOString().slice(0, 10); -} - -interface ExistingModel { - name?: string; - family?: string; - attachment?: boolean; - reasoning?: boolean; - tool_call?: boolean; - structured_output?: boolean; - temperature?: boolean; - knowledge?: string; - release_date?: string; - last_updated?: string; - open_weights?: boolean; - interleaved?: boolean | { field: string }; - status?: string; - cost?: { - input?: number; - output?: number; - reasoning?: number; - cache_read?: number; - cache_write?: number; - }; - limit?: { - context?: number; - input?: number; - output?: number; - }; - modalities?: { - input?: string[]; - output?: string[]; - }; - provider?: { - npm?: string; - api?: string; - }; -} - -async function loadExistingModel( - filePath: string, -): Promise { - try { - const file = Bun.file(filePath); - if (!(await file.exists())) { - return null; - } - const toml = await import(filePath, { with: { type: "toml" } }).then( - (mod) => mod.default, - ); - return toml as ExistingModel; - } catch (e) { - console.warn(`Warning: Failed to parse existing file ${filePath}:`, e); - return null; - } -} - -interface MergedModel { - name: string; - family?: string; - attachment: boolean; - reasoning: boolean; - tool_call: boolean; - structured_output?: boolean; - temperature: boolean; - knowledge?: string; - release_date: string; - last_updated: string; - open_weights: boolean; - interleaved?: boolean | { field: string }; - status?: string; - cost?: { - input: number; - output: number; - }; - limit: { - context: number; - output: number; - }; - modalities: { - input: string[]; - output: string[]; - }; -} - -function mergeModel( - apiModel: z.infer, - existing: ExistingModel | null, -): MergedModel { - const contextTokens = apiModel.context_length; - const outputTokens = apiModel.max_completion_tokens; - - const openWeights = Boolean(apiModel.hugging_face_url); - - const merged: MergedModel = { - // Always from API - name: extractModelName(apiModel.name), - attachment: false, // All Friendli models are text-only currently - reasoning: isReasoningModel(apiModel.id), - tool_call: apiModel.functionality.tool_call, - temperature: true, - release_date: timestampToDate(apiModel.created), - last_updated: getTodayDate(), - open_weights: openWeights, - limit: { - context: contextTokens, - output: outputTokens, - }, - modalities: { - input: ["text"], - output: ["text"], - }, - }; - - // structured_output only if true - if (apiModel.functionality.structured_output === true) { - merged.structured_output = true; - } - - // Cost from API - ONLY include if unit_type is TOKEN - if (apiModel.pricing.unit_type === "TOKEN") { - merged.cost = { - input: apiModel.pricing.input, - output: apiModel.pricing.output, - }; - } else { - console.log( - ` Note: ${apiModel.id} uses ${apiModel.pricing.unit_type} pricing - cost section omitted`, - ); - } - - // Preserve from existing OR infer - if (existing?.family) { - merged.family = existing.family; - } else { - const inferred = inferFamily(apiModel.id, apiModel.name); - if (inferred) { - merged.family = inferred; - } - } - - // Preserve manual fields from existing - if (existing?.knowledge) { - merged.knowledge = existing.knowledge; - } - if (existing?.interleaved !== undefined) { - merged.interleaved = existing.interleaved; - } - if (existing?.status !== undefined) { - merged.status = existing.status; - } - - return merged; -} - -function formatToml(model: MergedModel): string { - const lines: string[] = []; - - // Basic fields - lines.push(`name = "${model.name.replace(/"/g, '\\"')}"`); - if (model.family) { - lines.push(`family = "${model.family}"`); - } - lines.push(`attachment = ${model.attachment}`); - lines.push(`reasoning = ${model.reasoning}`); - lines.push(`tool_call = ${model.tool_call}`); - if (model.structured_output !== undefined) { - lines.push(`structured_output = ${model.structured_output}`); - } - lines.push(`temperature = ${model.temperature}`); - if (model.knowledge) { - lines.push(`knowledge = "${model.knowledge}"`); - } - lines.push(`release_date = "${model.release_date}"`); - lines.push(`last_updated = "${model.last_updated}"`); - lines.push(`open_weights = ${model.open_weights}`); - if (model.status) { - lines.push(`status = "${model.status}"`); - } - - // Interleaved section (if present) - if (model.interleaved !== undefined) { - lines.push(""); - if (model.interleaved === true) { - lines.push(`interleaved = true`); - } else if (typeof model.interleaved === "object") { - lines.push(`[interleaved]`); - lines.push(`field = "${model.interleaved.field}"`); - } - } - - // Cost section (only if present) - if (model.cost) { - lines.push(""); - lines.push(`[cost]`); - lines.push(`input = ${model.cost.input}`); - lines.push(`output = ${model.cost.output}`); - } - - // Limit section - lines.push(""); - lines.push(`[limit]`); - lines.push(`context = ${formatNumber(model.limit.context)}`); - lines.push(`output = ${formatNumber(model.limit.output)}`); - - // Modalities section - lines.push(""); - lines.push(`[modalities]`); - lines.push( - `input = [${model.modalities.input.map((m) => `"${m}"`).join(", ")}]`, - ); - lines.push( - `output = [${model.modalities.output.map((m) => `"${m}"`).join(", ")}]`, - ); - - return lines.join("\n") + "\n"; -} - -interface Changes { - field: string; - oldValue: string; - newValue: string; -} - -function detectChanges( - existing: ExistingModel | null, - merged: MergedModel, -): Changes[] { - if (!existing) return []; - - const changes: Changes[] = []; - - const compare = (field: string, oldVal: unknown, newVal: unknown) => { - const oldStr = JSON.stringify(oldVal); - const newStr = JSON.stringify(newVal); - if (oldStr !== newStr) { - changes.push({ - field, - oldValue: formatValue(oldVal), - newValue: formatValue(newVal), - }); - } - }; - - const formatValue = (val: unknown): string => { - if (typeof val === "number") return formatNumber(val); - if (Array.isArray(val)) return `[${val.join(", ")}]`; - if (val === undefined) return "(none)"; - return String(val); - }; - - compare("name", existing.name, merged.name); - compare("family", existing.family, merged.family); - compare("attachment", existing.attachment, merged.attachment); - compare("reasoning", existing.reasoning, merged.reasoning); - compare("tool_call", existing.tool_call, merged.tool_call); - compare( - "structured_output", - existing.structured_output, - merged.structured_output, - ); - compare("open_weights", existing.open_weights, merged.open_weights); - compare("release_date", existing.release_date, merged.release_date); - compare("cost.input", existing.cost?.input, merged.cost?.input); - compare("cost.output", existing.cost?.output, merged.cost?.output); - compare("limit.context", existing.limit?.context, merged.limit.context); - compare("limit.output", existing.limit?.output, merged.limit.output); - compare("modalities.input", existing.modalities?.input, merged.modalities.input); - - return changes; -} - -async function main() { - const args = process.argv.slice(2); - const dryRun = args.includes("--dry-run"); - - const modelsDir = path.join( - import.meta.dirname, - "..", - "..", - "..", - "providers", - "friendli", - "models", - ); - - if (dryRun) { - console.log(`[DRY RUN] Fetching Friendli models from API...`); - } else { - console.log(`Fetching Friendli models from API...`); - } - - // Fetch API data - const res = await fetch(API_ENDPOINT); - if (!res.ok) { - console.error(`Failed to fetch API: ${res.status} ${res.statusText}`); - process.exit(1); - } - - const json = await res.json(); - const parsed = FriendliResponse.safeParse(json); - if (!parsed.success) { - console.error("Invalid API response:", parsed.error.errors); - process.exit(1); - } - - const apiModels = parsed.data.data; - - // Get existing files (recursively) - const existingFiles = new Set(); - try { - for await (const file of new Bun.Glob("**/*.toml").scan({ - cwd: modelsDir, - absolute: false, - })) { - existingFiles.add(file); - } - } catch { - // Directory might not exist yet - } - - console.log( - `Found ${apiModels.length} models in API, ${existingFiles.size} existing files\n`, - ); - - // Track API model IDs for orphan detection - const apiModelIds = new Set(); - - let created = 0; - let updated = 0; - let unchanged = 0; - - for (const apiModel of apiModels) { - const relativePath = `${apiModel.id}.toml`; - const filePath = path.join(modelsDir, relativePath); - const dirPath = path.dirname(filePath); - - apiModelIds.add(relativePath); - - const existing = await loadExistingModel(filePath); - const merged = mergeModel(apiModel, existing); - const tomlContent = formatToml(merged); - - if (existing === null) { - created++; - if (dryRun) { - console.log(`[DRY RUN] Would create: ${relativePath}`); - console.log(` name = "${merged.name}"`); - if (merged.family) { - console.log(` family = "${merged.family}" (inferred)`); - } - console.log(""); - } else { - await mkdir(dirPath, { recursive: true }); - await Bun.write(filePath, tomlContent); - console.log(`Created: ${relativePath}`); - } - } else { - const changes = detectChanges(existing, merged); - - if (changes.length > 0) { - updated++; - if (dryRun) { - console.log(`[DRY RUN] Would update: ${relativePath}`); - } else { - await Bun.write(filePath, tomlContent); - console.log(`Updated: ${relativePath}`); - } - for (const change of changes) { - console.log(` ${change.field}: ${change.oldValue} → ${change.newValue}`); - } - console.log(""); - } else { - unchanged++; - } - } - } - - // Check for orphaned files - const orphaned: string[] = []; - for (const file of existingFiles) { - if (!apiModelIds.has(file)) { - orphaned.push(file); - console.log(`Warning: Orphaned file (not in API): ${file}`); - } - } - - // Summary - console.log(""); - if (dryRun) { - console.log( - `Summary: ${created} would be created, ${updated} would be updated, ${unchanged} unchanged, ${orphaned.length} orphaned`, - ); - } else { - console.log( - `Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} orphaned`, - ); - } -} - -await main(); diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index e6c30874235..6cd005e977c 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -18,6 +18,7 @@ import { deepinfra } from "./providers/deepinfra.js"; import { digitalocean } from "./providers/digitalocean.js"; import { edenai } from "./providers/edenai.js"; import { empiriolabs } from "./providers/empiriolabs.js"; +import { friendli } from "./providers/friendli.js"; import { githubCopilot } from "./providers/github-copilot.js"; import { google } from "./providers/google.js"; import { hyper } from "./providers/hyper.js"; @@ -143,6 +144,7 @@ export const providers: { digitalocean: SyncProvider; edenai: SyncProvider; empiriolabs: SyncProvider; + friendli: SyncProvider; "github-copilot": SyncProvider; google: SyncProvider; hyper: SyncProvider; @@ -179,6 +181,7 @@ export const providers: { digitalocean, edenai, empiriolabs, + friendli, "github-copilot": githubCopilot, google, hyper, @@ -222,7 +225,7 @@ export const groups = { "vercel", ], cloudflare: ["cloudflare-ai-gateway", "cloudflare-workers-ai"], - direct: ["ambient", "anthropic", "baseten", "chutes", "cortecs", "deepinfra", "digitalocean", "github-copilot", "google", "hyper", "meta", "ollama-cloud", "openai", "ovhcloud", "pioneer", "tinfoil", "venice", "wandb", "xai"], + direct: ["ambient", "anthropic", "baseten", "chutes", "cortecs", "deepinfra", "digitalocean", "friendli", "github-copilot", "google", "hyper", "meta", "ollama-cloud", "openai", "ovhcloud", "pioneer", "tinfoil", "venice", "wandb", "xai"], } as const; type ProviderID = keyof typeof providers; diff --git a/packages/core/src/sync/providers/friendli.ts b/packages/core/src/sync/providers/friendli.ts new file mode 100644 index 00000000000..fa03707261d --- /dev/null +++ b/packages/core/src/sync/providers/friendli.ts @@ -0,0 +1,543 @@ +import path from "node:path"; +import { readdirSync } from "node:fs"; +import { z } from "zod"; + +import { describeModel } from "../../describe.js"; +import { inferKimiFamily, ModelFamilyValues } from "../../family.js"; +import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js"; +import { factorBaseModel } from "./openrouter.js"; + +const API_ENDPOINT = "https://api.friendli.ai/serverless/v1/models"; +const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models"); + +// Friendli catalog pricing is USD per-token; catalog cost is USD per-million. +const PER_TOKEN_TO_PER_MILLION = 1_000_000; + +const InterleavedField = z.enum(["reasoning_content", "reasoning_details"]); + +// Friendli's /v1/models `interleaved` flag is unreliable for some models: it +// reports `false` for deepseek-ai/DeepSeek-V3.2 even though a live +// POST /chat/completions request (chat_template_kwargs.enable_thinking=true) +// returns both `reasoning` and `reasoning_content` in the response. Relying on +// `existing?.interleaved` to carry this forward is fragile — if the on-disk +// file ever loses the field for any reason, the live verification is silently +// forgotten on the next sync with no trace. This map is the durable source of +// truth for models where a live request has verified a real field the +// catalog API misreports; translateInterleaved consults it before falling +// back to the existing on-disk value. +const VERIFIED_INTERLEAVED_OVERRIDES: Record = { + "deepseek-ai/DeepSeek-V3.2": { field: "reasoning_content" }, +}; + +// Raw API reasoning_options shape, including budget_tokens (a real +// reasoning-budget control on Friendli: min = -1 means unlimited, max +// corresponds to max_completion_tokens). Confirmed via the live /v1/models +// response and https://friendli.ai/docs/openapi/model-apis/chat-completions +// (reasoning_budget is a documented request field). The catalog's min/max +// are not safe published bounds (see translateReasoningOptions below), so +// they are parsed but never carried into the synced model. +const FriendliReasoningOption = z + .discriminatedUnion("type", [ + z.object({ type: z.literal("toggle") }).passthrough(), + z + .object({ type: z.literal("effort"), values: z.array(z.string()) }) + .passthrough(), + z + .object({ + type: z.literal("budget_tokens"), + min: z.number().optional(), + max: z.number().optional(), + }) + .passthrough(), + ]) + .optional(); + +export const FriendliModel = z + .object({ + id: z.string(), + hugging_face_id: z.string().optional(), + name: z.string(), + created: z.number(), + context_length: z.number(), + max_completion_tokens: z.number(), + functionality: z + .object({ + tool_call: z.boolean(), + parallel_tool_call: z.boolean().optional(), + structured_output: z.boolean(), + tool_choice: z.boolean().optional(), + system_messages: z.boolean().optional(), + }) + .passthrough(), + pricing: z + .object({ + input: z.union([z.string(), z.number()]), + output: z.union([z.string(), z.number()]), + prompt: z.union([z.string(), z.number()]).optional(), + completion: z.union([z.string(), z.number()]).optional(), + input_cache_read: z.union([z.string(), z.number()]).optional(), + input_cache_write: z.union([z.string(), z.number()]).optional(), + // The pre-SyncProvider generator validated this field and authored + // cost only for TOKEN pricing. The current catalog always omits it + // for the 7 live models, but Friendli has served SECOND-priced + // entries before — passthrough would silently x1,000,000 a + // per-second rate into the catalog's USD/MTok cost. + unit_type: z.enum(["TOKEN", "SECOND"]).optional(), + }) + .passthrough(), + description: z.string().optional(), + hugging_face_url: z.string().optional(), + license: z.string().optional(), + policy: z.string().nullable().optional(), + deprecation_date: z.string().nullable().optional(), + reasoning: z.boolean().optional(), + reasoning_options: z.array(FriendliReasoningOption).optional(), + interleaved: z.union([InterleavedField, z.boolean()]).optional(), + input_modalities: z.array(z.string()).optional(), + output_modalities: z.array(z.string()).optional(), + base_model: z.string().optional(), + mode: z.string().optional(), + }) + .passthrough(); + +export const FriendliResponse = z + .object({ + data: z.array(FriendliModel), + }) + .passthrough(); + +export type FriendliModel = z.infer; + +// HuggingFace-style API orgs that are not catalog lab ids. Map them onto the +// catalog metadata tree so self-referential or HF-style base_model values +// resolve to the right lab directory. +const LAB_PREFIX_MAP: Record = { + "zai-org": "zhipuai", + "deepseek-ai": "deepseek", + "LGAI-EXAONE": "lgai-exaone", + "MiniMaxAI": "minimax", + "meta-llama": "meta", + "mistralai": "mistral", + "Qwen": "alibaba", +}; + +// Resolve an API `base_model` id to the on-disk `models//.toml` id. +// Friendli declares a base_model for most entries, but only models with an +// existing lab metadata file can be factored (override-only). Self-referential +// base_model values (==id) resolve to the model's own lab id when a metadata +// file exists under the mapped lab prefix. +// +// Case-insensitive lookup: the API lowercases some ids (e.g. +// "minimax/minimax-m2.5") that exist on disk as mixed-case +// ("minimax/MiniMax-M2.5.toml"), so we never trust a raw API id and always +// read the directory. +const baseModelCache = new Map(); + +function resolveBaseModelID(baseModel: string | undefined): string | undefined { + if (baseModel === undefined || baseModel.length === 0) return undefined; + const cached = baseModelCache.get(baseModel); + if (cached !== undefined) return cached ?? undefined; + + let resolved = lookupLabFile(baseModel); + if (resolved === undefined) { + const [org, ...parts] = baseModel.split("/"); + const mapped = org !== undefined ? LAB_PREFIX_MAP[org] : undefined; + if (mapped !== undefined && parts.length > 0) { + resolved = lookupLabFile(`${mapped}/${parts.join("/")}`); + } + } + + baseModelCache.set(baseModel, resolved ?? null); + return resolved; +} + +// Resolve a Friendli entry to its catalog lab metadata id. +// 1) API-declared base_model (handles HF id → catalog slug mismatches) +// 2) self-referential fallback: some entries (e.g. deepseek-ai/DeepSeek-V3.2) +// omit base_model entirely even though a matching lab metadata file +// exists under the mapped lab prefix — resolve against the model's own id. +function resolveLabModelSync(model: FriendliModel): string | undefined { + return resolveBaseModelID(model.base_model) ?? resolveBaseModelID(model.id); +} + +function lookupLabFile(baseModel: string): string | undefined { + const [lab, ...modelParts] = baseModel.split("/"); + const modelSlug = modelParts.join("/"); + if (lab === undefined || modelSlug.length === 0) return undefined; + + let labDir: string | undefined; + try { + const dirs = readdirSync(MODELS_DIR, { withFileTypes: true }) + .filter((entry) => entry.isDirectory()) + .map((entry) => entry.name); + labDir = dirs.find((dir) => dir.toLowerCase() === lab.toLowerCase()); + } catch { + return undefined; + } + if (labDir === undefined) return undefined; + + const expected = `${modelSlug}.toml`.toLowerCase(); + let fileMatch: string | undefined; + try { + fileMatch = readdirSync(path.join(MODELS_DIR, labDir)) + .filter((file) => file.endsWith(".toml")) + .find((file) => file.toLowerCase() === expected); + } catch { + // fall through + } + if (fileMatch === undefined) return undefined; + + return `${labDir}/${fileMatch.slice(0, -".toml".length)}`; +} + +export const friendli = { + id: "friendli", + name: "Friendli", + modelsDir: "providers/friendli/models", + // Friendli occasionally rotates models in and out of its catalog; retain + // local files for entries the API no longer advertises instead of deleting. + deleteMissing: false, + // Friendli's catalog describes real reasoning controls and limits directly; + // do not carry over a stale base_model when a model switches lab → full inline. + preserveBaseModels: false, + // The runner's default preserveDescription re-injects the resolved base + // description when the translator omits it, recreating an identical + // override. Friendli descriptions come from the API verbatim and match the + // lab's, so drop the re-injection. + preserveDescriptions: false, + // Leading wire-path comments (Toggle/Effort/Budget + doc URLs) always + // refresh from reasoningHeader() below instead of freezing whatever + // comment happened to be on disk the first time a file was created. + authoritativeHeaders: true, + async fetchModels() { + const response = await fetch(API_ENDPOINT); + if (!response.ok) { + throw new Error( + `Friendli request failed: ${response.status} ${response.statusText}`, + ); + } + return response.json(); + }, + parseModels(raw: unknown) { + return FriendliResponse.parse(raw).data; + }, + translateModel(model: FriendliModel, context) { + const existing = context.existing(model.id); + // Mirror the deepinfra pattern: a brand-new deprecated model is skipped + // outright (nothing to author), but a model we already track keeps its + // file and gets marked `status = "deprecated"` instead of silently + // falling out of translateModel — that left it retained via + // deleteMissing but stuck live in the catalog with no lifecycle marker. + if (isDeprecated(model) && existing === undefined) return undefined; + const factorBase = resolveLabModelSync(model); + // Friendli is a multi-lab relay, so a brand-new remote model with no + // provider-agnostic lab metadata to factor onto must not be authored + // full-inline by an hourly sync (AGENTS.md: full inline is reserved for + // first-party labs or true host-unique aliases). Mirror the deepinfra + // create gate: already-tracked files keep updating; unresolvable new + // IDs are skipped with a notice until lab metadata exists or a true + // host-unique alias is hand-authored. + if (existing === undefined && factorBase === undefined) return undefined; + const built = buildFriendliModel( + model, + existing, + factorBase, + isDeprecated(model), + ); + return { + id: model.id, + model: built, + header: reasoningHeader(built), + }; + }, + sourceID(model: FriendliModel) { + return model.id; + }, + skippedNotice(ids: string[]) { + if (ids.length === 0) return []; + return [ + `${ids.length} remote model(s) skipped: no provider-agnostic lab metadata to factor onto (full-inline creates are not authored for a multi-lab relay — add models//.toml, then re-sync) or deprecation_date passed before the model was ever tracked: ${ids.join(", ")}`, + ]; + }, + missingNotice(paths: string[]) { + if (paths.length === 0) return []; + return [ + `${paths.length} local model(s) retained after being removed from the Friendli API: ${paths.join(", ")}`, + ]; + }, +} satisfies SyncProvider; + +// Leading wire-path comments for every reasoning control type this host +// authors on a file, matching the wire paths documented in +// providers/friendli/provider.toml. A comment per authored control type +// (toggle, effort, budget) keeps the rationale attached even for budget-only +// files such as MiniMax-M2.5 — a toggle-only header would lose it. With +// authoritativeHeaders enabled, this header always replaces whatever was on +// disk, so it never goes stale. +const REASONING_GUIDE_URL = "https://friendli.ai/docs/guides/reasoning"; +const EFFORT_DOC_URL = + "https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0"; +const BUDGET_DOC_URL = + "https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0"; + +function reasoningHeader(model: SyncedModel): string | undefined { + const options = model.reasoning_options; + if (options === undefined || options.length === 0) return undefined; + const lines: string[] = []; + for (const option of options) { + if (option.type === "toggle") { + lines.push("# Toggle: chat_template_kwargs.enable_thinking = true | false"); + lines.push(`# ${REASONING_GUIDE_URL}`); + } + if (option.type === "effort") { + if (option.values.length > 0) { + const values = option.values.map((value) => `"${value}"`).join(" | "); + lines.push(`# Effort: reasoning_effort = ${values}`); + } else { + lines.push("# Effort: reasoning_effort (model-specific accepted values)"); + } + lines.push(`# ${EFFORT_DOC_URL}`); + } + if (option.type === "budget_tokens") { + lines.push( + "# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited)", + ); + lines.push(`# ${BUDGET_DOC_URL}`); + } + } + return lines.length > 0 ? `${lines.join("\n")}\n` : undefined; +} + +type Modality = "text" | "audio" | "image" | "video" | "pdf"; + +const ALLOWED_MODALITIES: Record = { + text: true, + audio: true, + image: true, + video: true, + pdf: true, +}; + +function translateModalities(values: string[] | undefined): Modality[] { + const result = [...new Set( + (values ?? ["text"]) + .map((value) => value.toLowerCase()) + .filter((value): value is Modality => ALLOWED_MODALITIES[value] === true), + )]; + return result.length > 0 ? result : ["text"]; +} + +// Skip models whose deprecation_date has passed. Friendli returns an ISO +// timestamp (e.g. "2026-08-20T00:00:00Z"); we compare against now at sync time. +function isDeprecated(model: FriendliModel): boolean { + if (model.deprecation_date === undefined || model.deprecation_date === null) return false; + const dep = Date.parse(model.deprecation_date); + return Number.isFinite(dep) && dep <= Date.now(); +} + +function perMillion(value: string | number | undefined): number | undefined { + if (value === undefined) return undefined; + const number = Number(value); + if (!Number.isFinite(number) || number < 0) return undefined; + const perM = number * PER_TOKEN_TO_PER_MILLION; + return Math.round(perM * 1_000_000) / 1_000_000; +} + +function buildCost( + model: FriendliModel, + existing: ExistingModel["cost"] | undefined, +): NonNullable | undefined { + // TOKEN-priced per-token USD rates are converted to USD/MTok. Any other + // unit (e.g. SECOND) is not a token rate: do not author a cost section for + // it instead of publishing an invented per-million price. + if (model.pricing.unit_type !== undefined && model.pricing.unit_type !== "TOKEN") { + return existing; + } + const input = perMillion(model.pricing.input); + const output = perMillion(model.pricing.output); + if (input === undefined || output === undefined) return existing; + return { + input, + output, + cache_read: perMillion(model.pricing.input_cache_read) ?? existing?.cache_read, + cache_write: perMillion(model.pricing.input_cache_write) ?? existing?.cache_write, + }; +} + +// Translate API reasoning_options into host-accurate catalog options. +// budget_tokens IS a real reasoning-budget control on Friendli, confirmed by +// the live /v1/models response and the OpenAPI chat-completions docs +// (reasoning_budget is a documented request field, min = -1 means unlimited). +// The control itself is real (verified with live requests across GLM-5.3, +// gemma-4-31B-it, and DeepSeek-V3.2), but the catalog's min/max values are +// not safe published range constraints: GLM-5.3 accepted +// reasoning_budget=1_048_577 despite reporting max=1_048_576. Preserve the +// capability without publishing unverified bounds — never author min/max. +function translateReasoningOptions( + api: FriendliModel["reasoning_options"], +): SyncedFullModel["reasoning_options"] { + if (api === undefined) return undefined; + const options: NonNullable = []; + for (const option of api) { + if (option === undefined) continue; + if (option.type === "budget_tokens") { + options.push({ type: "budget_tokens" }); + continue; + } + options.push(option as NonNullable[number]); + } + return options.length > 0 ? options : []; +} + +function translateInterleaved( + modelID: string, + value: FriendliModel["interleaved"], + existing: SyncedFullModel["interleaved"] | undefined, +): SyncedFullModel["interleaved"] { + const verified = VERIFIED_INTERLEAVED_OVERRIDES[modelID]; + if (verified !== undefined) return verified; + if (value === undefined) return existing; + // The models endpoint can be stale/wrong for this field (verified live + // against deepseek-ai/DeepSeek-V3.2, see VERIFIED_INTERLEAVED_OVERRIDES) — + // trust an existing authored value over an API false rather than clearing it. + if (value === false) return existing; + if (value === true) return true; + return { field: value }; +} + +function inferFamily(modelID: string, name: string): SyncedFullModel["family"] { + const kimiFamily = inferKimiFamily(modelID, name); + if (kimiFamily !== undefined) return kimiFamily; + const target = `${modelID} ${name}`.toLowerCase(); + return [...ModelFamilyValues] + .sort((a, b) => b.length - a.length) + .find((family) => { + const escaped = family.toLowerCase().replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); + if (family === "o") { + return new RegExp(`(^|[^a-z0-9])${escaped}(?=\\d|$|[^a-z0-9])`).test(target); + } + return new RegExp(`(^|[^a-z0-9])${escaped}(?=$|[^a-z0-9])`).test(target); + }); +} + +function buildFriendliModel( + model: FriendliModel, + existing: ExistingModel | undefined, + factorBase: string | undefined, + deprecated: boolean, +): SyncedModel { + // Mirror deepinfra/pioneer lifecycle behavior: mark a live-catalog model + // deprecated when its deprecation_date has passed, but do not let a retained + // deleteMissing:false file stay permanently deprecated if it later returns + // to the catalog active again. Preserve other hand-authored lifecycle + // statuses (e.g. beta) unchanged. + const status = deprecated + ? "deprecated" as const + : existing?.status === "deprecated" + ? undefined + : existing?.status; + + // Only override modalities when the API explicitly provides them; otherwise + // omit the override so lab metadata (e.g. gemma vision) is inherited. + const apiInput = model.input_modalities !== undefined ? translateModalities(model.input_modalities) : undefined; + const apiOutput = model.output_modalities !== undefined ? translateModalities(model.output_modalities) : undefined; + // For a factored provider entry, only write the modality sides Friendli + // actually supplied. Plain-object inheritance deep-merges, so an omitted + // side must remain omitted to preserve the lab's canonical modality list + // rather than replacing it with an empty array. + const modalities = apiInput !== undefined || apiOutput !== undefined + ? { + ...(apiInput !== undefined ? { input: apiInput } : {}), + ...(apiOutput !== undefined ? { output: apiOutput } : {}), + } + : undefined; + // undefined when the API omits input_modalities so factorBaseModel + // inherits the lab attachment; only override when explicitly provided. + const attachment = apiInput !== undefined ? apiInput.some((value) => value !== "text") : undefined; + // Completion-length cap: when a base_model exists, defer to the lab's own + // limit.output instead of forcing Friendli's max_completion_tokens onto it. + // Friendli's max_completion_tokens equals context_length for every one of + // the 7 live models, and blindly asserting that as the completion cap would + // overwrite lab-verified, genuinely tighter completion limits (e.g. + // DeepSeek-V3.2's lab file documents output=64_000 out of a 128_000 + // context, not "same as context"). Only fall back to Friendli's own + // reported value when there is no base_model to inherit a real + // completion-cap policy from (full-inline entries). + const limit = { + context: model.context_length, + input: existing?.limit?.input, + output: factorBase !== undefined ? undefined : model.max_completion_tokens, + }; + const reasoning = model.reasoning === true; + const reasoningOptions = reasoning ? translateReasoningOptions(model.reasoning_options) : undefined; + const interleaved = translateInterleaved(model.id, model.interleaved, existing?.interleaved); + const structuredOutput = model.functionality.structured_output; + const cost = buildCost(model, existing?.cost); + const releaseDate = existing?.release_date ?? new Date(model.created * 1000).toISOString().slice(0, 10); + const today = new Date().toISOString().slice(0, 10); + const lastUpdated = existing?.last_updated ?? today; + + if (factorBase !== undefined) { + return factorBaseModel( + factorBase, + { + attachment, + reasoning, + reasoning_options: reasoningOptions, + interleaved, + structured_output: structuredOutput, + // A factored entry inherits the lab description. Friendli's catalog + // description is host metadata, not a new model identity, and its + // generic text can be weaker than the lab's canonical description. + // Keep it only for full-inline entries below. + description: undefined, + limit, + modalities, + cost, + status, + }, + limit, + existing?.base_model === factorBase ? existing.base_model_omit : undefined, + ); + } + + const name = existing?.name ?? (model.name.split("/").at(-1) ?? model.name); + return { + name, + description: + existing?.description ?? + model.description ?? + describeModel({ + id: model.id, + providerId: "friendli", + name, + family: existing?.family, + reasoning, + tool_call: model.functionality.tool_call, + structured_output: structuredOutput, + open_weights: Boolean(model.hugging_face_url), + // Full-inline entries have no lab modalities to inherit. Default only + // sides omitted by the API to text rather than constructing empty + // arrays, which would advertise an impossible no-output/no-input model. + modalities: { input: apiInput ?? ["text"], output: apiOutput ?? ["text"] }, + }), + family: existing?.family ?? inferFamily(model.id, name), + // Full-inline has no lab to inherit from; default text-only when the API + // omits modalities. Earlier `attachment` is undefined in that case. + attachment: attachment ?? false, + reasoning, + reasoning_options: reasoningOptions, + tool_call: model.functionality.tool_call, + structured_output: structuredOutput, + temperature: existing?.temperature ?? true, + release_date: releaseDate, + last_updated: lastUpdated, + open_weights: Boolean(model.hugging_face_url), + interleaved, + knowledge: existing?.knowledge, + cost, + limit: { context: model.context_length, output: model.max_completion_tokens }, + modalities: { input: apiInput ?? ["text"], output: apiOutput ?? ["text"] }, + status, + }; +} diff --git a/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml b/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml index 9fa904e270b..82bcc78bb46 100644 --- a/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml +++ b/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml @@ -1,13 +1,14 @@ +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "minimax/MiniMax-M2.5" -name = "MiniMax-M2.5" -release_date = "2026-02-12" -last_updated = "2026-02-12" structured_output = true -reasoning_options = [] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.3 output = 1.2 @@ -15,4 +16,3 @@ cache_read = 0.06 [limit] context = 196_608 -output = 196_608 diff --git a/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml b/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml index 4714efddca6..4dfa7122d90 100644 --- a/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml +++ b/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml @@ -1,19 +1,18 @@ -name = "DeepSeek-V3.2" -description = "DeepSeek chat model for instruction following, coding, and analysis" -family = "deepseek" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -tool_call = true -structured_output = true -temperature = true -release_date = "2025-12-01" -last_updated = "2025-12-01" -open_weights = true +# Toggle: chat_template_kwargs.enable_thinking = true | false +# https://friendli.ai/docs/guides/reasoning +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 +base_model = "deepseek/deepseek-v3.2" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.5 output = 1.5 @@ -21,8 +20,3 @@ cache_read = 0.25 [limit] context = 163_840 -output = 163_840 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/friendli/models/google/gemma-4-31B-it.toml b/providers/friendli/models/google/gemma-4-31B-it.toml index cd945119017..e6094afcc7e 100644 --- a/providers/friendli/models/google/gemma-4-31B-it.toml +++ b/providers/friendli/models/google/gemma-4-31B-it.toml @@ -1,5 +1,17 @@ +# Toggle: chat_template_kwargs.enable_thinking = true | false +# https://friendli.ai/docs/guides/reasoning +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "google/gemma-4-31b-it" -reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 0.14 diff --git a/providers/friendli/models/zai-org/GLM-5.1.toml b/providers/friendli/models/zai-org/GLM-5.1.toml index 1ee4131636b..d97a8d26dda 100644 --- a/providers/friendli/models/zai-org/GLM-5.1.toml +++ b/providers/friendli/models/zai-org/GLM-5.1.toml @@ -1,13 +1,18 @@ +# Toggle: chat_template_kwargs.enable_thinking = true | false +# https://friendli.ai/docs/guides/reasoning +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.1" -name = "GLM-5.1" -release_date = "2026-04-07" -last_updated = "2026-04-07" -structured_output = true -reasoning_options = [] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 1.4 output = 4.4 @@ -15,4 +20,3 @@ cache_read = 0.26 [limit] context = 202_752 -output = 202_752 diff --git a/providers/friendli/models/zai-org/GLM-5.2.toml b/providers/friendli/models/zai-org/GLM-5.2.toml index 81fbbb90c49..187d5de39d0 100644 --- a/providers/friendli/models/zai-org/GLM-5.2.toml +++ b/providers/friendli/models/zai-org/GLM-5.2.toml @@ -1,17 +1,28 @@ -name = "GLM-5.2" -# Friendli documents only $.chat_template_kwargs.enable_thinking = true | false -# for this model, not the configured effort values "high" and "max". -# https://friendli.ai/docs/guides/reasoning (accessed 2026-06-25) +# Toggle: chat_template_kwargs.enable_thinking = true | false +# https://friendli.ai/docs/guides/reasoning +# Effort: reasoning_effort = "high" | "max" +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.2" +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + [[reasoning_options]] type = "effort" values = ["high", "max"] -[interleaved] -field = "reasoning_content" +[[reasoning_options]] +type = "budget_tokens" [cost] input = 1.4 output = 4.4 -cache_read = 0.26 \ No newline at end of file +cache_read = 0.26 + +[limit] +context = 1_048_576 diff --git a/providers/friendli/models/zai-org/GLM-5.3-Flash.toml b/providers/friendli/models/zai-org/GLM-5.3-Flash.toml new file mode 100644 index 00000000000..7d25fca9c7b --- /dev/null +++ b/providers/friendli/models/zai-org/GLM-5.3-Flash.toml @@ -0,0 +1,26 @@ +# Effort: reasoning_effort = "low" | "high" | "max" +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 +base_model = "zhipuai/glm-5.3-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.03 + +[limit] +context = 1_048_576 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/friendli/models/zai-org/GLM-5.3.toml b/providers/friendli/models/zai-org/GLM-5.3.toml index bf8c8ce9c16..baad00a0786 100644 --- a/providers/friendli/models/zai-org/GLM-5.3.toml +++ b/providers/friendli/models/zai-org/GLM-5.3.toml @@ -1,42 +1,23 @@ -# Friendli's chat-completions endpoint documents reasoning_effort (generic -# enum) and reasoning_budget (integer token cap) as real request fields; GLM-5.3 -# on Z.AI's own API always reasons and only exposes low|high|max, so we mirror -# those graded levels here instead of Friendli's full generic effort enum. -# -# budget_tokens is a REAL, independently-enforced reasoning-token budget, not -# derived from max_tokens/context capacity: two live POST /chat/completions -# requests against this exact model (reasoning_budget=30 and =50, generous -# max_tokens=3000/4000) cut reasoning_content mid-sentence at the requested -# cap while completion_tokens continued to 1553/177 respectively — proof the -# reasoning phase and the completion phase are capped independently. -# min=-1/max=1_048_576 are Friendli's own reported values, read verbatim from -# GET /serverless/v1/models -> reasoning_options -> {type: budget_tokens} for -# zai-org/GLM-5.3 (not computed by us from max_completion_tokens; that field -# happens to equal this model's max_completion_tokens/context_length, but we -# pass through the catalog's own budget_tokens object, we do not derive it). +# Effort: reasoning_effort = "low" | "high" | "max" # https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) # https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 -# (accessed 2026-08-29) base_model = "zhipuai/glm-5.3" +[interleaved] +field = "reasoning_content" + [[reasoning_options]] type = "effort" values = ["low", "high", "max"] [[reasoning_options]] type = "budget_tokens" -min = -1 -max = 1_048_576 - -[interleaved] -field = "reasoning_content" [cost] -input = 1.4 -output = 4.4 -cache_read = 0.26 +input = 1.26 +output = 3.96 +cache_read = 0.234 [limit] context = 1_048_576 -output = 1_048_576 - From 5b56f0f6b4769a5e81f0ca078c2d80362565a509 Mon Sep 17 00:00:00 2001 From: siyoon Date: Sun, 13 Sep 2026 22:17:05 +0900 Subject: [PATCH 2/6] fix(friendli): tri-state reasoning, tool_call delta, MiniMax budget verification --- packages/core/src/sync/providers/friendli.ts | 33 ++++++++++++++------ 1 file changed, 24 insertions(+), 9 deletions(-) diff --git a/packages/core/src/sync/providers/friendli.ts b/packages/core/src/sync/providers/friendli.ts index fa03707261d..5c89c41234d 100644 --- a/packages/core/src/sync/providers/friendli.ts +++ b/packages/core/src/sync/providers/friendli.ts @@ -368,11 +368,15 @@ function buildCost( // budget_tokens IS a real reasoning-budget control on Friendli, confirmed by // the live /v1/models response and the OpenAPI chat-completions docs // (reasoning_budget is a documented request field, min = -1 means unlimited). -// The control itself is real (verified with live requests across GLM-5.3, -// gemma-4-31B-it, and DeepSeek-V3.2), but the catalog's min/max values are -// not safe published range constraints: GLM-5.3 accepted -// reasoning_budget=1_048_577 despite reporting max=1_048_576. Preserve the -// capability without publishing unverified bounds — never author min/max. +// The control is verified with live requests on GLM-5.3, gemma-4-31B-it, +// DeepSeek-V3.2, and MiniMax-M2.5 (2026-09-13: MiniMax reasoning_budget=10 +// truncated reasoning_content at 46 chars while completion_tokens continued +// to a full answer; budget=2000 produced 1011 chars of reasoning under the +// same prompt and max_tokens — the two phases cap independently). The +// catalog's min/max values are not safe published range constraints: +// GLM-5.3 accepted reasoning_budget=1_048_577 despite reporting +// max=1_048_576. Preserve the capability without publishing unverified +// bounds — never author min/max. function translateReasoningOptions( api: FriendliModel["reasoning_options"], ): SyncedFullModel["reasoning_options"] { @@ -468,8 +472,12 @@ function buildFriendliModel( input: existing?.limit?.input, output: factorBase !== undefined ? undefined : model.max_completion_tokens, }; - const reasoning = model.reasoning === true; - const reasoningOptions = reasoning ? translateReasoningOptions(model.reasoning_options) : undefined; + // Reasoning is tri-state: Friendli omits the flag for some reasoners, and + // treating "absent" as `false` would publish an explicit reasoning=false + // override on factored entries and strip their reasoning_options. Only + // override when the API is authoritative; otherwise the lab wins. + const reasoning = model.reasoning === true ? true : model.reasoning === false ? false : undefined; + const reasoningOptions = reasoning === false ? undefined : translateReasoningOptions(model.reasoning_options); const interleaved = translateInterleaved(model.id, model.interleaved, existing?.interleaved); const structuredOutput = model.functionality.structured_output; const cost = buildCost(model, existing?.cost); @@ -485,6 +493,9 @@ function buildFriendliModel( reasoning, reasoning_options: reasoningOptions, interleaved, + // Friendli is authoritative for this host's tool-call surface; a + // real delta vs the lab (either direction) must be published. + tool_call: model.functionality.tool_call, structured_output: structuredOutput, // A factored entry inherits the lab description. Friendli's catalog // description is host metadata, not a new model identity, and its @@ -502,6 +513,10 @@ function buildFriendliModel( } const name = existing?.name ?? (model.name.split("/").at(-1) ?? model.name); + // Full-inline has no lab to inherit from: a reasoning flag the API omits + // defaults to false here (describeModel needs a boolean), while factored + // entries above leave it unset so the lab's value stands. + const inlineReasoning = reasoning ?? false; return { name, description: @@ -512,7 +527,7 @@ function buildFriendliModel( providerId: "friendli", name, family: existing?.family, - reasoning, + reasoning: inlineReasoning, tool_call: model.functionality.tool_call, structured_output: structuredOutput, open_weights: Boolean(model.hugging_face_url), @@ -525,7 +540,7 @@ function buildFriendliModel( // Full-inline has no lab to inherit from; default text-only when the API // omits modalities. Earlier `attachment` is undefined in that case. attachment: attachment ?? false, - reasoning, + reasoning: inlineReasoning, reasoning_options: reasoningOptions, tool_call: model.functionality.tool_call, structured_output: structuredOutput, From 1fca2b7b38e20a755daa93060a181549358a923c Mon Sep 17 00:00:00 2001 From: siyoon Date: Sun, 13 Sep 2026 22:39:04 +0900 Subject: [PATCH 3/6] fix(friendli): delete removed/deprecated models, drop budget_tokens, align effort with lab --- packages/core/src/sync/providers/friendli.ts | 81 +++++++------------ .../models/MiniMaxAI/MiniMax-M2.5.toml | 6 +- .../models/deepseek-ai/DeepSeek-V3.2.toml | 5 -- .../models/google/gemma-4-31B-it.toml | 5 -- .../friendli/models/zai-org/GLM-5.1.toml | 5 -- .../friendli/models/zai-org/GLM-5.2.toml | 5 -- .../models/zai-org/GLM-5.3-Flash.toml | 5 -- .../friendli/models/zai-org/GLM-5.3.toml | 5 -- 8 files changed, 30 insertions(+), 87 deletions(-) diff --git a/packages/core/src/sync/providers/friendli.ts b/packages/core/src/sync/providers/friendli.ts index 5c89c41234d..ed1086d6abe 100644 --- a/packages/core/src/sync/providers/friendli.ts +++ b/packages/core/src/sync/providers/friendli.ts @@ -194,9 +194,12 @@ export const friendli = { id: "friendli", name: "Friendli", modelsDir: "providers/friendli/models", - // Friendli occasionally rotates models in and out of its catalog; retain - // local files for entries the API no longer advertises instead of deleting. - deleteMissing: false, + // Friendli's /v1/models is authoritative for what this host serves: a model + // absent from the catalog (or past its deprecation_date) must not stay in + // the catalog as a live-looking route, so missing files are deleted rather + // than retained. Deprecation marking below only applies while the model is + // still listed; once it disappears, the file goes with it. + deleteMissing: true, // Friendli's catalog describes real reasoning controls and limits directly; // do not carry over a stale base_model when a model switches lab → full inline. preserveBaseModels: false, @@ -223,12 +226,12 @@ export const friendli = { }, translateModel(model: FriendliModel, context) { const existing = context.existing(model.id); - // Mirror the deepinfra pattern: a brand-new deprecated model is skipped - // outright (nothing to author), but a model we already track keeps its - // file and gets marked `status = "deprecated"` instead of silently - // falling out of translateModel — that left it retained via - // deleteMissing but stuck live in the catalog with no lifecycle marker. - if (isDeprecated(model) && existing === undefined) return undefined; + // A model past its deprecation_date is skipped outright (tracked or not): + // with deleteMissing enabled, skipping removes an already-tracked file on + // the next sync, so the catalog never keeps serving a dead route as a + // live-looking entry. Source-of-truth policy: a deprecation_date in the + // catalog means the same thing as the model disappearing from it. + if (isDeprecated(model)) return undefined; const factorBase = resolveLabModelSync(model); // Friendli is a multi-lab relay, so a brand-new remote model with no // provider-agnostic lab metadata to factor onto must not be authored @@ -242,7 +245,6 @@ export const friendli = { model, existing, factorBase, - isDeprecated(model), ); return { id: model.id, @@ -256,29 +258,24 @@ export const friendli = { skippedNotice(ids: string[]) { if (ids.length === 0) return []; return [ - `${ids.length} remote model(s) skipped: no provider-agnostic lab metadata to factor onto (full-inline creates are not authored for a multi-lab relay — add models//.toml, then re-sync) or deprecation_date passed before the model was ever tracked: ${ids.join(", ")}`, + `${ids.length} remote model(s) skipped: no provider-agnostic lab metadata to factor onto (full-inline creates are not authored for a multi-lab relay — add models//.toml, then re-sync) or deprecation_date passed: ${ids.join(", ")}`, ]; }, missingNotice(paths: string[]) { if (paths.length === 0) return []; return [ - `${paths.length} local model(s) retained after being removed from the Friendli API: ${paths.join(", ")}`, + `${paths.length} local model(s) deleted after being removed from the Friendli API (or past their deprecation_date): ${paths.join(", ")}`, ]; }, } satisfies SyncProvider; // Leading wire-path comments for every reasoning control type this host // authors on a file, matching the wire paths documented in -// providers/friendli/provider.toml. A comment per authored control type -// (toggle, effort, budget) keeps the rationale attached even for budget-only -// files such as MiniMax-M2.5 — a toggle-only header would lose it. With -// authoritativeHeaders enabled, this header always replaces whatever was on -// disk, so it never goes stale. +// providers/friendli/provider.toml. With authoritativeHeaders enabled, this +// header always replaces whatever was on disk, so it never goes stale. const REASONING_GUIDE_URL = "https://friendli.ai/docs/guides/reasoning"; const EFFORT_DOC_URL = "https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0"; -const BUDGET_DOC_URL = - "https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0"; function reasoningHeader(model: SyncedModel): string | undefined { const options = model.reasoning_options; @@ -298,12 +295,6 @@ function reasoningHeader(model: SyncedModel): string | undefined { } lines.push(`# ${EFFORT_DOC_URL}`); } - if (option.type === "budget_tokens") { - lines.push( - "# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited)", - ); - lines.push(`# ${BUDGET_DOC_URL}`); - } } return lines.length > 0 ? `${lines.join("\n")}\n` : undefined; } @@ -365,18 +356,14 @@ function buildCost( } // Translate API reasoning_options into host-accurate catalog options. -// budget_tokens IS a real reasoning-budget control on Friendli, confirmed by -// the live /v1/models response and the OpenAPI chat-completions docs -// (reasoning_budget is a documented request field, min = -1 means unlimited). -// The control is verified with live requests on GLM-5.3, gemma-4-31B-it, -// DeepSeek-V3.2, and MiniMax-M2.5 (2026-09-13: MiniMax reasoning_budget=10 -// truncated reasoning_content at 46 chars while completion_tokens continued -// to a full answer; budget=2000 produced 1011 chars of reasoning under the -// same prompt and max_tokens — the two phases cap independently). The -// catalog's min/max values are not safe published range constraints: -// GLM-5.3 accepted reasoning_budget=1_048_577 despite reporting -// max=1_048_576. Preserve the capability without publishing unverified -// bounds — never author min/max. +// budget_tokens is deliberately dropped: although reasoning_budget is a real, +// independently enforced Friendli control (live-verified on GLM-5.3, +// gemma-4-31B-it, DeepSeek-V3.2, and MiniMax-M2.5), the catalog only +// publishes toggle/effort controls for OpenAI-compatible hosts, and a +// budget-only reasoner (MiniMax-M2.5) is always-on — an empty option list +// matches the repo's established convention for always-on reasoners. The +// raw zod schema above still parses budget_tokens so the shape is validated, +// but it is never carried into the synced model. function translateReasoningOptions( api: FriendliModel["reasoning_options"], ): SyncedFullModel["reasoning_options"] { @@ -384,10 +371,7 @@ function translateReasoningOptions( const options: NonNullable = []; for (const option of api) { if (option === undefined) continue; - if (option.type === "budget_tokens") { - options.push({ type: "budget_tokens" }); - continue; - } + if (option.type === "budget_tokens") continue; options.push(option as NonNullable[number]); } return options.length > 0 ? options : []; @@ -428,18 +412,11 @@ function buildFriendliModel( model: FriendliModel, existing: ExistingModel | undefined, factorBase: string | undefined, - deprecated: boolean, ): SyncedModel { - // Mirror deepinfra/pioneer lifecycle behavior: mark a live-catalog model - // deprecated when its deprecation_date has passed, but do not let a retained - // deleteMissing:false file stay permanently deprecated if it later returns - // to the catalog active again. Preserve other hand-authored lifecycle - // statuses (e.g. beta) unchanged. - const status = deprecated - ? "deprecated" as const - : existing?.status === "deprecated" - ? undefined - : existing?.status; + // translateModel already skips models past their deprecation_date, so every + // model reaching this point is live. Carry hand-authored lifecycle statuses + // (e.g. beta) through unchanged. + const status = existing?.status; // Only override modalities when the API explicitly provides them; otherwise // omit the override so lab metadata (e.g. gemma vision) is inherited. diff --git a/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml b/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml index 82bcc78bb46..a47490bbf58 100644 --- a/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml +++ b/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml @@ -1,14 +1,10 @@ -# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) -# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "minimax/MiniMax-M2.5" structured_output = true +reasoning_options = [] [interleaved] field = "reasoning_content" -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 0.3 output = 1.2 diff --git a/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml b/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml index 4dfa7122d90..1550a8ec3f5 100644 --- a/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml +++ b/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml @@ -1,7 +1,5 @@ # Toggle: chat_template_kwargs.enable_thinking = true | false # https://friendli.ai/docs/guides/reasoning -# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) -# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "deepseek/deepseek-v3.2" [interleaved] @@ -10,9 +8,6 @@ field = "reasoning_content" [[reasoning_options]] type = "toggle" -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 0.5 output = 1.5 diff --git a/providers/friendli/models/google/gemma-4-31B-it.toml b/providers/friendli/models/google/gemma-4-31B-it.toml index e6094afcc7e..9d288167323 100644 --- a/providers/friendli/models/google/gemma-4-31B-it.toml +++ b/providers/friendli/models/google/gemma-4-31B-it.toml @@ -1,7 +1,5 @@ # Toggle: chat_template_kwargs.enable_thinking = true | false # https://friendli.ai/docs/guides/reasoning -# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) -# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "google/gemma-4-31b-it" [interleaved] @@ -10,9 +8,6 @@ field = "reasoning_content" [[reasoning_options]] type = "toggle" -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 0.14 output = 0.4 diff --git a/providers/friendli/models/zai-org/GLM-5.1.toml b/providers/friendli/models/zai-org/GLM-5.1.toml index d97a8d26dda..49d367259ca 100644 --- a/providers/friendli/models/zai-org/GLM-5.1.toml +++ b/providers/friendli/models/zai-org/GLM-5.1.toml @@ -1,7 +1,5 @@ # Toggle: chat_template_kwargs.enable_thinking = true | false # https://friendli.ai/docs/guides/reasoning -# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) -# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.1" [interleaved] @@ -10,9 +8,6 @@ field = "reasoning_content" [[reasoning_options]] type = "toggle" -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 1.4 output = 4.4 diff --git a/providers/friendli/models/zai-org/GLM-5.2.toml b/providers/friendli/models/zai-org/GLM-5.2.toml index 187d5de39d0..4fff04e7c0f 100644 --- a/providers/friendli/models/zai-org/GLM-5.2.toml +++ b/providers/friendli/models/zai-org/GLM-5.2.toml @@ -2,8 +2,6 @@ # https://friendli.ai/docs/guides/reasoning # Effort: reasoning_effort = "high" | "max" # https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 -# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) -# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.2" [interleaved] @@ -16,9 +14,6 @@ type = "toggle" type = "effort" values = ["high", "max"] -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 1.4 output = 4.4 diff --git a/providers/friendli/models/zai-org/GLM-5.3-Flash.toml b/providers/friendli/models/zai-org/GLM-5.3-Flash.toml index 7d25fca9c7b..bea59ccfb1b 100644 --- a/providers/friendli/models/zai-org/GLM-5.3-Flash.toml +++ b/providers/friendli/models/zai-org/GLM-5.3-Flash.toml @@ -1,7 +1,5 @@ # Effort: reasoning_effort = "low" | "high" | "max" # https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 -# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) -# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.3-flash" [interleaved] @@ -11,9 +9,6 @@ field = "reasoning_content" type = "effort" values = ["low", "high", "max"] -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 0.15 output = 0.5 diff --git a/providers/friendli/models/zai-org/GLM-5.3.toml b/providers/friendli/models/zai-org/GLM-5.3.toml index baad00a0786..25fae4ddb49 100644 --- a/providers/friendli/models/zai-org/GLM-5.3.toml +++ b/providers/friendli/models/zai-org/GLM-5.3.toml @@ -1,7 +1,5 @@ # Effort: reasoning_effort = "low" | "high" | "max" # https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 -# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) -# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.3" [interleaved] @@ -11,9 +9,6 @@ field = "reasoning_content" type = "effort" values = ["low", "high", "max"] -[[reasoning_options]] -type = "budget_tokens" - [cost] input = 1.26 output = 3.96 From 367a40e0095c7f7076a668198b94cd05e6dc7d70 Mon Sep 17 00:00:00 2001 From: siyoon Date: Sun, 13 Sep 2026 22:58:00 +0900 Subject: [PATCH 4/6] fix(friendli): restore unbounded budget_tokens per review --- packages/core/src/sync/providers/friendli.ts | 32 +++++++++++++------ .../models/MiniMaxAI/MiniMax-M2.5.toml | 6 +++- .../models/deepseek-ai/DeepSeek-V3.2.toml | 5 +++ .../models/google/gemma-4-31B-it.toml | 5 +++ .../friendli/models/zai-org/GLM-5.1.toml | 5 +++ .../friendli/models/zai-org/GLM-5.2.toml | 5 +++ .../models/zai-org/GLM-5.3-Flash.toml | 5 +++ .../friendli/models/zai-org/GLM-5.3.toml | 5 +++ 8 files changed, 58 insertions(+), 10 deletions(-) diff --git a/packages/core/src/sync/providers/friendli.ts b/packages/core/src/sync/providers/friendli.ts index ed1086d6abe..83e722c849c 100644 --- a/packages/core/src/sync/providers/friendli.ts +++ b/packages/core/src/sync/providers/friendli.ts @@ -276,6 +276,8 @@ export const friendli = { const REASONING_GUIDE_URL = "https://friendli.ai/docs/guides/reasoning"; const EFFORT_DOC_URL = "https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0"; +const BUDGET_DOC_URL = + "https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0"; function reasoningHeader(model: SyncedModel): string | undefined { const options = model.reasoning_options; @@ -295,6 +297,12 @@ function reasoningHeader(model: SyncedModel): string | undefined { } lines.push(`# ${EFFORT_DOC_URL}`); } + if (option.type === "budget_tokens") { + lines.push( + "# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited)", + ); + lines.push(`# ${BUDGET_DOC_URL}`); + } } return lines.length > 0 ? `${lines.join("\n")}\n` : undefined; } @@ -356,14 +364,17 @@ function buildCost( } // Translate API reasoning_options into host-accurate catalog options. -// budget_tokens is deliberately dropped: although reasoning_budget is a real, -// independently enforced Friendli control (live-verified on GLM-5.3, -// gemma-4-31B-it, DeepSeek-V3.2, and MiniMax-M2.5), the catalog only -// publishes toggle/effort controls for OpenAI-compatible hosts, and a -// budget-only reasoner (MiniMax-M2.5) is always-on — an empty option list -// matches the repo's established convention for always-on reasoners. The -// raw zod schema above still parses budget_tokens so the shape is validated, -// but it is never carried into the synced model. +// budget_tokens is kept as an unbounded `{ type = "budget_tokens" }`: +// reasoning_budget is a real, independently enforced Friendli control +// (live-verified on GLM-5.3, gemma-4-31B-it, DeepSeek-V3.2, and +// MiniMax-M2.5 — small budgets truncate reasoning_content mid-sentence while +// completion continues), and peers such as OpenRouter/Requesty publish it +// when the host supports it. The catalog's min/max values are not safe +// published range constraints — GLM-5.3 accepted reasoning_budget=1_048_577 +// despite reporting max=1_048_576 — so the capability is preserved without +// authoring bounds. A budget-only reasoner (MiniMax-M2.5) therefore publishes +// `[{ type = "budget_tokens" }]`, not []: [] would falsely claim no caller +// control on a host that documents reasoning_budget. function translateReasoningOptions( api: FriendliModel["reasoning_options"], ): SyncedFullModel["reasoning_options"] { @@ -371,7 +382,10 @@ function translateReasoningOptions( const options: NonNullable = []; for (const option of api) { if (option === undefined) continue; - if (option.type === "budget_tokens") continue; + if (option.type === "budget_tokens") { + options.push({ type: "budget_tokens" }); + continue; + } options.push(option as NonNullable[number]); } return options.length > 0 ? options : []; diff --git a/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml b/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml index a47490bbf58..82bcc78bb46 100644 --- a/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml +++ b/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml @@ -1,10 +1,14 @@ +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "minimax/MiniMax-M2.5" structured_output = true -reasoning_options = [] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.3 output = 1.2 diff --git a/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml b/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml index 1550a8ec3f5..4dfa7122d90 100644 --- a/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml +++ b/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml @@ -1,5 +1,7 @@ # Toggle: chat_template_kwargs.enable_thinking = true | false # https://friendli.ai/docs/guides/reasoning +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "deepseek/deepseek-v3.2" [interleaved] @@ -8,6 +10,9 @@ field = "reasoning_content" [[reasoning_options]] type = "toggle" +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.5 output = 1.5 diff --git a/providers/friendli/models/google/gemma-4-31B-it.toml b/providers/friendli/models/google/gemma-4-31B-it.toml index 9d288167323..e6094afcc7e 100644 --- a/providers/friendli/models/google/gemma-4-31B-it.toml +++ b/providers/friendli/models/google/gemma-4-31B-it.toml @@ -1,5 +1,7 @@ # Toggle: chat_template_kwargs.enable_thinking = true | false # https://friendli.ai/docs/guides/reasoning +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "google/gemma-4-31b-it" [interleaved] @@ -8,6 +10,9 @@ field = "reasoning_content" [[reasoning_options]] type = "toggle" +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.14 output = 0.4 diff --git a/providers/friendli/models/zai-org/GLM-5.1.toml b/providers/friendli/models/zai-org/GLM-5.1.toml index 49d367259ca..d97a8d26dda 100644 --- a/providers/friendli/models/zai-org/GLM-5.1.toml +++ b/providers/friendli/models/zai-org/GLM-5.1.toml @@ -1,5 +1,7 @@ # Toggle: chat_template_kwargs.enable_thinking = true | false # https://friendli.ai/docs/guides/reasoning +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.1" [interleaved] @@ -8,6 +10,9 @@ field = "reasoning_content" [[reasoning_options]] type = "toggle" +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 1.4 output = 4.4 diff --git a/providers/friendli/models/zai-org/GLM-5.2.toml b/providers/friendli/models/zai-org/GLM-5.2.toml index 4fff04e7c0f..187d5de39d0 100644 --- a/providers/friendli/models/zai-org/GLM-5.2.toml +++ b/providers/friendli/models/zai-org/GLM-5.2.toml @@ -2,6 +2,8 @@ # https://friendli.ai/docs/guides/reasoning # Effort: reasoning_effort = "high" | "max" # https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.2" [interleaved] @@ -14,6 +16,9 @@ type = "toggle" type = "effort" values = ["high", "max"] +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 1.4 output = 4.4 diff --git a/providers/friendli/models/zai-org/GLM-5.3-Flash.toml b/providers/friendli/models/zai-org/GLM-5.3-Flash.toml index bea59ccfb1b..7d25fca9c7b 100644 --- a/providers/friendli/models/zai-org/GLM-5.3-Flash.toml +++ b/providers/friendli/models/zai-org/GLM-5.3-Flash.toml @@ -1,5 +1,7 @@ # Effort: reasoning_effort = "low" | "high" | "max" # https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.3-flash" [interleaved] @@ -9,6 +11,9 @@ field = "reasoning_content" type = "effort" values = ["low", "high", "max"] +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.15 output = 0.5 diff --git a/providers/friendli/models/zai-org/GLM-5.3.toml b/providers/friendli/models/zai-org/GLM-5.3.toml index 25fae4ddb49..baad00a0786 100644 --- a/providers/friendli/models/zai-org/GLM-5.3.toml +++ b/providers/friendli/models/zai-org/GLM-5.3.toml @@ -1,5 +1,7 @@ # Effort: reasoning_effort = "low" | "high" | "max" # https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.3" [interleaved] @@ -9,6 +11,9 @@ field = "reasoning_content" type = "effort" values = ["low", "high", "max"] +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 1.26 output = 3.96 From 175bfe801493113fd3ae35968c4b35d55252dd6d Mon Sep 17 00:00:00 2001 From: rekram1-node Date: Mon, 14 Sep 2026 16:36:16 +0000 Subject: [PATCH 5/6] feat(sync): track selectively skipped models --- packages/core/src/sync/index.ts | 13 +++++- packages/core/src/sync/missing-issues.ts | 2 +- packages/core/src/sync/providers/friendli.ts | 5 +++ packages/core/test/friendli.test.ts | 30 +++++++++++++ packages/core/test/missing-skips.test.ts | 46 ++++++++++++++++++++ sync.md | 3 ++ 6 files changed, 96 insertions(+), 3 deletions(-) create mode 100644 packages/core/test/friendli.test.ts create mode 100644 packages/core/test/missing-skips.test.ts diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index 6cd005e977c..0d11751b4c6 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -97,6 +97,11 @@ export interface SyncProvider { * undefined to skip silently (no notice, no missing-model issue). */ sourceID?(model: SourceModel): string | undefined; + /** + * Return the ID when a source model skipped by translateModel needs a + * missing-model issue. Return undefined for intentional skips. + */ + missingModelID?(model: SourceModel): string | undefined; skippedNotice?(ids: string[]): string[]; fetchModels(): Promise; parseModels(raw: unknown): SourceModel[]; @@ -258,6 +263,7 @@ export async function syncProvider( const caseNormalizedDesiredPaths = new Map(); const desiredMetadata = new Map; content: string }>(); const skippedRemote: string[] = []; + const missingRemote: string[] = []; const missingReasoning = new Map(); for (const sourceModel of sourceModels) { @@ -280,6 +286,8 @@ export async function syncProvider( if (translated === undefined) { const skippedID = provider.sourceID?.(sourceModel); if (skippedID !== undefined) skippedRemote.push(skippedID); + const missingID = provider.missingModelID?.(sourceModel); + if (missingID !== undefined) missingRemote.push(missingID); continue; } @@ -490,10 +498,11 @@ export async function syncProvider( ...provider.missingNotice?.(missingLocal) ?? [], ]; - const issueModels = [ + const issueModels = [...new Set([ + ...missingRemote, ...(provider.skipCreates === true ? skippedRemote : []), ...missingReasoning.keys(), - ]; + ])]; if ( provider.trackMissingModels !== false && issueModels.length > 0 diff --git a/packages/core/src/sync/missing-issues.ts b/packages/core/src/sync/missing-issues.ts index 81a01a3742e..3e841f4aec0 100644 --- a/packages/core/src/sync/missing-issues.ts +++ b/packages/core/src/sync/missing-issues.ts @@ -26,7 +26,7 @@ function issueBody(provider: MissingModelIssueTarget, modelId: string, reason?: `| Expected path | \`${provider.modelsDir}/${modelId}.toml\` |`, "", reason === undefined - ? "This provider uses `skipCreates` because the remote source is not enough to auto-author a full TOML." + ? "Automatic creation was skipped because the remote source is not enough to auto-author a complete catalog entry." : `Sync diagnostic: ${reason}`, "Add the model manually (prefer `base_model` when matching `models/` metadata exists).", ...(reason === undefined ? [] : [ diff --git a/packages/core/src/sync/providers/friendli.ts b/packages/core/src/sync/providers/friendli.ts index 83e722c849c..68929c32871 100644 --- a/packages/core/src/sync/providers/friendli.ts +++ b/packages/core/src/sync/providers/friendli.ts @@ -255,6 +255,11 @@ export const friendli = { sourceID(model: FriendliModel) { return model.id; }, + missingModelID(model: FriendliModel) { + // Active models only reach the skip path when their provider-agnostic lab + // metadata is missing. Deprecated models are intentional removals. + return isDeprecated(model) ? undefined : model.id; + }, skippedNotice(ids: string[]) { if (ids.length === 0) return []; return [ diff --git a/packages/core/test/friendli.test.ts b/packages/core/test/friendli.test.ts new file mode 100644 index 00000000000..db086d53b76 --- /dev/null +++ b/packages/core/test/friendli.test.ts @@ -0,0 +1,30 @@ +import { expect, test } from "bun:test"; + +import { friendli, FriendliModel } from "../src/sync/providers/friendli.js"; + +const model = FriendliModel.parse({ + id: "example/model", + name: "Example Model", + created: 1_775_088_000, + context_length: 128_000, + max_completion_tokens: 128_000, + functionality: { + tool_call: true, + structured_output: true, + }, + pricing: { + input: "0.000001", + output: "0.000002", + }, +}); + +test("tracks active Friendli models missing lab metadata", () => { + expect(friendli.missingModelID(model)).toBe(model.id); +}); + +test("does not track deprecated Friendli models as missing", () => { + expect(friendli.missingModelID({ + ...model, + deprecation_date: "2000-01-01T00:00:00Z", + })).toBeUndefined(); +}); diff --git a/packages/core/test/missing-skips.test.ts b/packages/core/test/missing-skips.test.ts new file mode 100644 index 00000000000..e3843008d7f --- /dev/null +++ b/packages/core/test/missing-skips.test.ts @@ -0,0 +1,46 @@ +import { expect, spyOn, test } from "bun:test"; +import { mkdir, mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +import { syncProvider, type SyncProvider } from "../src/sync/index.js"; +import * as missingIssues from "../src/sync/missing-issues.js"; + +test("opens issues for selectively skipped missing models", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "sync-missing-model-")); + const modelsDir = path.join(dir, "providers", "example", "models"); + await mkdir(modelsDir, { recursive: true }); + const issues = spyOn(missingIssues, "openMissingModelIssues").mockResolvedValue([]); + const provider: SyncProvider<{ id: string; missing: boolean }> = { + id: "example", + name: "Example", + modelsDir, + async fetchModels() { + return [ + { id: "needs-metadata", missing: true }, + { id: "intentional-skip", missing: false }, + ]; + }, + parseModels(raw) { + return raw as { id: string; missing: boolean }[]; + }, + translateModel() { + return undefined; + }, + sourceID(model) { + return model.id; + }, + missingModelID(model) { + return model.missing ? model.id : undefined; + }, + }; + + try { + await syncProvider(provider, { openIssues: true }); + expect(issues).toHaveBeenCalledTimes(1); + expect(issues.mock.calls[0]?.[1]).toEqual(["needs-metadata"]); + } finally { + issues.mockRestore(); + await rm(dir, { recursive: true, force: true }); + } +}); diff --git a/sync.md b/sync.md index 06ceb1b2a42..d7729562e0f 100644 --- a/sync.md +++ b/sync.md @@ -46,6 +46,7 @@ Sync runs also write `.sync/model-sync-report.md` for the automation workflow PR - Removes existing files that are no longer present in the desired synced set. - Writes `.sync/model-sync-report.md` for GitHub Actions. - When `skipCreates` is set and issue opens are enabled, opens one deduped GitHub issue per remote model missing from the local catalog (via `gh`). +- When a provider selectively skips only some new models, `missingModelID` can mark those skips for the same deduped issue flow without disabling safe automatic creates. Because the runner removes files missing from the desired set, a provider module should only skip source models when deleting existing local files for those skipped IDs is intentional. @@ -59,6 +60,8 @@ Providers that cannot safely auto-create TOMLs set `skipCreates: true`. In GitHu 4. Dispatches the Issue Fixer explicitly so issues created with `GITHUB_TOKEN` can still produce PRs 5. If listing fails, creates nothing (fail closed) +Providers that can auto-create most models may instead return an ID from `missingModelID` only for `translateModel` skips that need manual metadata. Intentional skips return `undefined` and do not open issues. + Requires `GH_TOKEN` on the sync workflow step. Local runs are notice-only unless `--open-issues`. Use `--no-issues` / `--dry-run` to skip creates. Each newly opened issue explicitly dispatches the issue-fixer workflow so an agent can research the missing metadata and open a model PR. The first Actions run may open a batch of issues per provider, including remote IDs the catalog intentionally omits (e.g. OpenAI whisper/tts/moderation surfaces, dated snapshots). This one-time volume is accepted by design: close unwanted issues once and the closed-title dedupe suppresses them permanently. If the dedupe list window (1000 labeled issues per provider) ever fills, the sync fails closed and creates nothing rather than risk duplicates. From 07509ba63af2d1a65b3272f5818d89f0e3bd4bfd Mon Sep 17 00:00:00 2001 From: rekram1-node Date: Mon, 14 Sep 2026 16:59:29 +0000 Subject: [PATCH 6/6] fix(friendli): fail closed on catalog gaps --- packages/core/src/sync/index.ts | 13 +++++++---- packages/core/src/sync/providers/friendli.ts | 24 ++++++++++++-------- packages/core/test/friendli.test.ts | 11 +++++++++ packages/core/test/missing-skips.test.ts | 6 ++++- sync.md | 4 ++-- 5 files changed, 42 insertions(+), 16 deletions(-) diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index 0d11751b4c6..39d87a5817e 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -99,7 +99,8 @@ export interface SyncProvider { sourceID?(model: SourceModel): string | undefined; /** * Return the ID when a source model skipped by translateModel needs a - * missing-model issue. Return undefined for intentional skips. + * missing-model issue. Existing local metadata for that ID is preserved. + * Return undefined for intentional skips. */ missingModelID?(model: SourceModel): string | undefined; skippedNotice?(ids: string[]): string[]; @@ -263,7 +264,7 @@ export async function syncProvider( const caseNormalizedDesiredPaths = new Map(); const desiredMetadata = new Map; content: string }>(); const skippedRemote: string[] = []; - const missingRemote: string[] = []; + const missingRemote = new Set(); const missingReasoning = new Map(); for (const sourceModel of sourceModels) { @@ -287,7 +288,7 @@ export async function syncProvider( const skippedID = provider.sourceID?.(sourceModel); if (skippedID !== undefined) skippedRemote.push(skippedID); const missingID = provider.missingModelID?.(sourceModel); - if (missingID !== undefined) missingRemote.push(missingID); + if (missingID !== undefined) missingRemote.add(missingID); continue; } @@ -467,6 +468,10 @@ export async function syncProvider( const missingLocal: string[] = []; for (const relativePath of new Set([...existing.keys(), ...brokenSymlinks])) { if (desired.has(relativePath)) continue; + if (missingRemote.has(relativePath.slice(0, -5))) { + unchanged++; + continue; + } if (missingReasoning.has(relativePath.slice(0, -5))) { unchanged++; continue; @@ -499,7 +504,7 @@ export async function syncProvider( ]; const issueModels = [...new Set([ - ...missingRemote, + ...missingRemote.values(), ...(provider.skipCreates === true ? skippedRemote : []), ...missingReasoning.keys(), ])]; diff --git a/packages/core/src/sync/providers/friendli.ts b/packages/core/src/sync/providers/friendli.ts index 68929c32871..d021f8ed484 100644 --- a/packages/core/src/sync/providers/friendli.ts +++ b/packages/core/src/sync/providers/friendli.ts @@ -222,10 +222,15 @@ export const friendli = { return response.json(); }, parseModels(raw: unknown) { - return FriendliResponse.parse(raw).data; + const models = FriendliResponse.parse(raw).data; + if (models.length === 0) { + throw new Error("Friendli returned an empty model catalog; refusing destructive sync"); + } + return models; }, translateModel(model: FriendliModel, context) { const existing = context.existing(model.id); + const authored = context.authored(model.id); // A model past its deprecation_date is skipped outright (tracked or not): // with deleteMissing enabled, skipping removes an already-tracked file on // the next sync, so the catalog never keeps serving a dead route as a @@ -233,14 +238,15 @@ export const friendli = { // catalog means the same thing as the model disappearing from it. if (isDeprecated(model)) return undefined; const factorBase = resolveLabModelSync(model); - // Friendli is a multi-lab relay, so a brand-new remote model with no - // provider-agnostic lab metadata to factor onto must not be authored - // full-inline by an hourly sync (AGENTS.md: full inline is reserved for - // first-party labs or true host-unique aliases). Mirror the deepinfra - // create gate: already-tracked files keep updating; unresolvable new - // IDs are skipped with a notice until lab metadata exists or a true - // host-unique alias is hand-authored. - if (existing === undefined && factorBase === undefined) return undefined; + // Friendli is a multi-lab relay, so models that need a canonical lab entry + // are handled by the missing-model issue flow. If an existing factored + // entry becomes temporarily unresolvable, skip it as well: the runner + // preserves its TOML rather than expanding or deleting it. Existing true + // host-unique full-inline entries can still update normally. + if ( + factorBase === undefined + && (existing === undefined || authored?.base_model !== undefined) + ) return undefined; const built = buildFriendliModel( model, existing, diff --git a/packages/core/test/friendli.test.ts b/packages/core/test/friendli.test.ts index db086d53b76..7c6afa49a91 100644 --- a/packages/core/test/friendli.test.ts +++ b/packages/core/test/friendli.test.ts @@ -22,6 +22,17 @@ test("tracks active Friendli models missing lab metadata", () => { expect(friendli.missingModelID(model)).toBe(model.id); }); +test("rejects an empty Friendli catalog", () => { + expect(() => friendli.parseModels({ data: [] })).toThrow("empty model catalog"); +}); + +test("skips a factored model when its lab metadata cannot be resolved", () => { + expect(friendli.translateModel(model, { + existing: () => ({ base_model: "example/missing" }), + authored: () => ({ base_model: "example/missing" }), + })).toBeUndefined(); +}); + test("does not track deprecated Friendli models as missing", () => { expect(friendli.missingModelID({ ...model, diff --git a/packages/core/test/missing-skips.test.ts b/packages/core/test/missing-skips.test.ts index e3843008d7f..ddb3351186f 100644 --- a/packages/core/test/missing-skips.test.ts +++ b/packages/core/test/missing-skips.test.ts @@ -10,6 +10,8 @@ test("opens issues for selectively skipped missing models", async () => { const dir = await mkdtemp(path.join(tmpdir(), "sync-missing-model-")); const modelsDir = path.join(dir, "providers", "example", "models"); await mkdir(modelsDir, { recursive: true }); + const existingPath = path.join(modelsDir, "needs-metadata.toml"); + await Bun.write(existingPath, 'name = "Keep me"\n'); const issues = spyOn(missingIssues, "openMissingModelIssues").mockResolvedValue([]); const provider: SyncProvider<{ id: string; missing: boolean }> = { id: "example", @@ -36,7 +38,9 @@ test("opens issues for selectively skipped missing models", async () => { }; try { - await syncProvider(provider, { openIssues: true }); + const result = await syncProvider(provider, { openIssues: true }); + expect(result).toMatchObject({ deleted: 0, unchanged: 1 }); + expect(await Bun.file(existingPath).text()).toBe('name = "Keep me"\n'); expect(issues).toHaveBeenCalledTimes(1); expect(issues.mock.calls[0]?.[1]).toEqual(["needs-metadata"]); } finally { diff --git a/sync.md b/sync.md index d7729562e0f..70a06dab55b 100644 --- a/sync.md +++ b/sync.md @@ -46,7 +46,7 @@ Sync runs also write `.sync/model-sync-report.md` for the automation workflow PR - Removes existing files that are no longer present in the desired synced set. - Writes `.sync/model-sync-report.md` for GitHub Actions. - When `skipCreates` is set and issue opens are enabled, opens one deduped GitHub issue per remote model missing from the local catalog (via `gh`). -- When a provider selectively skips only some new models, `missingModelID` can mark those skips for the same deduped issue flow without disabling safe automatic creates. +- When a provider selectively skips only some models, `missingModelID` can preserve existing metadata and mark those skips for the same deduped issue flow without disabling safe automatic creates. Because the runner removes files missing from the desired set, a provider module should only skip source models when deleting existing local files for those skipped IDs is intentional. @@ -60,7 +60,7 @@ Providers that cannot safely auto-create TOMLs set `skipCreates: true`. In GitHu 4. Dispatches the Issue Fixer explicitly so issues created with `GITHUB_TOKEN` can still produce PRs 5. If listing fails, creates nothing (fail closed) -Providers that can auto-create most models may instead return an ID from `missingModelID` only for `translateModel` skips that need manual metadata. Intentional skips return `undefined` and do not open issues. +Providers that can auto-create most models may instead return an ID from `missingModelID` only for `translateModel` skips that need manual metadata. The runner preserves an existing local entry for that ID while the issue is handled. Intentional skips return `undefined` and do not open issues. Requires `GH_TOKEN` on the sync workflow step. Local runs are notice-only unless `--open-issues`. Use `--no-issues` / `--dry-run` to skip creates. Each newly opened issue explicitly dispatches the issue-fixer workflow so an agent can research the missing metadata and open a model PR.