diff --git a/packages/core/script/generate-friendli.ts b/packages/core/script/generate-friendli.ts deleted file mode 100644 index d048f3d8369..00000000000 --- a/packages/core/script/generate-friendli.ts +++ /dev/null @@ -1,505 +0,0 @@ -#!/usr/bin/env bun - -import { mkdir } from "node:fs/promises"; -import path from "node:path"; -import { z } from "zod"; - -import { inferKimiFamily } from "../src/family.js"; - -// Friendli API endpoint -const API_ENDPOINT = "https://api.friendli.ai/serverless/v1/models"; - -// Zod schemas for API response validation -const Functionality = z.object({ - tool_call: z.boolean(), - parallel_tool_call: z.boolean(), - structured_output: z.boolean(), -}); - -const Pricing = z.object({ - input: z.number(), - output: z.number(), - response_time: z.number(), - unit_type: z.enum(["TOKEN", "SECOND"]), -}); - -const FriendliModel = z - .object({ - id: z.string(), - name: z.string(), - max_completion_tokens: z.number(), - context_length: z.number(), - functionality: Functionality, - pricing: Pricing, - hugging_face_url: z.string().optional(), - description: z.string().optional(), - license: z.string().optional(), - policy: z.string().optional().nullable(), - created: z.number(), // Unix timestamp - }) - .passthrough(); - -const FriendliResponse = z.object({ - data: z.array(FriendliModel), -}); - -// Family inference patterns -const familyPatterns: [RegExp, string][] = [ - [/qwen3/i, "qwen3"], - [/deepseek-r1/i, "deepseek-r1"], - [/glm-4/i, "glm-4"], - [/glm-5/i, "glm"], -]; - -function inferFamily(modelId: string, modelName: string): string | undefined { - const kimiFamily = inferKimiFamily(modelId, modelName); - if (kimiFamily !== undefined) return kimiFamily; - - for (const [pattern, family] of familyPatterns) { - if (pattern.test(modelId) || pattern.test(modelName)) { - return family; - } - } - return undefined; -} - -function extractModelName(fullName: string): string { - // "meta-llama/Llama-3.3-70B-Instruct" -> "Llama 3.3 70B Instruct" - const parts = fullName.split("/"); - const modelName = parts.at(-1) ?? fullName; - return modelName - .replace(/-/g, " ") - .replace(/\b\w/g, (l) => l.toUpperCase()); -} - -// TODO: Replace with functionality.parse_reasoning from API when available -function isReasoningModel(modelId: string): boolean { - const nonReasoningPatterns = [ - /qwen3.*instruct/i, - ]; - - for (const pattern of nonReasoningPatterns) { - if (pattern.test(modelId)) { - return false; - } - } - - // Everything else is reasoning or hybrid reasoning - return true; -} - -function formatNumber(n: number): string { - if (n >= 1000) { - // Format with underscores for readability (e.g., 131_072) - return n.toString().replace(/\B(?=(\d{3})+(?!\d))/g, "_"); - } - return n.toString(); -} - -function timestampToDate(timestamp: number): string { - const date = new Date(timestamp * 1000); - return date.toISOString().slice(0, 10); -} - -function getTodayDate(): string { - return new Date().toISOString().slice(0, 10); -} - -interface ExistingModel { - name?: string; - family?: string; - attachment?: boolean; - reasoning?: boolean; - tool_call?: boolean; - structured_output?: boolean; - temperature?: boolean; - knowledge?: string; - release_date?: string; - last_updated?: string; - open_weights?: boolean; - interleaved?: boolean | { field: string }; - status?: string; - cost?: { - input?: number; - output?: number; - reasoning?: number; - cache_read?: number; - cache_write?: number; - }; - limit?: { - context?: number; - input?: number; - output?: number; - }; - modalities?: { - input?: string[]; - output?: string[]; - }; - provider?: { - npm?: string; - api?: string; - }; -} - -async function loadExistingModel( - filePath: string, -): Promise { - try { - const file = Bun.file(filePath); - if (!(await file.exists())) { - return null; - } - const toml = await import(filePath, { with: { type: "toml" } }).then( - (mod) => mod.default, - ); - return toml as ExistingModel; - } catch (e) { - console.warn(`Warning: Failed to parse existing file ${filePath}:`, e); - return null; - } -} - -interface MergedModel { - name: string; - family?: string; - attachment: boolean; - reasoning: boolean; - tool_call: boolean; - structured_output?: boolean; - temperature: boolean; - knowledge?: string; - release_date: string; - last_updated: string; - open_weights: boolean; - interleaved?: boolean | { field: string }; - status?: string; - cost?: { - input: number; - output: number; - }; - limit: { - context: number; - output: number; - }; - modalities: { - input: string[]; - output: string[]; - }; -} - -function mergeModel( - apiModel: z.infer, - existing: ExistingModel | null, -): MergedModel { - const contextTokens = apiModel.context_length; - const outputTokens = apiModel.max_completion_tokens; - - const openWeights = Boolean(apiModel.hugging_face_url); - - const merged: MergedModel = { - // Always from API - name: extractModelName(apiModel.name), - attachment: false, // All Friendli models are text-only currently - reasoning: isReasoningModel(apiModel.id), - tool_call: apiModel.functionality.tool_call, - temperature: true, - release_date: timestampToDate(apiModel.created), - last_updated: getTodayDate(), - open_weights: openWeights, - limit: { - context: contextTokens, - output: outputTokens, - }, - modalities: { - input: ["text"], - output: ["text"], - }, - }; - - // structured_output only if true - if (apiModel.functionality.structured_output === true) { - merged.structured_output = true; - } - - // Cost from API - ONLY include if unit_type is TOKEN - if (apiModel.pricing.unit_type === "TOKEN") { - merged.cost = { - input: apiModel.pricing.input, - output: apiModel.pricing.output, - }; - } else { - console.log( - ` Note: ${apiModel.id} uses ${apiModel.pricing.unit_type} pricing - cost section omitted`, - ); - } - - // Preserve from existing OR infer - if (existing?.family) { - merged.family = existing.family; - } else { - const inferred = inferFamily(apiModel.id, apiModel.name); - if (inferred) { - merged.family = inferred; - } - } - - // Preserve manual fields from existing - if (existing?.knowledge) { - merged.knowledge = existing.knowledge; - } - if (existing?.interleaved !== undefined) { - merged.interleaved = existing.interleaved; - } - if (existing?.status !== undefined) { - merged.status = existing.status; - } - - return merged; -} - -function formatToml(model: MergedModel): string { - const lines: string[] = []; - - // Basic fields - lines.push(`name = "${model.name.replace(/"/g, '\\"')}"`); - if (model.family) { - lines.push(`family = "${model.family}"`); - } - lines.push(`attachment = ${model.attachment}`); - lines.push(`reasoning = ${model.reasoning}`); - lines.push(`tool_call = ${model.tool_call}`); - if (model.structured_output !== undefined) { - lines.push(`structured_output = ${model.structured_output}`); - } - lines.push(`temperature = ${model.temperature}`); - if (model.knowledge) { - lines.push(`knowledge = "${model.knowledge}"`); - } - lines.push(`release_date = "${model.release_date}"`); - lines.push(`last_updated = "${model.last_updated}"`); - lines.push(`open_weights = ${model.open_weights}`); - if (model.status) { - lines.push(`status = "${model.status}"`); - } - - // Interleaved section (if present) - if (model.interleaved !== undefined) { - lines.push(""); - if (model.interleaved === true) { - lines.push(`interleaved = true`); - } else if (typeof model.interleaved === "object") { - lines.push(`[interleaved]`); - lines.push(`field = "${model.interleaved.field}"`); - } - } - - // Cost section (only if present) - if (model.cost) { - lines.push(""); - lines.push(`[cost]`); - lines.push(`input = ${model.cost.input}`); - lines.push(`output = ${model.cost.output}`); - } - - // Limit section - lines.push(""); - lines.push(`[limit]`); - lines.push(`context = ${formatNumber(model.limit.context)}`); - lines.push(`output = ${formatNumber(model.limit.output)}`); - - // Modalities section - lines.push(""); - lines.push(`[modalities]`); - lines.push( - `input = [${model.modalities.input.map((m) => `"${m}"`).join(", ")}]`, - ); - lines.push( - `output = [${model.modalities.output.map((m) => `"${m}"`).join(", ")}]`, - ); - - return lines.join("\n") + "\n"; -} - -interface Changes { - field: string; - oldValue: string; - newValue: string; -} - -function detectChanges( - existing: ExistingModel | null, - merged: MergedModel, -): Changes[] { - if (!existing) return []; - - const changes: Changes[] = []; - - const compare = (field: string, oldVal: unknown, newVal: unknown) => { - const oldStr = JSON.stringify(oldVal); - const newStr = JSON.stringify(newVal); - if (oldStr !== newStr) { - changes.push({ - field, - oldValue: formatValue(oldVal), - newValue: formatValue(newVal), - }); - } - }; - - const formatValue = (val: unknown): string => { - if (typeof val === "number") return formatNumber(val); - if (Array.isArray(val)) return `[${val.join(", ")}]`; - if (val === undefined) return "(none)"; - return String(val); - }; - - compare("name", existing.name, merged.name); - compare("family", existing.family, merged.family); - compare("attachment", existing.attachment, merged.attachment); - compare("reasoning", existing.reasoning, merged.reasoning); - compare("tool_call", existing.tool_call, merged.tool_call); - compare( - "structured_output", - existing.structured_output, - merged.structured_output, - ); - compare("open_weights", existing.open_weights, merged.open_weights); - compare("release_date", existing.release_date, merged.release_date); - compare("cost.input", existing.cost?.input, merged.cost?.input); - compare("cost.output", existing.cost?.output, merged.cost?.output); - compare("limit.context", existing.limit?.context, merged.limit.context); - compare("limit.output", existing.limit?.output, merged.limit.output); - compare("modalities.input", existing.modalities?.input, merged.modalities.input); - - return changes; -} - -async function main() { - const args = process.argv.slice(2); - const dryRun = args.includes("--dry-run"); - - const modelsDir = path.join( - import.meta.dirname, - "..", - "..", - "..", - "providers", - "friendli", - "models", - ); - - if (dryRun) { - console.log(`[DRY RUN] Fetching Friendli models from API...`); - } else { - console.log(`Fetching Friendli models from API...`); - } - - // Fetch API data - const res = await fetch(API_ENDPOINT); - if (!res.ok) { - console.error(`Failed to fetch API: ${res.status} ${res.statusText}`); - process.exit(1); - } - - const json = await res.json(); - const parsed = FriendliResponse.safeParse(json); - if (!parsed.success) { - console.error("Invalid API response:", parsed.error.errors); - process.exit(1); - } - - const apiModels = parsed.data.data; - - // Get existing files (recursively) - const existingFiles = new Set(); - try { - for await (const file of new Bun.Glob("**/*.toml").scan({ - cwd: modelsDir, - absolute: false, - })) { - existingFiles.add(file); - } - } catch { - // Directory might not exist yet - } - - console.log( - `Found ${apiModels.length} models in API, ${existingFiles.size} existing files\n`, - ); - - // Track API model IDs for orphan detection - const apiModelIds = new Set(); - - let created = 0; - let updated = 0; - let unchanged = 0; - - for (const apiModel of apiModels) { - const relativePath = `${apiModel.id}.toml`; - const filePath = path.join(modelsDir, relativePath); - const dirPath = path.dirname(filePath); - - apiModelIds.add(relativePath); - - const existing = await loadExistingModel(filePath); - const merged = mergeModel(apiModel, existing); - const tomlContent = formatToml(merged); - - if (existing === null) { - created++; - if (dryRun) { - console.log(`[DRY RUN] Would create: ${relativePath}`); - console.log(` name = "${merged.name}"`); - if (merged.family) { - console.log(` family = "${merged.family}" (inferred)`); - } - console.log(""); - } else { - await mkdir(dirPath, { recursive: true }); - await Bun.write(filePath, tomlContent); - console.log(`Created: ${relativePath}`); - } - } else { - const changes = detectChanges(existing, merged); - - if (changes.length > 0) { - updated++; - if (dryRun) { - console.log(`[DRY RUN] Would update: ${relativePath}`); - } else { - await Bun.write(filePath, tomlContent); - console.log(`Updated: ${relativePath}`); - } - for (const change of changes) { - console.log(` ${change.field}: ${change.oldValue} → ${change.newValue}`); - } - console.log(""); - } else { - unchanged++; - } - } - } - - // Check for orphaned files - const orphaned: string[] = []; - for (const file of existingFiles) { - if (!apiModelIds.has(file)) { - orphaned.push(file); - console.log(`Warning: Orphaned file (not in API): ${file}`); - } - } - - // Summary - console.log(""); - if (dryRun) { - console.log( - `Summary: ${created} would be created, ${updated} would be updated, ${unchanged} unchanged, ${orphaned.length} orphaned`, - ); - } else { - console.log( - `Summary: ${created} created, ${updated} updated, ${unchanged} unchanged, ${orphaned.length} orphaned`, - ); - } -} - -await main(); diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index e6c30874235..39d87a5817e 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -18,6 +18,7 @@ import { deepinfra } from "./providers/deepinfra.js"; import { digitalocean } from "./providers/digitalocean.js"; import { edenai } from "./providers/edenai.js"; import { empiriolabs } from "./providers/empiriolabs.js"; +import { friendli } from "./providers/friendli.js"; import { githubCopilot } from "./providers/github-copilot.js"; import { google } from "./providers/google.js"; import { hyper } from "./providers/hyper.js"; @@ -96,6 +97,12 @@ export interface SyncProvider { * undefined to skip silently (no notice, no missing-model issue). */ sourceID?(model: SourceModel): string | undefined; + /** + * Return the ID when a source model skipped by translateModel needs a + * missing-model issue. Existing local metadata for that ID is preserved. + * Return undefined for intentional skips. + */ + missingModelID?(model: SourceModel): string | undefined; skippedNotice?(ids: string[]): string[]; fetchModels(): Promise; parseModels(raw: unknown): SourceModel[]; @@ -143,6 +150,7 @@ export const providers: { digitalocean: SyncProvider; edenai: SyncProvider; empiriolabs: SyncProvider; + friendli: SyncProvider; "github-copilot": SyncProvider; google: SyncProvider; hyper: SyncProvider; @@ -179,6 +187,7 @@ export const providers: { digitalocean, edenai, empiriolabs, + friendli, "github-copilot": githubCopilot, google, hyper, @@ -222,7 +231,7 @@ export const groups = { "vercel", ], cloudflare: ["cloudflare-ai-gateway", "cloudflare-workers-ai"], - direct: ["ambient", "anthropic", "baseten", "chutes", "cortecs", "deepinfra", "digitalocean", "github-copilot", "google", "hyper", "meta", "ollama-cloud", "openai", "ovhcloud", "pioneer", "tinfoil", "venice", "wandb", "xai"], + direct: ["ambient", "anthropic", "baseten", "chutes", "cortecs", "deepinfra", "digitalocean", "friendli", "github-copilot", "google", "hyper", "meta", "ollama-cloud", "openai", "ovhcloud", "pioneer", "tinfoil", "venice", "wandb", "xai"], } as const; type ProviderID = keyof typeof providers; @@ -255,6 +264,7 @@ export async function syncProvider( const caseNormalizedDesiredPaths = new Map(); const desiredMetadata = new Map; content: string }>(); const skippedRemote: string[] = []; + const missingRemote = new Set(); const missingReasoning = new Map(); for (const sourceModel of sourceModels) { @@ -277,6 +287,8 @@ export async function syncProvider( if (translated === undefined) { const skippedID = provider.sourceID?.(sourceModel); if (skippedID !== undefined) skippedRemote.push(skippedID); + const missingID = provider.missingModelID?.(sourceModel); + if (missingID !== undefined) missingRemote.add(missingID); continue; } @@ -456,6 +468,10 @@ export async function syncProvider( const missingLocal: string[] = []; for (const relativePath of new Set([...existing.keys(), ...brokenSymlinks])) { if (desired.has(relativePath)) continue; + if (missingRemote.has(relativePath.slice(0, -5))) { + unchanged++; + continue; + } if (missingReasoning.has(relativePath.slice(0, -5))) { unchanged++; continue; @@ -487,10 +503,11 @@ export async function syncProvider( ...provider.missingNotice?.(missingLocal) ?? [], ]; - const issueModels = [ + const issueModels = [...new Set([ + ...missingRemote.values(), ...(provider.skipCreates === true ? skippedRemote : []), ...missingReasoning.keys(), - ]; + ])]; if ( provider.trackMissingModels !== false && issueModels.length > 0 diff --git a/packages/core/src/sync/missing-issues.ts b/packages/core/src/sync/missing-issues.ts index 81a01a3742e..3e841f4aec0 100644 --- a/packages/core/src/sync/missing-issues.ts +++ b/packages/core/src/sync/missing-issues.ts @@ -26,7 +26,7 @@ function issueBody(provider: MissingModelIssueTarget, modelId: string, reason?: `| Expected path | \`${provider.modelsDir}/${modelId}.toml\` |`, "", reason === undefined - ? "This provider uses `skipCreates` because the remote source is not enough to auto-author a full TOML." + ? "Automatic creation was skipped because the remote source is not enough to auto-author a complete catalog entry." : `Sync diagnostic: ${reason}`, "Add the model manually (prefer `base_model` when matching `models/` metadata exists).", ...(reason === undefined ? [] : [ diff --git a/packages/core/src/sync/providers/friendli.ts b/packages/core/src/sync/providers/friendli.ts new file mode 100644 index 00000000000..d021f8ed484 --- /dev/null +++ b/packages/core/src/sync/providers/friendli.ts @@ -0,0 +1,560 @@ +import path from "node:path"; +import { readdirSync } from "node:fs"; +import { z } from "zod"; + +import { describeModel } from "../../describe.js"; +import { inferKimiFamily, ModelFamilyValues } from "../../family.js"; +import type { ExistingModel, SyncProvider, SyncedFullModel, SyncedModel } from "../index.js"; +import { factorBaseModel } from "./openrouter.js"; + +const API_ENDPOINT = "https://api.friendli.ai/serverless/v1/models"; +const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models"); + +// Friendli catalog pricing is USD per-token; catalog cost is USD per-million. +const PER_TOKEN_TO_PER_MILLION = 1_000_000; + +const InterleavedField = z.enum(["reasoning_content", "reasoning_details"]); + +// Friendli's /v1/models `interleaved` flag is unreliable for some models: it +// reports `false` for deepseek-ai/DeepSeek-V3.2 even though a live +// POST /chat/completions request (chat_template_kwargs.enable_thinking=true) +// returns both `reasoning` and `reasoning_content` in the response. Relying on +// `existing?.interleaved` to carry this forward is fragile — if the on-disk +// file ever loses the field for any reason, the live verification is silently +// forgotten on the next sync with no trace. This map is the durable source of +// truth for models where a live request has verified a real field the +// catalog API misreports; translateInterleaved consults it before falling +// back to the existing on-disk value. +const VERIFIED_INTERLEAVED_OVERRIDES: Record = { + "deepseek-ai/DeepSeek-V3.2": { field: "reasoning_content" }, +}; + +// Raw API reasoning_options shape, including budget_tokens (a real +// reasoning-budget control on Friendli: min = -1 means unlimited, max +// corresponds to max_completion_tokens). Confirmed via the live /v1/models +// response and https://friendli.ai/docs/openapi/model-apis/chat-completions +// (reasoning_budget is a documented request field). The catalog's min/max +// are not safe published bounds (see translateReasoningOptions below), so +// they are parsed but never carried into the synced model. +const FriendliReasoningOption = z + .discriminatedUnion("type", [ + z.object({ type: z.literal("toggle") }).passthrough(), + z + .object({ type: z.literal("effort"), values: z.array(z.string()) }) + .passthrough(), + z + .object({ + type: z.literal("budget_tokens"), + min: z.number().optional(), + max: z.number().optional(), + }) + .passthrough(), + ]) + .optional(); + +export const FriendliModel = z + .object({ + id: z.string(), + hugging_face_id: z.string().optional(), + name: z.string(), + created: z.number(), + context_length: z.number(), + max_completion_tokens: z.number(), + functionality: z + .object({ + tool_call: z.boolean(), + parallel_tool_call: z.boolean().optional(), + structured_output: z.boolean(), + tool_choice: z.boolean().optional(), + system_messages: z.boolean().optional(), + }) + .passthrough(), + pricing: z + .object({ + input: z.union([z.string(), z.number()]), + output: z.union([z.string(), z.number()]), + prompt: z.union([z.string(), z.number()]).optional(), + completion: z.union([z.string(), z.number()]).optional(), + input_cache_read: z.union([z.string(), z.number()]).optional(), + input_cache_write: z.union([z.string(), z.number()]).optional(), + // The pre-SyncProvider generator validated this field and authored + // cost only for TOKEN pricing. The current catalog always omits it + // for the 7 live models, but Friendli has served SECOND-priced + // entries before — passthrough would silently x1,000,000 a + // per-second rate into the catalog's USD/MTok cost. + unit_type: z.enum(["TOKEN", "SECOND"]).optional(), + }) + .passthrough(), + description: z.string().optional(), + hugging_face_url: z.string().optional(), + license: z.string().optional(), + policy: z.string().nullable().optional(), + deprecation_date: z.string().nullable().optional(), + reasoning: z.boolean().optional(), + reasoning_options: z.array(FriendliReasoningOption).optional(), + interleaved: z.union([InterleavedField, z.boolean()]).optional(), + input_modalities: z.array(z.string()).optional(), + output_modalities: z.array(z.string()).optional(), + base_model: z.string().optional(), + mode: z.string().optional(), + }) + .passthrough(); + +export const FriendliResponse = z + .object({ + data: z.array(FriendliModel), + }) + .passthrough(); + +export type FriendliModel = z.infer; + +// HuggingFace-style API orgs that are not catalog lab ids. Map them onto the +// catalog metadata tree so self-referential or HF-style base_model values +// resolve to the right lab directory. +const LAB_PREFIX_MAP: Record = { + "zai-org": "zhipuai", + "deepseek-ai": "deepseek", + "LGAI-EXAONE": "lgai-exaone", + "MiniMaxAI": "minimax", + "meta-llama": "meta", + "mistralai": "mistral", + "Qwen": "alibaba", +}; + +// Resolve an API `base_model` id to the on-disk `models//.toml` id. +// Friendli declares a base_model for most entries, but only models with an +// existing lab metadata file can be factored (override-only). Self-referential +// base_model values (==id) resolve to the model's own lab id when a metadata +// file exists under the mapped lab prefix. +// +// Case-insensitive lookup: the API lowercases some ids (e.g. +// "minimax/minimax-m2.5") that exist on disk as mixed-case +// ("minimax/MiniMax-M2.5.toml"), so we never trust a raw API id and always +// read the directory. +const baseModelCache = new Map(); + +function resolveBaseModelID(baseModel: string | undefined): string | undefined { + if (baseModel === undefined || baseModel.length === 0) return undefined; + const cached = baseModelCache.get(baseModel); + if (cached !== undefined) return cached ?? undefined; + + let resolved = lookupLabFile(baseModel); + if (resolved === undefined) { + const [org, ...parts] = baseModel.split("/"); + const mapped = org !== undefined ? LAB_PREFIX_MAP[org] : undefined; + if (mapped !== undefined && parts.length > 0) { + resolved = lookupLabFile(`${mapped}/${parts.join("/")}`); + } + } + + baseModelCache.set(baseModel, resolved ?? null); + return resolved; +} + +// Resolve a Friendli entry to its catalog lab metadata id. +// 1) API-declared base_model (handles HF id → catalog slug mismatches) +// 2) self-referential fallback: some entries (e.g. deepseek-ai/DeepSeek-V3.2) +// omit base_model entirely even though a matching lab metadata file +// exists under the mapped lab prefix — resolve against the model's own id. +function resolveLabModelSync(model: FriendliModel): string | undefined { + return resolveBaseModelID(model.base_model) ?? resolveBaseModelID(model.id); +} + +function lookupLabFile(baseModel: string): string | undefined { + const [lab, ...modelParts] = baseModel.split("/"); + const modelSlug = modelParts.join("/"); + if (lab === undefined || modelSlug.length === 0) return undefined; + + let labDir: string | undefined; + try { + const dirs = readdirSync(MODELS_DIR, { withFileTypes: true }) + .filter((entry) => entry.isDirectory()) + .map((entry) => entry.name); + labDir = dirs.find((dir) => dir.toLowerCase() === lab.toLowerCase()); + } catch { + return undefined; + } + if (labDir === undefined) return undefined; + + const expected = `${modelSlug}.toml`.toLowerCase(); + let fileMatch: string | undefined; + try { + fileMatch = readdirSync(path.join(MODELS_DIR, labDir)) + .filter((file) => file.endsWith(".toml")) + .find((file) => file.toLowerCase() === expected); + } catch { + // fall through + } + if (fileMatch === undefined) return undefined; + + return `${labDir}/${fileMatch.slice(0, -".toml".length)}`; +} + +export const friendli = { + id: "friendli", + name: "Friendli", + modelsDir: "providers/friendli/models", + // Friendli's /v1/models is authoritative for what this host serves: a model + // absent from the catalog (or past its deprecation_date) must not stay in + // the catalog as a live-looking route, so missing files are deleted rather + // than retained. Deprecation marking below only applies while the model is + // still listed; once it disappears, the file goes with it. + deleteMissing: true, + // Friendli's catalog describes real reasoning controls and limits directly; + // do not carry over a stale base_model when a model switches lab → full inline. + preserveBaseModels: false, + // The runner's default preserveDescription re-injects the resolved base + // description when the translator omits it, recreating an identical + // override. Friendli descriptions come from the API verbatim and match the + // lab's, so drop the re-injection. + preserveDescriptions: false, + // Leading wire-path comments (Toggle/Effort/Budget + doc URLs) always + // refresh from reasoningHeader() below instead of freezing whatever + // comment happened to be on disk the first time a file was created. + authoritativeHeaders: true, + async fetchModels() { + const response = await fetch(API_ENDPOINT); + if (!response.ok) { + throw new Error( + `Friendli request failed: ${response.status} ${response.statusText}`, + ); + } + return response.json(); + }, + parseModels(raw: unknown) { + const models = FriendliResponse.parse(raw).data; + if (models.length === 0) { + throw new Error("Friendli returned an empty model catalog; refusing destructive sync"); + } + return models; + }, + translateModel(model: FriendliModel, context) { + const existing = context.existing(model.id); + const authored = context.authored(model.id); + // A model past its deprecation_date is skipped outright (tracked or not): + // with deleteMissing enabled, skipping removes an already-tracked file on + // the next sync, so the catalog never keeps serving a dead route as a + // live-looking entry. Source-of-truth policy: a deprecation_date in the + // catalog means the same thing as the model disappearing from it. + if (isDeprecated(model)) return undefined; + const factorBase = resolveLabModelSync(model); + // Friendli is a multi-lab relay, so models that need a canonical lab entry + // are handled by the missing-model issue flow. If an existing factored + // entry becomes temporarily unresolvable, skip it as well: the runner + // preserves its TOML rather than expanding or deleting it. Existing true + // host-unique full-inline entries can still update normally. + if ( + factorBase === undefined + && (existing === undefined || authored?.base_model !== undefined) + ) return undefined; + const built = buildFriendliModel( + model, + existing, + factorBase, + ); + return { + id: model.id, + model: built, + header: reasoningHeader(built), + }; + }, + sourceID(model: FriendliModel) { + return model.id; + }, + missingModelID(model: FriendliModel) { + // Active models only reach the skip path when their provider-agnostic lab + // metadata is missing. Deprecated models are intentional removals. + return isDeprecated(model) ? undefined : model.id; + }, + skippedNotice(ids: string[]) { + if (ids.length === 0) return []; + return [ + `${ids.length} remote model(s) skipped: no provider-agnostic lab metadata to factor onto (full-inline creates are not authored for a multi-lab relay — add models//.toml, then re-sync) or deprecation_date passed: ${ids.join(", ")}`, + ]; + }, + missingNotice(paths: string[]) { + if (paths.length === 0) return []; + return [ + `${paths.length} local model(s) deleted after being removed from the Friendli API (or past their deprecation_date): ${paths.join(", ")}`, + ]; + }, +} satisfies SyncProvider; + +// Leading wire-path comments for every reasoning control type this host +// authors on a file, matching the wire paths documented in +// providers/friendli/provider.toml. With authoritativeHeaders enabled, this +// header always replaces whatever was on disk, so it never goes stale. +const REASONING_GUIDE_URL = "https://friendli.ai/docs/guides/reasoning"; +const EFFORT_DOC_URL = + "https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0"; +const BUDGET_DOC_URL = + "https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0"; + +function reasoningHeader(model: SyncedModel): string | undefined { + const options = model.reasoning_options; + if (options === undefined || options.length === 0) return undefined; + const lines: string[] = []; + for (const option of options) { + if (option.type === "toggle") { + lines.push("# Toggle: chat_template_kwargs.enable_thinking = true | false"); + lines.push(`# ${REASONING_GUIDE_URL}`); + } + if (option.type === "effort") { + if (option.values.length > 0) { + const values = option.values.map((value) => `"${value}"`).join(" | "); + lines.push(`# Effort: reasoning_effort = ${values}`); + } else { + lines.push("# Effort: reasoning_effort (model-specific accepted values)"); + } + lines.push(`# ${EFFORT_DOC_URL}`); + } + if (option.type === "budget_tokens") { + lines.push( + "# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited)", + ); + lines.push(`# ${BUDGET_DOC_URL}`); + } + } + return lines.length > 0 ? `${lines.join("\n")}\n` : undefined; +} + +type Modality = "text" | "audio" | "image" | "video" | "pdf"; + +const ALLOWED_MODALITIES: Record = { + text: true, + audio: true, + image: true, + video: true, + pdf: true, +}; + +function translateModalities(values: string[] | undefined): Modality[] { + const result = [...new Set( + (values ?? ["text"]) + .map((value) => value.toLowerCase()) + .filter((value): value is Modality => ALLOWED_MODALITIES[value] === true), + )]; + return result.length > 0 ? result : ["text"]; +} + +// Skip models whose deprecation_date has passed. Friendli returns an ISO +// timestamp (e.g. "2026-08-20T00:00:00Z"); we compare against now at sync time. +function isDeprecated(model: FriendliModel): boolean { + if (model.deprecation_date === undefined || model.deprecation_date === null) return false; + const dep = Date.parse(model.deprecation_date); + return Number.isFinite(dep) && dep <= Date.now(); +} + +function perMillion(value: string | number | undefined): number | undefined { + if (value === undefined) return undefined; + const number = Number(value); + if (!Number.isFinite(number) || number < 0) return undefined; + const perM = number * PER_TOKEN_TO_PER_MILLION; + return Math.round(perM * 1_000_000) / 1_000_000; +} + +function buildCost( + model: FriendliModel, + existing: ExistingModel["cost"] | undefined, +): NonNullable | undefined { + // TOKEN-priced per-token USD rates are converted to USD/MTok. Any other + // unit (e.g. SECOND) is not a token rate: do not author a cost section for + // it instead of publishing an invented per-million price. + if (model.pricing.unit_type !== undefined && model.pricing.unit_type !== "TOKEN") { + return existing; + } + const input = perMillion(model.pricing.input); + const output = perMillion(model.pricing.output); + if (input === undefined || output === undefined) return existing; + return { + input, + output, + cache_read: perMillion(model.pricing.input_cache_read) ?? existing?.cache_read, + cache_write: perMillion(model.pricing.input_cache_write) ?? existing?.cache_write, + }; +} + +// Translate API reasoning_options into host-accurate catalog options. +// budget_tokens is kept as an unbounded `{ type = "budget_tokens" }`: +// reasoning_budget is a real, independently enforced Friendli control +// (live-verified on GLM-5.3, gemma-4-31B-it, DeepSeek-V3.2, and +// MiniMax-M2.5 — small budgets truncate reasoning_content mid-sentence while +// completion continues), and peers such as OpenRouter/Requesty publish it +// when the host supports it. The catalog's min/max values are not safe +// published range constraints — GLM-5.3 accepted reasoning_budget=1_048_577 +// despite reporting max=1_048_576 — so the capability is preserved without +// authoring bounds. A budget-only reasoner (MiniMax-M2.5) therefore publishes +// `[{ type = "budget_tokens" }]`, not []: [] would falsely claim no caller +// control on a host that documents reasoning_budget. +function translateReasoningOptions( + api: FriendliModel["reasoning_options"], +): SyncedFullModel["reasoning_options"] { + if (api === undefined) return undefined; + const options: NonNullable = []; + for (const option of api) { + if (option === undefined) continue; + if (option.type === "budget_tokens") { + options.push({ type: "budget_tokens" }); + continue; + } + options.push(option as NonNullable[number]); + } + return options.length > 0 ? options : []; +} + +function translateInterleaved( + modelID: string, + value: FriendliModel["interleaved"], + existing: SyncedFullModel["interleaved"] | undefined, +): SyncedFullModel["interleaved"] { + const verified = VERIFIED_INTERLEAVED_OVERRIDES[modelID]; + if (verified !== undefined) return verified; + if (value === undefined) return existing; + // The models endpoint can be stale/wrong for this field (verified live + // against deepseek-ai/DeepSeek-V3.2, see VERIFIED_INTERLEAVED_OVERRIDES) — + // trust an existing authored value over an API false rather than clearing it. + if (value === false) return existing; + if (value === true) return true; + return { field: value }; +} + +function inferFamily(modelID: string, name: string): SyncedFullModel["family"] { + const kimiFamily = inferKimiFamily(modelID, name); + if (kimiFamily !== undefined) return kimiFamily; + const target = `${modelID} ${name}`.toLowerCase(); + return [...ModelFamilyValues] + .sort((a, b) => b.length - a.length) + .find((family) => { + const escaped = family.toLowerCase().replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); + if (family === "o") { + return new RegExp(`(^|[^a-z0-9])${escaped}(?=\\d|$|[^a-z0-9])`).test(target); + } + return new RegExp(`(^|[^a-z0-9])${escaped}(?=$|[^a-z0-9])`).test(target); + }); +} + +function buildFriendliModel( + model: FriendliModel, + existing: ExistingModel | undefined, + factorBase: string | undefined, +): SyncedModel { + // translateModel already skips models past their deprecation_date, so every + // model reaching this point is live. Carry hand-authored lifecycle statuses + // (e.g. beta) through unchanged. + const status = existing?.status; + + // Only override modalities when the API explicitly provides them; otherwise + // omit the override so lab metadata (e.g. gemma vision) is inherited. + const apiInput = model.input_modalities !== undefined ? translateModalities(model.input_modalities) : undefined; + const apiOutput = model.output_modalities !== undefined ? translateModalities(model.output_modalities) : undefined; + // For a factored provider entry, only write the modality sides Friendli + // actually supplied. Plain-object inheritance deep-merges, so an omitted + // side must remain omitted to preserve the lab's canonical modality list + // rather than replacing it with an empty array. + const modalities = apiInput !== undefined || apiOutput !== undefined + ? { + ...(apiInput !== undefined ? { input: apiInput } : {}), + ...(apiOutput !== undefined ? { output: apiOutput } : {}), + } + : undefined; + // undefined when the API omits input_modalities so factorBaseModel + // inherits the lab attachment; only override when explicitly provided. + const attachment = apiInput !== undefined ? apiInput.some((value) => value !== "text") : undefined; + // Completion-length cap: when a base_model exists, defer to the lab's own + // limit.output instead of forcing Friendli's max_completion_tokens onto it. + // Friendli's max_completion_tokens equals context_length for every one of + // the 7 live models, and blindly asserting that as the completion cap would + // overwrite lab-verified, genuinely tighter completion limits (e.g. + // DeepSeek-V3.2's lab file documents output=64_000 out of a 128_000 + // context, not "same as context"). Only fall back to Friendli's own + // reported value when there is no base_model to inherit a real + // completion-cap policy from (full-inline entries). + const limit = { + context: model.context_length, + input: existing?.limit?.input, + output: factorBase !== undefined ? undefined : model.max_completion_tokens, + }; + // Reasoning is tri-state: Friendli omits the flag for some reasoners, and + // treating "absent" as `false` would publish an explicit reasoning=false + // override on factored entries and strip their reasoning_options. Only + // override when the API is authoritative; otherwise the lab wins. + const reasoning = model.reasoning === true ? true : model.reasoning === false ? false : undefined; + const reasoningOptions = reasoning === false ? undefined : translateReasoningOptions(model.reasoning_options); + const interleaved = translateInterleaved(model.id, model.interleaved, existing?.interleaved); + const structuredOutput = model.functionality.structured_output; + const cost = buildCost(model, existing?.cost); + const releaseDate = existing?.release_date ?? new Date(model.created * 1000).toISOString().slice(0, 10); + const today = new Date().toISOString().slice(0, 10); + const lastUpdated = existing?.last_updated ?? today; + + if (factorBase !== undefined) { + return factorBaseModel( + factorBase, + { + attachment, + reasoning, + reasoning_options: reasoningOptions, + interleaved, + // Friendli is authoritative for this host's tool-call surface; a + // real delta vs the lab (either direction) must be published. + tool_call: model.functionality.tool_call, + structured_output: structuredOutput, + // A factored entry inherits the lab description. Friendli's catalog + // description is host metadata, not a new model identity, and its + // generic text can be weaker than the lab's canonical description. + // Keep it only for full-inline entries below. + description: undefined, + limit, + modalities, + cost, + status, + }, + limit, + existing?.base_model === factorBase ? existing.base_model_omit : undefined, + ); + } + + const name = existing?.name ?? (model.name.split("/").at(-1) ?? model.name); + // Full-inline has no lab to inherit from: a reasoning flag the API omits + // defaults to false here (describeModel needs a boolean), while factored + // entries above leave it unset so the lab's value stands. + const inlineReasoning = reasoning ?? false; + return { + name, + description: + existing?.description ?? + model.description ?? + describeModel({ + id: model.id, + providerId: "friendli", + name, + family: existing?.family, + reasoning: inlineReasoning, + tool_call: model.functionality.tool_call, + structured_output: structuredOutput, + open_weights: Boolean(model.hugging_face_url), + // Full-inline entries have no lab modalities to inherit. Default only + // sides omitted by the API to text rather than constructing empty + // arrays, which would advertise an impossible no-output/no-input model. + modalities: { input: apiInput ?? ["text"], output: apiOutput ?? ["text"] }, + }), + family: existing?.family ?? inferFamily(model.id, name), + // Full-inline has no lab to inherit from; default text-only when the API + // omits modalities. Earlier `attachment` is undefined in that case. + attachment: attachment ?? false, + reasoning: inlineReasoning, + reasoning_options: reasoningOptions, + tool_call: model.functionality.tool_call, + structured_output: structuredOutput, + temperature: existing?.temperature ?? true, + release_date: releaseDate, + last_updated: lastUpdated, + open_weights: Boolean(model.hugging_face_url), + interleaved, + knowledge: existing?.knowledge, + cost, + limit: { context: model.context_length, output: model.max_completion_tokens }, + modalities: { input: apiInput ?? ["text"], output: apiOutput ?? ["text"] }, + status, + }; +} diff --git a/packages/core/test/friendli.test.ts b/packages/core/test/friendli.test.ts new file mode 100644 index 00000000000..7c6afa49a91 --- /dev/null +++ b/packages/core/test/friendli.test.ts @@ -0,0 +1,41 @@ +import { expect, test } from "bun:test"; + +import { friendli, FriendliModel } from "../src/sync/providers/friendli.js"; + +const model = FriendliModel.parse({ + id: "example/model", + name: "Example Model", + created: 1_775_088_000, + context_length: 128_000, + max_completion_tokens: 128_000, + functionality: { + tool_call: true, + structured_output: true, + }, + pricing: { + input: "0.000001", + output: "0.000002", + }, +}); + +test("tracks active Friendli models missing lab metadata", () => { + expect(friendli.missingModelID(model)).toBe(model.id); +}); + +test("rejects an empty Friendli catalog", () => { + expect(() => friendli.parseModels({ data: [] })).toThrow("empty model catalog"); +}); + +test("skips a factored model when its lab metadata cannot be resolved", () => { + expect(friendli.translateModel(model, { + existing: () => ({ base_model: "example/missing" }), + authored: () => ({ base_model: "example/missing" }), + })).toBeUndefined(); +}); + +test("does not track deprecated Friendli models as missing", () => { + expect(friendli.missingModelID({ + ...model, + deprecation_date: "2000-01-01T00:00:00Z", + })).toBeUndefined(); +}); diff --git a/packages/core/test/missing-skips.test.ts b/packages/core/test/missing-skips.test.ts new file mode 100644 index 00000000000..ddb3351186f --- /dev/null +++ b/packages/core/test/missing-skips.test.ts @@ -0,0 +1,50 @@ +import { expect, spyOn, test } from "bun:test"; +import { mkdir, mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +import { syncProvider, type SyncProvider } from "../src/sync/index.js"; +import * as missingIssues from "../src/sync/missing-issues.js"; + +test("opens issues for selectively skipped missing models", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "sync-missing-model-")); + const modelsDir = path.join(dir, "providers", "example", "models"); + await mkdir(modelsDir, { recursive: true }); + const existingPath = path.join(modelsDir, "needs-metadata.toml"); + await Bun.write(existingPath, 'name = "Keep me"\n'); + const issues = spyOn(missingIssues, "openMissingModelIssues").mockResolvedValue([]); + const provider: SyncProvider<{ id: string; missing: boolean }> = { + id: "example", + name: "Example", + modelsDir, + async fetchModels() { + return [ + { id: "needs-metadata", missing: true }, + { id: "intentional-skip", missing: false }, + ]; + }, + parseModels(raw) { + return raw as { id: string; missing: boolean }[]; + }, + translateModel() { + return undefined; + }, + sourceID(model) { + return model.id; + }, + missingModelID(model) { + return model.missing ? model.id : undefined; + }, + }; + + try { + const result = await syncProvider(provider, { openIssues: true }); + expect(result).toMatchObject({ deleted: 0, unchanged: 1 }); + expect(await Bun.file(existingPath).text()).toBe('name = "Keep me"\n'); + expect(issues).toHaveBeenCalledTimes(1); + expect(issues.mock.calls[0]?.[1]).toEqual(["needs-metadata"]); + } finally { + issues.mockRestore(); + await rm(dir, { recursive: true, force: true }); + } +}); diff --git a/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml b/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml index 9fa904e270b..82bcc78bb46 100644 --- a/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml +++ b/providers/friendli/models/MiniMaxAI/MiniMax-M2.5.toml @@ -1,13 +1,14 @@ +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "minimax/MiniMax-M2.5" -name = "MiniMax-M2.5" -release_date = "2026-02-12" -last_updated = "2026-02-12" structured_output = true -reasoning_options = [] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.3 output = 1.2 @@ -15,4 +16,3 @@ cache_read = 0.06 [limit] context = 196_608 -output = 196_608 diff --git a/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml b/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml index 4714efddca6..4dfa7122d90 100644 --- a/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml +++ b/providers/friendli/models/deepseek-ai/DeepSeek-V3.2.toml @@ -1,19 +1,18 @@ -name = "DeepSeek-V3.2" -description = "DeepSeek chat model for instruction following, coding, and analysis" -family = "deepseek" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -tool_call = true -structured_output = true -temperature = true -release_date = "2025-12-01" -last_updated = "2025-12-01" -open_weights = true +# Toggle: chat_template_kwargs.enable_thinking = true | false +# https://friendli.ai/docs/guides/reasoning +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 +base_model = "deepseek/deepseek-v3.2" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 0.5 output = 1.5 @@ -21,8 +20,3 @@ cache_read = 0.25 [limit] context = 163_840 -output = 163_840 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/friendli/models/google/gemma-4-31B-it.toml b/providers/friendli/models/google/gemma-4-31B-it.toml index cd945119017..e6094afcc7e 100644 --- a/providers/friendli/models/google/gemma-4-31B-it.toml +++ b/providers/friendli/models/google/gemma-4-31B-it.toml @@ -1,5 +1,17 @@ +# Toggle: chat_template_kwargs.enable_thinking = true | false +# https://friendli.ai/docs/guides/reasoning +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "google/gemma-4-31b-it" -reasoning_options = [{ type = "toggle" }] + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" [cost] input = 0.14 diff --git a/providers/friendli/models/zai-org/GLM-5.1.toml b/providers/friendli/models/zai-org/GLM-5.1.toml index 1ee4131636b..d97a8d26dda 100644 --- a/providers/friendli/models/zai-org/GLM-5.1.toml +++ b/providers/friendli/models/zai-org/GLM-5.1.toml @@ -1,13 +1,18 @@ +# Toggle: chat_template_kwargs.enable_thinking = true | false +# https://friendli.ai/docs/guides/reasoning +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.1" -name = "GLM-5.1" -release_date = "2026-04-07" -last_updated = "2026-04-07" -structured_output = true -reasoning_options = [] [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + [cost] input = 1.4 output = 4.4 @@ -15,4 +20,3 @@ cache_read = 0.26 [limit] context = 202_752 -output = 202_752 diff --git a/providers/friendli/models/zai-org/GLM-5.2.toml b/providers/friendli/models/zai-org/GLM-5.2.toml index 81fbbb90c49..187d5de39d0 100644 --- a/providers/friendli/models/zai-org/GLM-5.2.toml +++ b/providers/friendli/models/zai-org/GLM-5.2.toml @@ -1,17 +1,28 @@ -name = "GLM-5.2" -# Friendli documents only $.chat_template_kwargs.enable_thinking = true | false -# for this model, not the configured effort values "high" and "max". -# https://friendli.ai/docs/guides/reasoning (accessed 2026-06-25) +# Toggle: chat_template_kwargs.enable_thinking = true | false +# https://friendli.ai/docs/guides/reasoning +# Effort: reasoning_effort = "high" | "max" +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 base_model = "zhipuai/glm-5.2" +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + [[reasoning_options]] type = "effort" values = ["high", "max"] -[interleaved] -field = "reasoning_content" +[[reasoning_options]] +type = "budget_tokens" [cost] input = 1.4 output = 4.4 -cache_read = 0.26 \ No newline at end of file +cache_read = 0.26 + +[limit] +context = 1_048_576 diff --git a/providers/friendli/models/zai-org/GLM-5.3-Flash.toml b/providers/friendli/models/zai-org/GLM-5.3-Flash.toml new file mode 100644 index 00000000000..7d25fca9c7b --- /dev/null +++ b/providers/friendli/models/zai-org/GLM-5.3-Flash.toml @@ -0,0 +1,26 @@ +# Effort: reasoning_effort = "low" | "high" | "max" +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) +# https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 +base_model = "zhipuai/glm-5.3-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.15 +output = 0.5 +cache_read = 0.03 + +[limit] +context = 1_048_576 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/friendli/models/zai-org/GLM-5.3.toml b/providers/friendli/models/zai-org/GLM-5.3.toml index bf8c8ce9c16..baad00a0786 100644 --- a/providers/friendli/models/zai-org/GLM-5.3.toml +++ b/providers/friendli/models/zai-org/GLM-5.3.toml @@ -1,42 +1,23 @@ -# Friendli's chat-completions endpoint documents reasoning_effort (generic -# enum) and reasoning_budget (integer token cap) as real request fields; GLM-5.3 -# on Z.AI's own API always reasons and only exposes low|high|max, so we mirror -# those graded levels here instead of Friendli's full generic effort enum. -# -# budget_tokens is a REAL, independently-enforced reasoning-token budget, not -# derived from max_tokens/context capacity: two live POST /chat/completions -# requests against this exact model (reasoning_budget=30 and =50, generous -# max_tokens=3000/4000) cut reasoning_content mid-sentence at the requested -# cap while completion_tokens continued to 1553/177 respectively — proof the -# reasoning phase and the completion phase are capped independently. -# min=-1/max=1_048_576 are Friendli's own reported values, read verbatim from -# GET /serverless/v1/models -> reasoning_options -> {type: budget_tokens} for -# zai-org/GLM-5.3 (not computed by us from max_completion_tokens; that field -# happens to equal this model's max_completion_tokens/context_length, but we -# pass through the catalog's own budget_tokens object, we do not derive it). +# Effort: reasoning_effort = "low" | "high" | "max" # https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-effort-one-of-0 +# Budget: reasoning_budget = positive integer reasoning-token cap (-1 = unlimited) # https://friendli.ai/docs/openapi/model-apis/chat-completions#body-reasoning-budget-one-of-0 -# (accessed 2026-08-29) base_model = "zhipuai/glm-5.3" +[interleaved] +field = "reasoning_content" + [[reasoning_options]] type = "effort" values = ["low", "high", "max"] [[reasoning_options]] type = "budget_tokens" -min = -1 -max = 1_048_576 - -[interleaved] -field = "reasoning_content" [cost] -input = 1.4 -output = 4.4 -cache_read = 0.26 +input = 1.26 +output = 3.96 +cache_read = 0.234 [limit] context = 1_048_576 -output = 1_048_576 - diff --git a/sync.md b/sync.md index 06ceb1b2a42..70a06dab55b 100644 --- a/sync.md +++ b/sync.md @@ -46,6 +46,7 @@ Sync runs also write `.sync/model-sync-report.md` for the automation workflow PR - Removes existing files that are no longer present in the desired synced set. - Writes `.sync/model-sync-report.md` for GitHub Actions. - When `skipCreates` is set and issue opens are enabled, opens one deduped GitHub issue per remote model missing from the local catalog (via `gh`). +- When a provider selectively skips only some models, `missingModelID` can preserve existing metadata and mark those skips for the same deduped issue flow without disabling safe automatic creates. Because the runner removes files missing from the desired set, a provider module should only skip source models when deleting existing local files for those skipped IDs is intentional. @@ -59,6 +60,8 @@ Providers that cannot safely auto-create TOMLs set `skipCreates: true`. In GitHu 4. Dispatches the Issue Fixer explicitly so issues created with `GITHUB_TOKEN` can still produce PRs 5. If listing fails, creates nothing (fail closed) +Providers that can auto-create most models may instead return an ID from `missingModelID` only for `translateModel` skips that need manual metadata. The runner preserves an existing local entry for that ID while the issue is handled. Intentional skips return `undefined` and do not open issues. + Requires `GH_TOKEN` on the sync workflow step. Local runs are notice-only unless `--open-issues`. Use `--no-issues` / `--dry-run` to skip creates. Each newly opened issue explicitly dispatches the issue-fixer workflow so an agent can research the missing metadata and open a model PR. The first Actions run may open a batch of issues per provider, including remote IDs the catalog intentionally omits (e.g. OpenAI whisper/tts/moderation surfaces, dated snapshots). This one-time volume is accepted by design: close unwanted issues once and the closed-title dedupe suppresses them permanently. If the dedupe list window (1000 labeled issues per provider) ever fills, the sync fails closed and creates nothing rather than risk duplicates.