diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index 031341b56cd..05889f5314f 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -30,6 +30,7 @@ import { kilo } from "./providers/kilo.js"; import { llmgateway, llmgatewayProviders } from "./providers/llmgateway.js"; import { mergeGateway } from "./providers/merge-gateway.js"; import { meta } from "./providers/meta.js"; +import { nebul } from "./providers/nebul.js"; import { nanoGpt } from "./providers/nano-gpt.js"; import { ollamaCloud } from "./providers/ollama-cloud.js"; import { openai } from "./providers/openai.js"; @@ -165,6 +166,7 @@ export const providers: { "llmgateway-providers": SyncProvider; "merge-gateway": SyncProvider; meta: SyncProvider; + nebul: SyncProvider; "nano-gpt": SyncProvider; ofox: SyncProvider; "ollama-cloud": SyncProvider; @@ -204,6 +206,7 @@ export const providers: { "llmgateway-providers": llmgatewayProviders, "merge-gateway": mergeGateway, meta, + nebul, "nano-gpt": nanoGpt, ofox, "ollama-cloud": ollamaCloud, @@ -237,7 +240,7 @@ export const groups = { "vercel", ], cloudflare: ["cloudflare-ai-gateway", "cloudflare-workers-ai"], - direct: ["aiand", "ambient", "anthropic", "baseten", "chutes", "cortecs", "deepinfra", "digitalocean", "fireworks-ai", "friendli", "github-copilot", "google", "hyper", "meta", "ollama-cloud", "openai", "ovhcloud", "pioneer", "tinfoil", "venice", "wandb", "xai"], + direct: ["aiand", "ambient", "anthropic", "baseten", "chutes", "cortecs", "deepinfra", "digitalocean", "fireworks-ai", "friendli", "github-copilot", "google", "hyper", "meta", "nebul", "ollama-cloud", "openai", "ovhcloud", "pioneer", "tinfoil", "venice", "wandb", "xai"], } as const; type ProviderID = keyof typeof providers; diff --git a/packages/core/src/sync/providers/nebul.ts b/packages/core/src/sync/providers/nebul.ts new file mode 100644 index 00000000000..2923ed2ed39 --- /dev/null +++ b/packages/core/src/sync/providers/nebul.ts @@ -0,0 +1,264 @@ +import { existsSync, readdirSync } from "node:fs"; +import path from "node:path"; +import { z } from "zod"; + +import type { ExistingModel, SyncProvider, SyncedModel } from "../index.js"; +import { MissingReasoningOptionsError } from "../missing-reasoning-options.js"; +import { factorBaseModel, modelMetadata } from "./openrouter.js"; + +const API_ENDPOINT = "https://api.inference.nebul.io/v1/model/info"; +const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models"); + +// Maps the org prefix of a served ID to the lab namespace under models/. +// The Hugging Face org name and the lab name in models/ often differ. +// Keys are lowercase, and the lookup lowercases the org the same way. Hugging Face +// org paths are case-insensitive in URLs, so "qwen/Qwen3.8-27B-FP8" is a valid ID shape. +const ORG_TO_MODEL_PROVIDER: Record = { + "deepseek-ai": "deepseek", + google: "google", + "meta-models": "meta", + mistralai: "mistral", + moonshotai: "moonshotai", + nvidia: "nvidia", + openai: "openai", + qwen: "alibaba", + "zai-org": "zhipuai", +}; + +// Served IDs whose lab metadata in models/ lives under a different model name. +const BASE_MODEL_ALIASES: Record = { + "mistralai/Mistral-Large-3-675B-Instruct-2512": "mistral/mistral-large-2512", +}; + +// The sync scope is general chat models. The catalog also lists specialized +// OCR (document text recognition) models by name and flags private, internal, +// and safety entries with display_tags. The filter keeps all of those out, +// because they are not chat models and models.dev has no matching lab metadata for them. +const OUT_OF_SCOPE_PATTERNS = [/OCR/i]; +const OUT_OF_SCOPE_TAGS = new Set(["Guard Model", "Content Safety", "Private", "Internal"]); + +// The sync fails closed (it throws on bad data instead of syncing it) on a +// partial catalog. The in-scope chat catalog has ~14 models as of 2026-09-24. +// A truncated response (a per-lab serving outage or a half-written deploy) can +// pass the non-empty checks in parseModels. The code must not treat it as the +// real catalog. As a second guard, the provider runs with deleteMissing: false, +// so even a bad catalog cannot delete curated local files. A catalog with less +// than half the known-good size is structurally incomplete. Raise this number +// deliberately as the catalog grows. +const MIN_CHAT_MODELS = 6; + +const ModelInfo = z.object({ + description: z.string().nullable().optional(), + huggingface_id: z.string().nullable().optional(), + input_cost_per_1m_tokens: z.number().nullable().optional(), + output_cost_per_1m_tokens: z.number().nullable().optional(), + cache_read_input_cost_per_1m_tokens: z.number().nullable().optional(), + display_tags: z.array(z.string()).nullable().optional(), + max_input_tokens: z.number().nullable().optional(), + mode: z.string().nullable(), + model_type: z.string().nullable(), + // The catalog only advertises this list, and the sync never copies it into an + // entry, because probes proved the list unreliable. The schema accepts any + // string. A strict enum throws on a future unknown value. That failure stops + // the hourly run and blocks cost/context refreshes for curated models. + reasoning_efforts: z.array(z.string()).nullable().optional(), + superseded_by_model_name: z.string().nullable().optional(), +}).passthrough(); + +export const NebulEntry = z.object({ + model_info: ModelInfo, + model_name: z.string().min(1), +}).passthrough(); + +export const NebulResponse = z.object({ + data: z.array(NebulEntry), +}).passthrough(); + +export type NebulEntry = z.infer; + +export const nebul = { + id: "nebul", + name: "Nebul", + modelsDir: "providers/nebul/models", + // Nebul is a curated provider, so only the hand-authored flagship models ship. + // The catalog is authoritative for their live cost and context, but it must + // never grow or shrink the local set. skipCreates keeps any other in-scope chat + // model out, and deleteMissing: false keeps a curated model that drops out of + // the catalog (zai-org/GLM-5.3 is intermittently absent) instead of deleting it. + skipCreates: true, + deleteMissing: false, + // Skipped reasoners (the fail-closed path below) and chat models outside the + // curation are expected. Missing-model issues for them ask for models that + // this provider deliberately does not curate. + trackMissingModels: false, + async fetchModels() { + const response = await fetch(API_ENDPOINT); + if (!response.ok) { + throw new Error(`Nebul models request failed: ${response.status} ${response.statusText}`); + } + return response.json(); + }, + parseModels(raw) { + const data = NebulResponse.parse(raw).data; + // An empty catalog is an upstream fault. The delete-missing pass deletes + // every local model file when the sync accepts it, so the code throws instead. + if (data.length === 0) { + throw new Error("Nebul returned an empty model catalog"); + } + // The same failure applies when the response shape drifts and no entry + // matches the chat-model filter anymore, for example after renamed + // model_type or mode values. + if (!data.some(isCatalogChatModel)) { + throw new Error("Nebul returned no usable chat models"); + } + const chatCount = data.filter(isCatalogChatModel).length; + if (chatCount < MIN_CHAT_MODELS) { + throw new Error( + `Nebul returned only ${chatCount} usable chat models (expected at least ${MIN_CHAT_MODELS}); treating the catalog as a partial fault and skipping this run`, + ); + } + return data; + }, + // The unauthenticated /v1/model/info endpoint is authoritative for the live + // cost and context of the curated entries. The provider runs with skipCreates + // and deleteMissing: false, so translateModel only refreshes existing files. + // skippedNotice reports any other in-scope chat model, and missingNotice + // reports a curated model that the catalog no longer lists. Whole-catalog + // faults still fail closed in parseModels. + translateModel(entry, context) { + if (!isCatalogChatModel(entry)) return undefined; + const id = entry.model_name; + const info = entry.model_info; + const existing = context.existing(id); + // Existing entries must survive incomplete source data. A transient null + // price or an unresolved alias otherwise deletes the hand-authored TOML on + // the next run. Existing entries keep their authored base_model and + // cost/limit, and only new models need a source entry with complete pricing + // and a resolvable base. + const baseModel = existing?.base_model ?? resolveBaseModel(id, info.huggingface_id ?? undefined); + const cost = info.input_cost_per_1m_tokens != null && info.output_cost_per_1m_tokens != null + ? { + input: info.input_cost_per_1m_tokens, + output: info.output_cost_per_1m_tokens, + cache_read: info.cache_read_input_cost_per_1m_tokens ?? undefined, + } + : existing?.cost; + const limit = info.max_input_tokens != null ? { context: info.max_input_tokens } : existing?.limit; + if (existing === undefined && (baseModel === undefined || cost === undefined || limit === undefined)) return undefined; + // A hand-authored reasoning = false marks a served ID whose lab model + // reasons but that this host serves with thinking disabled. The catalog + // reports supports_reasoning = false and no reasoning_efforts for it. Keep + // the override and suppress the reasoning controls and traces. The entry + // then needs no reasoning_options, and no interleaved side channel applies + // when the host returns no traces. + const reasoningDisabled = existing?.reasoning === false; + // Fail closed unless the caller control is hand-authored from live probe + // evidence. Probes proved the advertised reasoning_efforts wrong in one + // case: on 2026-09-23 the catalog advertised low|medium|high|max for one + // model, and its engine rejects every value but high. The runner skips the + // ID so that the options can be hand-authored from live probes. + const isReasoner = !reasoningDisabled && (baseModel !== undefined + ? modelMetadata(baseModel).reasoning === true + : existing?.reasoning === true); + if (isReasoner && existing?.reasoning_options === undefined) { + throw new MissingReasoningOptionsError( + id, + `${id} is a reasoning model, but the catalog entry has no probe-verified reasoning_options; hand-author them instead of trusting the advertised reasoning_efforts`, + ); + } + const values = reasoningDisabled + ? { reasoning: false, interleaved: undefined, reasoning_options: undefined, cost, limit } + : { interleaved: existing?.interleaved, reasoning_options: existing?.reasoning_options, cost, limit }; + if (baseModel !== undefined) { + return { + id, + model: factorBaseModel(baseModel, values, limit) as SyncedModel, + }; + } + // The existing standalone definition has a served alias that no longer + // resolves. Keep the authored fields, and refresh only what /model/info + // still provides. + return { id, model: { ...existing, ...values } as SyncedModel }; + }, + // Report only in-scope chat models whose base_model does not resolve. Filtered + // entries (embeddings, rerankers, out-of-scope specialized models, superseded + // IDs) skip silently. + sourceID(entry: NebulEntry) { + return isCatalogChatModel(entry) ? entry.model_name : undefined; + }, + skippedNotice(ids) { + if (ids.length === 0) return []; + return [ + `Nebul serves these in-scope chat models, but only the curated catalog is shipped (skipCreates); not added:`, + ids.map((id) => `\`${id}\``).join(", "), + ]; + }, + missingNotice(paths) { + if (paths.length === 0) return []; + return [ + `Nebul models absent from the source catalog were retained, not deleted:`, + paths.map((p) => `\`${p.replace(/\.toml$/, "")}\``).join(", "), + ]; + }, +} satisfies SyncProvider; + +function isCatalogChatModel(entry: NebulEntry): boolean { + const info = entry.model_info; + return info.model_type === "llm" && info.mode === "chat" + && info.superseded_by_model_name == null && !OUT_OF_SCOPE_PATTERNS.some((pattern) => pattern.test(entry.model_name)) + && !(info.display_tags ?? []).some((tag) => OUT_OF_SCOPE_TAGS.has(tag)); +} + +// Nebul documents exactly one reasoning control: reasoning_effort. The sync +// copies only hand-authored options into an entry, because probes proved the +// advertised reasoning_efforts unreliable (on 2026-09-23 the catalog advertised +// low|medium|high|max for one model, and its engine rejects every value but +// high). translateModel rejects a reasoner with no authored options, and a +// non-reasoner carries no options. The API supports no lab-style toggles or +// budget controls unless a probe of this host shows them. + +function resolveBaseModel(servedID: string, huggingfaceID: string | undefined): string | undefined { + return baseModelCandidates(servedID, huggingfaceID).find(canonicalExists); +} + +// existsSync is case-insensitive on Windows and macOS. readdirSync reads the +// real on-disk filename case, so the resolved base_model matches the models/ +// filename exactly, and CI on Linux passes. +function canonicalExists(candidate: string): boolean { + const file = path.join(MODELS_DIR, `${candidate}.toml`); + if (!existsSync(file)) return false; + try { + return readdirSync(path.dirname(file)).includes(path.basename(file)); + } catch { + return false; + } +} + +function baseModelCandidates(servedID: string, huggingfaceID: string | undefined): string[] { + const alias = BASE_MODEL_ALIASES[servedID]; + const servedCandidate = mapOrgToCandidate(servedID); + const hfCandidate = huggingfaceID === undefined ? undefined : mapOrgToCandidate(huggingfaceID); + return [ + ...new Set([alias, servedCandidate, hfCandidate, ...quantizationStripped(hfCandidate), ...quantizationStripped(servedCandidate)]).values(), + ].filter((candidate): candidate is string => candidate !== undefined); +} + +function mapOrgToCandidate(id: string): string | undefined { + const [org, ...modelParts] = id.split("/"); + if (org === undefined || modelParts.length === 0) return undefined; + const provider = ORG_TO_MODEL_PROVIDER[org.toLowerCase()]; + if (provider === undefined) return undefined; + return `${provider}/${modelParts.join("/").toLowerCase()}`; +} + +// Hosts serve quantized checkpoints (the same weights in a smaller number +// format, for example -FP8, -BF16), and the lab publishes the metadata under +// the base-precision name. The resolver also tries the names without the +// quantization suffix. NVIDIA prefixes checkpoint names with "NVIDIA-", and the +// metadata names drop that prefix. +function quantizationStripped(candidate: string | undefined): string[] { + if (candidate === undefined) return []; + const withoutQuant = candidate.replace(/-(fp8|bf16|fp4|int8)$/i, ""); + const withoutPrefix = withoutQuant.replace(/nvidia-/, ""); + return withoutQuant === candidate ? [] : [...new Set([withoutQuant, withoutPrefix])].filter((value) => value !== candidate); +} diff --git a/packages/core/test/nebul.test.ts b/packages/core/test/nebul.test.ts new file mode 100644 index 00000000000..354aa3a4de1 --- /dev/null +++ b/packages/core/test/nebul.test.ts @@ -0,0 +1,299 @@ +import { expect, test } from "bun:test"; +import { readdirSync } from "node:fs"; +import path from "node:path"; + +import type { ExistingModel } from "../src/sync/index.js"; +import { MissingReasoningOptionsError } from "../src/sync/missing-reasoning-options.js"; +import { + NebulEntry, + NebulResponse, + nebul, +} from "../src/sync/providers/nebul.js"; + +function nebulEntry(model_name?: string, model_info: Record = {}): NebulEntry { + return NebulEntry.parse({ + model_name: model_name ?? "zai-org/GLM-5.3", + model_info: { + description: "test", + huggingface_id: model_name ?? "zai-org/GLM-5.3", + input_cost_per_1m_tokens: 1.47, + output_cost_per_1m_tokens: 4.62, + cache_read_input_cost_per_1m_tokens: 0.35, + max_input_tokens: 1_000_000, + mode: "chat", + model_type: "llm", + reasoning_efforts: ["low", "high", "max"], + ...model_info, + }, + }); +} + +function existingWith(reasoning_options: ExistingModel["reasoning_options"]): ExistingModel { + return { reasoning_options } as ExistingModel; +} + +const context = (existing: ExistingModel | undefined) => ({ existing: () => existing }); + +test("syncs Nebul's factored overrides against resolved lab metadata", () => { + // The lab model here does not reason. A new entry keeps the factored pricing + // and context and carries no options. + const translated = nebul.translateModel( + nebulEntry("mistralai/Mistral-Large-3-675B-Instruct-2512", { max_input_tokens: 1_048_576 }), + context(undefined), + ); + expect(translated).toMatchObject({ + id: "mistralai/Mistral-Large-3-675B-Instruct-2512", + model: { + base_model: "mistral/mistral-large-2512", + cost: { input: 1.47, output: 4.62, cache_read: 0.35 }, + limit: { context: 1_048_576 }, + }, + }); + expect(translated?.model.reasoning_options).toBeUndefined(); +}); + +test("fails closed for a new reasoner that only advertises efforts", () => { + // Probes proved the advertised list unreliable: one catalog model advertised + // low|medium|high|max, and its engine rejected every value but high. A new + // reasoner must never inherit the list. The options need live probe evidence + // first. + const entry = nebulEntry("zai-org/GLM-5.3"); + expect(() => nebul.translateModel(entry, context(undefined))).toThrow(MissingReasoningOptionsError); +}); + +test("preserves authored reasoning controls when the host exposes no efforts", () => { + const authored = [{ type: "toggle" as const }]; + const translated = nebul.translateModel(nebulEntry("zai-org/GLM-5.3", { reasoning_efforts: [] }), context(existingWith(authored))); + expect(translated?.model.reasoning_options).toEqual(authored); +}); + +test("keeps authored probe-verified controls over the advertised effort list", () => { + // Live probes on 2026-09-23 showed that the advertised list can be wrong: one + // catalog model advertised low|medium|high|max, and its served engine rejected + // every value but high. + const authored = [{ type: "effort" as const, values: ["none", "high"] }]; + const translated = nebul.translateModel(nebulEntry("zai-org/GLM-5.3"), context(existingWith(authored))); + expect(translated?.model.reasoning_options).toEqual(authored); +}); + +test("carries authored interleaved through sync", () => { + const inline = nebul.translateModel( + nebulEntry("zai-org/GLM-5.3"), + context({ interleaved: true, reasoning_options: [{ type: "toggle" }] } as ExistingModel), + ); + expect(inline?.model.interleaved).toBe(true); + + const named = nebul.translateModel( + nebulEntry("someorg/Some-Model"), + context({ interleaved: { field: "reasoning_content" }, reasoning_options: [{ type: "toggle" }] } as ExistingModel), + ); + expect(named?.model.interleaved).toEqual({ field: "reasoning_content" }); +}); + +test("keeps an authored reasoning = false override for a lab reasoner the host serves without thinking", () => { + const existing = { + base_model: "alibaba/qwen3.5-397b-a17b", + reasoning: false, + reasoning_options: [{ type: "effort" as const, values: ["low"] }], + interleaved: { field: "reasoning_content" as const }, + } as ExistingModel; + const translated = nebul.translateModel( + nebulEntry("Qwen/Qwen3.5-397B-A17B", { reasoning_efforts: undefined }), + context(existing), + ); + expect(translated?.model.reasoning).toBe(false); + expect(translated?.model.reasoning_options).toBeUndefined(); + expect(translated?.model.interleaved).toBeUndefined(); +}); + +test("fails closed when a reasoner advertises no efforts and none are authored", () => { + const entry = nebulEntry("zai-org/GLM-5.3", { reasoning_efforts: [] }); + expect(() => nebul.translateModel(entry, context(undefined))).toThrow(MissingReasoningOptionsError); + expect(() => + nebul.translateModel(entry, context({ base_model: "zhipuai/glm-5.3" } as ExistingModel)), + ).toThrow(MissingReasoningOptionsError); +}); + +test("keeps existing entries when the source pricing or context is temporarily null", () => { + const existing = { + base_model: "zhipuai/glm-5.3", + cost: { input: 1.47, output: 4.62 }, + limit: { context: 1_048_576 }, + reasoning_options: [{ type: "effort" as const, values: ["low", "high"] }], + } as ExistingModel; + const translated = nebul.translateModel( + nebulEntry("zai-org/GLM-5.3", { input_cost_per_1m_tokens: null, output_cost_per_1m_tokens: null, max_input_tokens: null }), + context(existing), + ); + expect(translated).toMatchObject({ + id: "zai-org/GLM-5.3", + model: { + base_model: "zhipuai/glm-5.3", + cost: { input: 1.47, output: 4.62 }, + limit: { context: 1_048_576 }, + reasoning_options: [{ type: "effort", values: ["low", "high"] }], + }, + }); +}); + +test("keeps existing entries when the served alias no longer resolves to lab metadata", () => { + const existing = { + base_model: "zhipuai/glm-5.3", + cost: { input: 1.47, output: 4.62 }, + limit: { context: 1_048_576 }, + reasoning_options: [{ type: "effort" as const, values: ["low", "high"] }], + } as ExistingModel; + const translated = nebul.translateModel(nebulEntry("someorg/Unknown-Model", { huggingface_id: null }), context(existing)); + expect(translated?.model.base_model).toBe("zhipuai/glm-5.3"); +}); + +test("resolves base models across org renames and quantization suffixes", () => { + const cases: [string, string | null, string][] = [ + ["zai-org/GLM-5.3", "zai-org/GLM-5.3", "zhipuai/glm-5.3"], + ["Qwen/Qwen3.5-397B-A17B", "Qwen/Qwen3.5-397B-A17B", "alibaba/qwen3.5-397b-a17b"], + // Hugging Face org paths are case-insensitive, so a lowercase org must + // resolve identically. + ["qwen/qwen3.5-397b-a17b", "qwen/qwen3.5-397b-a17b", "alibaba/qwen3.5-397b-a17b"], + ["nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16", "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16", "nvidia/nemotron-3-super-120b-a12b"], + ["mistralai/Mistral-Large-3-675B-Instruct-2512", "mistralai/Mistral-Large-3-675B-Instruct-2512", "mistral/mistral-large-2512"], + ]; + // The authored options in the context keep reasoner labs out of the + // fail-closed path. This case asserts base_model resolution only. + for (const [model_name, huggingface_id, expected] of cases) { + const entry = nebulEntry(model_name, { huggingface_id }); + expect( + nebul.translateModel(entry, context(existingWith([{ type: "effort" as const, values: ["high"] }])))?.model.base_model, + ).toBe(expected); + } +}); + +test("skips out-of-scope specialized models and superseded entries silently", () => { + for (const model_name of ["someorg/Doc-OCR-Tool", "someorg/ScanOCR-Large"]) { + const entry = nebulEntry(model_name, {}); + expect(nebul.translateModel(entry, context(undefined))).toBeUndefined(); + expect(nebul.sourceID(entry)).toBeUndefined(); + } + for (const display_tags of [["Guard Model"], ["Content Safety"], ["Private"], ["Internal"]]) { + const entry = nebulEntry("someorg/Some-Model", { display_tags }); + expect(nebul.translateModel(entry, context(undefined))).toBeUndefined(); + expect(nebul.sourceID(entry)).toBeUndefined(); + } + for (const model_name of ["someorg/Retired-Model-A", "someorg/Retired-Model-B"]) { + const entry = nebulEntry(model_name, { superseded_by_model_name: "zai-org/GLM-5.3" }); + expect(nebul.translateModel(entry, context(undefined))).toBeUndefined(); + expect(nebul.sourceID(entry)).toBeUndefined(); + } +}); + +test("skips embeddings and rerankers while reporting unresolvable chat models", () => { + const embedding = nebulEntry("someorg/Some-Embedding", { model_type: "embedding" }); + expect(nebul.translateModel(embedding, context(undefined))).toBeUndefined(); + expect(nebul.sourceID(embedding)).toBeUndefined(); + + const chat = nebulEntry("mistralai/Mistral-Large-3-675B-Instruct-2512", { huggingface_id: null }); + expect(nebul.translateModel(chat, context(undefined))).toBeDefined(); + expect(nebul.sourceID(chat)).toBe("mistralai/Mistral-Large-3-675B-Instruct-2512"); +}); + +test("skips chat models whose pricing or context is absent instead of crashing", () => { + const unpriced = nebulEntry("zai-org/GLM-5.3", { input_cost_per_1m_tokens: null, output_cost_per_1m_tokens: null, max_input_tokens: null }); + expect(nebul.translateModel(unpriced, context(undefined))).toBeUndefined(); + expect(nebul.sourceID(unpriced)).toBe("zai-org/GLM-5.3"); +}); + +test("parses nullable serving artifacts and unknown-host metadata from /model/info", () => { + const parsed = NebulResponse.parse({ + data: [ + { model_name: "Some/Embedding", model_info: { mode: null, model_type: "embedding" } }, + { model_name: "Some/Chat", model_info: { mode: "chat", model_type: "llm", unknown_host_field: true } }, + ], + }); + expect(parsed.data).toHaveLength(2); +}); + +test("fails closed on an empty catalog so sync cannot delete every local file", () => { + expect(() => nebul.parseModels({ data: [] })).toThrow("Nebul returned an empty model catalog"); +}); + +test("fails closed when no entry matches the chat-model filter", () => { + expect(() => + nebul.parseModels({ + data: [ + { model_name: "Some/Embedding", model_info: { mode: null, model_type: "embedding" } }, + { model_name: "Some/Reranker", model_info: { mode: null, model_type: "rerank" } }, + ], + }), + ).toThrow("Nebul returned no usable chat models"); +}); + +test("parseModels keeps chat entries alongside filtered serving artifacts", () => { + const chat = { model_info: { mode: "chat", model_type: "llm" } }; + const parsed = nebul.parseModels({ + data: [ + { model_name: "Some/Embedding", model_info: { mode: null, model_type: "embedding" } }, + { model_name: "Some/Chat-A", ...chat }, + { model_name: "Some/Chat-B", ...chat }, + { model_name: "Some/Chat-C", ...chat }, + { model_name: "Some/Chat-D", ...chat }, + { model_name: "Some/Chat-E", ...chat }, + { model_name: "Some/Chat-F", ...chat }, + ], + }); + expect(parsed).toHaveLength(7); +}); + +test("fails closed on a partial catalog so sync cannot prune healthy local files", () => { + expect(() => + nebul.parseModels({ + data: [{ model_name: "Some/Chat", model_info: { mode: "chat", model_type: "llm" } }], + }), + ).toThrow("treating the catalog as a partial fault"); +}); + +test("accepts and ignores unknown advertised reasoning effort values", () => { + // The sync never copies the advertised list, so an unrecognized value must + // not abort parsing. A strict enum fails the hourly run and blocks + // cost/context refreshes for the curated models. + const parsed = NebulResponse.parse({ + data: [{ model_name: "Some/Chat", model_info: { mode: "chat", model_type: "llm", reasoning_efforts: ["ultra"] } }], + }); + expect(parsed.data[0]?.model_info.reasoning_efforts).toEqual(["ultra"]); + + const authored = [{ type: "effort" as const, values: ["high"] }]; + const translated = nebul.translateModel( + nebulEntry("zai-org/GLM-5.3", { reasoning_efforts: ["ultra"] }), + context(existingWith(authored)), + ); + expect(translated?.model.reasoning_options).toEqual(authored); +}); + +// Nebul is a curated provider. Exactly the four requested flagship models ship, +// and the catalog sync must never add to or remove from that set. +const CURATED_MODEL_IDS = [ + "Qwen/Qwen3.5-397B-A17B", + "mistralai/Mistral-Large-3-675B-Instruct-2512", + "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16", + "zai-org/GLM-5.3", +]; + +function modelFiles(dir: string): string[] { + return readdirSync(dir, { withFileTypes: true }).flatMap((entry) => { + const full = path.join(dir, entry.name); + if (entry.isDirectory()) return modelFiles(full); + return entry.name.endsWith(".toml") ? [full] : []; + }); +} + +test("ships exactly the four curated models", () => { + const modelsDir = path.join(import.meta.dirname, "..", "..", "..", "providers", "nebul", "models"); + const ids = modelFiles(modelsDir) + .map((file) => path.relative(modelsDir, file).replace(/\.toml$/, "").split(path.sep).join("/")) + .sort(); + expect(ids).toEqual([...CURATED_MODEL_IDS].sort()); +}); + +test("sync cannot grow or shrink the curated catalog", () => { + expect(nebul.skipCreates).toBe(true); + expect(nebul.deleteMissing).toBe(false); + expect(nebul.trackMissingModels).toBe(false); +}); diff --git a/providers/nebul/logo.svg b/providers/nebul/logo.svg new file mode 100644 index 00000000000..33daff15792 --- /dev/null +++ b/providers/nebul/logo.svg @@ -0,0 +1,5 @@ + + + + + diff --git a/providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml b/providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml new file mode 100644 index 00000000000..949f76302ce --- /dev/null +++ b/providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = none|low|medium|high. Thinking is ON by default, and +# none turns it off. Live probes on 2026-09-23 returned 200 OK for every listed +# level. The set matches the same-surface peer baseline (ovhcloud, scaleway, crof, +# digitalocean), because minimal, xhigh, and max showed no distinct effect. +# The catalog reports supports_reasoning = false for this ID, and that is a false +# negative. Traces arrive in reasoning_content. The costs match the Nebul API on +# 2026-09-23. +base_model = "alibaba/qwen3.5-397b-a17b" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.63 +output = 3.78 +cache_read = 0.15 diff --git a/providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml b/providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml new file mode 100644 index 00000000000..d34868d3cb8 --- /dev/null +++ b/providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml @@ -0,0 +1,7 @@ +# The costs match https://api.inference.nebul.io/v1/model/info on 2026-09-23. +base_model = "mistral/mistral-large-2512" + +[cost] +input = 0.6 +output = 1.73 +cache_read = 0.15 diff --git a/providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml b/providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml new file mode 100644 index 00000000000..f8295a2f74f --- /dev/null +++ b/providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml @@ -0,0 +1,22 @@ +# Effort: reasoning_effort = none|low|medium|high|max. Thinking is ON by default, +# and none turns it off. Live probes on 2026-09-23 returned 200 OK for every +# listed level. The set matches the same-surface peer baseline, because minimal +# and xhigh showed no distinct effect. The catalog reports supports_reasoning = +# false for this ID, and that is a false negative. Traces arrive in +# reasoning_content. The context and costs match the Nebul API on 2026-09-23. +base_model = "nvidia/nemotron-3-super-120b-a12b" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[cost] +input = 0.32 +output = 0.69 +cache_read = 0.08 + +[limit] +context = 1_000_000 diff --git a/providers/nebul/models/zai-org/GLM-5.3.toml b/providers/nebul/models/zai-org/GLM-5.3.toml new file mode 100644 index 00000000000..423092675cf --- /dev/null +++ b/providers/nebul/models/zai-org/GLM-5.3.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = low|high|max. The set matches the catalog and the zai +# and OpenRouter entries for the same model. Traces arrive in reasoning_content. +# The costs and context match https://api.inference.nebul.io/v1/model/info on +# 2026-09-23. +base_model = "zhipuai/glm-5.3" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 1.47 +output = 4.62 +cache_read = 0.35 + +[limit] +context = 1_048_576 diff --git a/providers/nebul/provider.toml b/providers/nebul/provider.toml new file mode 100644 index 00000000000..dc44e31d87d --- /dev/null +++ b/providers/nebul/provider.toml @@ -0,0 +1,5 @@ +name = "Nebul" +npm = "@ai-sdk/openai-compatible" +api = "https://api.inference.nebul.io/v1" +env = ["NEBUL_API_KEY"] +doc = "https://docs.nebul.io"