From 93fb473dd4942a30aaf175399858dad5cca924a9 Mon Sep 17 00:00:00 2001 From: Wynand Huizinga Date: Thu, 24 Sep 2026 13:06:12 +0200 Subject: [PATCH 1/2] feat(providers): add Nebul provider with 4 curated chat models Add the Nebul provider (NEBUL-AI) with exactly four curated chat models, each as a base_model override: - zai-org/GLM-5.3 - nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16 - Qwen/Qwen3.5-397B-A17B - mistralai/Mistral-Large-3-675B-Instruct-2512 The sync module reads the unauthenticated /v1/model/info catalog as authoritative for the curated entries' cost/context, but is constrained so it can never grow or shrink the shipped set: skipCreates keeps other in-scope chat models out, deleteMissing: false retains a curated model that drops out of the catalog (surfaced via missingNotice), and trackMissingModels: false avoids missing-model issues for models this provider deliberately does not curate. Reasoning options are only ever hand-authored from live probes; the catalog's advertised reasoning_efforts are probe-proven unreliable, and a reasoner without authored options fail-closes. Whole-catalog faults are rejected before any file is written or deleted. --- packages/core/src/sync/index.ts | 5 +- packages/core/src/sync/providers/nebul.ts | 256 +++++++++++++++ packages/core/test/nebul.test.ts | 296 ++++++++++++++++++ providers/nebul/logo.svg | 5 + .../nebul/models/Qwen/Qwen3.5-397B-A17B.toml | 19 ++ .../Mistral-Large-3-675B-Instruct-2512.toml | 7 + ...VIDIA-Nemotron-3-Super-120B-A12B-BF16.toml | 21 ++ providers/nebul/models/zai-org/GLM-5.3.toml | 19 ++ providers/nebul/provider.toml | 5 + 9 files changed, 632 insertions(+), 1 deletion(-) create mode 100644 packages/core/src/sync/providers/nebul.ts create mode 100644 packages/core/test/nebul.test.ts create mode 100644 providers/nebul/logo.svg create mode 100644 providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml create mode 100644 providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml create mode 100644 providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml create mode 100644 providers/nebul/models/zai-org/GLM-5.3.toml create mode 100644 providers/nebul/provider.toml diff --git a/packages/core/src/sync/index.ts b/packages/core/src/sync/index.ts index 031341b56cd..05889f5314f 100644 --- a/packages/core/src/sync/index.ts +++ b/packages/core/src/sync/index.ts @@ -30,6 +30,7 @@ import { kilo } from "./providers/kilo.js"; import { llmgateway, llmgatewayProviders } from "./providers/llmgateway.js"; import { mergeGateway } from "./providers/merge-gateway.js"; import { meta } from "./providers/meta.js"; +import { nebul } from "./providers/nebul.js"; import { nanoGpt } from "./providers/nano-gpt.js"; import { ollamaCloud } from "./providers/ollama-cloud.js"; import { openai } from "./providers/openai.js"; @@ -165,6 +166,7 @@ export const providers: { "llmgateway-providers": SyncProvider; "merge-gateway": SyncProvider; meta: SyncProvider; + nebul: SyncProvider; "nano-gpt": SyncProvider; ofox: SyncProvider; "ollama-cloud": SyncProvider; @@ -204,6 +206,7 @@ export const providers: { "llmgateway-providers": llmgatewayProviders, "merge-gateway": mergeGateway, meta, + nebul, "nano-gpt": nanoGpt, ofox, "ollama-cloud": ollamaCloud, @@ -237,7 +240,7 @@ export const groups = { "vercel", ], cloudflare: ["cloudflare-ai-gateway", "cloudflare-workers-ai"], - direct: ["aiand", "ambient", "anthropic", "baseten", "chutes", "cortecs", "deepinfra", "digitalocean", "fireworks-ai", "friendli", "github-copilot", "google", "hyper", "meta", "ollama-cloud", "openai", "ovhcloud", "pioneer", "tinfoil", "venice", "wandb", "xai"], + direct: ["aiand", "ambient", "anthropic", "baseten", "chutes", "cortecs", "deepinfra", "digitalocean", "fireworks-ai", "friendli", "github-copilot", "google", "hyper", "meta", "nebul", "ollama-cloud", "openai", "ovhcloud", "pioneer", "tinfoil", "venice", "wandb", "xai"], } as const; type ProviderID = keyof typeof providers; diff --git a/packages/core/src/sync/providers/nebul.ts b/packages/core/src/sync/providers/nebul.ts new file mode 100644 index 00000000000..8a7579eedfb --- /dev/null +++ b/packages/core/src/sync/providers/nebul.ts @@ -0,0 +1,256 @@ +import { existsSync, readdirSync } from "node:fs"; +import path from "node:path"; +import { z } from "zod"; + +import type { ExistingModel, SyncProvider, SyncedModel } from "../index.js"; +import { MissingReasoningOptionsError } from "../missing-reasoning-options.js"; +import { factorBaseModel, modelMetadata } from "./openrouter.js"; + +const API_ENDPOINT = "https://api.inference.nebul.io/v1/model/info"; +const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models"); + +// Served org prefix -> models/ metadata namespace (HF org names differ from catalog labs). +// Keys are lowercase; lookups normalize the org the same way (Hugging Face orgs are +// case-insensitive in URLs, so e.g. "qwen/Qwen3.8-27B-FP8" is a valid ID shape). +const ORG_TO_MODEL_PROVIDER: Record = { + "deepseek-ai": "deepseek", + google: "google", + "meta-models": "meta", + mistralai: "mistral", + moonshotai: "moonshotai", + nvidia: "nvidia", + openai: "openai", + qwen: "alibaba", + "zai-org": "zhipuai", +}; + +// Served IDs whose canonical metadata lives under a differently-named lab entry. +const BASE_MODEL_ALIASES: Record = { + "mistralai/Mistral-Large-3-675B-Instruct-2512": "mistral/mistral-large-2512", +}; + +// Catalog scope is general chat models. The public catalog also lists specialized +// document-OCR models by name, and flags internal-only or safety-infrastructure +// entries via display_tags; keep all of those out (not coding/chat targets, and +// models.dev carries no matching lab metadata for them). +const OUT_OF_SCOPE_PATTERNS = [/OCR/i]; +const OUT_OF_SCOPE_TAGS = new Set(["Guard Model", "Content Safety", "Private", "Internal"]); + +// Fail-closed floor against partial catalog faults. The in-scope chat catalog is +// ~14 models as of 2026-09-24; a truncated response (per-lab serving outage, +// half-written deploy) that still passes the non-empty checks should not be +// treated as the real catalog. Defense in depth: the provider also runs with +// deleteMissing: false, so even a bad catalog cannot prune curated local files. +// Any catalog showing less than half the known-good size is treated as +// structurally incomplete. Raise this deliberately as the catalog grows. +const MIN_CHAT_MODELS = 6; + +const ModelInfo = z.object({ + description: z.string().nullable().optional(), + huggingface_id: z.string().nullable().optional(), + input_cost_per_1m_tokens: z.number().nullable().optional(), + output_cost_per_1m_tokens: z.number().nullable().optional(), + cache_read_input_cost_per_1m_tokens: z.number().nullable().optional(), + display_tags: z.array(z.string()).nullable().optional(), + max_input_tokens: z.number().nullable().optional(), + mode: z.string().nullable(), + model_type: z.string().nullable(), + // Advertised only; never synced (probes proved the list unreliable), so accept + // any string. A strict enum here would let a future unknown value throw and + // fail the whole hourly run, blocking cost/context refreshes for curated models. + reasoning_efforts: z.array(z.string()).nullable().optional(), + superseded_by_model_name: z.string().nullable().optional(), +}).passthrough(); + +export const NebulEntry = z.object({ + model_info: ModelInfo, + model_name: z.string().min(1), +}).passthrough(); + +export const NebulResponse = z.object({ + data: z.array(NebulEntry), +}).passthrough(); + +export type NebulEntry = z.infer; + +export const nebul = { + id: "nebul", + name: "Nebul", + modelsDir: "providers/nebul/models", + // Nebul is a curated provider: only the hand-authored flagship models ship. + // The catalog is authoritative for their live cost/context, but must never + // grow or shrink the local set: skipCreates keeps any other in-scope chat + // model out, and deleteMissing: false keeps a curated model that drops out of + // the catalog (zai-org/GLM-5.3 has been intermittently absent) instead of + // silently removing it. + skipCreates: true, + deleteMissing: false, + // Skipped reasoners (fail-closed below) and out-of-catalog chat models are + // expected; opening missing-model issues for them would request models this + // provider deliberately does not curate. + trackMissingModels: false, + async fetchModels() { + const response = await fetch(API_ENDPOINT); + if (!response.ok) { + throw new Error(`Nebul models request failed: ${response.status} ${response.statusText}`); + } + return response.json(); + }, + parseModels(raw) { + const data = NebulResponse.parse(raw).data; + // An empty catalog is an upstream fault; syncing it would delete every + // local model file via the delete-missing pass, so fail loudly instead. + if (data.length === 0) { + throw new Error("Nebul returned an empty model catalog"); + } + // Same failure mode if the response shape drifts and no entry matches the + // chat-model filter anymore (e.g. renamed model_type/mode values). + if (!data.some(isCatalogChatModel)) { + throw new Error("Nebul returned no usable chat models"); + } + const chatCount = data.filter(isCatalogChatModel).length; + if (chatCount < MIN_CHAT_MODELS) { + throw new Error( + `Nebul returned only ${chatCount} usable chat models (expected at least ${MIN_CHAT_MODELS}); treating the catalog as a partial fault and skipping this run`, + ); + } + return data; + }, + // Unauthenticated /v1/model/info is authoritative for the curated entries' + // live cost/context. Because the provider runs with skipCreates and + // deleteMissing: false, translateModel only ever refreshes existing files; any + // other in-scope chat model is reported via skippedNotice, and a curated model + // missing from the catalog is retained and reported via missingNotice. Whole- + // catalog faults still fail closed in parseModels. + translateModel(entry, context) { + if (!isCatalogChatModel(entry)) return undefined; + const id = entry.model_name; + const info = entry.model_info; + const existing = context.existing(id); + // Existing entries must survive incomplete source data — a transient null + // price or an unresolved alias would otherwise delete the hand-authored + // TOML on the next run. They keep their authored base_model and cost/limit; + // only brand-new models need a fully-priced, resolvable source entry. + const baseModel = existing?.base_model ?? resolveBaseModel(id, info.huggingface_id ?? undefined); + const cost = info.input_cost_per_1m_tokens != null && info.output_cost_per_1m_tokens != null + ? { + input: info.input_cost_per_1m_tokens, + output: info.output_cost_per_1m_tokens, + cache_read: info.cache_read_input_cost_per_1m_tokens ?? undefined, + } + : existing?.cost; + const limit = info.max_input_tokens != null ? { context: info.max_input_tokens } : existing?.limit; + if (existing === undefined && (baseModel === undefined || cost === undefined || limit === undefined)) return undefined; + // A hand-authored reasoning = false marks a served ID whose lab model reasons + // but which this host runs with thinking disabled (the catalog reports + // supports_reasoning = false and no reasoning_efforts). Keep the override and + // suppress the control/trace machinery entirely: no reasoning_options to + // require, and no interleaved side channel when no traces are returned. + const reasoningDisabled = existing?.reasoning === false; + // Fail closed unless caller control is probe-verified and hand-authored: + // publishing the catalog's advertised reasoning_efforts unreviewed would + // sync proven-wrong controls (2026-09-23: it advertised low|medium|high|max + // for one model, whose engine rejects every value but high). The runner + // skips the ID so the options can be hand-authored from live probes. + const isReasoner = !reasoningDisabled && (baseModel !== undefined + ? modelMetadata(baseModel).reasoning === true + : existing?.reasoning === true); + if (isReasoner && existing?.reasoning_options === undefined) { + throw new MissingReasoningOptionsError( + id, + `${id} is a reasoning model, but the catalog entry has no probe-verified reasoning_options; hand-author them instead of trusting the advertised reasoning_efforts`, + ); + } + const values = reasoningDisabled + ? { reasoning: false, interleaved: undefined, reasoning_options: undefined, cost, limit } + : { interleaved: existing?.interleaved, reasoning_options: existing?.reasoning_options, cost, limit }; + if (baseModel !== undefined) { + return { + id, + model: factorBaseModel(baseModel, values, limit) as SyncedModel, + }; + } + // Existing standalone definition whose served alias no longer resolves: + // keep the authored fields, refreshing only what /model/info still provides. + return { id, model: { ...existing, ...values } as SyncedModel }; + }, + // Only report in-scope chat models whose base_model could not be resolved; filtered + // entries (embeddings, rerankers, out-of-scope specialized models, superseded IDs) skip silently. + sourceID(entry: NebulEntry) { + return isCatalogChatModel(entry) ? entry.model_name : undefined; + }, + skippedNotice(ids) { + if (ids.length === 0) return []; + return [ + `Nebul serves these in-scope chat models, but only the curated catalog is shipped (skipCreates); not added:`, + ids.map((id) => `\`${id}\``).join(", "), + ]; + }, + missingNotice(paths) { + if (paths.length === 0) return []; + return [ + `Nebul models absent from the source catalog were retained, not deleted:`, + paths.map((p) => `\`${p.replace(/\.toml$/, "")}\``).join(", "), + ]; + }, +} satisfies SyncProvider; + +function isCatalogChatModel(entry: NebulEntry): boolean { + const info = entry.model_info; + return info.model_type === "llm" && info.mode === "chat" + && info.superseded_by_model_name == null && !OUT_OF_SCOPE_PATTERNS.some((pattern) => pattern.test(entry.model_name)) + && !(info.display_tags ?? []).some((tag) => OUT_OF_SCOPE_TAGS.has(tag)); +} + +// Nebul documents exactly one reasoning control: reasoning_effort. Authored +// options are the only options ever synced: they are live-probe evidence for +// what the served engine accepts, while the catalog's advertised +// reasoning_efforts are probe-proven unreliable (2026-09-23: it advertised +// low|medium|high|max for one model, whose engine rejects every value but +// high). The advertised list is therefore never copied into an entry — a +// reasoner with nothing authored is rejected above, and a non-reasoner carries +// no options. Lab-style toggles or budgets are not supported on this API +// unless a probe of this host shows them. + +function resolveBaseModel(servedID: string, huggingfaceID: string | undefined): string | undefined { + return baseModelCandidates(servedID, huggingfaceID).find(canonicalExists); +} + +// existsSync is case-insensitive on Windows/macOS; verify the real on-disk filename case +// so the resolved base_model matches the canonical metadata exactly (and CI on Linux). +function canonicalExists(candidate: string): boolean { + const file = path.join(MODELS_DIR, `${candidate}.toml`); + if (!existsSync(file)) return false; + try { + return readdirSync(path.dirname(file)).includes(path.basename(file)); + } catch { + return false; + } +} + +function baseModelCandidates(servedID: string, huggingfaceID: string | undefined): string[] { + const alias = BASE_MODEL_ALIASES[servedID]; + const servedCandidate = mapOrgToCandidate(servedID); + const hfCandidate = huggingfaceID === undefined ? undefined : mapOrgToCandidate(huggingfaceID); + return [ + ...new Set([alias, servedCandidate, hfCandidate, ...quantizationStripped(hfCandidate), ...quantizationStripped(servedCandidate)]).values(), + ].filter((candidate): candidate is string => candidate !== undefined); +} + +function mapOrgToCandidate(id: string): string | undefined { + const [org, ...modelParts] = id.split("/"); + if (org === undefined || modelParts.length === 0) return undefined; + const provider = ORG_TO_MODEL_PROVIDER[org.toLowerCase()]; + if (provider === undefined) return undefined; + return `${provider}/${modelParts.join("/").toLowerCase()}`; +} + +// Hosts serve quantized checkpoints (e.g. -FP8, -BF16) of weights whose canonical +// metadata is published for the base precision; try those names without the suffix. +// NVIDIA also prefixes checkpoints with "NVIDIA-", which the metadata names drop. +function quantizationStripped(candidate: string | undefined): string[] { + if (candidate === undefined) return []; + const withoutQuant = candidate.replace(/-(fp8|bf16|fp4|int8)$/i, ""); + const withoutPrefix = withoutQuant.replace(/nvidia-/, ""); + return withoutQuant === candidate ? [] : [...new Set([withoutQuant, withoutPrefix])].filter((value) => value !== candidate); +} diff --git a/packages/core/test/nebul.test.ts b/packages/core/test/nebul.test.ts new file mode 100644 index 00000000000..aa9a4304ccc --- /dev/null +++ b/packages/core/test/nebul.test.ts @@ -0,0 +1,296 @@ +import { expect, test } from "bun:test"; +import { readdirSync } from "node:fs"; +import path from "node:path"; + +import type { ExistingModel } from "../src/sync/index.js"; +import { MissingReasoningOptionsError } from "../src/sync/missing-reasoning-options.js"; +import { + NebulEntry, + NebulResponse, + nebul, +} from "../src/sync/providers/nebul.js"; + +function nebulEntry(model_name?: string, model_info: Record = {}): NebulEntry { + return NebulEntry.parse({ + model_name: model_name ?? "zai-org/GLM-5.3", + model_info: { + description: "test", + huggingface_id: model_name ?? "zai-org/GLM-5.3", + input_cost_per_1m_tokens: 1.47, + output_cost_per_1m_tokens: 4.62, + cache_read_input_cost_per_1m_tokens: 0.35, + max_input_tokens: 1_000_000, + mode: "chat", + model_type: "llm", + reasoning_efforts: ["low", "high", "max"], + ...model_info, + }, + }); +} + +function existingWith(reasoning_options: ExistingModel["reasoning_options"]): ExistingModel { + return { reasoning_options } as ExistingModel; +} + +const context = (existing: ExistingModel | undefined) => ({ existing: () => existing }); + +test("syncs Nebul's factored overrides against resolved lab metadata", () => { + // Non-reasoner lab: new entries keep factored pricing/context and carry no options. + const translated = nebul.translateModel( + nebulEntry("mistralai/Mistral-Large-3-675B-Instruct-2512", { max_input_tokens: 1_048_576 }), + context(undefined), + ); + expect(translated).toMatchObject({ + id: "mistralai/Mistral-Large-3-675B-Instruct-2512", + model: { + base_model: "mistral/mistral-large-2512", + cost: { input: 1.47, output: 4.62, cache_read: 0.35 }, + limit: { context: 1_048_576 }, + }, + }); + expect(translated?.model.reasoning_options).toBeUndefined(); +}); + +test("fails closed for a new reasoner that only advertises efforts", () => { + // The advertised list is probe-proven unreliable (live probes found a catalog + // model advertising low|medium|high|max whose engine rejected every value but + // high), so a new reasoner must never inherit it — options need live probes first. + const entry = nebulEntry("zai-org/GLM-5.3"); + expect(() => nebul.translateModel(entry, context(undefined))).toThrow(MissingReasoningOptionsError); +}); + +test("preserves authored reasoning controls when the host exposes no efforts", () => { + const authored = [{ type: "toggle" as const }]; + const translated = nebul.translateModel(nebulEntry("zai-org/GLM-5.3", { reasoning_efforts: [] }), context(existingWith(authored))); + expect(translated?.model.reasoning_options).toEqual(authored); +}); + +test("keeps authored probe-verified controls over the advertised effort list", () => { + // Live probes (2026-09-23) showed the advertised list can be wrong: + // one catalog model advertised low|medium|high|max while its served engine + // rejected every value but high. + const authored = [{ type: "effort" as const, values: ["none", "high"] }]; + const translated = nebul.translateModel(nebulEntry("zai-org/GLM-5.3"), context(existingWith(authored))); + expect(translated?.model.reasoning_options).toEqual(authored); +}); + +test("carries authored interleaved through sync", () => { + const inline = nebul.translateModel( + nebulEntry("zai-org/GLM-5.3"), + context({ interleaved: true, reasoning_options: [{ type: "toggle" }] } as ExistingModel), + ); + expect(inline?.model.interleaved).toBe(true); + + const named = nebul.translateModel( + nebulEntry("someorg/Some-Model"), + context({ interleaved: { field: "reasoning_content" }, reasoning_options: [{ type: "toggle" }] } as ExistingModel), + ); + expect(named?.model.interleaved).toEqual({ field: "reasoning_content" }); +}); + +test("keeps an authored reasoning = false override for a lab reasoner the host serves without thinking", () => { + const existing = { + base_model: "alibaba/qwen3.5-397b-a17b", + reasoning: false, + reasoning_options: [{ type: "effort" as const, values: ["low"] }], + interleaved: { field: "reasoning_content" as const }, + } as ExistingModel; + const translated = nebul.translateModel( + nebulEntry("Qwen/Qwen3.5-397B-A17B", { reasoning_efforts: undefined }), + context(existing), + ); + expect(translated?.model.reasoning).toBe(false); + expect(translated?.model.reasoning_options).toBeUndefined(); + expect(translated?.model.interleaved).toBeUndefined(); +}); + +test("fails closed when a reasoner advertises no efforts and none are authored", () => { + const entry = nebulEntry("zai-org/GLM-5.3", { reasoning_efforts: [] }); + expect(() => nebul.translateModel(entry, context(undefined))).toThrow(MissingReasoningOptionsError); + expect(() => + nebul.translateModel(entry, context({ base_model: "zhipuai/glm-5.3" } as ExistingModel)), + ).toThrow(MissingReasoningOptionsError); +}); + +test("keeps existing entries when the source pricing or context is temporarily null", () => { + const existing = { + base_model: "zhipuai/glm-5.3", + cost: { input: 1.47, output: 4.62 }, + limit: { context: 1_048_576 }, + reasoning_options: [{ type: "effort" as const, values: ["low", "high"] }], + } as ExistingModel; + const translated = nebul.translateModel( + nebulEntry("zai-org/GLM-5.3", { input_cost_per_1m_tokens: null, output_cost_per_1m_tokens: null, max_input_tokens: null }), + context(existing), + ); + expect(translated).toMatchObject({ + id: "zai-org/GLM-5.3", + model: { + base_model: "zhipuai/glm-5.3", + cost: { input: 1.47, output: 4.62 }, + limit: { context: 1_048_576 }, + reasoning_options: [{ type: "effort", values: ["low", "high"] }], + }, + }); +}); + +test("keeps existing entries when the served alias no longer resolves to lab metadata", () => { + const existing = { + base_model: "zhipuai/glm-5.3", + cost: { input: 1.47, output: 4.62 }, + limit: { context: 1_048_576 }, + reasoning_options: [{ type: "effort" as const, values: ["low", "high"] }], + } as ExistingModel; + const translated = nebul.translateModel(nebulEntry("someorg/Unknown-Model", { huggingface_id: null }), context(existing)); + expect(translated?.model.base_model).toBe("zhipuai/glm-5.3"); +}); + +test("resolves base models across org renames and quantization suffixes", () => { + const cases: [string, string | null, string][] = [ + ["zai-org/GLM-5.3", "zai-org/GLM-5.3", "zhipuai/glm-5.3"], + ["Qwen/Qwen3.5-397B-A17B", "Qwen/Qwen3.5-397B-A17B", "alibaba/qwen3.5-397b-a17b"], + // Hugging Face org paths are case-insensitive; a lowercase org must resolve identically. + ["qwen/qwen3.5-397b-a17b", "qwen/qwen3.5-397b-a17b", "alibaba/qwen3.5-397b-a17b"], + ["nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16", "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16", "nvidia/nemotron-3-super-120b-a12b"], + ["mistralai/Mistral-Large-3-675B-Instruct-2512", "mistralai/Mistral-Large-3-675B-Instruct-2512", "mistral/mistral-large-2512"], + ]; + // Authored options in context keep reasoner labs out of the fail-closed path + // (this case asserts base_model resolution only). + for (const [model_name, huggingface_id, expected] of cases) { + const entry = nebulEntry(model_name, { huggingface_id }); + expect( + nebul.translateModel(entry, context(existingWith([{ type: "effort" as const, values: ["high"] }])))?.model.base_model, + ).toBe(expected); + } +}); + +test("skips out-of-scope specialized models and superseded entries silently", () => { + for (const model_name of ["someorg/Doc-OCR-Tool", "someorg/ScanOCR-Large"]) { + const entry = nebulEntry(model_name, {}); + expect(nebul.translateModel(entry, context(undefined))).toBeUndefined(); + expect(nebul.sourceID(entry)).toBeUndefined(); + } + for (const display_tags of [["Guard Model"], ["Content Safety"], ["Private"], ["Internal"]]) { + const entry = nebulEntry("someorg/Some-Model", { display_tags }); + expect(nebul.translateModel(entry, context(undefined))).toBeUndefined(); + expect(nebul.sourceID(entry)).toBeUndefined(); + } + for (const model_name of ["someorg/Retired-Model-A", "someorg/Retired-Model-B"]) { + const entry = nebulEntry(model_name, { superseded_by_model_name: "zai-org/GLM-5.3" }); + expect(nebul.translateModel(entry, context(undefined))).toBeUndefined(); + expect(nebul.sourceID(entry)).toBeUndefined(); + } +}); + +test("skips embeddings and rerankers while reporting unresolvable chat models", () => { + const embedding = nebulEntry("someorg/Some-Embedding", { model_type: "embedding" }); + expect(nebul.translateModel(embedding, context(undefined))).toBeUndefined(); + expect(nebul.sourceID(embedding)).toBeUndefined(); + + const chat = nebulEntry("mistralai/Mistral-Large-3-675B-Instruct-2512", { huggingface_id: null }); + expect(nebul.translateModel(chat, context(undefined))).toBeDefined(); + expect(nebul.sourceID(chat)).toBe("mistralai/Mistral-Large-3-675B-Instruct-2512"); +}); + +test("skips chat models whose pricing or context is absent instead of crashing", () => { + const unpriced = nebulEntry("zai-org/GLM-5.3", { input_cost_per_1m_tokens: null, output_cost_per_1m_tokens: null, max_input_tokens: null }); + expect(nebul.translateModel(unpriced, context(undefined))).toBeUndefined(); + expect(nebul.sourceID(unpriced)).toBe("zai-org/GLM-5.3"); +}); + +test("parses nullable serving artifacts and unknown-host metadata from /model/info", () => { + const parsed = NebulResponse.parse({ + data: [ + { model_name: "Some/Embedding", model_info: { mode: null, model_type: "embedding" } }, + { model_name: "Some/Chat", model_info: { mode: "chat", model_type: "llm", unknown_host_field: true } }, + ], + }); + expect(parsed.data).toHaveLength(2); +}); + +test("fails closed on an empty catalog so sync cannot delete every local file", () => { + expect(() => nebul.parseModels({ data: [] })).toThrow("Nebul returned an empty model catalog"); +}); + +test("fails closed when no entry matches the chat-model filter", () => { + expect(() => + nebul.parseModels({ + data: [ + { model_name: "Some/Embedding", model_info: { mode: null, model_type: "embedding" } }, + { model_name: "Some/Reranker", model_info: { mode: null, model_type: "rerank" } }, + ], + }), + ).toThrow("Nebul returned no usable chat models"); +}); + +test("parseModels keeps chat entries alongside filtered serving artifacts", () => { + const chat = { model_info: { mode: "chat", model_type: "llm" } }; + const parsed = nebul.parseModels({ + data: [ + { model_name: "Some/Embedding", model_info: { mode: null, model_type: "embedding" } }, + { model_name: "Some/Chat-A", ...chat }, + { model_name: "Some/Chat-B", ...chat }, + { model_name: "Some/Chat-C", ...chat }, + { model_name: "Some/Chat-D", ...chat }, + { model_name: "Some/Chat-E", ...chat }, + { model_name: "Some/Chat-F", ...chat }, + ], + }); + expect(parsed).toHaveLength(7); +}); + +test("fails closed on a partial catalog so sync cannot prune healthy local files", () => { + expect(() => + nebul.parseModels({ + data: [{ model_name: "Some/Chat", model_info: { mode: "chat", model_type: "llm" } }], + }), + ).toThrow("treating the catalog as a partial fault"); +}); + +test("accepts and ignores unknown advertised reasoning effort values", () => { + // The advertised list is never synced, so an unrecognized value must not abort + // parsing: a strict enum would fail the hourly run and block cost/context + // refreshes for the curated models. + const parsed = NebulResponse.parse({ + data: [{ model_name: "Some/Chat", model_info: { mode: "chat", model_type: "llm", reasoning_efforts: ["ultra"] } }], + }); + expect(parsed.data[0]?.model_info.reasoning_efforts).toEqual(["ultra"]); + + const authored = [{ type: "effort" as const, values: ["high"] }]; + const translated = nebul.translateModel( + nebulEntry("zai-org/GLM-5.3", { reasoning_efforts: ["ultra"] }), + context(existingWith(authored)), + ); + expect(translated?.model.reasoning_options).toEqual(authored); +}); + +// Nebul is a curated provider: exactly the four requested flagship models ship, +// and the catalog sync must never add to or remove from that set. +const CURATED_MODEL_IDS = [ + "Qwen/Qwen3.5-397B-A17B", + "mistralai/Mistral-Large-3-675B-Instruct-2512", + "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16", + "zai-org/GLM-5.3", +]; + +function modelFiles(dir: string): string[] { + return readdirSync(dir, { withFileTypes: true }).flatMap((entry) => { + const full = path.join(dir, entry.name); + if (entry.isDirectory()) return modelFiles(full); + return entry.name.endsWith(".toml") ? [full] : []; + }); +} + +test("ships exactly the four curated models", () => { + const modelsDir = path.join(import.meta.dirname, "..", "..", "..", "providers", "nebul", "models"); + const ids = modelFiles(modelsDir) + .map((file) => path.relative(modelsDir, file).replace(/\.toml$/, "").split(path.sep).join("/")) + .sort(); + expect(ids).toEqual([...CURATED_MODEL_IDS].sort()); +}); + +test("sync cannot grow or shrink the curated catalog", () => { + expect(nebul.skipCreates).toBe(true); + expect(nebul.deleteMissing).toBe(false); + expect(nebul.trackMissingModels).toBe(false); +}); diff --git a/providers/nebul/logo.svg b/providers/nebul/logo.svg new file mode 100644 index 00000000000..33daff15792 --- /dev/null +++ b/providers/nebul/logo.svg @@ -0,0 +1,5 @@ + + + + + diff --git a/providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml b/providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml new file mode 100644 index 00000000000..e74f8d592f1 --- /dev/null +++ b/providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml @@ -0,0 +1,19 @@ +# Effort: reasoning_effort = none|low|medium|high; thinking is ON by default +# and none turns it off (all listed levels probe-verified 200 OK on 2026-09-23; set +# narrowed to the same-surface peer baseline (ovhcloud/scaleway/crof/digitalocean) — +# minimal/xhigh/max lacked distinct-effect evidence). The catalog's +# supports_reasoning = false for this ID is a false negative. +# Traces arrive in reasoning_content. Costs live-verified. +base_model = "alibaba/qwen3.5-397b-a17b" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] + +[cost] +input = 0.63 +output = 3.78 +cache_read = 0.15 diff --git a/providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml b/providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml new file mode 100644 index 00000000000..a0fd9e2b729 --- /dev/null +++ b/providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml @@ -0,0 +1,7 @@ +# Costs live-verified via https://api.inference.nebul.io/v1/model/info on 2026-09-23. +base_model = "mistral/mistral-large-2512" + +[cost] +input = 0.6 +output = 1.73 +cache_read = 0.15 diff --git a/providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml b/providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml new file mode 100644 index 00000000000..598e335349f --- /dev/null +++ b/providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml @@ -0,0 +1,21 @@ +# Effort: reasoning_effort = none|low|medium|high|max; thinking is ON by default +# and none turns it off (all listed levels probe-verified 200 OK on 2026-09-23; set +# narrowed to the same-surface peer baseline — minimal/xhigh lacked distinct-effect +# evidence). The catalog's supports_reasoning = false for this ID is a false negative. +# Traces arrive in reasoning_content. Context/costs live-verified. +base_model = "nvidia/nemotron-3-super-120b-a12b" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "max"] + +[cost] +input = 0.32 +output = 0.69 +cache_read = 0.08 + +[limit] +context = 1_000_000 diff --git a/providers/nebul/models/zai-org/GLM-5.3.toml b/providers/nebul/models/zai-org/GLM-5.3.toml new file mode 100644 index 00000000000..b61b191c937 --- /dev/null +++ b/providers/nebul/models/zai-org/GLM-5.3.toml @@ -0,0 +1,19 @@ +# Effort: reasoning_effort = low|high|max (catalog + zai/OpenRouter parity). +# Traces arrive in reasoning_content. +# Costs/context live-verified via https://api.inference.nebul.io/v1/model/info on 2026-09-23. +base_model = "zhipuai/glm-5.3" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 1.47 +output = 4.62 +cache_read = 0.35 + +[limit] +context = 1_048_576 diff --git a/providers/nebul/provider.toml b/providers/nebul/provider.toml new file mode 100644 index 00000000000..dc44e31d87d --- /dev/null +++ b/providers/nebul/provider.toml @@ -0,0 +1,5 @@ +name = "Nebul" +npm = "@ai-sdk/openai-compatible" +api = "https://api.inference.nebul.io/v1" +env = ["NEBUL_API_KEY"] +doc = "https://docs.nebul.io" From fbee3679594c97010c146deec9abccdfbb9e795f Mon Sep 17 00:00:00 2001 From: Wynand Huizinga Date: Thu, 24 Sep 2026 14:00:35 +0200 Subject: [PATCH 2/2] docs(nebul): rewrite comments in plain English Rewrite the comments added with the Nebul provider for simple, active-voice English: short sentences, no semicolons or em-dashes, no speculative modals. No code or data changes. --- packages/core/src/sync/providers/nebul.ts | 144 +++++++++--------- packages/core/test/nebul.test.ts | 31 ++-- .../nebul/models/Qwen/Qwen3.5-397B-A17B.toml | 13 +- .../Mistral-Large-3-675B-Instruct-2512.toml | 2 +- ...VIDIA-Nemotron-3-Super-120B-A12B-BF16.toml | 11 +- providers/nebul/models/zai-org/GLM-5.3.toml | 7 +- 6 files changed, 111 insertions(+), 97 deletions(-) diff --git a/packages/core/src/sync/providers/nebul.ts b/packages/core/src/sync/providers/nebul.ts index 8a7579eedfb..2923ed2ed39 100644 --- a/packages/core/src/sync/providers/nebul.ts +++ b/packages/core/src/sync/providers/nebul.ts @@ -9,9 +9,10 @@ import { factorBaseModel, modelMetadata } from "./openrouter.js"; const API_ENDPOINT = "https://api.inference.nebul.io/v1/model/info"; const MODELS_DIR = path.join(import.meta.dirname, "..", "..", "..", "..", "..", "models"); -// Served org prefix -> models/ metadata namespace (HF org names differ from catalog labs). -// Keys are lowercase; lookups normalize the org the same way (Hugging Face orgs are -// case-insensitive in URLs, so e.g. "qwen/Qwen3.8-27B-FP8" is a valid ID shape). +// Maps the org prefix of a served ID to the lab namespace under models/. +// The Hugging Face org name and the lab name in models/ often differ. +// Keys are lowercase, and the lookup lowercases the org the same way. Hugging Face +// org paths are case-insensitive in URLs, so "qwen/Qwen3.8-27B-FP8" is a valid ID shape. const ORG_TO_MODEL_PROVIDER: Record = { "deepseek-ai": "deepseek", google: "google", @@ -24,25 +25,26 @@ const ORG_TO_MODEL_PROVIDER: Record = { "zai-org": "zhipuai", }; -// Served IDs whose canonical metadata lives under a differently-named lab entry. +// Served IDs whose lab metadata in models/ lives under a different model name. const BASE_MODEL_ALIASES: Record = { "mistralai/Mistral-Large-3-675B-Instruct-2512": "mistral/mistral-large-2512", }; -// Catalog scope is general chat models. The public catalog also lists specialized -// document-OCR models by name, and flags internal-only or safety-infrastructure -// entries via display_tags; keep all of those out (not coding/chat targets, and -// models.dev carries no matching lab metadata for them). +// The sync scope is general chat models. The catalog also lists specialized +// OCR (document text recognition) models by name and flags private, internal, +// and safety entries with display_tags. The filter keeps all of those out, +// because they are not chat models and models.dev has no matching lab metadata for them. const OUT_OF_SCOPE_PATTERNS = [/OCR/i]; const OUT_OF_SCOPE_TAGS = new Set(["Guard Model", "Content Safety", "Private", "Internal"]); -// Fail-closed floor against partial catalog faults. The in-scope chat catalog is -// ~14 models as of 2026-09-24; a truncated response (per-lab serving outage, -// half-written deploy) that still passes the non-empty checks should not be -// treated as the real catalog. Defense in depth: the provider also runs with -// deleteMissing: false, so even a bad catalog cannot prune curated local files. -// Any catalog showing less than half the known-good size is treated as -// structurally incomplete. Raise this deliberately as the catalog grows. +// The sync fails closed (it throws on bad data instead of syncing it) on a +// partial catalog. The in-scope chat catalog has ~14 models as of 2026-09-24. +// A truncated response (a per-lab serving outage or a half-written deploy) can +// pass the non-empty checks in parseModels. The code must not treat it as the +// real catalog. As a second guard, the provider runs with deleteMissing: false, +// so even a bad catalog cannot delete curated local files. A catalog with less +// than half the known-good size is structurally incomplete. Raise this number +// deliberately as the catalog grows. const MIN_CHAT_MODELS = 6; const ModelInfo = z.object({ @@ -55,9 +57,10 @@ const ModelInfo = z.object({ max_input_tokens: z.number().nullable().optional(), mode: z.string().nullable(), model_type: z.string().nullable(), - // Advertised only; never synced (probes proved the list unreliable), so accept - // any string. A strict enum here would let a future unknown value throw and - // fail the whole hourly run, blocking cost/context refreshes for curated models. + // The catalog only advertises this list, and the sync never copies it into an + // entry, because probes proved the list unreliable. The schema accepts any + // string. A strict enum throws on a future unknown value. That failure stops + // the hourly run and blocks cost/context refreshes for curated models. reasoning_efforts: z.array(z.string()).nullable().optional(), superseded_by_model_name: z.string().nullable().optional(), }).passthrough(); @@ -77,17 +80,16 @@ export const nebul = { id: "nebul", name: "Nebul", modelsDir: "providers/nebul/models", - // Nebul is a curated provider: only the hand-authored flagship models ship. - // The catalog is authoritative for their live cost/context, but must never - // grow or shrink the local set: skipCreates keeps any other in-scope chat + // Nebul is a curated provider, so only the hand-authored flagship models ship. + // The catalog is authoritative for their live cost and context, but it must + // never grow or shrink the local set. skipCreates keeps any other in-scope chat // model out, and deleteMissing: false keeps a curated model that drops out of - // the catalog (zai-org/GLM-5.3 has been intermittently absent) instead of - // silently removing it. + // the catalog (zai-org/GLM-5.3 is intermittently absent) instead of deleting it. skipCreates: true, deleteMissing: false, - // Skipped reasoners (fail-closed below) and out-of-catalog chat models are - // expected; opening missing-model issues for them would request models this - // provider deliberately does not curate. + // Skipped reasoners (the fail-closed path below) and chat models outside the + // curation are expected. Missing-model issues for them ask for models that + // this provider deliberately does not curate. trackMissingModels: false, async fetchModels() { const response = await fetch(API_ENDPOINT); @@ -98,13 +100,14 @@ export const nebul = { }, parseModels(raw) { const data = NebulResponse.parse(raw).data; - // An empty catalog is an upstream fault; syncing it would delete every - // local model file via the delete-missing pass, so fail loudly instead. + // An empty catalog is an upstream fault. The delete-missing pass deletes + // every local model file when the sync accepts it, so the code throws instead. if (data.length === 0) { throw new Error("Nebul returned an empty model catalog"); } - // Same failure mode if the response shape drifts and no entry matches the - // chat-model filter anymore (e.g. renamed model_type/mode values). + // The same failure applies when the response shape drifts and no entry + // matches the chat-model filter anymore, for example after renamed + // model_type or mode values. if (!data.some(isCatalogChatModel)) { throw new Error("Nebul returned no usable chat models"); } @@ -116,21 +119,22 @@ export const nebul = { } return data; }, - // Unauthenticated /v1/model/info is authoritative for the curated entries' - // live cost/context. Because the provider runs with skipCreates and - // deleteMissing: false, translateModel only ever refreshes existing files; any - // other in-scope chat model is reported via skippedNotice, and a curated model - // missing from the catalog is retained and reported via missingNotice. Whole- - // catalog faults still fail closed in parseModels. + // The unauthenticated /v1/model/info endpoint is authoritative for the live + // cost and context of the curated entries. The provider runs with skipCreates + // and deleteMissing: false, so translateModel only refreshes existing files. + // skippedNotice reports any other in-scope chat model, and missingNotice + // reports a curated model that the catalog no longer lists. Whole-catalog + // faults still fail closed in parseModels. translateModel(entry, context) { if (!isCatalogChatModel(entry)) return undefined; const id = entry.model_name; const info = entry.model_info; const existing = context.existing(id); - // Existing entries must survive incomplete source data — a transient null - // price or an unresolved alias would otherwise delete the hand-authored - // TOML on the next run. They keep their authored base_model and cost/limit; - // only brand-new models need a fully-priced, resolvable source entry. + // Existing entries must survive incomplete source data. A transient null + // price or an unresolved alias otherwise deletes the hand-authored TOML on + // the next run. Existing entries keep their authored base_model and + // cost/limit, and only new models need a source entry with complete pricing + // and a resolvable base. const baseModel = existing?.base_model ?? resolveBaseModel(id, info.huggingface_id ?? undefined); const cost = info.input_cost_per_1m_tokens != null && info.output_cost_per_1m_tokens != null ? { @@ -141,17 +145,18 @@ export const nebul = { : existing?.cost; const limit = info.max_input_tokens != null ? { context: info.max_input_tokens } : existing?.limit; if (existing === undefined && (baseModel === undefined || cost === undefined || limit === undefined)) return undefined; - // A hand-authored reasoning = false marks a served ID whose lab model reasons - // but which this host runs with thinking disabled (the catalog reports - // supports_reasoning = false and no reasoning_efforts). Keep the override and - // suppress the control/trace machinery entirely: no reasoning_options to - // require, and no interleaved side channel when no traces are returned. + // A hand-authored reasoning = false marks a served ID whose lab model + // reasons but that this host serves with thinking disabled. The catalog + // reports supports_reasoning = false and no reasoning_efforts for it. Keep + // the override and suppress the reasoning controls and traces. The entry + // then needs no reasoning_options, and no interleaved side channel applies + // when the host returns no traces. const reasoningDisabled = existing?.reasoning === false; - // Fail closed unless caller control is probe-verified and hand-authored: - // publishing the catalog's advertised reasoning_efforts unreviewed would - // sync proven-wrong controls (2026-09-23: it advertised low|medium|high|max - // for one model, whose engine rejects every value but high). The runner - // skips the ID so the options can be hand-authored from live probes. + // Fail closed unless the caller control is hand-authored from live probe + // evidence. Probes proved the advertised reasoning_efforts wrong in one + // case: on 2026-09-23 the catalog advertised low|medium|high|max for one + // model, and its engine rejects every value but high. The runner skips the + // ID so that the options can be hand-authored from live probes. const isReasoner = !reasoningDisabled && (baseModel !== undefined ? modelMetadata(baseModel).reasoning === true : existing?.reasoning === true); @@ -170,12 +175,14 @@ export const nebul = { model: factorBaseModel(baseModel, values, limit) as SyncedModel, }; } - // Existing standalone definition whose served alias no longer resolves: - // keep the authored fields, refreshing only what /model/info still provides. + // The existing standalone definition has a served alias that no longer + // resolves. Keep the authored fields, and refresh only what /model/info + // still provides. return { id, model: { ...existing, ...values } as SyncedModel }; }, - // Only report in-scope chat models whose base_model could not be resolved; filtered - // entries (embeddings, rerankers, out-of-scope specialized models, superseded IDs) skip silently. + // Report only in-scope chat models whose base_model does not resolve. Filtered + // entries (embeddings, rerankers, out-of-scope specialized models, superseded + // IDs) skip silently. sourceID(entry: NebulEntry) { return isCatalogChatModel(entry) ? entry.model_name : undefined; }, @@ -202,22 +209,21 @@ function isCatalogChatModel(entry: NebulEntry): boolean { && !(info.display_tags ?? []).some((tag) => OUT_OF_SCOPE_TAGS.has(tag)); } -// Nebul documents exactly one reasoning control: reasoning_effort. Authored -// options are the only options ever synced: they are live-probe evidence for -// what the served engine accepts, while the catalog's advertised -// reasoning_efforts are probe-proven unreliable (2026-09-23: it advertised -// low|medium|high|max for one model, whose engine rejects every value but -// high). The advertised list is therefore never copied into an entry — a -// reasoner with nothing authored is rejected above, and a non-reasoner carries -// no options. Lab-style toggles or budgets are not supported on this API -// unless a probe of this host shows them. +// Nebul documents exactly one reasoning control: reasoning_effort. The sync +// copies only hand-authored options into an entry, because probes proved the +// advertised reasoning_efforts unreliable (on 2026-09-23 the catalog advertised +// low|medium|high|max for one model, and its engine rejects every value but +// high). translateModel rejects a reasoner with no authored options, and a +// non-reasoner carries no options. The API supports no lab-style toggles or +// budget controls unless a probe of this host shows them. function resolveBaseModel(servedID: string, huggingfaceID: string | undefined): string | undefined { return baseModelCandidates(servedID, huggingfaceID).find(canonicalExists); } -// existsSync is case-insensitive on Windows/macOS; verify the real on-disk filename case -// so the resolved base_model matches the canonical metadata exactly (and CI on Linux). +// existsSync is case-insensitive on Windows and macOS. readdirSync reads the +// real on-disk filename case, so the resolved base_model matches the models/ +// filename exactly, and CI on Linux passes. function canonicalExists(candidate: string): boolean { const file = path.join(MODELS_DIR, `${candidate}.toml`); if (!existsSync(file)) return false; @@ -245,9 +251,11 @@ function mapOrgToCandidate(id: string): string | undefined { return `${provider}/${modelParts.join("/").toLowerCase()}`; } -// Hosts serve quantized checkpoints (e.g. -FP8, -BF16) of weights whose canonical -// metadata is published for the base precision; try those names without the suffix. -// NVIDIA also prefixes checkpoints with "NVIDIA-", which the metadata names drop. +// Hosts serve quantized checkpoints (the same weights in a smaller number +// format, for example -FP8, -BF16), and the lab publishes the metadata under +// the base-precision name. The resolver also tries the names without the +// quantization suffix. NVIDIA prefixes checkpoint names with "NVIDIA-", and the +// metadata names drop that prefix. function quantizationStripped(candidate: string | undefined): string[] { if (candidate === undefined) return []; const withoutQuant = candidate.replace(/-(fp8|bf16|fp4|int8)$/i, ""); diff --git a/packages/core/test/nebul.test.ts b/packages/core/test/nebul.test.ts index aa9a4304ccc..354aa3a4de1 100644 --- a/packages/core/test/nebul.test.ts +++ b/packages/core/test/nebul.test.ts @@ -35,7 +35,8 @@ function existingWith(reasoning_options: ExistingModel["reasoning_options"]): Ex const context = (existing: ExistingModel | undefined) => ({ existing: () => existing }); test("syncs Nebul's factored overrides against resolved lab metadata", () => { - // Non-reasoner lab: new entries keep factored pricing/context and carry no options. + // The lab model here does not reason. A new entry keeps the factored pricing + // and context and carries no options. const translated = nebul.translateModel( nebulEntry("mistralai/Mistral-Large-3-675B-Instruct-2512", { max_input_tokens: 1_048_576 }), context(undefined), @@ -52,9 +53,10 @@ test("syncs Nebul's factored overrides against resolved lab metadata", () => { }); test("fails closed for a new reasoner that only advertises efforts", () => { - // The advertised list is probe-proven unreliable (live probes found a catalog - // model advertising low|medium|high|max whose engine rejected every value but - // high), so a new reasoner must never inherit it — options need live probes first. + // Probes proved the advertised list unreliable: one catalog model advertised + // low|medium|high|max, and its engine rejected every value but high. A new + // reasoner must never inherit the list. The options need live probe evidence + // first. const entry = nebulEntry("zai-org/GLM-5.3"); expect(() => nebul.translateModel(entry, context(undefined))).toThrow(MissingReasoningOptionsError); }); @@ -66,9 +68,9 @@ test("preserves authored reasoning controls when the host exposes no efforts", ( }); test("keeps authored probe-verified controls over the advertised effort list", () => { - // Live probes (2026-09-23) showed the advertised list can be wrong: - // one catalog model advertised low|medium|high|max while its served engine - // rejected every value but high. + // Live probes on 2026-09-23 showed that the advertised list can be wrong: one + // catalog model advertised low|medium|high|max, and its served engine rejected + // every value but high. const authored = [{ type: "effort" as const, values: ["none", "high"] }]; const translated = nebul.translateModel(nebulEntry("zai-org/GLM-5.3"), context(existingWith(authored))); expect(translated?.model.reasoning_options).toEqual(authored); @@ -149,13 +151,14 @@ test("resolves base models across org renames and quantization suffixes", () => const cases: [string, string | null, string][] = [ ["zai-org/GLM-5.3", "zai-org/GLM-5.3", "zhipuai/glm-5.3"], ["Qwen/Qwen3.5-397B-A17B", "Qwen/Qwen3.5-397B-A17B", "alibaba/qwen3.5-397b-a17b"], - // Hugging Face org paths are case-insensitive; a lowercase org must resolve identically. + // Hugging Face org paths are case-insensitive, so a lowercase org must + // resolve identically. ["qwen/qwen3.5-397b-a17b", "qwen/qwen3.5-397b-a17b", "alibaba/qwen3.5-397b-a17b"], ["nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16", "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16", "nvidia/nemotron-3-super-120b-a12b"], ["mistralai/Mistral-Large-3-675B-Instruct-2512", "mistralai/Mistral-Large-3-675B-Instruct-2512", "mistral/mistral-large-2512"], ]; - // Authored options in context keep reasoner labs out of the fail-closed path - // (this case asserts base_model resolution only). + // The authored options in the context keep reasoner labs out of the + // fail-closed path. This case asserts base_model resolution only. for (const [model_name, huggingface_id, expected] of cases) { const entry = nebulEntry(model_name, { huggingface_id }); expect( @@ -248,9 +251,9 @@ test("fails closed on a partial catalog so sync cannot prune healthy local files }); test("accepts and ignores unknown advertised reasoning effort values", () => { - // The advertised list is never synced, so an unrecognized value must not abort - // parsing: a strict enum would fail the hourly run and block cost/context - // refreshes for the curated models. + // The sync never copies the advertised list, so an unrecognized value must + // not abort parsing. A strict enum fails the hourly run and blocks + // cost/context refreshes for the curated models. const parsed = NebulResponse.parse({ data: [{ model_name: "Some/Chat", model_info: { mode: "chat", model_type: "llm", reasoning_efforts: ["ultra"] } }], }); @@ -264,7 +267,7 @@ test("accepts and ignores unknown advertised reasoning effort values", () => { expect(translated?.model.reasoning_options).toEqual(authored); }); -// Nebul is a curated provider: exactly the four requested flagship models ship, +// Nebul is a curated provider. Exactly the four requested flagship models ship, // and the catalog sync must never add to or remove from that set. const CURATED_MODEL_IDS = [ "Qwen/Qwen3.5-397B-A17B", diff --git a/providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml b/providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml index e74f8d592f1..949f76302ce 100644 --- a/providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml +++ b/providers/nebul/models/Qwen/Qwen3.5-397B-A17B.toml @@ -1,9 +1,10 @@ -# Effort: reasoning_effort = none|low|medium|high; thinking is ON by default -# and none turns it off (all listed levels probe-verified 200 OK on 2026-09-23; set -# narrowed to the same-surface peer baseline (ovhcloud/scaleway/crof/digitalocean) — -# minimal/xhigh/max lacked distinct-effect evidence). The catalog's -# supports_reasoning = false for this ID is a false negative. -# Traces arrive in reasoning_content. Costs live-verified. +# Effort: reasoning_effort = none|low|medium|high. Thinking is ON by default, and +# none turns it off. Live probes on 2026-09-23 returned 200 OK for every listed +# level. The set matches the same-surface peer baseline (ovhcloud, scaleway, crof, +# digitalocean), because minimal, xhigh, and max showed no distinct effect. +# The catalog reports supports_reasoning = false for this ID, and that is a false +# negative. Traces arrive in reasoning_content. The costs match the Nebul API on +# 2026-09-23. base_model = "alibaba/qwen3.5-397b-a17b" [interleaved] diff --git a/providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml b/providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml index a0fd9e2b729..d34868d3cb8 100644 --- a/providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml +++ b/providers/nebul/models/mistralai/Mistral-Large-3-675B-Instruct-2512.toml @@ -1,4 +1,4 @@ -# Costs live-verified via https://api.inference.nebul.io/v1/model/info on 2026-09-23. +# The costs match https://api.inference.nebul.io/v1/model/info on 2026-09-23. base_model = "mistral/mistral-large-2512" [cost] diff --git a/providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml b/providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml index 598e335349f..f8295a2f74f 100644 --- a/providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml +++ b/providers/nebul/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16.toml @@ -1,8 +1,9 @@ -# Effort: reasoning_effort = none|low|medium|high|max; thinking is ON by default -# and none turns it off (all listed levels probe-verified 200 OK on 2026-09-23; set -# narrowed to the same-surface peer baseline — minimal/xhigh lacked distinct-effect -# evidence). The catalog's supports_reasoning = false for this ID is a false negative. -# Traces arrive in reasoning_content. Context/costs live-verified. +# Effort: reasoning_effort = none|low|medium|high|max. Thinking is ON by default, +# and none turns it off. Live probes on 2026-09-23 returned 200 OK for every +# listed level. The set matches the same-surface peer baseline, because minimal +# and xhigh showed no distinct effect. The catalog reports supports_reasoning = +# false for this ID, and that is a false negative. Traces arrive in +# reasoning_content. The context and costs match the Nebul API on 2026-09-23. base_model = "nvidia/nemotron-3-super-120b-a12b" [interleaved] diff --git a/providers/nebul/models/zai-org/GLM-5.3.toml b/providers/nebul/models/zai-org/GLM-5.3.toml index b61b191c937..423092675cf 100644 --- a/providers/nebul/models/zai-org/GLM-5.3.toml +++ b/providers/nebul/models/zai-org/GLM-5.3.toml @@ -1,6 +1,7 @@ -# Effort: reasoning_effort = low|high|max (catalog + zai/OpenRouter parity). -# Traces arrive in reasoning_content. -# Costs/context live-verified via https://api.inference.nebul.io/v1/model/info on 2026-09-23. +# Effort: reasoning_effort = low|high|max. The set matches the catalog and the zai +# and OpenRouter entries for the same model. Traces arrive in reasoning_content. +# The costs and context match https://api.inference.nebul.io/v1/model/info on +# 2026-09-23. base_model = "zhipuai/glm-5.3" [interleaved]