From 939b47fe10a6ce0a6fcfb6ca53a9d3eff5b418bc Mon Sep 17 00:00:00 2001 From: Roy Kollen Svendsen Date: Sun, 20 Sep 2026 01:56:17 +0200 Subject: [PATCH] feat(engy): add engy provider engy (api.engy.ai) relays open models over an OpenAI-compatible endpoint; each of the eight entries is override-only against its lab model. Prices, context and modalities come from the public GET /v1/models, the input/output split from the authenticated engy.ai/api/v1/models (context is their sum), and image input was checked by sending an image to each model. Reasoning controls were measured per model with paired prompts and a positive control, since they differ per deployment and no endpoint reports them: toggles on chat_template_kwargs (enable_thinking for glm-5.2 and Qwen, thinking for DeepSeek and Kimi), effort levels from the host's ladder and the same-surface peers, and no toggle on the GLM-5.3 pair, where nothing switches reasoning off. kimi-k3 overrides temperature = true, which engy honours and Moonshot's API does not. Each header carries its cap, sample sizes and p-values. Co-Authored-By: Claude Opus 5 Co-Authored-By: Claude Fable 5 --- providers/engy/logo.svg | 1 + .../engy/models/deepseek-v4-flash-0731.toml | 28 +++++++++++++++++ .../engy/models/deepseek-v4.1-flash.toml | 28 +++++++++++++++++ providers/engy/models/glm-5.2.toml | 27 +++++++++++++++++ providers/engy/models/glm-5.3-flash.toml | 27 +++++++++++++++++ providers/engy/models/glm-5.3.toml | 23 ++++++++++++++ providers/engy/models/kimi-k3.toml | 28 +++++++++++++++++ providers/engy/models/qwen3.6-35b-a3b.toml | 26 ++++++++++++++++ providers/engy/models/qwen3.8-27b.toml | 30 +++++++++++++++++++ providers/engy/provider.toml | 10 +++++++ 10 files changed, 228 insertions(+) create mode 100644 providers/engy/logo.svg create mode 100644 providers/engy/models/deepseek-v4-flash-0731.toml create mode 100644 providers/engy/models/deepseek-v4.1-flash.toml create mode 100644 providers/engy/models/glm-5.2.toml create mode 100644 providers/engy/models/glm-5.3-flash.toml create mode 100644 providers/engy/models/glm-5.3.toml create mode 100644 providers/engy/models/kimi-k3.toml create mode 100644 providers/engy/models/qwen3.6-35b-a3b.toml create mode 100644 providers/engy/models/qwen3.8-27b.toml create mode 100644 providers/engy/provider.toml diff --git a/providers/engy/logo.svg b/providers/engy/logo.svg new file mode 100644 index 00000000000..f3982100499 --- /dev/null +++ b/providers/engy/logo.svg @@ -0,0 +1 @@ + diff --git a/providers/engy/models/deepseek-v4-flash-0731.toml b/providers/engy/models/deepseek-v4-flash-0731.toml new file mode 100644 index 00000000000..faf04d33e6d --- /dev/null +++ b/providers/engy/models/deepseek-v4-flash-0731.toml @@ -0,0 +1,28 @@ +# Toggle: chat_template_kwargs.thinking = true|false +# Effort: reasoning_effort = low|high|max +# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-08-28/30). +# Reasoning is off unless asked; thinking=false and none suppress it 18/18. +# engy maps effort per template (engy.ai/docs, 2026-09-19): low/medium -> low, high -> high, xhigh/max -> max. +# Probe 2026-08-30, n=113-126/level at 64k: low, medium and high did not separate (p>=0.89); +# high vs max p=0.0064. +base_model = "deepseek/deepseek-v4-flash-0731" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.045 +output = 0.09 +cache_read = 0.009 + +[limit] +context = 1_048_576 +input = 920_576 +output = 128_000 diff --git a/providers/engy/models/deepseek-v4.1-flash.toml b/providers/engy/models/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..9a2ac0e3b67 --- /dev/null +++ b/providers/engy/models/deepseek-v4.1-flash.toml @@ -0,0 +1,28 @@ +# Toggle: chat_template_kwargs.thinking = true|false +# Effort: reasoning_effort = low|high|max +# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-09-12). +# Reasoning is off unless asked; thinking=false and none suppress it 40/40. +# engy maps effort per template (engy.ai/docs, 2026-09-19): low/medium -> low, high -> high, xhigh/max -> max. +# Probe 2026-09-12, n=40/level at 32k: low, medium and high did not separate (paired p>=0.04); +# high vs max paired p=1e-04. +base_model = "deepseek/deepseek-v4.1-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.04 +output = 0.08 +cache_read = 0.008 + +[limit] +context = 327_680 +input = 262_144 +output = 65_536 diff --git a/providers/engy/models/glm-5.2.toml b/providers/engy/models/glm-5.2.toml new file mode 100644 index 00000000000..44e65a8b1d1 --- /dev/null +++ b/providers/engy/models/glm-5.2.toml @@ -0,0 +1,27 @@ +# Toggle: chat_template_kwargs.enable_thinking = true|false +# Effort: reasoning_effort = high|max +# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-08-28/30). +# Live probe 2026-08-30, n=20/level at 8k: low, medium and high one rung (p>=0.38), xhigh and max +# another (p=0.19), the rungs apart at p<=2e-07 (MWU), xhigh and max hit the cap 6/20 and 5/20; +# toggle, none and minimal suppress reasoning 20/20. +base_model = "zhipuai/glm-5.2" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["high", "max"] + +[cost] +input = 0.68 +output = 1.5 +cache_read = 0.18 + +[limit] +context = 262_144 +input = 229_376 +output = 32_768 diff --git a/providers/engy/models/glm-5.3-flash.toml b/providers/engy/models/glm-5.3-flash.toml new file mode 100644 index 00000000000..b01900d0fc1 --- /dev/null +++ b/providers/engy/models/glm-5.3-flash.toml @@ -0,0 +1,27 @@ +# Effort: reasoning_effort = low|high|max +# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-08-28/30). +# Launch discount: page lists 0.15/0.50, billed 0.135/0.45. Served text+image, so input is narrowed. +# "(Ox Alpha)" is the codename it was served under before the name went public; status unset. +# Live probe 2026-08-30 at 8k, n=40 for low/medium/high, 20 for xhigh/max: low = medium (p=0.41) +# < high (p=1.8e-05) < xhigh = max; none/minimal/enable_thinking=false measure as low; 0/320 empty. +base_model = "zhipuai/glm-5.3-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.135 +output = 0.45 +cache_read = 0.027 + +[limit] +context = 262_144 +input = 229_376 +output = 32_768 + +[modalities] +input = ["text", "image"] diff --git a/providers/engy/models/glm-5.3.toml b/providers/engy/models/glm-5.3.toml new file mode 100644 index 00000000000..d8b78ac0bb2 --- /dev/null +++ b/providers/engy/models/glm-5.3.toml @@ -0,0 +1,23 @@ +# Effort: reasoning_effort = low|high|max +# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-08-28/31). +# Live probe 2026-08-30 at 8k, n=40 for low/medium/high, 20 for xhigh/max: low = medium (p=0.49) +# < high (p=3.4e-06) < xhigh = max (MWU). Nothing switches it off: none, minimal and +# enable_thinking=false measure as low; 0 of 320 empty. +base_model = "zhipuai/glm-5.3" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.98 +output = 3.08 +cache_read = 0.18 + +[limit] +context = 327_680 +input = 294_912 +output = 32_768 diff --git a/providers/engy/models/kimi-k3.toml b/providers/engy/models/kimi-k3.toml new file mode 100644 index 00000000000..4983705f007 --- /dev/null +++ b/providers/engy/models/kimi-k3.toml @@ -0,0 +1,28 @@ +# Toggle: chat_template_kwargs.thinking = true|false +# Prices: https://api.engy.ai/v1/models (2026-09-03, raised 30% from 08-28); limits: authed +# https://engy.ai/api/v1/models (2026-10-02; max_input was 983,040 on 2026-08-30; context is +# max_input + max_output, above the lab's 1M). engy serves this text+image only, so input is narrowed. +# Live probe 2026-08-30, n=40/level at 16k: reasoning_effort inert, every pair p>=0.61 (MWU); +# thinking=false suppresses reasoning 40/40. temperature honoured (0 vs 1.5, 12 prompts x 3 +# replicates, 12/12, paired p=0.0005, 2026-08-31); the lab's false is Moonshot's own API. +base_model = "moonshotai/kimi-k3" +temperature = true + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 1.95 +output = 9.75 +cache_read = 0.195 + +[limit] +context = 1_113_088 +input = 1_047_552 +output = 65_536 + +[modalities] +input = ["text", "image"] diff --git a/providers/engy/models/qwen3.6-35b-a3b.toml b/providers/engy/models/qwen3.6-35b-a3b.toml new file mode 100644 index 00000000000..afdc3d9c83c --- /dev/null +++ b/providers/engy/models/qwen3.6-35b-a3b.toml @@ -0,0 +1,26 @@ +# Toggle: chat_template_kwargs.enable_thinking = true|false +# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-08-28/30). +# engy serves this deployment text-only, so modalities and attachment are narrowed. +# Live probe 2026-08-31, n=36/level at 8,192 (the cap): reasoning_effort inert, every pair p>=0.39 +# (MWU); toggle and none suppress reasoning 36/36. No effort control is offered. +base_model = "alibaba/qwen3.6-35b-a3b" +attachment = false + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[cost] +input = 0.045 +output = 0.3 +cache_read = 0.015 + +[limit] +context = 208_192 +input = 200_000 +output = 8_192 + +[modalities] +input = ["text"] diff --git a/providers/engy/models/qwen3.8-27b.toml b/providers/engy/models/qwen3.8-27b.toml new file mode 100644 index 00000000000..ca2c6e87638 --- /dev/null +++ b/providers/engy/models/qwen3.8-27b.toml @@ -0,0 +1,30 @@ +# Toggle: chat_template_kwargs.enable_thinking = true|false +# Effort: reasoning_effort = low|medium|xhigh +# Prices: https://api.engy.ai/v1/models (2026-08-28/30); limits: authed https://engy.ai/api/v1/models (2026-09-13). +# engy serves this deployment text+image only, so modalities.input is narrowed. +# Live probe 2026-08-30, n=60/level at 32k: xhigh separates from low and medium (paired p<0.01), high +# and max behave as xhigh (p>=0.87), medium vs low unsettled (p=0.44); toggle and none suppress 20/20. +base_model = "alibaba/qwen3.8-27b" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 0.045 +output = 0.32 +cache_read = 0.015 + +[limit] +context = 1_001_536 +input = 936_000 +output = 65_536 + +[modalities] +input = ["text", "image"] diff --git a/providers/engy/provider.toml b/providers/engy/provider.toml new file mode 100644 index 00000000000..ef46175ba83 --- /dev/null +++ b/providers/engy/provider.toml @@ -0,0 +1,10 @@ +name = "engy" +env = ["ENGY_API_KEY"] +npm = "@ai-sdk/openai-compatible" +# Raw HTTP reasoning controls (live 2026-08-30), authored per model. POST /v1/chat/completions +# forwards chat_template_kwargs to the chat template (enable_thinking on glm-5.2 and Qwen, thinking +# on DeepSeek and Kimi; the GLM-5.3 templates do not switch it off) and takes reasoning_effort +# none|minimal|low|medium|high|xhigh|max; no budget field. Docs list no models; doc is pricing. +# Limits: authed GET https://engy.ai/api/v1/models; context = max_input + max_output. +api = "https://api.engy.ai/v1" +doc = "https://engy.ai/pricing"