Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions providers/engy/logo.svg
Loading
Sorry, something went wrong. Reload?
Sorry, we cannot display this file.
Sorry, this file is invalid so it cannot be displayed.
28 changes: 28 additions & 0 deletions providers/engy/models/deepseek-v4-flash-0731.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
# Toggle: chat_template_kwargs.thinking = true|false
# Effort: reasoning_effort = low|high|max
# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-08-28/30).
# Reasoning is off unless asked; thinking=false and none suppress it 18/18.
# engy maps effort per template (engy.ai/docs, 2026-09-19): low/medium -> low, high -> high, xhigh/max -> max.
# Probe 2026-08-30, n=113-126/level at 64k: low, medium and high did not separate (p>=0.89);
# high vs max p=0.0064.
base_model = "deepseek/deepseek-v4-flash-0731"

[interleaved]
field = "reasoning_content"

[[reasoning_options]]
type = "toggle"

[[reasoning_options]]
type = "effort"
values = ["low", "high", "max"]

[cost]
input = 0.045
output = 0.09
cache_read = 0.009

[limit]
context = 1_048_576
input = 920_576
output = 128_000
28 changes: 28 additions & 0 deletions providers/engy/models/deepseek-v4.1-flash.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
# Toggle: chat_template_kwargs.thinking = true|false
# Effort: reasoning_effort = low|high|max
# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-09-12).
# Reasoning is off unless asked; thinking=false and none suppress it 40/40.
# engy maps effort per template (engy.ai/docs, 2026-09-19): low/medium -> low, high -> high, xhigh/max -> max.
# Probe 2026-09-12, n=40/level at 32k: low, medium and high did not separate (paired p>=0.04);
# high vs max paired p=1e-04.
base_model = "deepseek/deepseek-v4.1-flash"

[interleaved]
field = "reasoning_content"

[[reasoning_options]]
type = "toggle"

[[reasoning_options]]
type = "effort"
values = ["low", "high", "max"]

[cost]
input = 0.04
output = 0.08
cache_read = 0.008

[limit]
context = 327_680
input = 262_144
output = 65_536
27 changes: 27 additions & 0 deletions providers/engy/models/glm-5.2.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
# Toggle: chat_template_kwargs.enable_thinking = true|false
# Effort: reasoning_effort = high|max
# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-08-28/30).
# Live probe 2026-08-30, n=20/level at 8k: low, medium and high one rung (p>=0.38), xhigh and max
# another (p=0.19), the rungs apart at p<=2e-07 (MWU), xhigh and max hit the cap 6/20 and 5/20;
# toggle, none and minimal suppress reasoning 20/20.
base_model = "zhipuai/glm-5.2"

[interleaved]
field = "reasoning_content"

[[reasoning_options]]
type = "toggle"

[[reasoning_options]]
type = "effort"
values = ["high", "max"]

[cost]
input = 0.68
output = 1.5
cache_read = 0.18

[limit]
context = 262_144
input = 229_376
output = 32_768
27 changes: 27 additions & 0 deletions providers/engy/models/glm-5.3-flash.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
# Effort: reasoning_effort = low|high|max
# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-08-28/30).
# Launch discount: page lists 0.15/0.50, billed 0.135/0.45. Served text+image, so input is narrowed.
# "(Ox Alpha)" is the codename it was served under before the name went public; status unset.
# Live probe 2026-08-30 at 8k, n=40 for low/medium/high, 20 for xhigh/max: low = medium (p=0.41)
# < high (p=1.8e-05) < xhigh = max; none/minimal/enable_thinking=false measure as low; 0/320 empty.
base_model = "zhipuai/glm-5.3-flash"

[interleaved]
field = "reasoning_content"

[[reasoning_options]]
type = "effort"
values = ["low", "high", "max"]

[cost]
input = 0.135
output = 0.45
cache_read = 0.027

[limit]
context = 262_144
input = 229_376
output = 32_768

[modalities]
input = ["text", "image"]
23 changes: 23 additions & 0 deletions providers/engy/models/glm-5.3.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
# Effort: reasoning_effort = low|high|max
# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-08-28/31).
# Live probe 2026-08-30 at 8k, n=40 for low/medium/high, 20 for xhigh/max: low = medium (p=0.49)
# < high (p=3.4e-06) < xhigh = max (MWU). Nothing switches it off: none, minimal and
# enable_thinking=false measure as low; 0 of 320 empty.
base_model = "zhipuai/glm-5.3"

[interleaved]
field = "reasoning_content"

[[reasoning_options]]
type = "effort"
values = ["low", "high", "max"]

[cost]
input = 0.98
output = 3.08
cache_read = 0.18

[limit]
context = 327_680
input = 294_912
output = 32_768
28 changes: 28 additions & 0 deletions providers/engy/models/kimi-k3.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
# Toggle: chat_template_kwargs.thinking = true|false
# Prices: https://api.engy.ai/v1/models (2026-09-03, raised 30% from 08-28); limits: authed
# https://engy.ai/api/v1/models (2026-10-02; max_input was 983,040 on 2026-08-30; context is
# max_input + max_output, above the lab's 1M). engy serves this text+image only, so input is narrowed.
# Live probe 2026-08-30, n=40/level at 16k: reasoning_effort inert, every pair p>=0.61 (MWU);
# thinking=false suppresses reasoning 40/40. temperature honoured (0 vs 1.5, 12 prompts x 3
# replicates, 12/12, paired p=0.0005, 2026-08-31); the lab's false is Moonshot's own API.
base_model = "moonshotai/kimi-k3"
temperature = true

[interleaved]
field = "reasoning_content"

[[reasoning_options]]
type = "toggle"

[cost]
input = 1.95
output = 9.75
cache_read = 0.195

[limit]
context = 1_113_088
input = 1_047_552
output = 65_536

[modalities]
input = ["text", "image"]
26 changes: 26 additions & 0 deletions providers/engy/models/qwen3.6-35b-a3b.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,26 @@
# Toggle: chat_template_kwargs.enable_thinking = true|false
# Prices: https://api.engy.ai/v1/models; limits: authed https://engy.ai/api/v1/models (2026-08-28/30).
# engy serves this deployment text-only, so modalities and attachment are narrowed.
# Live probe 2026-08-31, n=36/level at 8,192 (the cap): reasoning_effort inert, every pair p>=0.39
# (MWU); toggle and none suppress reasoning 36/36. No effort control is offered.
base_model = "alibaba/qwen3.6-35b-a3b"
attachment = false

[interleaved]
field = "reasoning_content"

[[reasoning_options]]
type = "toggle"

[cost]
input = 0.045
output = 0.3
cache_read = 0.015

[limit]
context = 208_192
input = 200_000
output = 8_192

[modalities]
input = ["text"]
30 changes: 30 additions & 0 deletions providers/engy/models/qwen3.8-27b.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
# Toggle: chat_template_kwargs.enable_thinking = true|false
# Effort: reasoning_effort = low|medium|xhigh
# Prices: https://api.engy.ai/v1/models (2026-08-28/30); limits: authed https://engy.ai/api/v1/models (2026-09-13).
# engy serves this deployment text+image only, so modalities.input is narrowed.
# Live probe 2026-08-30, n=60/level at 32k: xhigh separates from low and medium (paired p<0.01), high
# and max behave as xhigh (p>=0.87), medium vs low unsettled (p=0.44); toggle and none suppress 20/20.
base_model = "alibaba/qwen3.8-27b"

[interleaved]
field = "reasoning_content"

[[reasoning_options]]
type = "toggle"

[[reasoning_options]]
type = "effort"
values = ["low", "medium", "xhigh"]

[cost]
input = 0.045
output = 0.32
cache_read = 0.015

[limit]
context = 1_001_536
input = 936_000
output = 65_536

[modalities]
input = ["text", "image"]
10 changes: 10 additions & 0 deletions providers/engy/provider.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
name = "engy"
env = ["ENGY_API_KEY"]
npm = "@ai-sdk/openai-compatible"
# Raw HTTP reasoning controls (live 2026-08-30), authored per model. POST /v1/chat/completions
# forwards chat_template_kwargs to the chat template (enable_thinking on glm-5.2 and Qwen, thinking
# on DeepSeek and Kimi; the GLM-5.3 templates do not switch it off) and takes reasoning_effort
# none|minimal|low|medium|high|xhigh|max; no budget field. Docs list no models; doc is pricing.
# Limits: authed GET https://engy.ai/api/v1/models; context = max_input + max_output.
api = "https://api.engy.ai/v1"
doc = "https://engy.ai/pricing"
Loading