Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions providers/zerosignal/logo.svg
Loading
Sorry, something went wrong. Reload?
Sorry, we cannot display this file.
Sorry, this file is invalid so it cannot be displayed.
25 changes: 25 additions & 0 deletions providers/zerosignal/models/glm-5.2.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
# ZeroSignal is a multi-operator relay; operators set their own prices and serving
# limits. Values below were read from GET /v1/models on the local zs-proxy on 2026-09-16:
# cost is the lowest advertised operator rate, USD per 1M tokens, inclusive of the
# protocol fee (what the caller pays); the routed operator may charge differently.
# Effort values are the network's advertised `allowed_efforts` for this id. Any
# [limit] override is the advertised context_length / max_completion_tokens.
# https://docs.zerosignal.ai/for-users/pricing
# Measured 2026-09-16 (same prompt, temperature 0, max_tokens 4000, reasoning_content
# length): low and high return identical output (aliased); max is distinct; none
# still produces a full reasoning trace, so off is not available on this host. The
# operator advertises the full effort enum, but only high/max are distinct controls.
# Trace is returned in reasoning_content.
base_model = "zhipuai/glm-5.2"
reasoning_options = [{ type = "effort", values = ["high", "max"] }]

[interleaved]
field = "reasoning_content"

[cost]
input = 1.694
output = 5.324
cache_read = 0.3146

[limit]
output = 32_768
21 changes: 21 additions & 0 deletions providers/zerosignal/models/glm-5.3-flash.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
# ZeroSignal is a multi-operator relay; operators set their own prices and serving
# limits. Values below were read from GET /v1/models on the local zs-proxy on 2026-09-16:
# cost is the lowest advertised operator rate, USD per 1M tokens, inclusive of the
# protocol fee (what the caller pays); the routed operator may charge differently.
# Effort values are the network's advertised `allowed_efforts` for this id. Any
# [limit] override is the advertised context_length / max_completion_tokens.
# https://docs.zerosignal.ai/for-users/pricing
# Measured 2026-09-16 (same prompt, max_tokens 3000, reasoning tokens from usage):
# low/high/max accepted with graded traces; none, medium and xhigh are rejected by the
# host ("always engages in thinking and cannot be disabled"), so off is unavailable.
# Trace is returned in reasoning_content.
base_model = "zhipuai/glm-5.3-flash"
reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }]

[interleaved]
field = "reasoning_content"

[cost]
input = 0.09075
output = 0.3025
cache_read = 0.01815
24 changes: 24 additions & 0 deletions providers/zerosignal/models/glm-5.3.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
# ZeroSignal is a multi-operator relay; operators set their own prices and serving
# limits. Values below were read from GET /v1/models on the local zs-proxy on 2026-09-16:
# cost is the lowest advertised operator rate, USD per 1M tokens, inclusive of the
# protocol fee (what the caller pays); the routed operator may charge differently.
# Effort values are the network's advertised `allowed_efforts` for this id. Any
# [limit] override is the advertised context_length / max_completion_tokens.
# https://docs.zerosignal.ai/for-users/pricing
# Measured 2026-09-16 (same prompt, max_tokens 3000, reasoning tokens from usage):
# low/high/max accepted with graded traces; none, medium and xhigh are rejected by the
# host ("always engages in thinking and cannot be disabled"), so off is unavailable.
# Trace is returned in reasoning_content.
base_model = "zhipuai/glm-5.3"
reasoning_options = [{ type = "effort", values = ["low", "high", "max"] }]

[interleaved]
field = "reasoning_content"

[cost]
input = 1.694
output = 5.324
cache_read = 0.3146

[limit]
output = 32_768
19 changes: 19 additions & 0 deletions providers/zerosignal/models/google/gemini-3.7-flash.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
# ZeroSignal is a multi-operator relay; operators set their own prices and serving
# limits. Values below were read from GET /v1/models on the local zs-proxy on 2026-09-16:
# cost is the lowest advertised operator rate, USD per 1M tokens, inclusive of the
# protocol fee (what the caller pays); the routed operator may charge differently.
# Effort values are the network's advertised `allowed_efforts` for this id. Any
# [limit] override is the advertised context_length / max_completion_tokens.
# https://docs.zerosignal.ai/for-users/pricing
# Measured 2026-09-16: none is rejected ("Reasoning is mandatory for this endpoint"),
# so off is unavailable; low returns zero reasoning tokens, medium/high graded, and
# xhigh/max are accepted but do not exceed high, so the lab's set is kept. This
# operator returns the trace in a field named `reasoning`, which is not one of the
# schema's interleaved field values, so no [interleaved] is declared.
base_model = "google/gemini-3.7-flash"
reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }]

[cost]
input = 1.1
output = 5.5
cache_read = 0.11
19 changes: 19 additions & 0 deletions providers/zerosignal/models/google/gemini-3.8-flash.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
# ZeroSignal is a multi-operator relay; operators set their own prices and serving
# limits. Values below were read from GET /v1/models on the local zs-proxy on 2026-09-16:
# cost is the lowest advertised operator rate, USD per 1M tokens, inclusive of the
# protocol fee (what the caller pays); the routed operator may charge differently.
# Effort values are the network's advertised `allowed_efforts` for this id. Any
# [limit] override is the advertised context_length / max_completion_tokens.
# https://docs.zerosignal.ai/for-users/pricing
# Measured 2026-09-16: none is rejected ("Reasoning is mandatory for this endpoint"),
# so off is unavailable; low returns zero reasoning tokens, medium/high graded, and
# xhigh/max are accepted but do not exceed high, so the lab's set is kept. This
# operator returns the trace in a field named `reasoning`, which is not one of the
# schema's interleaved field values, so no [interleaved] is declared.
base_model = "google/gemini-3.8-flash"
reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }]

[cost]
input = 1.1
output = 5.5
cache_read = 0.11
23 changes: 23 additions & 0 deletions providers/zerosignal/models/grok-4.3.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
# ZeroSignal is a multi-operator relay; operators set their own prices and serving
# limits. Values below were read from GET /v1/models on the local zs-proxy on 2026-09-16:
# cost is the lowest advertised operator rate, USD per 1M tokens, inclusive of the
# protocol fee (what the caller pays); the routed operator may charge differently.
# Effort values are the network's advertised `allowed_efforts` for this id. Any
# [limit] override is the advertised context_length / max_completion_tokens.
# https://docs.zerosignal.ai/for-users/pricing
# Measured 2026-09-16: none returns no reasoning trace (off); graded levels think.
# Trace is returned in reasoning_content.
base_model = "xai/grok-4.3"
reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high"] }]

[interleaved]
field = "reasoning_content"

[cost]
input = 1.5125
output = 3.025
cache_read = 0.242

[limit]
context = 200_000
output = 32_768
23 changes: 23 additions & 0 deletions providers/zerosignal/models/grok-4.5.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
# ZeroSignal is a multi-operator relay; operators set their own prices and serving
# limits. Values below were read from GET /v1/models on the local zs-proxy on 2026-09-16:
# cost is the lowest advertised operator rate, USD per 1M tokens, inclusive of the
# protocol fee (what the caller pays); the routed operator may charge differently.
# Effort values are the network's advertised `allowed_efforts` for this id. Any
# [limit] override is the advertised context_length / max_completion_tokens.
# https://docs.zerosignal.ai/for-users/pricing
# Measured 2026-09-16: none and max are rejected by the host, so off is unavailable;
# the graded levels are accepted and all think (token counts within noise of each
# other, so the lab's set is kept). Trace is returned in reasoning_content.
base_model = "xai/grok-4.5"
reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }]

[interleaved]
field = "reasoning_content"

[cost]
input = 2.42
output = 7.26
cache_read = 0.363

[limit]
output = 32_768
23 changes: 23 additions & 0 deletions providers/zerosignal/models/grok-4.6.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
# ZeroSignal is a multi-operator relay; operators set their own prices and serving
# limits. Values below were read from GET /v1/models on the local zs-proxy on 2026-09-16:
# cost is the lowest advertised operator rate, USD per 1M tokens, inclusive of the
# protocol fee (what the caller pays); the routed operator may charge differently.
# Effort values are the network's advertised `allowed_efforts` for this id. Any
# [limit] override is the advertised context_length / max_completion_tokens.
# https://docs.zerosignal.ai/for-users/pricing
# Measured 2026-09-16: none and max are rejected by the host, so off is unavailable;
# the graded levels are accepted and all think (token counts within noise of each
# other, so the lab's set is kept). Trace is returned in reasoning_content.
base_model = "xai/grok-4.6"
reasoning_options = [{ type = "effort", values = ["low", "medium", "high", "xhigh"] }]

[interleaved]
field = "reasoning_content"

[cost]
input = 2.42
output = 7.26
cache_read = 0.363

[limit]
output = 32_768
25 changes: 25 additions & 0 deletions providers/zerosignal/models/kimi-k3.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
# ZeroSignal is a multi-operator relay; operators set their own prices and serving
# limits. Values below were read from GET /v1/models on the local zs-proxy on 2026-09-16:
# cost is the lowest advertised operator rate, USD per 1M tokens, inclusive of the
# protocol fee (what the caller pays); the routed operator may charge differently.
# Effort values are the network's advertised `allowed_efforts` for this id. Any
# [limit] override is the advertised context_length / max_completion_tokens.
# https://docs.zerosignal.ai/for-users/pricing
# Measured 2026-09-16: none returns no reasoning trace (off); low/high/max think.
# Off is effort=none, so no toggle. Moonshot's temperature rule applies through the
# proxy: thinking modes require temperature 1, none requires 0.6.
# Trace is returned in reasoning_content.
base_model = "moonshotai/kimi-k3"
reasoning_options = [{ type = "effort", values = ["none", "low", "high", "max"] }]

[interleaved]
field = "reasoning_content"

[cost]
input = 3.432
output = 17.16
cache_read = 0.3432

[limit]
context = 1_000_000
output = 32_768
18 changes: 18 additions & 0 deletions providers/zerosignal/models/openai/gpt-5.6-luna.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
# ZeroSignal is a multi-operator relay; operators set their own prices and serving
# limits. Values below were read from GET /v1/models on the local zs-proxy on 2026-09-16:
# cost is the lowest advertised operator rate, USD per 1M tokens, inclusive of the
# protocol fee (what the caller pays); the routed operator may charge differently.
# Effort values are the network's advertised `allowed_efforts` for this id. Any
# [limit] override is the advertised context_length / max_completion_tokens.
# https://docs.zerosignal.ai/for-users/pricing
# Measured 2026-09-16: none returns no reasoning trace (off); all graded levels are
# accepted by this host, so the set matches the lab entry.
# This operator returns the trace in a field named `reasoning`, not one of the schema's
# interleaved field values, so no [interleaved] is declared.
base_model = "openai/gpt-5.6-luna"
reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }]

[cost]
input = 0.319
output = 1.936
cache_read = 0.033
Loading
Loading