Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 6 additions & 5 deletions providers/alibaba-cn/models/deepseek-v4.1-flash.toml
Original file line number Diff line number Diff line change
@@ -1,11 +1,12 @@
# CNY→USD rate: 6.721845, 2026-09-18, source: https://open.er-api.com/v6/latest/USD
# Sources (accessed 2026-09-18):
# https://bailian.console.aliyun.com/cn-beijing?tab=model#/model-market/detail/deepseek-v4.1-flash?serviceSite=asia-pacific-china
# https://help.aliyun.com/zh/model-studio/deepseek-api — capability table,
# reasoning_effort ladder, max_tokens default (393_216, shared with thinking)
# https://help.aliyun.com/zh/model-studio/billing-for-model-studio — Beijing
# peak/off-peak pricing: CNY 2/1 input, 8/4 output per 1M tokens with context
# cache discount; peak price converted to USD following the existing
# deepseek-v4-flash.toml convention on this host (off-peak is half).
# cache discount; peak price converted to USD using the rate above (off-peak
# is half).
# Hybrid reasoning: enable_thinking = true|false toggles thinking;
# reasoning_effort = low|high|max (default high) — low is only supported on
# v4.1-flash / v4-flash-0731 / v4-pro-0813, and v4.1-flash is the current
Expand All @@ -21,6 +22,6 @@ reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "h
field = "reasoning_content"

[cost]
input = 0.278
output = 1.111
cache_read = 0.014
input = 0.29754
output = 1.19015
cache_read = 0.01488
Loading