Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
131 commits
Select commit Hold shift + click to select a range
4fa2a2f
feat(srt): run single-node AgentX on native srt-slurm
cquil11 Sep 25, 2026
765e70f
fix(amd): drop a duplicate DSV4 prefill key and restore the MI355X Qw…
cquil11 Sep 25, 2026
f0d4219
feat(srt): accept vLLM points in the single-node adapter
cquil11 Sep 25, 2026
e9667ed
feat(agentx): port the DSV4.1 Flash B300 SGLang AgentX config to srt-…
cquil11 Sep 25, 2026
bbe361c
feat(agentx): let srt-slurm AgentX recipes apply the chat template cl…
cquil11 Sep 25, 2026
4fce20b
feat(agentx): run GB200/GB300 single-node AgentX natively and port DS…
cquil11 Sep 25, 2026
90dc9ef
feat(agentx): let srt-slurm AgentX recipes apply the chat template cl…
cquil11 Sep 25, 2026
5374133
feat(agentx): let srt-slurm AgentX recipes apply the chat template cl…
cquil11 Sep 25, 2026
76e3cc5
feat(agentx): let srt-slurm AgentX recipes set the AIPerf benchmark g…
cquil11 Sep 25, 2026
307487b
feat(agentx): port the DSV4.1 Flash B200 vLLM AgentX config to srt-slurm
cquil11 Sep 25, 2026
a743687
feat(srt): force ATOM AgentX golden acceptance with its server flag
cquil11 Sep 25, 2026
0b6114d
feat(srt): accept draft_model speculation in the single-node adapter
cquil11 Sep 25, 2026
d0f755b
fix(mi355x): mount the shared HF cache for native AgentX checkpoints
cquil11 Sep 25, 2026
e664992
feat(agentx): port the Qwen3.5 MI300X SGLang AgentX config to srt-slurm
cquil11 Sep 25, 2026
69b5855
feat(agentx): port the Qwen3.5 MI325X SGLang AgentX config to srt-slurm
cquil11 Sep 25, 2026
b4931b8
feat(agentx): port the GLM-5.2 MI325X SGLang AgentX config to srt-slurm
cquil11 Sep 25, 2026
2857142
feat(agentx): port the DSV4.1 Flash H100 SGLang AgentX configs to srt…
cquil11 Sep 25, 2026
376cbc9
feat(agentx): port the Qwen3.5 H100 SGLang AgentX configs to srt-slurm
cquil11 Sep 25, 2026
428972e
feat(agentx): port the Qwen3.5 H200 SGLang HiCache EP1 AgentX config …
cquil11 Sep 25, 2026
e655a95
feat(srt): accept the draft_model label for single-node DSpark points
cquil11 Sep 25, 2026
72a734f
feat(agentx): port the DSV4.1 Flash MI300X vLLM AgentX config to srt-…
cquil11 Sep 25, 2026
4331a36
feat(agentx): port the DSV4.1 Flash MI325X vLLM AgentX config to srt-…
cquil11 Sep 25, 2026
fd37fe5
fix(srt): bind single-node draft_model and EAGLE3 AgentX points
cquil11 Sep 25, 2026
6d086e6
feat(srt): accept vLLM EAGLE3 speculation in the single-node adapter
cquil11 Sep 25, 2026
aa8d182
feat(srt): accept EAGLE3 speculation in the single-node adapter
cquil11 Sep 25, 2026
3266a31
feat(srt): accept EAGLE3 speculation in the single-node adapter
cquil11 Sep 25, 2026
fd82c45
feat(agentx): port the MiniMax-M3 MI325X vLLM AgentX config to srt-slurm
cquil11 Sep 25, 2026
0e99fba
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
5939066
feat(agentx): port the Qwen3.5 FP4 B300 SGLang AgentX config to srt-s…
cquil11 Sep 25, 2026
fa3b481
feat(agentx): port the Qwen3.5 FP8 B300 SGLang AgentX config to srt-s…
cquil11 Sep 25, 2026
d8e5ba5
feat(agentx): port the GLM-5.2 FP4 B300 SGLang AgentX config to srt-s…
cquil11 Sep 25, 2026
8f258c5
feat(agentx): port the GLM-5.2 FP8 B300 SGLang AgentX config to srt-s…
cquil11 Sep 25, 2026
931b3c6
feat(agentx): port the DSV4 Pro B300 SGLang AgentX config to srt-slurm
cquil11 Sep 25, 2026
20d3dd1
feat(agentx): port the MiniMax-M3 B300 TRT-LLM AgentX config to srt-s…
cquil11 Sep 25, 2026
05aa089
fix(b300): serve Qwen3.8-Flash-Next NVFP4 natively from the writable …
cquil11 Sep 25, 2026
ed2f0f0
feat(agentx): port the Qwen3.8-Flash-Next FP4 B300 SGLang AgentX conf…
cquil11 Sep 25, 2026
d5278fb
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
a5ecd11
feat(srt): accept draft_model speculation in the single-node adapter
cquil11 Sep 25, 2026
3f24127
feat(srt): force ATOM AgentX golden acceptance with its server flag
cquil11 Sep 25, 2026
84fdf52
fix(mi355x): mount the shared HF cache for native AgentX checkpoints
cquil11 Sep 25, 2026
6395aec
feat(srt): accept draft-model labeled speculation in the single-node …
cquil11 Sep 25, 2026
fab28a4
feat(agentx): port the DSV4.1 Flash H100 and H200 vLLM AgentX configs…
cquil11 Sep 25, 2026
f9f470b
feat(agentx): port the MiniMax-M3 H100 and H200 vLLM AgentX configs t…
cquil11 Sep 25, 2026
2298f0e
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
768e383
feat(agentx): port the Qwen3.5 MI355X SGLang AgentX config to srt-slurm
cquil11 Sep 25, 2026
dddcf1f
feat(agentx): port the DSV4.1 Flash MI355X SGLang AgentX config to sr…
cquil11 Sep 25, 2026
775e639
feat(agentx): port the GLM-5.2 FP4 MI355X SGLang AgentX config to srt…
cquil11 Sep 25, 2026
84d67d1
feat(agentx): port the GLM-5.2 FP8 MI355X SGLang AgentX config to srt…
cquil11 Sep 25, 2026
4625b67
feat(agentx): port the DSV4 MI355X SGLang AgentX config to srt-slurm
cquil11 Sep 25, 2026
31825ce
feat(agentx): port the DSV4 MI355X ATOM AgentX config to srt-slurm
cquil11 Sep 25, 2026
9cfaf0a
feat(agentx): port the GLM-5.2 MI355X ATOM AgentX config to srt-slurm
cquil11 Sep 25, 2026
d133cbc
feat(agentx): port the Kimi-K3 MI355X ATOM AgentX config to srt-slurm
cquil11 Sep 25, 2026
269e704
feat(agentx): port the MiniMax-M3 MI355X ATOM AgentX config to srt-slurm
cquil11 Sep 25, 2026
4d56892
feat(srt): accept EAGLE3 speculation in the single-node adapter
cquil11 Sep 25, 2026
208ef8a
feat(agentx): port the DSV4.1 Flash B300 vLLM AgentX config to srt-slurm
cquil11 Sep 25, 2026
5e9a32e
feat(agentx): port the MiniMax-M3 B200 and B300 vLLM AgentX configs t…
cquil11 Sep 25, 2026
a5ce1b7
feat(agentx): port the DSV4 B200 and B300 vLLM AgentX configs to srt-…
cquil11 Sep 25, 2026
556e3d0
feat(agentx): port the Qwen3.5 FP4 B200 SGLang AgentX config to srt-s…
cquil11 Sep 25, 2026
f2b229d
feat(agentx): port the Qwen3.5 FP8 B200 SGLang AgentX config to srt-s…
cquil11 Sep 25, 2026
f4c601d
feat(agentx): port the Qwen3.8 Next FP4 B200 SGLang AgentX config to …
cquil11 Sep 25, 2026
c084079
fix(h100): point native single-node uv caches at shared NFS
cquil11 Sep 25, 2026
b6d744f
feat(agentx): port the GLM-5.2 FP4 B200 SGLang AgentX config to srt-s…
cquil11 Sep 25, 2026
71c94da
feat(agentx): port the GLM-5.2 FP8 B200 SGLang AgentX config to srt-s…
cquil11 Sep 25, 2026
2166e4c
feat(agentx): port the DSV4 FP4 B200 SGLang AgentX config to srt-slurm
cquil11 Sep 25, 2026
a37f73c
feat(agentx): port the MiniMax-M3 FP4 B200 TRT-LLM AgentX config to s…
cquil11 Sep 25, 2026
d245dd1
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
aee6c29
feat(agentx): let srt-slurm AgentX recipes set the AIPerf benchmark g…
cquil11 Sep 25, 2026
ebdd4e6
feat(agentx): port the DSV4.1 Flash GB200 and GB300 vLLM AgentX confi…
cquil11 Sep 25, 2026
0ebfdad
feat(agentx): port the MiniMax-M3 MI300X vLLM AgentX config to srt-slurm
cquil11 Sep 25, 2026
1a53f2b
feat(agentx): port the DSV4.1 Flash MI355X vLLM AgentX config to srt-…
cquil11 Sep 25, 2026
286801b
feat(agentx): port the MiniMax-M3 MI355X vLLM AgentX config to srt-slurm
cquil11 Sep 25, 2026
68d9a24
feat(srt): accept vLLM decode context parallelism and non-drafting po…
cquil11 Sep 25, 2026
7152530
feat(agentx): port the Kimi-K3 B300 vLLM AgentX config to srt-slurm
cquil11 Sep 25, 2026
e7bb22b
feat(agentx): port the Kimi-K3 MI355X vLLM AgentX DCP1 points to srt-…
cquil11 Sep 25, 2026
2e668e6
fix(matrix): read node counts from named srt-slurm override variants
cquil11 Sep 25, 2026
e236d7f
refactor(agentx): consolidate DSV4 multi-node AgentX recipes into ove…
cquil11 Sep 25, 2026
c08474c
refactor(agentx): consolidate GLM-5.2 multi-node AgentX recipes into …
cquil11 Sep 25, 2026
658f8c0
refactor(agentx): consolidate Kimi-K3 GB200 AgentX recipes into overr…
cquil11 Sep 25, 2026
9b233f1
refactor(agentx): consolidate MiniMax-M3 multi-node AgentX recipes in…
cquil11 Sep 25, 2026
39960de
refactor(agentx): consolidate Qwen3.5 multi-node AgentX recipes into …
cquil11 Sep 25, 2026
9df3b31
docs(recipes): describe AgentX override-variant bundles
cquil11 Sep 25, 2026
94d12a6
feat(agentx): port the DSV4 MI355X vLLM AgentX config to srt-slurm
cquil11 Sep 25, 2026
73aa755
fix(agentx): give the MiniMax-M3 Hopper Mooncake master time to install
cquil11 Sep 25, 2026
386abc9
fix(agentx): give the AMD SGLang AgentX recipes an hour to become hea…
cquil11 Sep 25, 2026
3fe468e
fix(agentx): extend the DSV4.1 Flash H100 SGLang health window for co…
cquil11 Sep 25, 2026
d5bc165
fix(agentx): extend the MiniMax-M3 H100 vLLM load window for cold NFS…
cquil11 Sep 25, 2026
1e5db28
fix(h100): keep one uv cache per runner for native single-node jobs
cquil11 Sep 25, 2026
9cb97f9
fix(b300): give srt-slurm jobs the workflow time limit instead of srt…
cquil11 Sep 25, 2026
c673cdc
fix(agentx): report RDMA port states when the Kimi-K3 B300 Mooncake r…
cquil11 Sep 25, 2026
b93c407
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
ce64145
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
e6382ec
fix(mi355x): let native single-node points use a squash staged on /it…
cquil11 Sep 25, 2026
6cbbab5
fix(agentx): probe DSXE rdmap RDMA rails for the Kimi-K3 B300 Mooncak…
cquil11 Sep 25, 2026
2d43cdd
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
968f5e4
fix(srt): let single-node AgentX evals run without a fixed-sequence c…
cquil11 Sep 25, 2026
c0cc3cc
fix(srt): evaluate single-node AgentX points with the workflow's fram…
cquil11 Sep 25, 2026
9ec5207
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
c76c373
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
b4bb9de
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
acf6343
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
14da934
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
ecca6a0
fix(b300): download Qwen3.8-Flash-Next NVFP4 into the shared HF cache
cquil11 Sep 25, 2026
71465e4
fix(srt): stage single-node AgentX eval artifacts once
cquil11 Sep 25, 2026
6393719
fix(srt): stage single-node AgentX eval artifacts once
cquil11 Sep 25, 2026
1a4e8db
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
bd9b8a4
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
2b0e684
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
c979310
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
35b6dba
fix(b300): read models without node-local staging from the shared mod…
cquil11 Sep 25, 2026
0a720c5
fix(agentx): give GB200/GB300 DSV4.1 Flash servers a two-hour health …
cquil11 Sep 25, 2026
9656676
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
60fc401
Merge H100/H200 AgentX single-node ports
cquil11 Sep 25, 2026
98564d7
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
e84e267
Merge AMD vLLM and MI300X/MI325X SGLang AgentX single-node ports
cquil11 Sep 25, 2026
413883d
chore(agentx): keep the pruned-image MI355X vLLM configs on their leg…
cquil11 Sep 25, 2026
2cf876d
Merge B200 SGLang/TRT AgentX single-node ports
cquil11 Sep 25, 2026
9585eb1
Merge MI355X SGLang/ATOM AgentX single-node ports
cquil11 Sep 25, 2026
51f5c95
fix(agentx): give the DSV4.1 Flash GB200 and GB300 vLLM recipes a two…
cquil11 Sep 25, 2026
85700c3
fix(agentx): let Mooncake pick the GID on DSXE InfiniBand rails for K…
cquil11 Sep 25, 2026
5564685
Merge NVIDIA vLLM AgentX single-node ports
cquil11 Sep 25, 2026
9150c70
fix(agentx): give the MiniMax-M3 B300 TRT-LLM server a two-hour healt…
cquil11 Sep 25, 2026
02ca714
style(agentx): write srt-slurm recipe overrides as block YAML
cquil11 Sep 25, 2026
49efaa1
Merge remote-tracking branch 'origin/agent/srt-agentx-port' into agen…
cquil11 Sep 25, 2026
ff0859d
fix(agentx): give the Qwen3.8-Flash-Next B300 SGLang server a two-hou…
cquil11 Sep 25, 2026
2a88c4b
fix(srt): keep GLM-5.2's 150-step SWE-bench budget for single-node Ag…
cquil11 Sep 25, 2026
478d5c4
fix(agentx): give the MiniMax-M3 B200 TRT-LLM server a two-hour healt…
cquil11 Sep 25, 2026
997e9f4
fix(b200): default the Qwen3.8-Flash-Next NVFP4 path when no pool set…
cquil11 Sep 25, 2026
d7349b9
Merge B300 AgentX follow-ups: MiniMax-M3 TRT and Qwen3.8-Next health …
cquil11 Sep 25, 2026
730ffb9
chore(minimaxm3): drop the vLLM SimpleCPUOffload patch
cquil11 Sep 25, 2026
e599563
chore: merge main into AgentX SRT migration
adibarra Sep 26, 2026
8fc9551
style(srt): format single-node adapter
adibarra Sep 26, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
201 changes: 0 additions & 201 deletions benchmarks/multi_node/agentic/dsv4_fp4_mi355x_sglang-disagg.sh

This file was deleted.

4 changes: 2 additions & 2 deletions benchmarks/multi_node/amd_utils/trace_replay.sh
Original file line number Diff line number Diff line change
Expand Up @@ -73,7 +73,7 @@ PORT="${ROUTER_PORT}"
check_env_vars DURATION RESULT_FILENAME FLUSH_DRAIN_TIMEOUT CLEAR_CACHE_BETWEEN_CONC
export MODEL DURATION MAX_MODEL_LEN
# The workflow guard / upload steps expect one "${RESULT_FILENAME}_conc<N>.json" per
# concurrency, so each conc below is suffixed with _conc<N> (as agentic_srt.sh does).
# concurrency, so each conc below is suffixed with _conc<N> (as srt_agentic.sh does).
RESULT_FILENAME_BASE="${RESULT_FILENAME}"

mkdir -p "$RESULT_DIR"
Expand Down Expand Up @@ -103,7 +103,7 @@ for max_concurrency in "${chosen_concurrencies[@]}"; do

# benchmark-multinode-tmpl.yml expects the per-conc nesting (LOGS/agentic/conc_*/...)
# even though CI runs one concurrency per job; nesting also keeps local multi-conc
# sweeps from overwriting each other (same layout as agentic_srt.sh).
# sweeps from overwriting each other (same layout as srt_agentic.sh).
CONC_RESULT_DIR="$RESULT_DIR/conc_${max_concurrency}"
mkdir -p "$CONC_RESULT_DIR"

Expand Down
4 changes: 2 additions & 2 deletions benchmarks/multi_node/srt-slurm-recipes/RECIPES.md
Original file line number Diff line number Diff line change
Expand Up @@ -15,14 +15,14 @@ Store every recipe at `<model-prefix>/<engine>/<gpu>-<precision>/<workload>/<rec
```text
dsr1/sglang/b200-fp4/8k1k/disagg-stp-mtp-variants.yaml
glm5.2/sglang/h200-fp8/agentx/disagg-1p1d-pcp8-tp8-dp8-mtp6-hicache.yaml
qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p7d-dep4-tep8-c7-b1-mtp-kvoffload.yaml
qwen3.5/trtllm/gb300-fp4/agentx/disagg-variants.yaml
```

- Use the master config's `model-prefix` and `precision` labels. Engines are `sglang`, `vllm`, `trtllm`, and `tilert`; frontend selection remains explicit inside the recipe. Hardware directories use GPU types such as `b200` and `gb300`, rather than cluster names.
- Workloads are `1k1k`, `8k1k`, or `agentx`. Existing bundles spanning several fixed sequence lengths use `fixed-seq-len`; keep their override selectors intact.
- Use lowercase, hyphen-separated filenames beginning with `agg` or `disagg`. Include topology and the settings that distinguish sibling recipes, such as parallelism, batch size, concurrency, MTP, offload, or cache configuration. Avoid dates, numbered latency/throughput labels, and repeating the model or hardware already in the path.
- In topology names, `1p4d` denotes prefill/decode worker counts, not necessarily physical nodes. Role-qualified `p-tp4` and `d-tp8` identify prefill/decode TP; `b` denotes batch size and `c` concurrency. The YAML is authoritative for runtime settings.
- Name override bundles `*-variants.yaml`. Keep distinct sweep entry files separate even when their contents match: recipe paths participate in eval grouping. The Qwen3.5 `*-stp-sweep.yaml` and `*-mtp-sweep.yaml` pair preserves that existing distinction.
- Name override bundles `*-variants.yaml`. Multi-node AgentX recipes that differ only per configuration share one bundle per master-config entry, usually `agg-variants.yaml` or `disagg-variants.yaml`: `base` holds the shared settings and each former recipe becomes a named `override_<name>` block holding only its differences (plain overrides, not `zip_override_*`). Master entries select one with `CONFIG_FILE=recipes/<dir>/<bundle>.yaml:override_<name>`. Recipes read as text by a launcher, such as power recipes with top-level `telemetry:`, stay standalone. Keep distinct sweep entry files separate even when their contents match: recipe paths participate in eval grouping. The Qwen3.5 `*-stp-sweep.yaml` and `*-mtp-sweep.yaml` pair preserves that existing distinction.
- Update `CONFIG_FILE` and `EVAL_CONFIG_FILE` references in active and deprecated master configs, launcher path rules, workflow filters, and local documentation together when moving a file. Preserve upstream source URLs as provenance and leave historical performance-changelog entries unchanged. No aliases for the old layout are provided.

Shared runtime assets stay under `configs/` beside the model directories; they are not standalone recipes. The four files in `configs/dsv4-moe-load-balancer-configs/` are copied verbatim from NVIDIA/srt-slurm commit `deb1dfd9934398664f92d194169c183e009da83b`, preserving the EPLB initial expert assignments used by 17 DSV4 TRT recipes. `setup_srt_slurm()` stages them into the job checkout's `configs/` directory for the recipes' bind mounts. Keeping a recipe in this tree does not activate it; the master configs determine the benchmark matrix.
Expand Down
4 changes: 2 additions & 2 deletions benchmarks/multi_node/srt-slurm-recipes/RECIPES_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -15,14 +15,14 @@ InferenceX 要求 srt-slurm 2.0 或更新版本,且配置必须声明 `schema:
```text
dsr1/sglang/b200-fp4/8k1k/disagg-stp-mtp-variants.yaml
glm5.2/sglang/h200-fp8/agentx/disagg-1p1d-pcp8-tp8-dp8-mtp6-hicache.yaml
qwen3.5/trtllm/gb300-fp4/agentx/disagg-1p7d-dep4-tep8-c7-b1-mtp-kvoffload.yaml
qwen3.5/trtllm/gb300-fp4/agentx/disagg-variants.yaml
```

- 使用主配置中的 `model-prefix` 和 `precision` 标签。引擎目录为 `sglang`、`vllm`、`trtllm` 或 `tilert`;前端仍在配置内显式声明。硬件目录使用 `b200`、`gb300` 等 GPU 型号,不使用集群名称。
- 工作负载目录为 `1k1k`、`8k1k` 或 `agentx`。已有的跨序列长度配置集合放在 `fixed-seq-len` 下,保留其覆盖项选择器。
- 文件名使用小写字母和连字符,以 `agg` 或 `disagg` 开头。包含拓扑及用于区分同目录配置的关键参数,例如并行方式、批大小、并发数、MTP、卸载或缓存设置。避免日期、带序号的延迟/吞吐量标签,以及重复目录中已有的模型或硬件信息。
- 拓扑名中的 `1p4d` 表示预填充/解码 worker 数,不一定等于物理节点数。`p-tp4` 和 `d-tp8` 分别标识预填充和解码 TP;`b` 表示批大小,`c` 表示并发数。运行参数以 YAML 为准。
- 覆盖项集合使用 `*-variants.yaml` 命名。即使内容相同,也保留独立扫描入口:配置路径参与评估分组。Qwen3.5 的 `*-stp-sweep.yaml` 和 `*-mtp-sweep.yaml` 保留了这一既有区别。
- 覆盖项集合使用 `*-variants.yaml` 命名。仅在各配置间存在差异的多节点 AgentX 配置,按主配置条目合并为一个集合,通常为 `agg-variants.yaml` 或 `disagg-variants.yaml`:`base` 保存共享设置,每个原配置成为一个具名 `override_<name>` 块,只包含其差异(使用普通覆盖项,而非 `zip_override_*`)。主配置通过 `CONFIG_FILE=recipes/<dir>/<bundle>.yaml:override_<name>` 选择其一。启动器以文本方式读取的配置(例如带顶层 `telemetry:` 的功耗配置)保持独立文件。即使内容相同,也保留独立扫描入口:配置路径参与评估分组。Qwen3.5 的 `*-stp-sweep.yaml` 和 `*-mtp-sweep.yaml` 保留了这一既有区别。
- 移动文件时,同步更新当前及已弃用主配置中的 `CONFIG_FILE`、`EVAL_CONFIG_FILE`,以及启动器路径规则、工作流过滤器和本地文档。保留上游来源 URL,并保持历史性能变更日志不变。不为旧目录结构提供别名。

共享运行时资源保留在模型目录旁的 `configs/` 中,不属于独立基准测试配置。`configs/dsv4-moe-load-balancer-configs/` 中的四个文件原样取自 NVIDIA/srt-slurm 提交 `deb1dfd9934398664f92d194169c183e009da83b`,保留了 17 个 DSV4 TRT 配置使用的 EPLB 初始专家分配。`setup_srt_slurm()` 将这些文件复制到作业仓库的 `configs/` 目录,供配置中的绑定挂载使用。将配置文件放入本目录不会启用该配置;实际基准测试矩阵由主配置决定。
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
#!/usr/bin/env bash
# Install the measured H100 DeepSeek-V4.1-Flash block-32 tilings into the worker's SGLang.
set -euo pipefail
agentic=/infmax-workspace/benchmarks/single_node/agentic
python3 "$agentic/install_h100_block32_configs.py" "$agentic/kernel_configs/h100_dsv41_block32" /logs
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
#!/usr/bin/env bash
# Install the measured H200 DeepSeek-V4.1-Flash block-32 tilings into the worker's SGLang.
set -euo pipefail
agentic=/infmax-workspace/benchmarks/single_node/agentic
python3 "$agentic/install_h200_block32_configs.py" "$agentic/kernel_configs/h200_dsv41_block32" /logs "$DSV41_BLOCK32_TP"
Original file line number Diff line number Diff line change
@@ -0,0 +1,13 @@
#!/usr/bin/env bash
# Stage the Kimi-K3 DSpark draft before ATOM starts: with an uncached repo id
# every rank pulls the same 7 GB at once. Shared-cache downloads can hit
# transient stale handles, hence the retries.
set -euo pipefail
unset HTTP_PROXY HTTPS_PROXY http_proxy https_proxy
for attempt in 1 2 3 4 5; do
hf download Inferact/Kimi-K3-DSpark && exit 0
echo "hf download attempt $attempt failed; retrying in 60s" >&2
sleep 60
done
echo "hf download of Inferact/Kimi-K3-DSpark failed after 5 attempts" >&2
exit 1
Original file line number Diff line number Diff line change
@@ -0,0 +1,42 @@
#!/usr/bin/env bash
# Pin the worker's Mooncake client and point its store at one active RDMA rail.
set -euo pipefail
pip_install=(python3 -m pip install)
if python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages; then
pip_install+=(--break-system-packages)
fi
"${pip_install[@]}" --quiet --no-cache-dir --no-deps --force-reinstall \
mooncake-transfer-engine-cuda13==0.3.11.post1
python3 -c "from mooncake.store import MooncakeDistributedStore" >/dev/null

# Rail-isolated nodes: two RNICs cannot reach each other even within a node, so
# every rank uses one rail. mlx5_0 is down on some nodes, and topology discovery
# then finds no HCA, so take the first active rail at runtime. DSXE nodes name
# their rails rdmap*.
rail=""
for device in mlx5_0 mlx5_1 mlx5_2 mlx5_3 mlx5_4 mlx5_5 mlx5_8 mlx5_9 \
mlx5_10 mlx5_11 mlx5_16 mlx5_17 mlx5_20 mlx5_21 mlx5_22 mlx5_23 \
$(ls /sys/class/infiniband 2>/dev/null | grep '^rdmap' | sort -V); do
if grep -q ACTIVE "/sys/class/infiniband/$device/ports/1/state" 2>/dev/null; then
rail="$device"
break
fi
done
if [[ -z "$rail" ]]; then
echo "Error: no active RDMA rail on $(hostname); Mooncake cannot initialise" >&2
for state in /sys/class/infiniband/*/ports/*/state; do
echo "$state: $(cat "$state" 2>&1)" >&2
done
exit 1
fi
config="${MOONCAKE_CONFIG_PATH:-/logs/mooncake_store_config.json}"
python3 - "$config" "$rail" <<'PY'
import json, sys
path, rail = sys.argv[1:]
with open(path) as handle:
config = json.load(handle)
config["device_name"] = rail
with open(path, "w") as handle:
json.dump(config, handle, indent=2)
PY
echo "Mooncake rail: $rail"
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
#!/usr/bin/env bash
# Start one LMCache MP server per TP rank in the worker container before vLLM,
# as the legacy MI300X MiniMax-M3 AgentX script did. A variant opts in with
# LMCACHE_SHARDS and LMCACHE_L1_SHARD_GB; its kv-transfer-config lists
# tcp://127.0.0.1:5555 through 5555 + LMCACHE_SHARDS - 1.
set -euo pipefail
[[ -n "${LMCACHE_SHARDS:-}" ]] || exit 0
: "${LMCACHE_L1_SHARD_GB:?}"
version=0.5.3
pip_install=(python3 -m pip install)
if python3 -m pip install --help 2>/dev/null | grep -q -- --break-system-packages; then
pip_install+=(--break-system-packages)
fi
"${pip_install[@]}" --quiet --no-cache-dir --no-deps \
"sortedcontainers==2.4.0" \
"opentelemetry-exporter-prometheus==0.61b0" \
"cupy-rocm-7-0==14.1.1" \
"lmcache==${version}" \
--find-links "https://github.com/LMCache/LMCache/releases/expanded_assets/v${version}-rocm"
python3 -c "import cupy; import lmcache.integration.vllm.lmcache_mp_connector; import opentelemetry.exporter.prometheus"

pids=()
for ((shard = 0; shard < LMCACHE_SHARDS; shard++)); do
# Detached so the servers outlive this preamble and serve the vLLM step.
setsid lmcache server \
--host 127.0.0.1 --port $((5555 + shard)) \
--http-host 127.0.0.1 --http-port $((8080 + shard)) \
--l1-size-gb "$LMCACHE_L1_SHARD_GB" --l1-init-size-gb 10 \
--l1-read-ttl-seconds 7200 --chunk-size 256 --max-workers 2 \
--eviction-policy LRU --supported-transfer-mode lmcache_driven \
> "/logs/lmcache_server_${shard}.log" 2>&1 < /dev/null &
pids+=($!)
done
for ((shard = 0; shard < LMCACHE_SHARDS; shard++)); do
for ((attempt = 0; ; attempt++)); do
python3 -c 'import sys, urllib.request; urllib.request.urlopen(sys.argv[1], timeout=2)' \
"http://127.0.0.1:$((8080 + shard))/healthcheck" 2> /dev/null && break
if ! kill -0 "${pids[$shard]}" 2>/dev/null || (( attempt >= 600 )); then
echo "ERROR: LMCache server $shard did not become ready" >&2
tail -n 50 "/logs/lmcache_server_${shard}.log" >&2 || true
exit 1
fi
sleep 1
done
done
echo "LMCache: ${LMCACHE_SHARDS} servers ready, ${LMCACHE_L1_SHARD_GB} GB L1 each"
Loading
Loading