Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
49 commits
Select commit Hold shift + click to select a range
c304072
fix: preserve complete request outcomes and diagnostic batches
edwingao28 Sep 12, 2026
57ae800
feat: add scoped fixed-sequence power requirements
edwingao28 Sep 12, 2026
ec4a58f
fix: verify Slurm completion and exit llm-d workers normally
edwingao28 Sep 12, 2026
e27dcc3
chore: compose required Slurm completion contract
edwingao28 Sep 12, 2026
df442d8
chore: compose scoped power matrix contract
edwingao28 Sep 12, 2026
85986f4
fix: verify Slurm completion and exit llm-d workers normally
edwingao28 Sep 12, 2026
0d748c1
feat: support opt-in NVIDIA SRT measured power
edwingao28 Sep 12, 2026
5eb5ca1
chore: sync Slurm prerequisite delivery reference
edwingao28 Sep 12, 2026
d3a175a
fix: isolate SRT power snapshots from existing routes
edwingao28 Sep 12, 2026
86b8621
fix: parse Slurm allocation exit codes without derived status
edwingao28 Sep 12, 2026
d5cd65f
fix: integrate exact Slurm exit-code parsing
edwingao28 Sep 12, 2026
20e5e71
ci: run llm-d lifecycle regressions
edwingao28 Sep 12, 2026
c5d812f
ci: run fixed-sequence power regression tests
edwingao28 Sep 12, 2026
fb208c1
ci: integrate Slurm lifecycle regression coverage
edwingao28 Sep 12, 2026
671b0c1
fix: honor accepted power requirement field names
edwingao28 Sep 12, 2026
364f2c5
fix: integrate accepted power requirement field names
edwingao28 Sep 12, 2026
bc8684e
chore: sync request outcome contract with current main
edwingao28 Sep 12, 2026
87b9e45
chore: preserve Slurm diagnostics after current main sync
edwingao28 Sep 12, 2026
c7a5962
fix: synchronize merged Kimi power support
edwingao28 Sep 12, 2026
4b518b2
fix: synchronize result contract with merged main
edwingao28 Sep 12, 2026
fe3a0d4
test: synchronize shared replay signal readiness
edwingao28 Sep 12, 2026
e3c7405
test: integrate shared Slurm signal fixture repair
edwingao28 Sep 12, 2026
812caff
fix: preserve resolved models during power checkout
edwingao28 Sep 12, 2026
d141e71
fix: preserve diagnostic sidecars and legacy result processing
edwingao28 Sep 12, 2026
ed9805b
fix: integrate result sidecar compatibility handling
edwingao28 Sep 12, 2026
a620774
chore: merge main diagnostics into Slurm lifecycle fixes
edwingao28 Sep 12, 2026
95c7b79
chore: preserve main diagnostics in SRT power routing
edwingao28 Sep 12, 2026
8642ec8
chore: synchronize the Slurm prerequisite for SRT power
edwingao28 Sep 12, 2026
8f89342
fix: recheck llm-d completion after engine shutdown
edwingao28 Sep 12, 2026
8742b8b
chore: retain native power foundation in Slurm fixes
edwingao28 Sep 12, 2026
20eb5c0
chore: sync validated power foundation into SRT routing
edwingao28 Sep 12, 2026
cf2f442
chore: sync Slurm completion branch with main
edwingao28 Sep 13, 2026
837e49f
chore: sync SRT power routing with updated parent
edwingao28 Sep 13, 2026
db219ab
chore: sync Slurm completion with TileRT main updates
edwingao28 Sep 13, 2026
de3464c
chore: sync SRT routing with TileRT-aware Slurm parent
edwingao28 Sep 13, 2026
32bdb7a
fix: retain Slurm outcome with GB200 AgentX power collection
edwingao28 Sep 13, 2026
48d1999
fix: retain fixed-sequence and AgentX GB200 producer routes
edwingao28 Sep 13, 2026
7d330f8
fix: preserve GB300 native status when syncing main
edwingao28 Sep 13, 2026
774d70b
fix: retain fixed power routing alongside GB300 AgentX
edwingao28 Sep 13, 2026
8d7d93c
fix: preserve H200 native failure through power collection
edwingao28 Sep 13, 2026
8aae3b3
fix: retain fixed power isolation when syncing H200 collection
edwingao28 Sep 13, 2026
6d4b874
fix: retain B300 recovery in Slurm completion refresh
edwingao28 Sep 13, 2026
836bdb5
fix: refresh SRT power against Slurm completion
edwingao28 Sep 13, 2026
eaa2f4b
fix: sync Slurm completion with required H100 power
edwingao28 Sep 13, 2026
912972d
fix: sync SRT power with current parent
edwingao28 Sep 13, 2026
c5c38f6
feat: make NVIDIA SRT power qualification independent
edwingao28 Sep 14, 2026
331adf9
fix: preserve failed H200 signal status
edwingao28 Sep 14, 2026
f11e11a
fix: use the repaired H100 producer for required power
edwingao28 Sep 14, 2026
7837ac9
chore: sync H100 required-power qualification with main
edwingao28 Sep 14, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/workflows/benchmark-multinode-tmpl.yml
Original file line number Diff line number Diff line change
Expand Up @@ -415,6 +415,7 @@ jobs:
${{ env.RESULT_FILENAME }}_*.json
agg_${{ env.RESULT_FILENAME }}_*.json
power_validation_${{ env.RESULT_FILENAME }}_*.json
slurm_job_*_outcome.txt
LOGS/power/**
LOGS/native_power/**
result_processing_${{ env.RESULT_FILENAME }}.json
Expand Down
1 change: 1 addition & 0 deletions .github/workflows/benchmark-tmpl.yml
Original file line number Diff line number Diff line change
Expand Up @@ -392,6 +392,7 @@ jobs:
gpu_metrics_identity.json
gpu_metrics_identity.csv
power_validation_${{ env.RESULT_FILENAME }}.json
slurm_job_*_outcome.txt
results/gpu_metrics*.csv
results/gpu_metrics*_context.json
results/gpu_metrics_identity.json
Expand Down
1 change: 1 addition & 0 deletions .github/workflows/test-changelog-gate.yml
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@ on:
- "runners/launch_*.sh"
- "runners/slurm_utils.sh"
- "runners/test_slurm_utils.py"
- "benchmarks/multi_node/llm-d/**"
- "utils/ci_priority.py"
- "utils/test_ci_priority.py"
- ".github/workflows/reuse-sweep-comment.yml"
Expand Down
6 changes: 5 additions & 1 deletion .github/workflows/test-process-result.yml
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,10 @@ on:
- 'benchmarks/multi_node/tilert_utils/**'
- 'benchmarks/multi_node/glm5.1_fp8_b200_tilert-disagg.sh'
- 'runners/test_tilert_power_lifecycle.py'
- 'runners/powerx_8k1k.sh'
- 'runners/prepare_srt_power.py'
- 'runners/launch_h100-dgxc-slurm.sh'
- 'utils/test_prepare_srt_power.py'

permissions:
contents: read
Expand All @@ -66,7 +70,7 @@ jobs:
run: |
cd utils
uv run --no-project --exclude-newer PT12H --python 3.12 --with pytest --with pyyaml \
python -m pytest test_aggregate_power.py test_aggregate_power_multinode.py agentic/aggregation/ test_inject_srt_power_concurrencies.py test_process_result.py test_native_multinode_power.py ../runners/test_native_collector_barriers.py ../runners/test_native_collector_receipts.py ../runners/test_tilert_power_lifecycle.py -v
python -m pytest test_aggregate_power.py test_aggregate_power_multinode.py agentic/aggregation/ test_inject_srt_power_concurrencies.py test_process_result.py test_prepare_srt_power.py test_native_multinode_power.py ../runners/test_native_collector_barriers.py ../runners/test_native_collector_receipts.py ../runners/test_tilert_power_lifecycle.py -v

- name: Test serving client result persistence
run: |
Expand Down
7 changes: 7 additions & 0 deletions configs/nvidia-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -666,6 +666,7 @@ dsr1-fp8-b300-dynamo-trt:
fixed-seq-len:
- isl: 8192
osl: 1024
require-power: true
search-space:
- spec-decoding: "mtp"
conc-list: [40]
Expand Down Expand Up @@ -760,6 +761,7 @@ dsr1-fp8-b300-dynamo-trt:
# 8k1k STP configs
- isl: 8192
osl: 1024
require-power: true
search-space:
- conc-list: [64]
prefill:
Expand Down Expand Up @@ -1883,6 +1885,7 @@ dsr1-fp8-h200-dynamo-trt:
fixed-seq-len:
- isl: 8192
osl: 1024
require-power: true
search-space:
# MTP configurations
- spec-decoding: "mtp"
Expand Down Expand Up @@ -2344,6 +2347,7 @@ dsr1-fp8-h100-dynamo-sglang:
fixed-seq-len:
- isl: 8192
osl: 1024
require-power: true
search-space:
# # STP: Max throughput TEP (1 prefill, 1 decode)
# - conc-list: [1, 2, 4, 8, 16, 32, 64, 128]
Expand Down Expand Up @@ -3697,6 +3701,7 @@ dsr1-fp4-b200-dynamo-sglang:
fixed-seq-len:
- isl: 8192
osl: 1024
require-power: true
search-space:
# Non-MTP configurations
- conc-list: [64, 128]
Expand Down Expand Up @@ -5010,6 +5015,7 @@ qwen3.5-fp8-gb200-dynamo-sglang:
fixed-seq-len:
- isl: 8192
osl: 1024
require-power: true
search-space:
# 1P1D STP: TP4 prefill + TP4 decode (pure tensor parallel). 2 nodes (1+1).
- spec-decoding: "none"
Expand Down Expand Up @@ -6731,6 +6737,7 @@ qwen3.5-fp4-gb300-dynamo-trt-mtp:
fixed-seq-len:
- isl: 8192
osl: 1024
require-power: true
search-space:
# 1P2D: 1 prefill (TP1/EP1/dp-attn), 2 decode (TP8/EP8/no-dp-attn), conc=20
- spec-decoding: "mtp"
Expand Down
4 changes: 4 additions & 0 deletions docs/configuration-procedures.md
Original file line number Diff line number Diff line change
Expand Up @@ -134,6 +134,10 @@ For GLM-5.1 on B200 Nscale, `MODEL_PATH` can select an existing shared checkpoin

Only fixed 8192/1024 `glm5.1-fp8-b200-tilert` requires native power. TileRT runs inside its returned `salloc` allocation, retains both role exit codes and drains collectors before staging audits. Exactly one physical node per role is supported. Other sequence lengths, AgentX and eval-only do not enable this collector. Hardware qualification and publication remain pending.

## Opt-in NVIDIA SRT power

With `REQUIRE_POWER=1`, fixed 8192/1024 SRT launches select runtime `3f3b7af26e34acc8b62b39971bec839a19ac57a2`. Preparation resolves the selected override, injects matrix concurrencies, translates the DeepSeek-V4 tokenizer setting and requires DCGM telemetry. The complete DSR1 B200 FP4 SGLang, B300 FP8 TRT, H100 FP8 SGLang and H200 FP8 TRT curves, plus Qwen3.5 GB200 FP8 SGLang and GB300 FP4 TRT MTP, enable this path through `require-power: true`. Retired DSV4 8k1k scenarios are not enabled. AgentX and eval-only retain existing routing. Available logs and producer provenance survive failure. Each hardware/framework scope needs its own full sweep and applicable evals.

## Register an srt-slurm recipe

Mapping source: [`benchmarks/multi_node/srt-slurm-recipes/RECIPES.md`](../benchmarks/multi_node/srt-slurm-recipes/RECIPES.md). Checked-in recipes: [`benchmarks/multi_node/srt-slurm-recipes/`](../benchmarks/multi_node/srt-slurm-recipes/).
Expand Down
4 changes: 4 additions & 0 deletions docs/configuration-procedures_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -132,6 +132,10 @@ B200 Nscale 的 GLM-5.1 可用 `MODEL_PATH` 指定已有共享权重,覆盖默

仅固定 8192/1024 的 `glm5.1-fp8-b200-tilert` 要求原生功耗。TileRT 在 `salloc` 返回的分配内运行,保留两个角色的退出码,并在保存审计数据前等待采集器排空。每个角色仅支持一个物理节点。其他序列长度、AgentX 和 eval-only 不启用此采集器。硬件资格验证与发布仍待完成。

## NVIDIA SRT 可选功耗

固定 8192/1024 SRT 运行仅在 `REQUIRE_POWER=1` 时选择 runtime `3f3b7af26e34acc8b62b39971bec839a19ac57a2`。准备过程解析所选 override、注入矩阵并发度、转换 DeepSeek-V4 tokenizer 设置并要求 DCGM 遥测。通过 `require-power: true`,为 DSR1 B200 FP4 SGLang、B300 FP8 TRT、H100 FP8 SGLang、H200 FP8 TRT,以及 Qwen3.5 GB200 FP8 SGLang、GB300 FP4 TRT MTP 的完整曲线启用该路径;不启用已退役的 DSV4 8k1k 场景。AgentX 和 eval-only 保留原有路由。失败时保留已有日志与生产者版本。各硬件及框架范围仍需完整 sweep 和适用 eval。

## 注册 srt-slurm 配方

映射来源:[`benchmarks/multi_node/srt-slurm-recipes/RECIPES.md`](../benchmarks/multi_node/srt-slurm-recipes/RECIPES.md)。检入的配方:[`benchmarks/multi_node/srt-slurm-recipes/`](../benchmarks/multi_node/srt-slurm-recipes/)。
Expand Down
4 changes: 4 additions & 0 deletions docs/results-and-ingestion.md
Original file line number Diff line number Diff line change
Expand Up @@ -114,6 +114,10 @@ Processing and diagnostic power-audit uploads run after launcher or validation f

The native collector sets UTC and records context beside its CSV for portable replay; existing benchmark monitors keep their current behavior. Its launcher integration requires separate hardware qualification. The offline adapter accepts this context without changing producers. Unusable samples outside the formal window do not establish coverage; `boundary_degenerate_rows` retains their per-GPU counts.

### Slurm completion receipts

NVIDIA SRT launchers verify the terminal allocation state and exit code, consulting `scontrol` when `sacct` is missing or non-terminal and retaining `slurm_job_*_outcome.txt`. They stage available evidence before returning failure. The shared log-streaming helper enables this terminal check only when its caller explicitly requests it; llm-d retains its existing completion behavior.

## Eval artifacts

### Per-config identity and collection
Expand Down
4 changes: 4 additions & 0 deletions docs/results-and-ingestion_zh.md
Original file line number Diff line number Diff line change
Expand Up @@ -114,6 +114,10 @@ PR changelog 选择具有代表性的 NVIDIA 和 AMD 覆盖,并非所有受影

原生采集器单独设置 UTC,并在 CSV 旁记录上下文以支持跨环境回放;现有基准监控行为保持不变。启动器接入需要另行完成硬件验证。离线适配器接受该上下文,不改变现有生产端。正式窗口外的无效样本不能构成覆盖;`boundary_degenerate_rows` 保留其逐 GPU 计数。

### Slurm 完成状态文件

NVIDIA SRT 启动器检查分配的最终状态和退出码;当 `sacct` 记录缺失或尚未进入最终状态时查询 `scontrol`,并保留 `slurm_job_*_outcome.txt`。启动器先保存已有证据再返回失败。共享日志流辅助函数仅在调用方明确请求时启用最终状态检查;llm-d 保留现有的结束行为。

## 评测工件

### 单配置身份和收集
Expand Down
32 changes: 32 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -7646,3 +7646,35 @@
- "将 vLLM ROCm nightly 更新至包含 vllm-project/vllm#54855 的 eed1f3d0c6043bd494424a22443ee198dd56f657,并针对 MXFP4/FP8 混合权重关闭不兼容的 AITER shared-expert fusion。"
- "将单节点 DPA AgentX 扫描从并发 64 扩展为 64/128/192。手工验证显示,在 c64/c128/c256 三个点中 c128 吞吐量最高;新增 c192 用于补齐中间区间并定位 c256 性能回退前的吞吐拐点。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3071

- config-keys:
- dsr1-fp4-b200-dynamo-sglang
- dsr1-fp8-b300-dynamo-trt
- dsr1-fp8-h100-dynamo-sglang
- dsr1-fp8-h200-dynamo-trt
- qwen3.5-fp8-gb200-dynamo-sglang
- qwen3.5-fp4-gb300-dynamo-trt-mtp
scenario-type:
- fixed-seq-len
description:
- "Require measured power for six complete NVIDIA SRT 8k1k curves with self-contained Slurm completion checks."
- "为六条完整的 NVIDIA SRT 8k1k 曲线强制采集功耗,并包含所需的 Slurm 完成状态检查。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3027

- config-keys:
- kimik3-fp4-b200-dynamo-vllm-agentic-dspark
scenario-type:
- agentic-coding
description:
- "Qualify shared Slurm completion on the complete existing B200 native Kimi AgentX curve."
- "用现有完整 B200 原生 Kimi AgentX 曲线验证共享 Slurm 完成状态。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3027

- config-keys:
- dsr1-fp8-h100-dynamo-sglang
scenario-type:
- fixed-seq-len
description:
- "Pin H100 SRT 8k1k startup to the fixed-sequence producer with bounded dependency preparation and owned infrastructure readiness; stop before submission when setup fails."
- "固定 H100 SRT 8k1k 的 producer,保留固定长度功耗能力,限制依赖准备时间并在就绪检查前登记基础服务;准备失败时禁止提交作业。"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/3027
24 changes: 20 additions & 4 deletions runners/launch_b200-nscale-compat.sh
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,8 @@

# shellcheck source=runners/slurm_utils.sh
source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh"
# shellcheck source=runners/powerx_8k1k.sh
source "$(dirname "${BASH_SOURCE[0]}")/powerx_8k1k.sh"

# Compatibility launcher for B200 Nscale configurations that have not yet
# moved to the native srt-slurm path in launch_b200-nscale-slurm.sh.
Expand Down Expand Up @@ -129,6 +131,7 @@ if [[ "$IS_MULTINODE" == "true" ]]; then
fi

USES_DCGM_POWER=0
if powerx_fixed_8k1k; then USES_DCGM_POWER=1; fi
_POWER_CONFIG_FILE="${CONFIG_FILE:-}"
if [[ "${EVAL_ONLY:-false}" == "true" && -n "${EVAL_CONFIG_FILE:-}" ]]; then
_POWER_CONFIG_FILE="$EVAL_CONFIG_FILE"
Expand All @@ -144,7 +147,7 @@ if [[ "$IS_MULTINODE" == "true" ]]; then
' "$_RECIPE_SRC"; then
USES_DCGM_POWER=1
fi
if [[ "$USES_DCGM_POWER" == "1" && (
if ! powerx_fixed_8k1k && [[ "$USES_DCGM_POWER" == "1" && (
"${IS_AGENTIC:-0}" == "1" ||
"$MODEL_PREFIX" != "dsv4" ||
"$PRECISION" != "fp4" ||
Expand All @@ -166,7 +169,9 @@ if [[ "$IS_MULTINODE" == "true" ]]; then
# Kimi K3 aggregate profiles use the srt-slurm fork that supports direct
# multi-node vLLM. Pin the tested renderer so branch movement cannot change
# generated rank commands between sweep points.
if [[ "$USES_DCGM_POWER" == "1" ]]; then
if powerx_fixed_8k1k; then
powerx_clone_srt "$SRT_REPO_DIR" || exit 1
elif [[ "$USES_DCGM_POWER" == "1" ]]; then
git clone "$POWER_SRT_SLURM_URL" "$SRT_REPO_DIR" || exit 1
cd "$SRT_REPO_DIR" || exit 1
git checkout "$POWER_SRT_SLURM_PIN" || exit 1
Expand Down Expand Up @@ -387,6 +392,7 @@ EOF
exit 1
fi

powerx_prepare_srt || exit 1
# Override the job name in the config file with the runner name
sed -i "s/^name:.*/name: \"${RUNNER_NAME}\"/" "${CONFIG_FILE%%:*}"
# Bump recipe health-check timeout from 360×10s=3600s to 720×10s=7200s
Expand Down Expand Up @@ -423,6 +429,7 @@ EOF
# srtctl creates logs in outputs/JOB_ID/logs/
LOGS_DIR="outputs/$JOB_ID/logs"
LOG_FILE="$LOGS_DIR/sweep_${JOB_ID}.log"
trap powerx_snapshot_srt EXIT

# Wait for log file to appear (also check job is still alive)
while ! ls "$LOG_FILE" &>/dev/null; do
Expand All @@ -449,10 +456,12 @@ EOF
tail -F -s 2 -n+1 "$LOG_FILE" --pid=$POLL_PID 2>/dev/null

wait $POLL_PID
SRT_JOB_RC=0
verify_slurm_job_completion "$JOB_ID" || SRT_JOB_RC=$?

set -x

echo "Job $JOB_ID completed!"
echo "Job $JOB_ID finished with status $SRT_JOB_RC; collecting evidence"
echo "Collecting results..."

if [ ! -d "$LOGS_DIR" ]; then
Expand All @@ -468,7 +477,12 @@ EOF
cp "$GITHUB_WORKSPACE/power-producer-sha.txt" "$LOGS_DIR/power/power-producer-sha.txt"
fi

cp -r "$LOGS_DIR" "$GITHUB_WORKSPACE/LOGS"
if powerx_fixed_8k1k; then
mkdir -p "$GITHUB_WORKSPACE/LOGS"
cp -a "$LOGS_DIR/." "$GITHUB_WORKSPACE/LOGS/"
else
cp -r "$LOGS_DIR" "$GITHUB_WORKSPACE/LOGS"
fi
tar czf "$GITHUB_WORKSPACE/multinode_server_logs.tar.gz" -C "$LOGS_DIR" .

if [[ "${EVAL_ONLY:-false}" != "true" ]]; then
Expand Down Expand Up @@ -504,6 +518,8 @@ EOF
done
find . -name '.nfs*' -delete 2>/dev/null || true

if [[ "$SRT_JOB_RC" != "0" ]]; then exit "$SRT_JOB_RC"; fi

else

SQUASH_FILE="/data/home/sa-shared/containers/$(echo "$IMAGE" | sed 's/[\/:@#]/_/g').sqsh"
Expand Down
29 changes: 20 additions & 9 deletions runners/launch_b200-nscale-slurm.sh
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,8 @@ SQUASH_LOCK_TIMEOUT=3600

# shellcheck source=runners/slurm_utils.sh
source "$(dirname "${BASH_SOURCE[0]}")/slurm_utils.sh"
# shellcheck source=runners/powerx_8k1k.sh
source "$(dirname "${BASH_SOURCE[0]}")/powerx_8k1k.sh"

set -x

Expand Down Expand Up @@ -68,6 +70,7 @@ if [[ $FRAMEWORK != "dynamo-vllm" ]] &&
fi

USES_DCGM_POWER=0
if powerx_fixed_8k1k; then USES_DCGM_POWER=1; fi
USES_AGENTX_POWER=0
_POWER_CONFIG_FILE="${CONFIG_FILE:-}"
if [[ "${EVAL_ONLY:-false}" == "true" && -n "${EVAL_CONFIG_FILE:-}" ]]; then
Expand All @@ -87,7 +90,7 @@ fi
if [[ "$USES_DCGM_POWER" == "1" && "$IS_AGENTIC" == "1" &&
"$MODEL_PREFIX" == "kimik3" && "$PRECISION" == "fp4" && "$FRAMEWORK" == "dynamo-vllm" ]]; then
USES_AGENTX_POWER=1
elif [[ "$USES_DCGM_POWER" == "1" && (
elif ! powerx_fixed_8k1k && [[ "$USES_DCGM_POWER" == "1" && (
"${IS_AGENTIC:-0}" == "1" ||
"$PRECISION" != "fp4" ||
( "$MODEL_PREFIX" == "dsv4" && "$FRAMEWORK" != "dynamo-sglang" && "$FRAMEWORK" != "dynamo-vllm" ) ||
Expand All @@ -103,7 +106,9 @@ export SERVED_MODEL_NAME=$MODEL
echo "Cloning srt-slurm repository..."
SRT_REPO_DIR="srt-slurm"
rm -rf "$SRT_REPO_DIR"
if [[ "$USES_DCGM_POWER" == "1" ]]; then
if powerx_fixed_8k1k; then
powerx_clone_srt "$SRT_REPO_DIR" || exit 1
elif [[ "$USES_DCGM_POWER" == "1" ]]; then
SELECTED_POWER_SRT_SLURM_PIN="$POWER_SRT_SLURM_PIN"
if [[ "$USES_AGENTX_POWER" == "1" ]]; then
SELECTED_POWER_SRT_SLURM_PIN="$AGENTX_POWER_SRT_SLURM_PIN"
Expand Down Expand Up @@ -352,6 +357,7 @@ if [[ -z "$CONFIG_FILE" ]]; then
fi

# Strip any :override[N] selector so sed and the injector operate on the file.
powerx_prepare_srt || exit 1
CONFIG_PATH="${CONFIG_FILE%%:*}"

# Override the job name in the config file with the runner name
Expand Down Expand Up @@ -392,18 +398,16 @@ echo "Extracted JOB_ID: $JOB_ID"

LOGS_DIR="outputs/$JOB_ID/logs"
LOG_FILE="$LOGS_DIR/sweep_${JOB_ID}.log"
trap powerx_snapshot_srt EXIT

# Waits for the log file to appear, fails fast if the job dies first, then
# streams until the job leaves the queue.
SRT_JOB_RC=0
stream_slurm_job_log "$JOB_ID" "$LOG_FILE" || SRT_JOB_RC=$?
if [[ "$SRT_JOB_RC" != "0" && "$USES_AGENTX_POWER" != "1" ]]; then
exit "$SRT_JOB_RC"
fi
stream_slurm_job_log "$JOB_ID" "$LOG_FILE" true || SRT_JOB_RC=$?

set -x

echo "Job $JOB_ID completed!"
echo "Job $JOB_ID finished with status $SRT_JOB_RC; collecting evidence"
echo "Collecting results..."

if [ ! -d "$LOGS_DIR" ]; then
Expand All @@ -425,10 +429,15 @@ if [[ "$USES_DCGM_POWER" == "1" ]]; then
cp "$GITHUB_WORKSPACE/power-producer-sha.txt" "$LOGS_DIR/power/power-producer-sha.txt"
fi

cp -r "$LOGS_DIR" "$GITHUB_WORKSPACE/LOGS"
if powerx_fixed_8k1k; then
mkdir -p "$GITHUB_WORKSPACE/LOGS"
cp -a "$LOGS_DIR/." "$GITHUB_WORKSPACE/LOGS/"
else
cp -r "$LOGS_DIR" "$GITHUB_WORKSPACE/LOGS"
fi
bundle_server_logs "$LOGS_DIR" "$GITHUB_WORKSPACE/multinode_server_logs.tar.gz"

if [[ "$AGENTX_POWER_RC" != "0" ]]; then
if [[ "$AGENTX_POWER_RC" != "0" && "$SRT_JOB_RC" == "0" ]]; then
echo "ERROR: AgentX power validation failed; available audit and server artifacts were staged" >&2
exit "$AGENTX_POWER_RC"
fi
Expand Down Expand Up @@ -487,3 +496,5 @@ for i in 1 2 3 4 5; do
sleep 10
done
find . -name '.nfs*' -delete 2>/dev/null || true

if [[ "$SRT_JOB_RC" != "0" ]]; then exit "$SRT_JOB_RC"; fi
Loading
Loading