Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
70 commits
Select commit Hold shift + click to select a range
0641575
docs(amd): define srt-slurm bring-up contract
cquil11 Aug 10, 2026
3a3e723
feat(amd): add MI300X srt-slurm aggregate scaffold
cquil11 Aug 10, 2026
fc07a3e
fix(amd): align MI300X recipe with vLLM 0.26
cquil11 Aug 10, 2026
b95b4b8
fix(amd): honor MI300X Pyxis home policy
cquil11 Aug 10, 2026
aa681ff
fix(amd): align MI300X worker launch policy
cquil11 Aug 10, 2026
1b256b9
feat(amd): use InferenceX fixed-sequence benchmark
cquil11 Aug 10, 2026
16c5efc
fix(amd): exclude unsuitable MI300X nodes
cquil11 Aug 10, 2026
b1aa29a
feat(amd): add MI300X disaggregated validation target
cquil11 Aug 10, 2026
181459d
fix(amd): preserve custom benchmark arguments
cquil11 Aug 10, 2026
9b728ab
Stage MI300X runtime config on compute nodes
cquil11 Aug 10, 2026
542ddda
Reuse stable ROCm container artifact
cquil11 Aug 10, 2026
2df8508
fix(amd): align disagg runtime dependencies
cquil11 Aug 10, 2026
b0376b3
docs(amd): pin runtime source override head
cquil11 Aug 10, 2026
9afc9a1
fix: route MI300X orchestration over private fabric
cquil11 Aug 10, 2026
8e139bd
fix: discover MI300X private fabric addresses
cquil11 Aug 10, 2026
6f25810
feat: validate MI300X MoRI-IO routing
cquil11 Aug 10, 2026
5b45946
fix: pin published vllm router image
cquil11 Aug 10, 2026
568327b
docs: update AMD runtime pin
cquil11 Aug 10, 2026
1a6c28e
fix(amd): use contiguous AITER cache for MoRI
cquil11 Aug 10, 2026
5f0e320
docs(amd): pin dynamic readiness runtime
cquil11 Aug 10, 2026
8a59c1a
docs(amd): pin CI-compatible runtime
cquil11 Aug 10, 2026
7476a92
docs(amd): pin formatted runtime head
cquil11 Aug 10, 2026
738471e
docs: update AMD srt-slurm pin
cquil11 Aug 10, 2026
64652fe
ci(amd): validate srt-slurm disaggregation
cquil11 Aug 10, 2026
d548a4f
ci(amd): bootstrap srt-slurm compute runtime
cquil11 Aug 10, 2026
8a1f8cc
ci(amd): validate node-local runtime images
cquil11 Aug 10, 2026
03c480d
ci(amd): stage srt-slurm on compute nodes
cquil11 Aug 10, 2026
ba5e7c7
ci(amd): validate aggregate srt-slurm serving
cquil11 Aug 10, 2026
d795641
ci(amd): route aggregate through srt-slurm
cquil11 Aug 10, 2026
524b1a2
fix(amd): select aggregate srt recipe explicitly
cquil11 Aug 10, 2026
71e908a
ci(amd): add MI355X srt-slurm validation
cquil11 Aug 10, 2026
89eb32e
Merge remote-tracking branch 'origin/main' into agent/srt-slurm-amd-i…
cquil11 Aug 10, 2026
33f2cfc
fix(amd): use writable MI355X shared root
cquil11 Aug 10, 2026
9a7d233
add production MI355X srt-slurm validation
cquil11 Aug 10, 2026
cc0b9f7
fix MI355X model staging input
cquil11 Aug 10, 2026
1d60fb1
bump AMD runtime cache fix
cquil11 Aug 10, 2026
87933a9
fix(amd): select Qwen3.5 text loader
cquil11 Aug 10, 2026
a55bed7
fix(amd): allocate full MI355X CPU topology
cquil11 Aug 10, 2026
d3ed345
fix(amd): align Qwen3.5 Hugging Face cache root
cquil11 Aug 10, 2026
3b3acff
fix(amd): use large-MR-capable MI355X runtime
cquil11 Aug 10, 2026
c5ce0a4
fix(amd): make runtime staging reproducible
cquil11 Aug 10, 2026
e22bf36
fix(amd): pin staged hub cache path
cquil11 Aug 10, 2026
147b2ec
chore(amd): pin rebased srt-slurm runtime
cquil11 Aug 11, 2026
dd49816
fix(amd): pin native router infra correction
cquil11 Aug 11, 2026
accf806
Merge remote-tracking branch 'origin/main' into agent/srt-slurm-amd-i…
cquil11 Aug 11, 2026
7b3ee00
fix(amd): select the native SGLang Router frontend
cquil11 Aug 11, 2026
9403f0b
test(amd): enforce the native SGLang frontend contract
cquil11 Aug 11, 2026
72ee996
test: drop redundant AMD integration contract file
cquil11 Aug 11, 2026
3105070
feat: add ATOM and Infera validation lanes
cquil11 Aug 11, 2026
226a875
docs: link ATOM validation pull request
cquil11 Aug 11, 2026
dd597d5
fix: use writable Enroot staging runtime
cquil11 Aug 11, 2026
98ac36d
fix: retry transient container imports
cquil11 Aug 11, 2026
c16663b
fix: install orchestration binaries on MI300X
cquil11 Aug 11, 2026
73d5e8e
fix: stage srt-slurm infrastructure binaries
cquil11 Aug 11, 2026
e27c76f
fix(srt): pin Infera worker-registry readiness
cquil11 Aug 11, 2026
5021e0e
fix(atom): validate upstream Infera KV event decoder
cquil11 Aug 11, 2026
f307b55
fix(mi300x): stage every eligible srt node
cquil11 Aug 11, 2026
e5066e6
fix(mi300x): cover the complete eligible node set
cquil11 Aug 11, 2026
f6b200f
fix(mi300x): recover incomplete runtime checkouts
cquil11 Aug 11, 2026
bb97fd7
test(mi355x): add ATOM and Infera validation lanes
cquil11 Aug 11, 2026
85f55d6
fix(atom): keep Mooncake transfer paths reusable
cquil11 Aug 11, 2026
1a5b36f
fix(atom): validate Mooncake TCP transfer path
cquil11 Aug 11, 2026
6c5311a
fix(atom): verify installed framework versions
cquil11 Aug 11, 2026
d71afb3
docs(atom): record validated TCP transport
cquil11 Aug 11, 2026
98b9b50
Merge remote-tracking branch 'origin/main' into agent/atom-infera-val…
cquil11 Aug 11, 2026
2e2f375
fix(atom): pin tcp-only Mooncake source overlay
cquil11 Aug 11, 2026
bdc0e20
fix: pin fetchable stable ATOM overlay
cquil11 Aug 11, 2026
f9bdbe1
fix: pin Mooncake TCP transport guard
cquil11 Aug 11, 2026
eb814bb
fix(ci): restore e2e eval matrix output
cquil11 Aug 11, 2026
4a98aa9
fix(amd): serialize ATOM source staging
cquil11 Aug 17, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/workflows/e2e-tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -261,6 +261,7 @@ jobs:
MULTI_AGENTIC_EVAL=$(echo "$CONFIG_JSON" | python3 -c "import sys,json; d=json.load(sys.stdin); print(json.dumps([x for x in d if x.get('scenario-type') == 'agentic-coding' and 'prefill' in x and x.get('run-eval', False)]))" | score_matrix multi-agentic-eval)
SINGLE=$(echo "$CONFIG_JSON" | python3 -c "import sys,json; d=json.load(sys.stdin); print(json.dumps([x for x in d if 'prefill' not in x and x.get('scenario-type') != 'agentic-coding' and not x.get('eval-only', False)]))" | score_matrix single)
MULTI=$(echo "$CONFIG_JSON" | python3 -c "import sys,json; d=json.load(sys.stdin); print(json.dumps([x for x in d if 'prefill' in x and x.get('scenario-type') != 'agentic-coding' and not x.get('eval-only', False)]))" | score_matrix multi)
EVALS=$(echo "$CONFIG_JSON" | python3 -c "import sys,json; d=json.load(sys.stdin); print(json.dumps([x for x in d if 'prefill' not in x and x.get('scenario-type') != 'agentic-coding' and x.get('run-eval', False)]))" | score_matrix eval)
MULTI_EVAL=$(echo "$CONFIG_JSON" | python3 -c "import sys,json; d=json.load(sys.stdin); print(json.dumps([x for x in d if 'prefill' in x and x.get('scenario-type') != 'agentic-coding' and x.get('run-eval', False)]))" | score_matrix multi-eval)
{
echo "agentic-config=$AGENTIC"
Expand Down
111 changes: 111 additions & 0 deletions benchmarks/multi_node/srt-slurm-recipes/AMD_BRINGUP.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,111 @@
# AMD srt-slurm bring-up

This document tracks the work-in-progress integration of
[`SemiAnalysisAI/srt-slurm`](https://github.com/SemiAnalysisAI/srt-slurm) with
InferenceX AMD Slurm clusters. The project is a functional orchestration
bring-up, not a performance-tuning exercise.

Current development pin for both AMD launchers:

- repository: `SemiAnalysisAI/srt-slurm`
- branch: `agent/amd-multinode-runtime`
- commit: `315e4b06a7e0806194a646ea21832e750e896a46`

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 New doc states the pinned srt-slurm commit is 315e4b06a7e0806194a646ea21832e750e896a46, but the launcher scripts added in the same PR (runners/launch_mi300x-amds-srt.sh:13, runners/launch_mi355x-amds-srt.sh:14) actually pin 5ecfb13d1ba0960045482f1ef006312d8729d37a.

Extended reasoning...

An engineer or on-call debugging a failed srt-slurm run reads AMD_BRINGUP.md to find the pinned commit, checks out 315e4b06... to reproduce, and investigates the wrong tree/behavior since the launchers actually run 5ecfb13d... — wasting debugging time and potentially reaching wrong conclusions about what code is under validation.

Verification: nit. Documentation/code inconsistency, all introduced in this PR. AMD_BRINGUP.md:12 states "Current development pin for both AMD launchers ... commit: 315e4b06a7e0806194a646ea21832e750e896a46", but both launchers pin a different SHA: runners/launch_mi300x-amds-srt.sh:7 SRT_SLURM_COMMIT="5ecfb13d1ba0960045482f1ef006312d8729d37a" and runners/launch_mi355x-amds-srt.sh:8 same value. The…


The MI300X launcher uses srt-slurm's supported `--no-preflight` submission mode
because the immutable squashfs files live on compute-node-local RAID rather
than the login node. Before submission, the launcher stages the benchmark
runtime across the same eligible node pool. Missing engine and router images
are imported atomically under per-image locks from the pinned public
`vllm/vllm-openai-rocm:v0.26.0` and
`vllm/vllm-router:nightly-20260809-d2ba586` images.

The MI300X login and compute nodes also do not share the Actions checkout. The
staging allocation checks out the exact pinned srt-slurm commit on every
eligible compute node, installs its compute-only runtime, and injects that
node-local path through srt-slurm's `SRTCTL_RUNTIME_SOURCE_DIR` transport
override. The submitter continues to validate against its local pinned checkout.

## Scope

1. Prove a single-node aggregate vLLM deployment on MI300X.
2. Prove a multi-node vLLM Router prefill/decode deployment on MI300X using
AMD's supported MoRI-IO KV connector.
3. Exercise both paths with fixed input/output sequence lengths and lightweight
models before introducing production-size models.
4. Validate the same paths through the upstream InferenceX GitHub Actions
runner infrastructure.
5. Port a representative existing MI355X disaggregated configuration after the
MI300X runtime contract is stable.

## MI300X cluster contract under validation

- eight AMD GPUs per healthy compute node;
- Slurm GPU allocation through `--gres=gpu:<count>` rather than the current
srt-slurm `--gpus-per-node` default;
- no site-specific `--segment` directive;
- ROCm device access through `/dev/kfd` and `/dev/dri`;
- Pyxis/Enroot writable, remap-root, and mount-home behavior matching the
established MI300X launcher;
- the shared Hugging Face cache and runner workspace remain user-owned;
- fixed-sequence validation runs InferenceX's existing
`utils/bench_serving/benchmark_serving.py` through srt-slurm's `custom`
benchmark hook rather than maintaining a second benchmark copy in
srt-slurm;
- the routable inter-node network interface is selected from live cluster
evidence rather than copied from an NVIDIA recipe.

The launcher will continue to exclude compute nodes already documented as
unsuitable. It must not resume down nodes, cancel or preempt existing jobs, or
alter unrelated shared software.

## Acceptance criteria

### Aggregate

- one srt-slurm allocation starts one aggregate vLLM service;
- every requested GPU is visible to ROCm and vLLM exactly once;
- the OpenAI-compatible health/model endpoint becomes ready;
- a fixed-sequence request completes successfully;
- srt-slurm tears down all owned processes and exits successfully.

### Disaggregated

- one allocation places distinct prefill and decode roles across multiple
MI300X nodes;
- vLLM Router and direct vLLM workers become healthy without Dynamo, NATS,
etcd, or bespoke per-recipe orchestration;
- role endpoints use routable node addresses and unique ports;
- KV transfer completes across AMD nodes and a fixed-sequence request succeeds;
- teardown removes only processes owned by the allocation.

### Regression safety

- NVIDIA remains the default accelerator runtime in srt-slurm;
- existing NVIDIA recipes and device binding tests remain green;
- AMD-specific mounts, Slurm directives, and environment variables live in a
reusable cluster profile rather than duplicated recipe shell fragments.

## Current status

The srt-slurm branch now contains the first accelerator-aware runtime slice:
cluster configuration accepts `accelerator_vendor: amd`, partial-GPU workers
use Linux ROCm's `ROCR_VISIBLE_DEVICES`, and legacy NVIDIA/CUDA behavior remains
the default. It also supports `gpu_sbatch_directive: gres` without changing the
legacy NVIDIA `--gpus-per-node` default. The initial MI300X cluster profile and
small-model aggregate recipe are checked in alongside this document. The first
aggregate path uses a direct private `vllm serve` endpoint. The two-node
1-prefill/1-decode path uses the official vLLM Router and vLLM's ROCm-only
`MoRIIOConnector`: srt-slurm owns the router discovery port and generates
role-aware worker registration config from the realized Slurm topology. The
control-plane endpoints use automatic RFC1918-preferring discovery because
private NIC names vary across MI300X node generations. The earlier
Dynamo/NIXL experiment reached KV-cache initialization but failed ROCm memory
registration; that NVIDIA-oriented data plane is now explicitly out of scope
rather than patched into the AMD implementation.

The aggregate recipe has completed end to end on MI300X with both fixed-length
concurrency points. The disaggregated recipe pins ROCm's supported AITER Flash
Attention backend for both roles. This is required by the released MoRI-IO
connector's registered-memory contract: AITER exposes a contiguous logical KV
cache tensor, whereas the default Triton NHD view is strided and cannot be
registered by `mori.io` without copying or patching vLLM.
Original file line number Diff line number Diff line change
@@ -0,0 +1,99 @@
# Small two-worker aggregate correctness lane for the srt-slurm ATOM backend
# and Infera's dynamic KV-aware router. This validates orchestration, not tuning.

name: "mi300x-atom-qwen3-0.6b-agg-2w-fixed-seq"

model:
path: "hf:Qwen/Qwen3-0.6B"
container: "infera-atom-v0.1.1"
precision: "fp16"

identity:
model:
repo: "Qwen/Qwen3-0.6B"
container:
image: "rocm/infera:atom-v0.1.1"
frameworks:
atom: "0.1.4.dev113+g5837907f3"
infera: "0.0.0"

slurm:
time_limit: "00:45:00"

resources:
gpu_type: "mi300x"
gpus_per_node: 8
agg_nodes: 1
agg_workers: 2
gpus_per_agg: 1

frontend:
type: infera
enable_multiple_frontends: false
env:
PYTHONPATH: "/atom-source:/infera-source"
args:
router-policy: kv-aware

backend:
type: atom
enable_kv_events: true
aggregated_environment:
HF_HOME: "/hf_hub_cache"
HF_HUB_CACHE: "/hf_hub_cache"
HUGGINGFACE_HUB_CACHE: "/hf_hub_cache"
PYTHONPATH: "/atom-source:/infera-source"
PYTHONUNBUFFERED: "1"
OMP_NUM_THREADS: "1"
atom_config:
aggregated:
kv_cache_dtype: fp8
gpu-memory-utilization: 0.50
max-model-len: 2048
max-num-seqs: 8
block-size: 16
enforce-eager: true

srun_options:
container-writable: ""
container-remap-root: ""
mem: "0"

health_check:
max_attempts: 240
interval_seconds: 5

benchmark:
type: custom
command: |
set -euo pipefail
result_root="/results/${SLURM_JOB_ID}"
mkdir -p "${result_root}/fixed-seq"
trap 'tar -C /logs -czf "'"${result_root}"'/runtime-logs.tar.gz" . 2>/dev/null || true' EXIT
for concurrency in 1 4; do
python3 /infmax-workspace/utils/bench_serving/benchmark_serving.py \
--backend openai \
--base-url "http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}" \
--endpoint /v1/completions \
--model Qwen/Qwen3-0.6B \
--tokenizer Qwen/Qwen3-0.6B \
--dataset-name random \
--random-input-len 128 \
--random-output-len 32 \
--random-prefix-len 96 \
--random-range-ratio 1.0 \
--random-num-workers 1 \
--num-warmups "${concurrency}" \
--num-prompts "$((concurrency * 4))" \
--max-concurrency "${concurrency}" \
--request-rate inf \
--ignore-eos \
--disable-tqdm \
--save-result \
--result-dir "${result_root}/fixed-seq" \
--result-filename "qwen3-0.6b-atom-agg-isl128-osl32-c${concurrency}.json";
done
env:
HF_HOME: /hf_hub_cache
HF_HUB_CACHE: /hf_hub_cache
HUGGINGFACE_HUB_CACHE: /hf_hub_cache
Original file line number Diff line number Diff line change
@@ -0,0 +1,110 @@
# Small two-node ATOM P/D correctness lane. Infera discovers both workers,
# routes the completions API, and ATOM transfers KV through Mooncake.

name: "mi300x-atom-qwen3-0.6b-disagg-1p1d-fixed-seq"

model:
path: "hf:Qwen/Qwen3-0.6B"
container: "infera-atom-v0.1.1"
precision: "fp16"

identity:
model:
repo: "Qwen/Qwen3-0.6B"
container:
image: "rocm/infera:atom-v0.1.1"
frameworks:
atom: "0.1.4.dev113+g5837907f3"
infera: "0.0.0"

slurm:
time_limit: "00:45:00"

resources:
gpu_type: "mi300x"
gpus_per_node: 8
prefill_nodes: 1
decode_nodes: 1
prefill_workers: 1
decode_workers: 1
gpus_per_prefill: 1
gpus_per_decode: 1

frontend:
type: infera
enable_multiple_frontends: false
env:
PYTHONPATH: "/atom-source:/infera-source"
args:
router-policy: kv-aware

backend:
type: atom
connector: mooncake
# The stable atom-v0.1.1 image bundles Mooncake before ROCm DMA-BUF memory
# registration support. Use its supported TCP transport for this correctness
# lane; connection pooling is supplied by the srt-slurm ATOM adapter.
mooncake_protocol: tcp
enable_kv_events: true
prefill_environment: &worker_environment
HF_HOME: "/hf_hub_cache"
HF_HUB_CACHE: "/hf_hub_cache"
HUGGINGFACE_HUB_CACHE: "/hf_hub_cache"
PYTHONPATH: "/atom-source:/infera-source"
PYTHONUNBUFFERED: "1"
OMP_NUM_THREADS: "1"
decode_environment: *worker_environment
atom_config:
prefill: &worker_config
kv_cache_dtype: fp8
gpu-memory-utilization: 0.50
max-model-len: 2048
max-num-seqs: 8
block-size: 16
enforce-eager: true
no-enable_prefix_caching: true
decode: *worker_config

srun_options:
container-writable: ""
container-remap-root: ""
mem: "0"

health_check:
max_attempts: 240
interval_seconds: 5

benchmark:
type: custom
command: |
set -euo pipefail
result_root="/results/${SLURM_JOB_ID}"
mkdir -p "${result_root}/fixed-seq"
trap 'tar -C /logs -czf "'"${result_root}"'/runtime-logs.tar.gz" . 2>/dev/null || true' EXIT
for concurrency in 1 4; do
python3 /infmax-workspace/utils/bench_serving/benchmark_serving.py \
--backend openai \
--base-url "http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}" \
--endpoint /v1/completions \
--model Qwen/Qwen3-0.6B \
--tokenizer Qwen/Qwen3-0.6B \
--dataset-name random \
--random-input-len 128 \
--random-output-len 32 \
--random-prefix-len 96 \
--random-range-ratio 1.0 \
--random-num-workers 1 \
--num-warmups "${concurrency}" \
--num-prompts "$((concurrency * 4))" \
--max-concurrency "${concurrency}" \
--request-rate inf \
--ignore-eos \
--disable-tqdm \
--save-result \
--result-dir "${result_root}/fixed-seq" \
--result-filename "qwen3-0.6b-atom-disagg-isl128-osl32-c${concurrency}.json";
done
env:
HF_HOME: /hf_hub_cache
HF_HUB_CACHE: /hf_hub_cache
HUGGINGFACE_HUB_CACHE: /hf_hub_cache
Loading