From 3db7de579f9faeb80d644bbca347fd305b7480ea Mon Sep 17 00:00:00 2001 From: Cemberk Date: Thu, 13 Aug 2026 17:17:13 +0000 Subject: [PATCH 01/92] feat(vllm_dissag): TP-aware wideEP topology + multi-node pools + niah selector The xPyD harness already models pools that span multiple nodes (xP/yD count NODES, roles are MASTER/CHILD via --data-parallel-start-rank + --headless), but the wideEP path hardcoded TP=1: dp_size was xP*GPUS_PER_NODE and the connector emitted a literal `-tp 1`. That makes any model whose REPLICATED (non-expert) weights exceed one GPU inexpressible - e.g. Kimi-K3 on MI300X, where TP1/DP16 needs 190.7 GiB/GPU (> 192 GB HBM) and TP2 is required to fit. - vllm_disagg.sh: topology math is TP-aware. dp_per_node = GPUS_PER_NODE/TP_SIZE feeds dp_size, dp_size_local and both start-rank calculations. TP_SIZE defaults to 1 and is validated to divide GPUS_PER_NODE. - moriio.sh: emit -tp ${TP_SIZE}; clamp --api-server-count to dp_size (it must be <= data-parallel-size or the frontend DP balancer routes to ranks that do not exist - only reachable once TP>1 shrinks dp_size below GPUS_PER_NODE). - moriio.sh: advertise the peer pool's node IPs as moriio_pod_hosts when a pool spans >1 node. Without it the connector falls back to the peer MASTER only and KV writes aimed at ranks on a peer CHILD node silently miss, so those ranks decode with no context. Emitted only when xP>1 || yD>1. - run_xPyD_models.slurm: forward TP_SIZE/GPUS_PER_NODE, wire BENCHMARK_SCRIPT=niah to the already-present benchmark_niah.sh, forward NIAH_* knobs. Default-neutral: at xP=1 yD=1 TP_SIZE=1 - the topology every registered entry uses - DRY_RUN output is byte-identical to develop across all four node roles. At xP=2 yD=2 TP_SIZE=2 it emits tp=2, dp_size=8, dp_local=4, child start-rank 4 (= EP16 per pool over 16 GPUs). Co-Authored-By: Claude Opus 5 (1M context) --- scripts/vllm_dissag/connectors/moriio.sh | 24 +++++++++-- scripts/vllm_dissag/run_xPyD_models.slurm | 19 +++++++-- scripts/vllm_dissag/vllm_disagg.sh | 49 ++++++++++++++++++++--- 3 files changed, 79 insertions(+), 13 deletions(-) diff --git a/scripts/vllm_dissag/connectors/moriio.sh b/scripts/vllm_dissag/connectors/moriio.sh index 44c64d68..4c740c25 100644 --- a/scripts/vllm_dissag/connectors/moriio.sh +++ b/scripts/vllm_dissag/connectors/moriio.sh @@ -120,7 +120,15 @@ connector_setup_env() { _moriio_build_kv_transfer_config() { local kv_role="$1" - echo '{"kv_connector":"MoRIIOConnector","kv_role":"'"${kv_role}"'","kv_port":"'"${KV_PORT}"'","kv_connector_extra_config":{"proxy_ip":"'"${MASTER_ADDR}"'","proxy_port":"'"${PROXY_PORT}"'","proxy_ping_port":"'"${PROXY_PING_PORT}"'","http_port":"'"${SERVE_PORT}"'","local_ping_port":"'"${LOCAL_PING_PORT}"'","handshake_port":"'"${HANDSHAKE_PORT}"'","notify_port":"'"${NOTIFY_PORT}"'"}}' + # Peer-pool node list. A kv_producer (prefill) handshakes the DECODE pool, a + # kv_consumer (decode) notifies the PREFILL pool. The driver leaves both empty + # for single-node pools (xP=1 && yD=1), in which case the key is omitted and + # the emitted config is byte-identical to the historical one. + local _peer="" + if [[ "${kv_role}" == "kv_producer" ]]; then _peer="${DECODE_POD_HOSTS:-}"; else _peer="${PREFILL_POD_HOSTS:-}"; fi + local _pod_hosts="" + [[ -n "${_peer}" ]] && _pod_hosts=',"moriio_pod_hosts":"'"${_peer}"'"' + echo '{"kv_connector":"MoRIIOConnector","kv_role":"'"${kv_role}"'","kv_port":"'"${KV_PORT}"'","kv_connector_extra_config":{"proxy_ip":"'"${MASTER_ADDR}"'","proxy_port":"'"${PROXY_PORT}"'","proxy_ping_port":"'"${PROXY_PING_PORT}"'","http_port":"'"${SERVE_PORT}"'","local_ping_port":"'"${LOCAL_PING_PORT}"'","handshake_port":"'"${HANDSHAKE_PORT}"'","notify_port":"'"${NOTIFY_PORT}"'"'"${_pod_hosts}"'}}' } connector_runtime_patch() { @@ -184,7 +192,15 @@ connector_launch_worker() { local extra_args=() kv_args=() if [[ "$role" == "master" ]]; then - extra_args+=(--api-server-count=${_GPUS_PER_NODE}) + # api-server-count MUST be <= data-parallel-size: the frontend DP + # load-balancer round-robins over data_parallel_rank [0, count), so a + # count above dp_size sends requests to ranks that do not exist. At + # TP1 dp_size >= GPUS_PER_NODE and this resolves to GPUS_PER_NODE + # exactly as before; at TP>1 (dp_size = GPUS_PER_NODE/TP per node) it + # correctly clamps down. + local _api_servers="${_GPUS_PER_NODE}" + [ "${dp_size}" -lt "${_api_servers}" ] && _api_servers="${dp_size}" + extra_args+=(--api-server-count=${_api_servers}) local kv_config; kv_config=$(_moriio_build_kv_transfer_config "${kv_role}") kv_args+=(--kv-transfer-config "${kv_config}") else @@ -204,7 +220,7 @@ connector_launch_worker() { if [[ "${DRY_RUN:-0}" == "1" ]]; then _dryrun_emit "moriio" "${log_prefix}" "${role}" \ vllm serve "${MODEL_PATH}" \ - -tp 1 \ + -tp "${TP_SIZE:-1}" \ --data-parallel-size "${dp_size}" \ --data-parallel-size-local "${DP_PARALLEL_SIZE_LOCAL}" \ --data-parallel-address "${dp_addr}" \ @@ -224,7 +240,7 @@ connector_launch_worker() { fi vllm serve ${MODEL_PATH} \ - -tp 1 \ + -tp "${TP_SIZE:-1}" \ --data-parallel-size "${dp_size}" \ --data-parallel-size-local ${DP_PARALLEL_SIZE_LOCAL} \ --data-parallel-address "${dp_addr}" \ diff --git a/scripts/vllm_dissag/run_xPyD_models.slurm b/scripts/vllm_dissag/run_xPyD_models.slurm index c71fc7e8..8a5b8f32 100755 --- a/scripts/vllm_dissag/run_xPyD_models.slurm +++ b/scripts/vllm_dissag/run_xPyD_models.slurm @@ -247,8 +247,12 @@ export DOCKER_IMAGE_NAME NIXL_REPO_DIR=$(pwd) LOG_PATH="${LOG_PATH:-/shared_inference/${USER}/model_blog_logs}" -xP="${xP:-1}" #-> Number of Prefill Servers -yD="${yD:-1}" #-> Number of Decode Servers +xP="${xP:-1}" #-> Number of Prefill NODES (a pool may span >1 node) +yD="${yD:-1}" #-> Number of Decode NODES (a pool may span >1 node) +# Tensor-parallel degree within each DP rank on the wideEP path. Default 1 keeps +# the historical one-DP-rank-per-GPU layout; set >1 only when the replicated +# (non-expert) weights do not fit a single GPU. See vllm_disagg.sh topology math. +TP_SIZE="${TP_SIZE:-1}" MODEL_DIR="${MODEL_DIR:-"/shared_inference/models_blog/"}" @@ -328,6 +332,8 @@ echo "" # Calculate NUM_NODES based on xP and yD NUM_NODES=$((xP + yD)) echo "Calculated NUM_NODES: $NUM_NODES (xP=$xP + yD=$yD, proxy co-located on prefill master)" +echo "Parallelism: TP_SIZE=$TP_SIZE -> DP ranks/node = ${GPUS_PER_NODE:-8}/$TP_SIZE" +export TP_SIZE # DeepEP configuration (only exported when RUN_DEEPEP=1) if [[ "$_run_deepep" == "1" ]]; then @@ -427,11 +433,13 @@ BENCHMARK_COMBINATIONS="${BENCHMARK_COMBINATIONS:-}" # Benchmark script selector: BENCHMARK_SCRIPT tag -> file run by the launcher. # sweep (default) -> benchmark_xPyD.sh (general concurrency sweep) # long_context -> benchmark_long_context.sh (per-shape warmup, c=1-first) +# niah -> benchmark_niah.sh (long-context retrieval/accuracy) BENCHMARK_SCRIPT="${BENCHMARK_SCRIPT:-sweep}" case "$BENCHMARK_SCRIPT" in sweep) BENCHMARK_SCRIPT_FILE="benchmark_xPyD.sh" ;; long_context) BENCHMARK_SCRIPT_FILE="benchmark_long_context.sh" ;; - *) echo "Error: invalid BENCHMARK_SCRIPT='$BENCHMARK_SCRIPT' (valid: sweep, long_context)" >&2; exit 1 ;; + niah) BENCHMARK_SCRIPT_FILE="benchmark_niah.sh" ;; + *) echo "Error: invalid BENCHMARK_SCRIPT='$BENCHMARK_SCRIPT' (valid: sweep, long_context, niah)" >&2; exit 1 ;; esac if [[ ! -f "$BENCHMARK_SCRIPT_FILE" ]]; then echo "Error: selected benchmark script '$BENCHMARK_SCRIPT_FILE' not found in $(pwd)." >&2 @@ -566,6 +574,8 @@ docker run --rm \ -e NIXL_COOKBOOK_PATH=$NIXL_COOKBOOK_PATH \ -e xP=$xP \ -e yD=$yD \ + ${TP_SIZE:+-e TP_SIZE=$TP_SIZE} \ + ${GPUS_PER_NODE:+-e GPUS_PER_NODE=$GPUS_PER_NODE} \ -e USER_NAME=$USER_NAME \ -e MODEL_NAME=$MODEL_NAME \ -e BENCHMARK_ITR=$BENCHMARK_ITR \ @@ -590,6 +600,9 @@ docker run --rm \ ${KV_CACHE_DTYPE:+-e KV_CACHE_DTYPE=$KV_CACHE_DTYPE} \ ${MORIIO_TOY_PROXY:+-e MORIIO_TOY_PROXY=$MORIIO_TOY_PROXY} \ ${BENCHMARK_SCRIPT_FILE:+-e BENCHMARK_SCRIPT_FILE=$BENCHMARK_SCRIPT_FILE} \ + ${NIAH_WORDS:+-e NIAH_WORDS="$NIAH_WORDS"} \ + ${NIAH_MAXTOK:+-e NIAH_MAXTOK=$NIAH_MAXTOK} \ + ${NIAH_TIMEOUT:+-e NIAH_TIMEOUT=$NIAH_TIMEOUT} \ ${PREFILL_CUDAGRAPH_MODE:+-e PREFILL_CUDAGRAPH_MODE=$PREFILL_CUDAGRAPH_MODE} \ ${DECODE_CUDAGRAPH_MODE:+-e DECODE_CUDAGRAPH_MODE=$DECODE_CUDAGRAPH_MODE} \ ${CUDAGRAPH_CAPTURE_SIZES:+-e CUDAGRAPH_CAPTURE_SIZES="$CUDAGRAPH_CAPTURE_SIZES"} \ diff --git a/scripts/vllm_dissag/vllm_disagg.sh b/scripts/vllm_dissag/vllm_disagg.sh index 06fbf84f..1bd94191 100755 --- a/scripts/vllm_dissag/vllm_disagg.sh +++ b/scripts/vllm_dissag/vllm_disagg.sh @@ -79,7 +79,7 @@ NNODES="${NNODES:-1}" MODEL_NAME="${MODEL_NAME:-}" xP="${xP:-1}" yD="${yD:-1}" -echo "[vllm_disagg] topology: xP=${xP} yD=${yD} (total nodes=$((xP + yD)))" +echo "[vllm_disagg] topology: xP=${xP} yD=${yD} (total nodes=$((xP + yD))) TP_SIZE=${TP_SIZE:-1}" IPADDRS="${IPADDRS:-localhost}" IFS=',' read -ra IP_ARRAY <<< "${IPADDRS}" @@ -92,14 +92,49 @@ host_name=$(hostname) # ============================================================================= # Topology math # ============================================================================= +# TP_SIZE is the tensor-parallel degree WITHIN each DP rank on the wideEP path. +# Historically this path was TP1 (one DP rank per GPU), so TP_SIZE defaults to 1 +# and every pre-existing model resolves to exactly the old numbers. +# +# TP>1 is needed when the REPLICATED (non-expert) weights do not fit one GPU: +# e.g. Kimi-K3 on MI300X has 106.5 GiB of replicated attn + shared-expert weight, +# so TP1/DP16 would need 190.7 GiB/GPU (> 192 GB HBM once the MoRI heap and KV +# cache are counted). TP2 halves that to 53.3 GiB/GPU and the model fits. +# +# dp_per_node = GPUS_PER_NODE / TP_SIZE (DP ranks hosted on one node) +# pool DP size = nodes_in_pool * dp_per_node (EP width = pool DP size * TP_SIZE) _GPUS_PER_NODE="${GPUS_PER_NODE:-8}" -PREFILL_DP_SIZE=$((xP * _GPUS_PER_NODE)) -DECODE_DP_SIZE=$((yD * _GPUS_PER_NODE)) -DP_PARALLEL_SIZE_LOCAL=${_GPUS_PER_NODE} -PREFILL_DP_START_RANK=$(( NODE_RANK * _GPUS_PER_NODE )) +TP_SIZE="${TP_SIZE:-1}" +if ! [[ "$TP_SIZE" =~ ^[0-9]+$ ]] || [ "$TP_SIZE" -lt 1 ]; then + echo "Error: invalid TP_SIZE='${TP_SIZE}' (expected a positive integer)." >&2; exit 1 +fi +if [ $(( _GPUS_PER_NODE % TP_SIZE )) -ne 0 ]; then + echo "Error: TP_SIZE=${TP_SIZE} does not divide GPUS_PER_NODE=${_GPUS_PER_NODE}." >&2; exit 1 +fi +_DP_PER_NODE=$(( _GPUS_PER_NODE / TP_SIZE )) +PREFILL_DP_SIZE=$((xP * _DP_PER_NODE)) +DECODE_DP_SIZE=$((yD * _DP_PER_NODE)) +DP_PARALLEL_SIZE_LOCAL=${_DP_PER_NODE} +PREFILL_DP_START_RANK=$(( NODE_RANK * _DP_PER_NODE )) PREFILL_MASTER_ADDR=$(echo "$IPADDRS" | awk -F',' '{print $1}') -DECODE_DP_START_RANK=$(( (NODE_RANK - xP) * _GPUS_PER_NODE )) +DECODE_DP_START_RANK=$(( (NODE_RANK - xP) * _DP_PER_NODE )) DECODE_MASTER_ADDR=$(echo "$IPADDRS" | awk -F',' -v pos="$xP" '{print $(pos+1)}') +export TP_SIZE + +# Peer-pool node IPs, ordered by pod index (= global_dp_rank / dp_per_node). +# A pool that spans MORE THAN ONE node must advertise every peer node to the KV +# connector: otherwise the connector falls back to the peer pool's MASTER only, +# and KV writes/notifies aimed at DP ranks living on a peer CHILD node silently +# miss -> those ranks decode with no context. Single-node pools (xP=1 && yD=1, +# the historical case) do not need this, so it stays empty there and the emitted +# command line is unchanged. +PREFILL_POD_HOSTS="" +DECODE_POD_HOSTS="" +if [ "$xP" -gt 1 ] || [ "$yD" -gt 1 ]; then + PREFILL_POD_HOSTS=$(printf '%s\n' "${IP_ARRAY[@]:0:$xP}" | paste -sd, -) + DECODE_POD_HOSTS=$(printf '%s\n' "${IP_ARRAY[@]:$xP:$yD}" | paste -sd, -) +fi +export PREFILL_POD_HOSTS DECODE_POD_HOSTS # ============================================================================= # Driver helper functions (shared by all connectors) @@ -214,7 +249,9 @@ echo "-----------------------------Printing node specific details -------------- echo "IPADDRS = ${IPADDRS}" echo "MASTER_ADDR=${MASTER_ADDR}" echo "PREFILL_DP_SIZE=${PREFILL_DP_SIZE} DECODE_DP_SIZE=${DECODE_DP_SIZE}" +echo "TP_SIZE=${TP_SIZE} DP_PER_NODE=${DP_PARALLEL_SIZE_LOCAL} (EP width per pool: prefill=$((PREFILL_DP_SIZE * TP_SIZE)) decode=$((DECODE_DP_SIZE * TP_SIZE)))" echo "PREFILL_MASTER_ADDR=${PREFILL_MASTER_ADDR} DECODE_MASTER_ADDR=${DECODE_MASTER_ADDR}" +[ -n "${PREFILL_POD_HOSTS}" ] && echo "PREFILL_POD_HOSTS=${PREFILL_POD_HOSTS} DECODE_POD_HOSTS=${DECODE_POD_HOSTS}" # ============================================================================= # Container barrier + runtime patches (skipped under DRY_RUN) From a5d7eb6e563269facde679c81ee92aa78b59e765 Mon Sep 17 00:00:00 2001 From: Cemberk Date: Thu, 13 Aug 2026 17:23:48 +0000 Subject: [PATCH 02/92] feat(vllm_dissag): honor models.yaml dp: flags on moriio wideEP; add Kimi-K3 recipe Two parts. 1) moriio.sh wideEP dropped ${model_args[@]} on the floor. models.yaml documents per-model dp: tuning as supported ("both connectors now append it") and rixl's deepep path does append it, but the moriio wideEP branch parsed the flags, logged them, and then emitted an argv without them. Verified with a sentinel flag: present in the log line, absent from the command. Now appended, matching rixl. Every existing wideEP model has an empty dp: block, so emitted argv is unchanged for all of them. 2) Kimi-K3 recipe, entirely as data: env: block for the gfx942 knobs (VLLM_ROCM_USE_AITER_MLA=0 - the AITER MLA kernel is gfx950-only - AITER_SITUV2_A8W4=1 for the packed-int4 SiTUv2 MoE path, KDA conv state layout, 40e9 KV cache for >600K contexts, per-role cudagraph + MoRI backends), and a dp: block for the serve flags the connector does not emit (reasoning parser, 1M max-model-len, batched-tokens pinned at 2048, MoE quantization-config). Registered in MORI_EP_VALID_MODELS and WIDE_EP_ONLY_MODELS. The JSON quantization-config carries its own shell quotes: the launcher word-splits dp: via eval, which would otherwise strip the JSON's double quotes and hand vLLM invalid JSON. Verified the emitted value parses as JSON. Co-Authored-By: Claude Opus 5 (1M context) --- scripts/vllm_dissag/connectors/moriio.sh | 3 +- scripts/vllm_dissag/models.yaml | 68 +++++++++++++++++++++++ scripts/vllm_dissag/run_xPyD_models.slurm | 3 +- 3 files changed, 72 insertions(+), 2 deletions(-) diff --git a/scripts/vllm_dissag/connectors/moriio.sh b/scripts/vllm_dissag/connectors/moriio.sh index 4c740c25..8c8be2a1 100644 --- a/scripts/vllm_dissag/connectors/moriio.sh +++ b/scripts/vllm_dissag/connectors/moriio.sh @@ -235,7 +235,7 @@ connector_launch_worker() { --all2all-backend "${_all2all}" \ --trust-remote-code \ --distributed-timeout-seconds "${DISTRIBUTED_TIMEOUT_SECONDS:-7200}" \ - "${exec_args[@]}" "${extra_args[@]}" "${kv_args[@]}" + "${exec_args[@]}" "${model_args[@]}" "${extra_args[@]}" "${kv_args[@]}" WORKER_PID=0; return 0 fi @@ -256,6 +256,7 @@ connector_launch_worker() { --trust-remote-code \ --distributed-timeout-seconds ${DISTRIBUTED_TIMEOUT_SECONDS:-7200} \ "${exec_args[@]}" \ + "${model_args[@]}" \ "${extra_args[@]}" \ "${kv_args[@]}" \ 2>&1 | tee /run_logs/${SLURM_JOB_ID}/${log_prefix}_NODE${NODE_RANK}.log >/dev/null & diff --git a/scripts/vllm_dissag/models.yaml b/scripts/vllm_dissag/models.yaml index 23d66059..131766e6 100644 --- a/scripts/vllm_dissag/models.yaml +++ b/scripts/vllm_dissag/models.yaml @@ -178,3 +178,71 @@ DeepSeek-R1: dp: "" decode: dp: "" + +# ============================ Kimi-K3 (MXFP4) on MI300X / gfx942 ============================ +# wideEP-only, and the first entry that needs TP>1 on the wideEP path. +# +# WHY TP2. K3 is ~2.8T params / 896 experts with a hybrid MLA + Kimi-Delta-Attention +# stack. On MI300X (192 GB/GPU) the REPLICATED (attn + shared-expert) weight is +# 106.5 GiB. At TP1/DP16 that lands 190.7 GiB/GPU before KV cache or the MoRI heap +# -> does not fit. TP2 shards it to 53.3 GiB/GPU; with 84.2 GiB of expert shard that +# is 137.5 GiB + 16 GiB MoRI heap, leaving room for KV. So the pool is TP2 x DP8 = +# EP16 over 16 GPUs (2 nodes), i.e. xP=2 yD=2 TP_SIZE=2 (see models.json). +# +# gfx942 specifics: +# - VLLM_ROCM_USE_AITER_MLA=0 is REQUIRED: the AITER MLA kernel is gfx950-only. +# - gfx942 has no scaled-MXFP4 MFMA and the a16w4 SiTUv2 heuristic FlyDSL kernel +# cannot codegen there, so the MoE is requantized to packed-int4 and run through +# SiTUv2 (AITER_SITUV2_A8W4=1 + the --quantization-config below). +# - MAX_NUM_BATCHED_TOKENS stays at 2048: 8192 corrupts generation on this stack, +# and the giant profiling shape at 16384 crashes LLVM codegen in the heuristic +# FlyDSL kernel. +# - KV_CACHE_MEMORY_BYTES=40e9 gives a 2.84M-token GPU KV cache, required for +# single requests beyond ~600K tokens. It also skips the boot profiling forward. +_kimi_k3_recipe_env: &kimi_k3_recipe_env + VLLM_USE_V1: "1" + VLLM_ROCM_USE_AITER: "1" + VLLM_ROCM_USE_AITER_MOE: "1" + VLLM_ROCM_USE_AITER_MLA: "0" # REQUIRED on gfx942 (AITER MLA is gfx950-only) + VLLM_ROCM_USE_AITER_RMSNORM: "1" + VLLM_ROCM_USE_AITER_FUSION_SHARED_EXPERTS: "0" + VLLM_USE_AITER_TRITON_SILU_MUL: "0" + AITER_SITUV2_A8W4: "1" # a8w4 (fp8-act x int4-wt) SiTU MoE path + VLLM_SSM_CONV_STATE_LAYOUT: "DS" # KDA (Kimi-Delta-Attention) conv state layout + KV_BLOCK_SIZE: "16" + KV_CACHE_DTYPE: "fp8" + KV_CACHE_MEMORY_BYTES: "40000000000" # 2.84M-token KV cache; needed for >600K ctx + GPU_MEMORY_UTILIZATION: "0.88" # 0.88 leaves the 16 GiB MoRI shmem heap + PREFILL_CUDAGRAPH_MODE: "NONE" + DECODE_CUDAGRAPH_MODE: "PIECEWISE" + VLLM_ALL2ALL_BACKEND: "mori_high_throughput" + PREFILL_MORI_BACKEND: "mori_high_throughput" + DECODE_MORI_BACKEND: "mori_low_latency" + MORI_SHMEM_HEAP_SIZE: "17179869184" + MORI_NUM_QP_PER_PE: "8" + VLLM_MORIIO_QP_PER_TRANSFER: "2" + VLLM_MORIIO_NUM_WORKERS: "4" + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + DISTRIBUTED_TIMEOUT_SECONDS: "7200" + +# Serve flags the connector does not already emit. --max-model-len 1000000 is K3's +# full native context; drop it to 10240 for a quick smoke run. +# +# NOTE: these strings are word-split by the launcher via `eval "model_args=(...)"`, +# so any value containing JSON must carry its OWN shell quotes - otherwise shell +# quote-removal strips the JSON's double quotes and vLLM gets invalid JSON. +_kimi_k3_dp_flags: &kimi_k3_dp_flags >- + --reasoning-parser kimi_k3 + --mm-encoder-tp-mode data + --safetensors-load-strategy prefetch + --max-model-len 1000000 + --max-num-seqs 8 + --max-num-batched-tokens 2048 + --quantization-config '{"moe":{"weight":"int4_per_group_32"}}' + +Kimi-K3: + env: *kimi_k3_recipe_env + prefill: + dp: *kimi_k3_dp_flags + decode: + dp: *kimi_k3_dp_flags diff --git a/scripts/vllm_dissag/run_xPyD_models.slurm b/scripts/vllm_dissag/run_xPyD_models.slurm index 8a5b8f32..1753f8b4 100755 --- a/scripts/vllm_dissag/run_xPyD_models.slurm +++ b/scripts/vllm_dissag/run_xPyD_models.slurm @@ -94,6 +94,7 @@ MORI_EP_VALID_MODELS=( \ "DeepSeek-V3" \ "DeepSeek-V3-5layer" \ "DeepSeek-R1" \ + "Kimi-K3" \ ) # Models allowed for CONNECTOR=rixl WIDE_EP=1 EP_BACKEND=deepep (legacy RUN_DEEPEP=1) @@ -179,7 +180,7 @@ WIDE_EP="${WIDE_EP:-0}" # the MoRI-EP / DeepEP recipe (block=16, MLA off, per-role cudagraph). Running them # in TP mode is unsupported — the TP argv would double the model's own # --compilation-config and drop the mandatory +quant_fp8 op. Reject early. -WIDE_EP_ONLY_MODELS=( "DeepSeek-V3" "DeepSeek-V3-5layer" "DeepSeek-R1" ) +WIDE_EP_ONLY_MODELS=( "DeepSeek-V3" "DeepSeek-V3-5layer" "DeepSeek-R1" "Kimi-K3" ) model_is_wide_ep_only() { local m="$1" for x in "${WIDE_EP_ONLY_MODELS[@]}"; do [[ "$m" == "$x" ]] && return 0; done From 01c56bb8dc32d3136c419c7b56dcc65914cbe004 Mon Sep 17 00:00:00 2001 From: Cemberk Date: Thu, 13 Aug 2026 17:32:06 +0000 Subject: [PATCH 03/92] feat(kimi-k3): MI300X image + colocated multi-node vLLM harness docker/pyt_vllm_kimi_k3_mi300x.ubuntu.amd.Dockerfile The shared vllm_disagg_inference stack with the pins K3 needs on gfx942 (MoRI 1.2.2, AITER 0.1.19 + flydsl 0.2.4, the K3+MoRIIO vLLM commit) plus the K3-aware AITER graft from the public vendor image - without that graft the K3 MoE profiling shape finds no tuned FlyDSL config, falls back to a heuristic kernel and aborts LLVM inside determine_available_memory. Differences from the recipe this is ported from: - every source pinned to an immutable SHA, not a fork branch name (branches on personal forks can be force-pushed; MAD needs the image rebuildable later) - GH_TOKEN build-arg dropped: all three repos are public, and a token passed this way is recorded in image metadata - the flydsl 0.2.4 re-pin happens at build time, so the launcher no longer runs pip install inside every container at serve time - WITH_NIXL defaults to 0 (K3 uses the moriio connector only) - no runtime patchers: all connector/KDA fixes are committed in the pinned vLLM scripts/vllm_multinode/ Colocated (single-instance) multi-node serving - the counterpart to vllm_dissag, for models that do not fit one node but want lowest single-request latency rather than disagg throughput. TP within a node, PP across nodes. It owns only node discovery, the container launch and the head/worker split; models.yaml, socket_barrier.py, benchmark_xPyD.sh, benchmark_niah.sh and parse_to_csv.py are reused from vllm_dissag (scripts/ is mounted whole), so the model recipe and the CSV pipeline have one home. The three K3 colocated variants collapse to this one script: dry-run confirms pp2xtp8 / allgather / moriep argv differ only by --enable-expert-parallel and --all2all-backend, so they become three models.json entries rather than three near-identical run.sh copies. Not built here - this environment has no docker and one 8-GPU node; the harness is verified by DRY_RUN argv inspection only. Co-Authored-By: Claude Opus 5 (1M context) --- ..._vllm_kimi_k3_mi300x.ubuntu.amd.Dockerfile | 346 ++++++++++++++++++ scripts/vllm_multinode/run_multinode.slurm | 215 +++++++++++ scripts/vllm_multinode/serve_colocated.sh | 166 +++++++++ 3 files changed, 727 insertions(+) create mode 100644 docker/pyt_vllm_kimi_k3_mi300x.ubuntu.amd.Dockerfile create mode 100755 scripts/vllm_multinode/run_multinode.slurm create mode 100755 scripts/vllm_multinode/serve_colocated.sh diff --git a/docker/pyt_vllm_kimi_k3_mi300x.ubuntu.amd.Dockerfile b/docker/pyt_vllm_kimi_k3_mi300x.ubuntu.amd.Dockerfile new file mode 100644 index 00000000..17b8d3f7 --- /dev/null +++ b/docker/pyt_vllm_kimi_k3_mi300x.ubuntu.amd.Dockerfile @@ -0,0 +1,346 @@ +# CONTEXT {'gpu_vendor': 'AMD', 'guest_os': 'UBUNTU'} +############################################################################### +# +# MIT License +# +# Copyright (c) 2026 Advanced Micro Devices, Inc. +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. +# +################################################################################# +# ============================================================================= +# pyt_vllm_kimi_k3_mi300x.ubuntu.amd.Dockerfile +# +# Kimi-K3 (MXFP4) wide-EP + MoRIIO disaggregated serving on MI300X / gfx942. +# +# This is docker/vllm_disagg_inference.ubuntu.amd.Dockerfile's stack with the +# component pins Kimi-K3 requires on gfx942, plus one extra step (5b) that +# grafts a K3-aware AITER. It is a SEPARATE image rather than a build-arg on +# the shared one because three pins move together (MoRI 1.2.2, AITER 0.1.19 + +# flydsl 0.2.4, a different vLLM commit) and the shared image is live-proven +# for DeepSeek at its own pins. +# +# Build (context = repo root): +# docker build -f docker/pyt_vllm_kimi_k3_mi300x.ubuntu.amd.Dockerfile \ +# -t /vllm-kimi-k3-mi300x:local . +# export DOCKER_IMAGE_NAME=/vllm-kimi-k3-mi300x:local +# +# WITH_NIXL defaults to 0 here: K3 runs on the moriio connector (MoRI-EP wideEP), +# so the UCX/RIXL/rocSHMEM/DeepEP layer is dead weight (~30-45 min of build). +# Set --build-arg WITH_NIXL=1 only if you also want the rixl connector present. +# +# WHY K3 NEEDS ITS OWN PINS ON gfx942 +# - AITER MLA is gfx950-only, so VLLM_ROCM_USE_AITER_MLA=0 is mandatory. That is +# a RUNTIME knob and lives in scripts/vllm_dissag/models.yaml, not here. +# - gfx942 has no scaled-MXFP4 MFMA, and the a16w4 SiTUv2 heuristic FlyDSL kernel +# cannot codegen there. The MoE therefore runs requantized to packed-int4 +# through SiTUv2, which needs AITER >= 0.1.19 AND flydsl >= 0.2.4 AND the +# K3-tuned FlyDSL MoE configs grafted in step 5b. +# - K3's vision tower imports aiter.ops.triton.conv.conv2d, first present in +# 0.1.19 (0.1.16.post3 raises "No module named 'aiter.ops.triton.conv'"). +# +# NO RUNTIME PATCHERS +# All Kimi-K3 MoRIIO connector fixes (4-KV-cache-group block routing, multi-chunk +# compute-progress prefill gate, KDA gather sync-free) are committed in the vLLM +# source this builds (VLLM_REF below), exactly as the shared disagg image handles +# its own MoRIIO fixes. Nothing patches site-packages at container start. +# +# PINNING +# Every source is pinned to an immutable commit SHA, not a branch name: these are +# personal forks whose branches can be force-pushed or deleted, and MAD needs the +# image to be rebuildable to the same bits a year from now. The human-readable +# branch each SHA came from is in the comment above it. +# +# STATUS +# Ported from the recipe in PR #193 (validated there from a scratch build: +# single-needle NIAH passing 10K-900K on 2 prefill + 2 decode MI300X nodes). +# NOT yet rebuilt from this file in MAD CI - the pins are carried over verbatim +# apart from branch->SHA, WITH_NIXL default, and GH_TOKEN removal. +# ============================================================================= + +# Image ARGs consumed by FROM must be declared before the first FROM (buildkit +# global scope); declaring them later scopes them to a single stage and the second +# FROM resolves blank. +# +# K3-aware AITER donor (step 5b). Public, anonymously pullable. Same ROCm 7.2.3 + +# torch 2.11 base as BASE_IMAGE, so the grafted trees are ABI-compatible. +ARG PROVEN_K3_IMAGE=amdsiloai/vllm:kimi-k3-mi325x-release-v2 +# Open ROCm vLLM CI base - same one the shared disagg image builds on. +ARG BASE_IMAGE=rocm/vllm-dev:ci_base-0fcd9b99cc9d63202da4c858d8ebc6582c9e2491 + +FROM ${PROVEN_K3_IMAGE} AS proven_k3_aiter + +FROM ${BASE_IMAGE} + +ENTRYPOINT [] +WORKDIR /app + +ARG GFX_COMPILATION_ARCH="gfx942" +ARG PYTORCH_ROCM_ARCH="gfx942" +ARG MAX_JOBS=32 +ARG NVCC_THREADS=8 +# K3 uses the moriio connector only; default the NIXL/RIXL transport layer OFF. +ARG WITH_NIXL=0 +ARG NIC_COMPILATION_ARCH="cx7" + +# ----------------------------------------------------------------------------- +# 1. MoRI v1.2.2 (the K3 recipe's pin; the shared disagg image pins v1.2.1). +# JIT-built, so this swaps the sources the EP kernels compile from at runtime. +# Do NOT pass USE_IONIC=OFF / USE_BNXT=OFF - disabling NIC backends produced a +# MoRI that deadlocked at the cross-node EP all-to-all init. +# ----------------------------------------------------------------------------- +ARG MORI_REPO=https://github.com/ROCm/mori.git +# tag v1.2.2 +ARG MORI_REF=fe12a11a7d6c6acd0771b772366ed9ed5e0d3d44 +ENV MORI_GPU_ARCHS=gfx942 +# UMBP needs gRPC headers absent from this base and is unrelated to EP dispatch. +ENV BUILD_UMBP=OFF BUILD_UMBP_SPDK=OFF +RUN sed -i 's|http://|https://|g' /etc/apt/sources.list 2>/dev/null || true && \ + sed -i 's|http://|https://|g' /etc/apt/sources.list.d/*.list 2>/dev/null || true && \ + apt-get update && apt-get install -y --no-install-recommends \ + git build-essential cmake ninja-build ccache libssl-dev pkg-config curl ca-certificates && \ + pip install meson==0.64.0 "pybind11[global]" tqdm prettytable && \ + pip uninstall -y amd_mori amd-mori amd-mori-nightly mori 2>/dev/null || true && \ + rm -rf /tmp/mori-src && \ + git clone --recursive "${MORI_REPO}" /tmp/mori-src && \ + cd /tmp/mori-src && git checkout "${MORI_REF}" && git submodule update --init --recursive && \ + BUILD_UMBP=OFF pip install . && \ + python3 -c "import mori, mori.io, mori.ops; print('MoRI OK at', mori.__path__[0])" && \ + mkdir -p /app && echo "MORI_REF=${MORI_REF}" >> /app/versions.txt && \ + rm -rf /tmp/mori-src + +# ----------------------------------------------------------------------------- +# 2. AITER 0.1.19 + flydsl 0.2.4 (K3 pins; see header for why 0.1.16.post3 fails). +# 0.1.19 also carries the #3658 top_k_top_p HSA-fault fix needed for DP-EP disagg. +# Then invalidate the prewarmed JIT cache compiled against the old .so. +# ----------------------------------------------------------------------------- +ARG AITER_VERSION=0.1.19 +ARG AITER_WHEEL_URL="https://github.com/ROCm/aiter/releases/download/v0.1.19/amd_aiter-0.1.19%2Brocm7.2.manylinux.2.28-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl" +RUN echo "Bumping AITER to ${AITER_VERSION} from ${AITER_WHEEL_URL}" && \ + _W="/tmp/$(basename "${AITER_WHEEL_URL}" | sed 's/%2B/+/g')" && \ + curl -fL --retry 3 --retry-delay 2 -o "${_W}" "${AITER_WHEEL_URL}" && \ + (pip uninstall -y amd_aiter amd-aiter aiter 2>/dev/null || true) && \ + pip install --no-deps "${_W}" && \ + pip install "flydsl==0.2.4" && \ + rm -f "${_W}" && \ + python3 - <<'PYEOF' +from importlib.metadata import version as v, PackageNotFoundError +vm = None +for n in ("amd-aiter", "amd_aiter", "aiter"): + try: vm = v(n); break + except PackageNotFoundError: pass +assert vm and vm.split("+", 1)[0].startswith("0.1.19"), f"AITER not 0.1.19: {vm!r}" +print("AITER OK:", vm) +PYEOF +RUN rm -rf /opt/vllm_cache/aiter_jit /root/.aiter && echo "cleared stale AITER JIT cache" && \ + echo "AITER_VERSION=${AITER_VERSION}" >> /app/versions.txt + +# ----------------------------------------------------------------------------- +# 3. vLLM: full source compile of the K3 + MoRIIO branch. All K3 connector fixes +# are committed in this tree - there is no runtime patcher: +# - 4-KV-cache-group block routing (K3 allocates 3 KDA/mamba groups + 1 MLA; +# the stock connector hardcoded 2-group indices and sent MLA KV to mamba +# block ids, so decode read empty blocks and generated without context) +# - multi-chunk prefill transfer gated on compute progress, not block count +# (the block-count gate fired after chunk 1 for prompts fitting one padded +# block, so only max_num_batched_tokens of KV ever crossed) +# - KDA gather made sync-free (a per-layer per-chunk device->CPU sync that +# turned >500K-token prefills into an apparent hang) +# Repo is public; no credentials are needed or accepted here (the upstream +# recipe took a GH_TOKEN build-arg, which would bake the token into image +# metadata - removed). +# ----------------------------------------------------------------------------- +ARG VLLM_REPO=https://github.com/raviguptaamd/vllm.git +# branch kimi-k3-wideep-disagg-fullsource-v3 @ 2026-08-11 +ARG VLLM_REF=862bfd8ca4db78b9cbcbcf9ec6013638e3ae6543 +ENV VLLM_TARGET_DEVICE=rocm \ + PYTORCH_ROCM_ARCH=${PYTORCH_ROCM_ARCH} \ + MAX_JOBS=${MAX_JOBS} +# MAX_JOBS/NVCC_THREADS are set INLINE on the pip line: under the legacy builder +# the ENV above does not reach the pip build subprocess, and vLLM's setup.py +# compute_num_jobs then dies on `int("")`. +RUN rm -rf /tmp/vllm-src && \ + git clone "${VLLM_REPO}" /tmp/vllm-src && \ + cd /tmp/vllm-src && git checkout "${VLLM_REF}" && \ + echo "VLLM_REF=${VLLM_REF}" >> /app/versions.txt && \ + pip uninstall -y vllm 2>/dev/null || true && \ + MAX_JOBS="${MAX_JOBS:-32}" NVCC_THREADS="${NVCC_THREADS:-8}" \ + pip install --no-deps --no-build-isolation -v . && \ + python3 -c "import vllm; print('vLLM', vllm.__version__, 'from', vllm.__file__)" && \ + rm -rf /tmp/vllm-src + +# Cross-check MoRI + AITER survived the vLLM install (no silent downgrade). +RUN python3 - <<'PYEOF' +from importlib.metadata import version as v, PackageNotFoundError +def get(names): + for n in names: + try: return v(n) + except PackageNotFoundError: pass + return None +av = get(("amd-aiter", "amd_aiter", "aiter")) +assert av and av.split("+", 1)[0].startswith("0.1.19"), f"AITER downgraded: {av!r}" +import mori, mori.io, mori.ops +print("Post-vLLM check OK: AITER", av, "+ MoRI importable") +PYEOF + +# ----------------------------------------------------------------------------- +# 4. vllm-router: DP-rank round-robin + the 2P2D KV-notify fix +# (remote_dp_rank_override + remote_dp_size). Without the notify fix a 2P2D +# EP16 run reproducibly wedges on "remote blocks never arrived" deferred-write +# expiries, because decode's notify targets the wrong DP rank. +# Same source the shared disagg image uses; pinned to its SHA here. +# ----------------------------------------------------------------------------- +ARG ROUTER_REPO=https://github.com/raviguptaamd/router.git +# branch ravgupta/discovery-dp-rank-roundrobin +ARG ROUTER_REF=6409ac1409410f54c5ed39791aadc69ce80b7bf3 +ARG RUST_TOOLCHAIN=1.88.0 +RUN if ! command -v cargo >/dev/null 2>&1; then \ + curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y --default-toolchain "${RUST_TOOLCHAIN}"; \ + fi && \ + export PATH="/root/.cargo/bin:${PATH}" && \ + rm -rf /tmp/vllm-router-src && \ + git clone --filter=blob:none "${ROUTER_REPO}" /tmp/vllm-router-src && \ + cd /tmp/vllm-router-src && git checkout "${ROUTER_REF}" && \ + cargo build --release && \ + install -m 755 target/release/vllm-router /usr/local/bin/vllm-router && \ + vllm-router --help 2>&1 | grep -q moriio && \ + echo "VLLM_ROUTER_REF=${ROUTER_REF}" >> /app/versions.txt && \ + rm -rf /tmp/vllm-router-src + +# ----------------------------------------------------------------------------- +# 4b. Optional NIXL/RIXL transport (WITH_NIXL=1). OFF by default for K3 - the +# recipe runs the moriio connector only. Identical to the shared disagg image. +# ----------------------------------------------------------------------------- +ENV _ROCM_DIR=/opt/rocm \ + _UCX_SOURCE=https://github.com/ROCm/ucx.git \ + _UCX_BRANCH=da3fac2a \ + _UCX_INSTALL_DIR=/usr/local/ucx/ \ + _RIXL_SOURCE=https://github.com/ROCm/RIXL.git \ + _RIXL_BRANCH=f33a5599 \ + _RIXL_INSTALL_DIR=/usr/local/RIXL/install \ + _NIXLBENCH_INSTALL_DIR=/usr/local/RIXL +RUN if [ "${WITH_NIXL}" != "1" ]; then \ + echo "WITH_NIXL=${WITH_NIXL}: skipping UCX/RIXL/rocSHMEM/DeepEP (MoRI-EP only)"; \ + else set -e && \ + echo "WITH_NIXL=1: building UCX + RIXL + rocSHMEM + DeepEP" && \ + apt-get update && apt-get install -y \ + autoconf automake libtool autogen pkg-config m4 gcc make \ + librdmacm-dev rdmacm-utils infiniband-diags ibverbs-utils perftest ethtool \ + libibverbs-dev rdma-core strace libgflags-dev \ + libaio-dev liburing-dev libcpprest-dev libgrpc-dev libgrpc++-dev \ + libprotobuf-dev protobuf-compiler-grpc wget && \ + pip install meson==0.64.0 "pybind11[global]" pyyaml && \ + cd /tmp && git clone "${_UCX_SOURCE}" && cd ucx && git checkout "${_UCX_BRANCH}" && \ + ./autogen.sh && mkdir -p build && cd build && \ + ../configure --prefix="${_UCX_INSTALL_DIR}" --with-rocm="${_ROCM_DIR}" \ + --disable-go --disable-java --disable-assertions --enable-mt && \ + make -j && make install && \ + cd /tmp && wget -q https://github.com/google/googletest/archive/refs/tags/v1.14.0.tar.gz && \ + tar -xzf v1.14.0.tar.gz && cd googletest-1.14.0 && mkdir -p build && cd build && \ + cmake -DBUILD_SHARED_LIBS=on .. && make -j && make install && \ + cd /tmp && git clone "${_RIXL_SOURCE}" && cd RIXL && git checkout "${_RIXL_BRANCH}" && \ + meson setup build/ --prefix="${_RIXL_INSTALL_DIR}" -Ducx_path="${_UCX_INSTALL_DIR}" \ + -Ddisable_gds_backend=true -Dcudapath_inc="${_ROCM_DIR}/include" -Dcudapath_lib="${_ROCM_DIR}/lib" && \ + cd build && ninja && ninja install && cd /tmp/RIXL && \ + pip install --config-settings=setup-args="-Dcudapath_inc=${_ROCM_DIR}/include" \ + --config-settings=setup-args="-Dcudapath_lib=${_ROCM_DIR}/lib" \ + --config-settings=setup-args="-Ducx_path=${_UCX_INSTALL_DIR}" \ + --config-settings=setup-args="-Ddisable_gds_backend=true" . && \ + cd /tmp && git clone --no-checkout --filter=blob:none https://github.com/ROCm/rocm-systems.git && \ + cd rocm-systems && git sparse-checkout set --cone projects/rocshmem && git checkout develop && \ + mkdir -p /tmp/rocshmem-build && cd /tmp/rocshmem-build && \ + /tmp/rocm-systems/projects/rocshmem/scripts/build_configs/all_backends \ + -DUSE_EXTERNAL_MPI=OFF -DGPU_TARGETS="${GFX_COMPILATION_ARCH}" && \ + cd /tmp && git clone https://github.com/ROCm/DeepEP.git && cd DeepEP && \ + PYTORCH_ROCM_ARCH="${GFX_COMPILATION_ARCH}" CFLAGS="-O3 -fPIC" \ + CXXFLAGS="-O3 -fPIC --offload-arch=${GFX_COMPILATION_ARCH}" HIP_CXX_FLAGS="-O3 -fPIC" \ + python3 setup.py --variant rocm --nic "${NIC_COMPILATION_ARCH}" build develop && \ + echo "WITH_NIXL build complete" >> /app/versions.txt && \ + rm -rf /tmp/ucx /tmp/googletest-1.14.0 /tmp/v1.14.0.tar.gz /tmp/rocm-systems /tmp/rocshmem-build; \ + fi +ENV LD_LIBRARY_PATH="/usr/local/ucx/lib:/usr/local/lib:/usr/local/RIXL/install/lib:${LD_LIBRARY_PATH}" \ + PATH="/usr/local/ucx/bin:${PATH}" + +# ----------------------------------------------------------------------------- +# 5. Cache locations (structural: WHERE the JIT/compile caches live in the image). +# Mount target for the launcher's persistent host JIT cache. +# +# This image ships NO runtime recipe / tuning / platform ENV, matching the +# shared disagg image. Everything run-tunable is applied at launch so the same +# image serves any cluster without a rebuild: +# - K3 serving recipe (VLLM_ROCM_USE_AITER_MLA=0, AITER_SITUV2_A8W4, +# KV_CACHE_MEMORY_BYTES, *_CUDAGRAPH_MODE, *_MORI_BACKEND, ...) +# -> scripts/vllm_dissag/models.yaml, entry "Kimi-K3" +# - ROCm-7.2.3 GPU-RDMA platform env + MoRI fabric tuning +# -> scripts/vllm_dissag/connectors/moriio.env +# The slurm launcher forwards both via `docker -e` (platform env must reach +# PID 1 - PyTorch reads alloc-conf at import). +# ----------------------------------------------------------------------------- +ENV AITER_JIT_DIR=/opt/vllm_cache/aiter_jit \ + VLLM_CACHE_ROOT=/opt/vllm_cache/vllm \ + TRITON_CACHE_DIR=/opt/vllm_cache/triton \ + COMGR_CACHE_DIR=/opt/vllm_cache/comgr + +# ----------------------------------------------------------------------------- +# 5b. K3-AWARE AITER GRAFT - the crux for K3 MXFP4 MoE on gfx942. +# The 0.1.19 release wheel has no Kimi-K3 MoE tuning. At K3's MoE profiling +# shape (M = EP x max_tokens = 131072, N=3584, K=3072, SiTUv2, mxfp4) it finds +# no tuned FlyDSL config and falls back to a heuristic kernel whose +# buffer.load.lds intrinsic aborts LLVM ("Do not know how to expand this +# operator's operand!"). The worker then dies natively inside +# determine_available_memory with no Python traceback and engine init fails. +# PROVEN_K3_IMAGE ships a K3-aware AITER: kimik3_{a8w4,fp4}_tuned_fmoe.csv +# model configs, working MXFP4->CK/int4 routing, and prebuilt hsaco in +# aiter_meta. Grafting its aiter + aiter_meta trees over the 0.1.19 install +# makes K3 MXFP4 MoE compile. +# Placed AFTER the vLLM/router layers so edits here do not invalidate the +# ~40-minute vLLM compile cache. +# ----------------------------------------------------------------------------- +RUN rm -rf /usr/local/lib/python3.12/dist-packages/aiter \ + /usr/local/lib/python3.12/dist-packages/aiter_meta \ + /usr/local/lib/python3.12/dist-packages/flydsl \ + /usr/local/lib/python3.12/dist-packages/aiter*.dist-info \ + /usr/local/lib/python3.12/dist-packages/amd_aiter*.dist-info 2>/dev/null || true +COPY --from=proven_k3_aiter /usr/local/lib/python3.12/dist-packages/aiter /usr/local/lib/python3.12/dist-packages/aiter +COPY --from=proven_k3_aiter /usr/local/lib/python3.12/dist-packages/aiter_meta /usr/local/lib/python3.12/dist-packages/aiter_meta +COPY --from=proven_k3_aiter /usr/local/lib/python3.12/dist-packages/flydsl /usr/local/lib/python3.12/dist-packages/flydsl +# The donor's flydsl is 0.2.2 and the COPY above lands it over the 0.2.4 from +# step 2. K3's int4 SiTUv2 path (_setup_kernel_k3_situ_gfx942 -> compile_moe_gemm1) +# hard-requires >= 0.2.4 or WorkerProc init dies on ImportError and the pool never +# starts. Re-pin AFTER the graft so 0.2.4 wins; the K3-tuned MoE configs live in +# aiter/aiter_meta (still grafted) and 0.2.4 is ABI-compatible with them. +# Doing this at BUILD time is what lets the launcher stay hermetic - the upstream +# recipe ran this pip install inside every container at serve time. +RUN pip install --no-cache-dir --force-reinstall "flydsl==0.2.4" && \ + python3 -c "import importlib.metadata as m; v=m.version('flydsl'); assert v=='0.2.4', f'flydsl {v}!=0.2.4'; print('flydsl OK', v)" && \ + echo "FLYDSL_REPIN=0.2.4 (after proven_k3 graft)" >> /app/versions.txt +RUN rm -rf /opt/vllm_cache/aiter_jit /root/.aiter && \ + echo "AITER_GRAFT=proven_k3 (kimik3 tuned fmoe configs)" >> /app/versions.txt + +# ----------------------------------------------------------------------------- +# 6. CRITICAL: scrub build-time MoRI JIT state. The `import mori` verifications +# above compile/lock MoRI EP kernels under /root/.mori/jit on the BUILD host, +# leaving stale .hsaco.lock files. At runtime MoriAll2AllManager finds those +# locks, waits on a build-in-progress whose owner PID is long gone, and +# DEADLOCKS at ep:0 init. A clean image ships /root/.mori empty. +# ----------------------------------------------------------------------------- +RUN rm -rf /root/.mori /tmp/mori_jit_* && mkdir -p /root/.mori && \ + echo "JIT_SCRUBBED: /root/.mori + /tmp/mori_jit_* cleared at build end" >> /app/versions.txt + +RUN cat /app/versions.txt 2>/dev/null | tail -20 || true diff --git a/scripts/vllm_multinode/run_multinode.slurm b/scripts/vllm_multinode/run_multinode.slurm new file mode 100755 index 00000000..2d6d9ba3 --- /dev/null +++ b/scripts/vllm_multinode/run_multinode.slurm @@ -0,0 +1,215 @@ +#!/bin/bash +#SBATCH --job-name=vllm-colocated +#SBATCH -N 2 # Default 2 nodes; override with sbatch -N +#SBATCH --ntasks-per-node=1 +#SBATCH --spread-job +#SBATCH --gres=gpu:8 +#SBATCH --time=24:00:00 +#SBATCH --output="/shared_inference/%u/model_blog_logs/slurm-%j.out" +#SBATCH --error="/shared_inference/%u/model_blog_logs/slurm-%j.err" +############################################################################### +# +# MIT License +# +# Copyright (c) 2026 Advanced Micro Devices, Inc. +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. +# +############################################################################### +# COLOCATED multi-node vLLM serving (one instance spanning N nodes). +# ============================================================================= +# The counterpart to scripts/vllm_dissag/run_xPyD_models.slurm: +# +# vllm_dissag prefill/decode DISAGGREGATED, KV connector, xP+yD nodes +# -> highest concurrent throughput +# vllm_multinode COLOCATED single instance, no KV connector, N nodes +# -> lowest single-request latency (one request uses all GPUs) +# +# Needed whenever a model does not fit one node: e.g. Kimi-K3 on MI300X, where +# the ~1.5 TB MXFP4 checkpoint cannot fit 8x192 GB, so TP8 within each node and +# PP across nodes puts ~102 GB/GPU on a 2-node allocation. +# +# This launcher deliberately owns very little. Node discovery, the container +# launch and the per-node role split live here; everything else is REUSED from +# scripts/vllm_dissag/ (models.yaml recipe, socket_barrier.py, benchmark_xPyD.sh, +# benchmark_niah.sh, parse_to_csv.py), which is why scripts/ is mounted whole. +# +# Usage: +# cd MAD/scripts/vllm_multinode && sbatch -N 2 run_multinode.slurm +# or via madengine (see models.json entries tagged vllm_multinode). +# +# Key env (madengine passes these through env_vars): +# MODEL_NAME key into scripts/vllm_dissag/models.yaml +# DOCKER_IMAGE_NAME image to run +# TP_SIZE tensor parallel within a node (default GPUS_PER_NODE) +# PP_SIZE pipeline parallel across nodes (default NNODES) +# ENABLE_EP 1 to add --enable-expert-parallel +# ALL2ALL_BACKEND e.g. allgather_reducescatter | mori_low_latency +# COLOCATED_EXTRA_ARGS extra serve flags (JSON values must be self-quoted) +# BENCHMARK_SCRIPT sweep (default) | long_context | niah +# ============================================================================= +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +MAD_SCRIPTS_DIR="$(cd "${SCRIPT_DIR}/.." && pwd)" +SHARED_DIR="${MAD_SCRIPTS_DIR}/vllm_dissag" + +REQUIRED_FILES=( + "${SCRIPT_DIR}/serve_colocated.sh" + "${SHARED_DIR}/models.yaml" + "${SHARED_DIR}/socket_barrier.py" + "${SHARED_DIR}/benchmark_xPyD.sh" + "${SHARED_DIR}/parse_to_csv.py" +) +for f in "${REQUIRED_FILES[@]}"; do + [[ -f "$f" ]] || { echo "Error: required file missing: $f" >&2; exit 1; } +done + +: "${DOCKER_IMAGE_NAME:?set DOCKER_IMAGE_NAME to the vLLM image to run}" +MODEL_NAME="${MODEL_NAME:?set MODEL_NAME (key into scripts/vllm_dissag/models.yaml)}" +MODEL_DIR="${MODEL_DIR:-/shared_inference/models_blog/}" +MODEL_PATH="${MODEL_PATH:-${MODEL_DIR%/}/${MODEL_NAME}}" +USER_NAME="${USER:-$(id -un)}" +LOG_PATH="${LOG_PATH:-/shared_inference/${USER_NAME}/model_blog_logs}" +GPUS_PER_NODE="${GPUS_PER_NODE:-8}" +SERVE_PORT="${SERVE_PORT:-8000}" +MASTER_PORT="${MASTER_PORT:-29500}" + +NUM_NODES="${SLURM_NNODES:-1}" +TP_SIZE="${TP_SIZE:-${GPUS_PER_NODE}}" +PP_SIZE="${PP_SIZE:-${NUM_NODES}}" +ENABLE_EP="${ENABLE_EP:-0}" +ALL2ALL_BACKEND="${ALL2ALL_BACKEND:-}" + +# Benchmark selector — same tags and same scripts as the disagg launcher. +BENCHMARK_SCRIPT="${BENCHMARK_SCRIPT:-sweep}" +case "$BENCHMARK_SCRIPT" in + sweep) BENCHMARK_SCRIPT_FILE="benchmark_xPyD.sh" ;; + long_context) BENCHMARK_SCRIPT_FILE="benchmark_long_context.sh" ;; + niah) BENCHMARK_SCRIPT_FILE="benchmark_niah.sh" ;; + *) echo "Error: invalid BENCHMARK_SCRIPT='$BENCHMARK_SCRIPT' (valid: sweep, long_context, niah)" >&2; exit 1 ;; +esac + +echo "=========================== colocated multi-node ===========================" +echo " model : ${MODEL_NAME} (${MODEL_PATH})" +echo " image : ${DOCKER_IMAGE_NAME}" +echo " nodes : ${NUM_NODES} x ${GPUS_PER_NODE} GPU" +echo " parallelism: TP=${TP_SIZE} PP=${PP_SIZE} EP=${ENABLE_EP} all2all=${ALL2ALL_BACKEND:-}" +echo " benchmark : ${BENCHMARK_SCRIPT} -> ${BENCHMARK_SCRIPT_FILE}" +echo "============================================================================" + +if [ $(( TP_SIZE * PP_SIZE )) -ne $(( NUM_NODES * GPUS_PER_NODE )) ]; then + echo "Error: TP_SIZE*PP_SIZE (${TP_SIZE}x${PP_SIZE}=$((TP_SIZE*PP_SIZE))) != total GPUs (${NUM_NODES}x${GPUS_PER_NODE}=$((NUM_NODES*GPUS_PER_NODE)))." >&2 + exit 1 +fi + +# ------------------------ +# Node discovery: rank-ordered IPs, master = first node +# ------------------------ +SELECTED_NODES=$(scontrol show hostnames "$SLURM_JOB_NODELIST" | head -n "$NUM_NODES") +SELECTED_NODELIST=$(echo "$SELECTED_NODES" | paste -sd,) +MASTER_NODE=$(echo "$SELECTED_NODES" | head -n 1) +MASTER_ADDR=$(srun --nodes=1 --ntasks=1 --time=00:20:00 --nodelist="$MASTER_NODE" bash -c 'hostname -I' | awk '{print $1}') + +IPS=() +for NODE in $SELECTED_NODES; do + IP=$(srun --nodes=1 --ntasks=1 --time=00:20:00 --nodelist="$NODE" bash -c 'hostname -I' | awk '{print $1}') + IPS+=("$IP") + echo " node ${NODE} -> ${IP}" +done +IPADDRS=$(echo "${IPS[*]}" | tr ' ' ',') + +mkdir -p "$LOG_PATH" + +export DOCKER_CONT_NAME="colocated_${MODEL_NAME}_${SLURM_JOB_ID}" +export MODEL_NAME MODEL_PATH MASTER_ADDR MASTER_PORT IPADDRS SERVE_PORT +export TP_SIZE PP_SIZE ENABLE_EP ALL2ALL_BACKEND GPUS_PER_NODE +export BENCHMARK_SCRIPT_FILE LOG_PATH USER_NAME +export NNODES="$NUM_NODES" +export MAD_SCRIPTS_MOUNT=/mad_scripts + +srun --nodelist="$SELECTED_NODELIST" bash -c ' +echo "Rank $SLURM_PROCID on $(hostname)" +docker rm -f $DOCKER_CONT_NAME 2>/dev/null || true +fuser -k 2223/tcp 2>/dev/null || true +sleep 2 +docker pull $DOCKER_IMAGE_NAME 2>/dev/null || true + +# Host RDMA userspace libs must match the host kernel driver; mount them over the +# image copies (same approach as the disagg launcher). +_RDMA_MOUNTS="" +_LIBDIR=/usr/lib/x86_64-linux-gnu +for _vlib in $_LIBDIR/libibverbs.so.1.* $_LIBDIR/librdmacm.so.1.*; do + [ -e "$_vlib" ] && _RDMA_MOUNTS="$_RDMA_MOUNTS -v $_vlib:$_vlib:ro" +done +for _pattern in libmlx5.so* libionic*.so* libbnxt_re*.so* libefa.so* libhns.so*; do + for _vlib in $_LIBDIR/${_pattern}; do + [ -e "$_vlib" ] && _RDMA_MOUNTS="$_RDMA_MOUNTS -v $_vlib:$_vlib:ro" + done +done +[ -d "$_LIBDIR/libibverbs" ] && _RDMA_MOUNTS="$_RDMA_MOUNTS -v $_LIBDIR/libibverbs:$_LIBDIR/libibverbs:ro" +[ -d /etc/libibverbs.d ] && _RDMA_MOUNTS="$_RDMA_MOUNTS -v /etc/libibverbs.d:/etc/libibverbs.d:ro" + +docker run --rm \ + --device /dev/dri --device /dev/kfd --device /dev/infiniband \ + --network host --ipc host --group-add video \ + --cap-add SYS_PTRACE --security-opt seccomp=unconfined --privileged \ + --shm-size ${DOCKER_SHM_SIZE:-256G} \ + --ulimit nofile=524288:524288 --ulimit memlock=-1:-1 \ + -v $LOG_PATH:/run_logs \ + -v '"$MAD_SCRIPTS_DIR"':$MAD_SCRIPTS_MOUNT \ + -v /shared_inference:/shared_inference \ + -v /tmp/vllm_cache:/tmp/vllm_cache \ + $_RDMA_MOUNTS \ + --entrypoint /bin/bash \ + -e SLURM_JOB_ID=$SLURM_JOB_ID \ + -e NODE_RANK=$SLURM_PROCID \ + -e NNODES=$NNODES \ + -e MASTER_ADDR=$MASTER_ADDR \ + -e MASTER_PORT=$MASTER_PORT \ + -e IPADDRS=$IPADDRS \ + -e MODEL_NAME=$MODEL_NAME \ + -e MODEL_PATH=$MODEL_PATH \ + -e SERVE_PORT=$SERVE_PORT \ + -e GPUS_PER_NODE=$GPUS_PER_NODE \ + -e TP_SIZE=$TP_SIZE \ + -e PP_SIZE=$PP_SIZE \ + -e ENABLE_EP=$ENABLE_EP \ + ${ALL2ALL_BACKEND:+-e ALL2ALL_BACKEND=$ALL2ALL_BACKEND} \ + ${COLOCATED_EXTRA_ARGS:+-e COLOCATED_EXTRA_ARGS="$COLOCATED_EXTRA_ARGS"} \ + -e BENCHMARK_SCRIPT_FILE=$BENCHMARK_SCRIPT_FILE \ + ${BENCHMARK_ITR:+-e BENCHMARK_ITR=$BENCHMARK_ITR} \ + ${BENCHMARK_CON:+-e BENCHMARK_CON="$BENCHMARK_CON"} \ + ${BENCHMARK_COMBINATIONS:+-e BENCHMARK_COMBINATIONS="$BENCHMARK_COMBINATIONS"} \ + ${NIAH_WORDS:+-e NIAH_WORDS="$NIAH_WORDS"} \ + ${NIAH_MAXTOK:+-e NIAH_MAXTOK=$NIAH_MAXTOK} \ + ${NIAH_TIMEOUT:+-e NIAH_TIMEOUT=$NIAH_TIMEOUT} \ + ${GPU_MEMORY_UTILIZATION:+-e GPU_MEMORY_UTILIZATION=$GPU_MEMORY_UTILIZATION} \ + ${MAX_MODEL_LEN:+-e MAX_MODEL_LEN=$MAX_MODEL_LEN} \ + --name $DOCKER_CONT_NAME \ + $DOCKER_IMAGE_NAME -c " + mkdir -p /run_logs/${SLURM_JOB_ID} + SHARED_DIR=$MAD_SCRIPTS_MOUNT/vllm_dissag \ + bash $MAD_SCRIPTS_MOUNT/vllm_multinode/serve_colocated.sh \ + 2>&1 | tee /run_logs/${SLURM_JOB_ID}/colocated_bench_NODE${SLURM_PROCID}.log + " +' + +srun --nodelist="$SELECTED_NODELIST" bash -c 'docker stop $DOCKER_CONT_NAME 2>/dev/null || true; docker rm $DOCKER_CONT_NAME 2>/dev/null || true' +echo "Colocated multi-node run complete. Logs: ${LOG_PATH}/${SLURM_JOB_ID}" diff --git a/scripts/vllm_multinode/serve_colocated.sh b/scripts/vllm_multinode/serve_colocated.sh new file mode 100755 index 00000000..5c808346 --- /dev/null +++ b/scripts/vllm_multinode/serve_colocated.sh @@ -0,0 +1,166 @@ +#!/bin/bash +# Colocated (single-instance) multi-node vLLM serve — per-node entry point. +# ============================================================================= +# Runs INSIDE the container, one process per node, launched by run_multinode.slurm. +# +# "Colocated" = one vLLM instance whose parallelism spans several nodes, with NO +# prefill/decode disaggregation and no KV connector. It is the other half of the +# multi-node story from scripts/vllm_dissag/: use this for lowest single-request +# latency (one request uses every GPU), use vllm_dissag for concurrent throughput. +# +# Node roles (by NODE_RANK): +# 0 -> head: serves the OpenAI API on SERVE_PORT, then benchmarks +# 1..N-1 -> worker: --headless, no API server +# +# The per-model recipe (env:) is read from the SAME scripts/vllm_dissag/models.yaml +# the disagg harness uses, so a model's gfx/quant/KV knobs have one home. Serve +# flags specific to a colocated variant come from COLOCATED_EXTRA_ARGS. +# +# Required env: MODEL_PATH, NODE_RANK, NNODES, MASTER_ADDR, IPADDRS +# Optional: MODEL_NAME, PP_SIZE, TP_SIZE, ENABLE_EP, ALL2ALL_BACKEND, +# SERVE_PORT, GPU_MEMORY_UTILIZATION, COLOCATED_EXTRA_ARGS, +# BENCHMARK_SCRIPT_FILE, DRY_RUN +# ============================================================================= +set -uo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +SHARED_DIR="${SHARED_DIR:-${SCRIPT_DIR}/../vllm_dissag}" + +: "${MODEL_PATH:?MODEL_PATH must be set (path to the model dir)}" +MODEL_NAME="${MODEL_NAME:-}" +NODE_RANK="${NODE_RANK:-0}" +NNODES="${NNODES:-1}" +MASTER_ADDR="${MASTER_ADDR:-localhost}" +MASTER_PORT="${MASTER_PORT:-29500}" +IPADDRS="${IPADDRS:-localhost}" +SERVE_PORT="${SERVE_PORT:-8000}" +GPUS_PER_NODE="${GPUS_PER_NODE:-8}" + +# Parallelism. Default is the natural colocated shape: TP within a node, PP across +# nodes (PP_SIZE = NNODES). Expert parallelism is opt-in; when on, the EP group is +# the TP group inside each node, so the expert all2all stays intra-node and the +# only cross-node traffic is the PP activation hand-off. +TP_SIZE="${TP_SIZE:-${GPUS_PER_NODE}}" +PP_SIZE="${PP_SIZE:-${NNODES}}" +ENABLE_EP="${ENABLE_EP:-0}" +ALL2ALL_BACKEND="${ALL2ALL_BACKEND:-}" + +echo "[colocated] node=$(hostname -s) rank=${NODE_RANK}/${NNODES} master=${MASTER_ADDR}" +echo "[colocated] TP=${TP_SIZE} PP=${PP_SIZE} EP=${ENABLE_EP} all2all=${ALL2ALL_BACKEND:-}" + +# ----------------------------------------------------------------------------- +# Per-model env from the shared models.yaml (same precedence rule as the disagg +# driver: connector default < models.yaml env: < submit-time -e). +# ----------------------------------------------------------------------------- +MODELS_YAML="${MODELS_YAML:-${SHARED_DIR}/models.yaml}" +if [[ -n "$MODEL_NAME" && -f "$MODELS_YAML" ]]; then + export MODELS_YAML MODEL_NAME + _yaml_env="$(python3 - <<'PY' +import os, yaml, shlex +m = yaml.safe_load(open(os.environ["MODELS_YAML"])) or {} +cfg = m.get(os.environ["MODEL_NAME"]) or {} +for k, v in (cfg.get("env") or {}).items(): + if k in os.environ: # submit-time -e wins + continue + print(f'export {k}={shlex.quote(str(v))}') +PY +)" + [[ -n "$_yaml_env" ]] && eval "$_yaml_env" + echo "[colocated] applied models.yaml env for '${MODEL_NAME}'" +fi + +# Recipe knobs that models.yaml may have supplied. +GPU_MEMORY_UTILIZATION="${GPU_MEMORY_UTILIZATION:-0.90}" + +# ----------------------------------------------------------------------------- +# Assemble argv +# ----------------------------------------------------------------------------- +serve_args=( + --served-model-name "${MODEL_NAME:-model}" + --tensor-parallel-size "${TP_SIZE}" + --pipeline-parallel-size "${PP_SIZE}" + --distributed-executor-backend mp + --nnodes "${NNODES}" + --node-rank "${NODE_RANK}" + --master-addr "${MASTER_ADDR}" + --master-port "${MASTER_PORT}" + --trust-remote-code + --gpu-memory-utilization "${GPU_MEMORY_UTILIZATION}" + --distributed-timeout-seconds "${DISTRIBUTED_TIMEOUT_SECONDS:-7200}" +) +[[ -n "${KV_CACHE_DTYPE:-}" ]] && serve_args+=(--kv-cache-dtype "${KV_CACHE_DTYPE}") +[[ -n "${KV_BLOCK_SIZE:-}" ]] && serve_args+=(--block-size "${KV_BLOCK_SIZE}") +[[ -n "${KV_CACHE_MEMORY_BYTES:-}" ]] && serve_args+=(--kv-cache-memory-bytes "${KV_CACHE_MEMORY_BYTES}") +if [[ "${ENABLE_EP}" == "1" ]]; then + serve_args+=(--enable-expert-parallel) + [[ -n "${ALL2ALL_BACKEND}" ]] && serve_args+=(--all2all-backend "${ALL2ALL_BACKEND}") +fi +if [[ "${NODE_RANK}" -eq 0 ]]; then + serve_args+=(--port "${SERVE_PORT}") +else + serve_args+=(--headless) +fi +# Per-variant serve flags (reasoning parser, max-model-len, quantization-config...). +# Word-split like the disagg driver does, so JSON values must carry their own quotes. +if [[ -n "${COLOCATED_EXTRA_ARGS:-}" ]]; then + eval "extra_args=(${COLOCATED_EXTRA_ARGS})" + serve_args+=("${extra_args[@]}") +fi + +if [[ "${DRY_RUN:-0}" == "1" ]]; then + echo "===DRYRUN colocated role=$([ "$NODE_RANK" -eq 0 ] && echo head || echo worker) NODE_RANK=${NODE_RANK}===" + printf '%s\n' vllm serve "${MODEL_PATH}" "${serve_args[@]}" + echo "===END===" + exit 0 +fi + +# ----------------------------------------------------------------------------- +# Container barrier — every node's container must exist before any rank dials the +# master, otherwise early ranks burn their connect retries against a dead port. +# Reuses the disagg harness's barrier rather than re-implementing a sleep. +# ----------------------------------------------------------------------------- +_BARRIER_PORT="${CONTAINER_BARRIER_PORT:-2223}" +host_ip=$(hostname -I | awk '{print $1}') +if [[ -f "${SHARED_DIR}/socket_barrier.py" ]]; then + echo "[colocated] waiting at container barrier on $(hostname -s)" + python3 "${SHARED_DIR}/socket_barrier.py" \ + --local-ip "${host_ip}" --local-port "${_BARRIER_PORT}" --enable-port \ + --node-ips "${IPADDRS}" --node-ports "${_BARRIER_PORT}" +fi + +mkdir -p "/run_logs/${SLURM_JOB_ID:-0}" +LOG="/run_logs/${SLURM_JOB_ID:-0}/colocated_NODE${NODE_RANK}.log" + +vllm serve "${MODEL_PATH}" "${serve_args[@]}" 2>&1 | tee "${LOG}" >/dev/null & +WORKER_PID=$! + +if [[ "${NODE_RANK}" -ne 0 ]]; then + # Workers have no API server: hold until the head tears the job down. + echo "[colocated] worker ${NODE_RANK} serving headless; log ${LOG}" + wait "${WORKER_PID}" + exit 0 +fi + +# ---- head: wait for readiness, benchmark, then shut down ------------------- +echo "[colocated] head waiting for 'Application startup complete.' in ${LOG}" +_TIMEOUT="${LOG_WAIT_TIMEOUT_SECONDS:-4000}"; _elapsed=0 +until grep -Fq "Application startup complete." "${LOG}" 2>/dev/null; do + if [ "${_elapsed}" -ge "${_TIMEOUT}" ]; then + echo "[colocated] TIMEOUT (${_TIMEOUT}s): server never became ready. Tail:" >&2 + tail -40 "${LOG}" >&2 + kill "${WORKER_PID}" 2>/dev/null + exit 1 + fi + sleep 10; _elapsed=$((_elapsed + 10)) +done +echo "[colocated] server ready after ${_elapsed}s" + +# Same benchmark scripts + CSV parser as the disagg harness. xP/yD are only used +# in their log filenames; a colocated run reports as 1 instance, 0 decode pools. +export BENCHMARK_PORT="${SERVE_PORT}" +export xP="${xP:-1}" yD="${yD:-0}" +bash "${SHARED_DIR}/${BENCHMARK_SCRIPT_FILE:-benchmark_xPyD.sh}" + +echo "[colocated] benchmark complete; stopping server" +pkill -P "${WORKER_PID}" 2>/dev/null; kill "${WORKER_PID}" 2>/dev/null || true +exit 0 From 13c2005a5b05f1248413032052dcae3c0a805c8d Mon Sep 17 00:00:00 2001 From: Cemberk Date: Thu, 13 Aug 2026 17:43:57 +0000 Subject: [PATCH 04/92] feat(kimi-k3): register MI300X models + emit perf CSV from NIAH runs models.json: four entries, all skip_gpu_arch gfx950 (the existing four Kimi-K3 entries are gfx950/TP8 single-node and all carry skip_gpu_arch gfx942, so K3 is currently skipped entirely on MI300X - this is that gap): pyt_vllm_disagg_mori_kimi-k3 -N 4 xP2 yD2 TP2 -> EP16/pool pyt_vllm_kimi-k3_mi300x_pp2xtp8 -N 2 TP8 x PP2, no EP pyt_vllm_kimi-k3_mi300x_wideep_allgather -N 2 + EP, allgather_reducescatter pyt_vllm_kimi-k3_mi300x_wideep_moriep -N 2 + EP, mori_low_latency The three colocated entries differ only in ENABLE_EP / ALL2ALL_BACKEND / AITER_SITUV2_A8W4, which is why they share one launcher. Result reporting: BENCHMARK_SCRIPT=niah previously produced logs and nothing else, so a model declaring multiple_results would have reported no metric. - parse_to_csv.py gains --niah: parses benchmark_niah.py output and writes the same 29-column madengine perf.csv as the throughput path (verified identical column sets). One row per context size, performance = needles found /10. A size that errored is written as a FAILURE row with performance 0 rather than dropped, so a pass->crash regression is visible instead of silent. - benchmark_niah.sh calls it. - Both slurm launchers copy /run_logs/$SLURM_JOB_ID/perf.csv to ./perf_$MODEL_NAME.csv on completion: madengine resolves multiple_results against the job directory it launched from, not the container log mount. No-op for entries that do not declare multiple_results. Verified: madengine discover goes 133 -> 137 with exactly these four added and none removed; every referenced dockerfile and script exists; -N matches each topology; the int4 quantization-config survives shell word-splitting as valid JSON; DeepSeek moriio and rixl/deepep DRY_RUN output is byte-identical to develop. Co-Authored-By: Claude Opus 5 (1M context) --- models.json | 139 +++++++++++++++++++++ scripts/vllm_dissag/benchmark_niah.sh | 8 ++ scripts/vllm_dissag/parse_to_csv.py | 74 +++++++++++ scripts/vllm_dissag/run_xPyD_models.slurm | 14 +++ scripts/vllm_multinode/run_multinode.slurm | 14 +++ 5 files changed, 249 insertions(+) diff --git a/models.json b/models.json index 1f95cb41..83c6dcea 100644 --- a/models.json +++ b/models.json @@ -367,6 +367,145 @@ "args": "--model_repo moonshotai/Kimi-K3 --config configs/default.yaml" }, + { + "name": "pyt_vllm_disagg_mori_kimi-k3", + "url": "", + "dockerfile": "docker/pyt_vllm_kimi_k3_mi300x", + "scripts": "scripts/vllm_dissag/run_xPyD_models.slurm", + "data": "huggingface", + "n_gpus": "-1", + "owner": "mad.support@amd.com", + "training_precision": "", + "multiple_results": "perf_Kimi-K3.csv", + "tags": [ + "pyt", + "vllm", + "vllm_disagg", + "mori_ep", + "inference" + ], + "timeout": -1, + "skip_gpu_arch": "gfx950", + "distributed": { + "launcher": "slurm_multi" + }, + "env_vars": { + "DOCKER_IMAGE_NAME": "", + "MODEL_NAME": "Kimi-K3", + "xP": "2", + "yD": "2", + "TP_SIZE": "2", + "RUN_MORI": "1", + "WIDE_EP": "1", + "BENCHMARK_SCRIPT": "niah", + "NIAH_WORDS": "10000,50000,100000,200000" + }, + "args": "-N 4 -n 4" + }, + { + "name": "pyt_vllm_kimi-k3_mi300x_pp2xtp8", + "url": "", + "dockerfile": "docker/pyt_vllm_kimi_k3_mi300x", + "scripts": "scripts/vllm_multinode/run_multinode.slurm", + "data": "huggingface", + "n_gpus": "-1", + "owner": "mad.support@amd.com", + "training_precision": "", + "multiple_results": "perf_Kimi-K3.csv", + "tags": [ + "pyt", + "vllm", + "vllm_multinode", + "inference" + ], + "timeout": -1, + "skip_gpu_arch": "gfx950", + "distributed": { + "launcher": "slurm_multi" + }, + "env_vars": { + "DOCKER_IMAGE_NAME": "", + "MODEL_NAME": "Kimi-K3", + "TP_SIZE": "8", + "PP_SIZE": "2", + "ENABLE_EP": "0", + "COLOCATED_EXTRA_ARGS": "--reasoning-parser kimi_k3 --mm-encoder-tp-mode data --safetensors-load-strategy prefetch --max-model-len 1000000 --max-num-seqs 8 --max-num-batched-tokens 2048", + "BENCHMARK_SCRIPT": "niah", + "NIAH_WORDS": "10000,50000,100000,200000" + }, + "args": "-N 2 -n 2" + }, + { + "name": "pyt_vllm_kimi-k3_mi300x_wideep_allgather", + "url": "", + "dockerfile": "docker/pyt_vllm_kimi_k3_mi300x", + "scripts": "scripts/vllm_multinode/run_multinode.slurm", + "data": "huggingface", + "n_gpus": "-1", + "owner": "mad.support@amd.com", + "training_precision": "", + "multiple_results": "perf_Kimi-K3.csv", + "tags": [ + "pyt", + "vllm", + "vllm_multinode", + "inference" + ], + "timeout": -1, + "skip_gpu_arch": "gfx950", + "distributed": { + "launcher": "slurm_multi" + }, + "env_vars": { + "DOCKER_IMAGE_NAME": "", + "MODEL_NAME": "Kimi-K3", + "TP_SIZE": "8", + "PP_SIZE": "2", + "ENABLE_EP": "1", + "ALL2ALL_BACKEND": "allgather_reducescatter", + "AITER_SITUV2_A8W4": "1", + "COLOCATED_EXTRA_ARGS": "--reasoning-parser kimi_k3 --mm-encoder-tp-mode data --safetensors-load-strategy prefetch --max-model-len 1000000 --max-num-seqs 8 --max-num-batched-tokens 2048 --quantization-config '{\"moe\":{\"weight\":\"int4_per_group_32\"}}'", + "BENCHMARK_SCRIPT": "niah", + "NIAH_WORDS": "10000,50000,100000,200000" + }, + "args": "-N 2 -n 2" + }, + { + "name": "pyt_vllm_kimi-k3_mi300x_wideep_moriep", + "url": "", + "dockerfile": "docker/pyt_vllm_kimi_k3_mi300x", + "scripts": "scripts/vllm_multinode/run_multinode.slurm", + "data": "huggingface", + "n_gpus": "-1", + "owner": "mad.support@amd.com", + "training_precision": "", + "multiple_results": "perf_Kimi-K3.csv", + "tags": [ + "pyt", + "vllm", + "vllm_multinode", + "mori_ep", + "inference" + ], + "timeout": -1, + "skip_gpu_arch": "gfx950", + "distributed": { + "launcher": "slurm_multi" + }, + "env_vars": { + "DOCKER_IMAGE_NAME": "", + "MODEL_NAME": "Kimi-K3", + "TP_SIZE": "8", + "PP_SIZE": "2", + "ENABLE_EP": "1", + "ALL2ALL_BACKEND": "mori_low_latency", + "AITER_SITUV2_A8W4": "1", + "COLOCATED_EXTRA_ARGS": "--reasoning-parser kimi_k3 --mm-encoder-tp-mode data --safetensors-load-strategy prefetch --max-model-len 1000000 --max-num-seqs 8 --max-num-batched-tokens 2048 --quantization-config '{\"moe\":{\"weight\":\"int4_per_group_32\"}}'", + "BENCHMARK_SCRIPT": "niah", + "NIAH_WORDS": "10000,50000,100000,200000" + }, + "args": "-N 2 -n 2" + }, { "name": "pyt_vllm_llama-3.1-8b", "url": "", diff --git a/scripts/vllm_dissag/benchmark_niah.sh b/scripts/vllm_dissag/benchmark_niah.sh index ba49a359..7e1fb381 100755 --- a/scripts/vllm_dissag/benchmark_niah.sh +++ b/scripts/vllm_dissag/benchmark_niah.sh @@ -25,4 +25,12 @@ NIAH_MAXTOK="${NIAH_MAXTOK:-2048}" \ NIAH_TIMEOUT="${NIAH_TIMEOUT:-1800}" \ python3 "${DIR}/benchmark_niah.py" 2>&1 | tee -a "${LOG}" +# Emit the madengine perf.csv so an accuracy run reports a metric like a +# throughput run does (one row per context size, needles found /10). Without +# this, BENCHMARK_SCRIPT=niah produces logs only and the model's +# multiple_results CSV is never written. +python3 "${DIR}/parse_to_csv.py" "${LOG}" --niah \ + --perf-csv "/run_logs/${SLURM_JOB_ID}/perf.csv" \ + --model-name "${MODEL_NAME}" 2>&1 | tee -a "${LOG}" + echo "NIAH results -> ${LOG}" diff --git a/scripts/vllm_dissag/parse_to_csv.py b/scripts/vllm_dissag/parse_to_csv.py index e772e394..d2db78ec 100644 --- a/scripts/vllm_dissag/parse_to_csv.py +++ b/scripts/vllm_dissag/parse_to_csv.py @@ -181,6 +181,68 @@ def save_perf_csv(results: Dict[Tuple[int, int, int], Dict], output_file: str, print(f"Saved {len(results)} rows to perf.csv: {output_file}") +def parse_niah_log(log_file: str): + """Parse a benchmark_niah.py log into {context_words: found_out_of_10}. + + benchmark_niah.py prints one line per size: + words= 20000 found= 9/10 [...] + and 'found=ERR' (via the summary) for a request that errored. Only the + per-size result lines are read; the trailing summary repeats them. + """ + results = {} + pat = re.compile(r'^words=\s*(\d+)\s+found=\s*(\d+)/10') + err = re.compile(r'^words=\s*(\d+)\s+ERROR') + with open(log_file, 'r', errors='replace') as f: + for line in f: + line = line.strip() + m = pat.match(line) + if m: + results[int(m.group(1))] = int(m.group(2)) + continue + m = err.match(line) + if m: + results.setdefault(int(m.group(1)), None) + return results + + +def save_niah_perf_csv(results, output_file: str, model_name: str = "", + pipeline: str = "vllm"): + """Save NIAH retrieval accuracy in madengine perf.csv format. + + One row per context size; performance is needles found out of 10. A size that + errored is reported as FAILURE with performance 0 rather than dropped, so a + regression that turns a pass into a crash is visible instead of silent. + """ + if not results: + print("No NIAH results to save to perf.csv.") + return + + meta = _get_run_metadata(pipeline) + fieldnames = [ + 'model', 'n_gpus', 'nnodes', 'gpus_per_node', 'training_precision', + 'pipeline', 'args', 'tags', 'docker_file', 'base_docker', 'docker_sha', + 'docker_image', 'git_commit', 'machine_name', 'deployment_type', 'launcher', + 'gpu_architecture', 'performance', 'metric', 'relative_change', 'status', + 'build_duration', 'test_duration', 'dataname', 'data_provider_type', + 'data_size', 'data_download_duration', 'build_number', + 'additional_docker_run_options', + ] + with open(output_file, 'w', newline='') as f: + writer = csv.DictWriter(f, fieldnames=fieldnames) + writer.writeheader() + for words in sorted(results): + found = results[words] + row = { + 'model': model_name, + 'performance': '0' if found is None else str(found), + 'metric': f'needles found /10 (NIAH ctx={words} words)', + 'status': 'FAILURE' if found is None else 'SUCCESS', + } + row.update(meta) + writer.writerow(row) + print(f"Saved {len(results)} NIAH rows to perf.csv: {output_file}") + + def main(): import sys import argparse @@ -190,6 +252,8 @@ def main(): parser.add_argument('-o', '--output', type=str, help='Output CSV file name (default: _results.csv)') parser.add_argument('--perf-csv', type=str, help='Also generate madengine perf.csv at this path') parser.add_argument('--model-name', type=str, default='', help='Model name for perf.csv') + parser.add_argument('--niah', action='store_true', + help='Parse a benchmark_niah.py log (retrieval accuracy) instead of a throughput sweep') args = parser.parse_args() @@ -201,6 +265,16 @@ def main(): print(f"Parsing log file: {log_file}") + if args.niah: + niah = parse_niah_log(log_file) + if not niah: + print("No NIAH results found in log file.") + return + if args.perf_csv: + save_niah_perf_csv(niah, args.perf_csv, args.model_name) + print(f"\nSummary:\n NIAH context sizes: {len(niah)}") + return + results = parse_benchmark_log(log_file) if not results: diff --git a/scripts/vllm_dissag/run_xPyD_models.slurm b/scripts/vllm_dissag/run_xPyD_models.slurm index 1753f8b4..48c9dfeb 100755 --- a/scripts/vllm_dissag/run_xPyD_models.slurm +++ b/scripts/vllm_dissag/run_xPyD_models.slurm @@ -655,3 +655,17 @@ docker run --rm \ ' srun --nodelist="$SELECTED_NODELIST_SRUN" bash -c 'docker stop $DOCKER_CONT_NAME 2>/dev/null || true; docker rm $DOCKER_CONT_NAME 2>/dev/null || true' +# Publish the perf CSV where madengine collects it. benchmark_xPyD.sh / +# benchmark_niah.sh write /run_logs/$SLURM_JOB_ID/perf.csv inside the container +# (= $LOG_PATH on the host), but madengine resolves a model's `multiple_results` +# against the JOB DIRECTORY it launched us from - so copy it there under the +# declared name. Harmless for entries that do not declare multiple_results. +_PERF_SRC="${LOG_PATH}/${SLURM_JOB_ID}/perf.csv" +_PERF_DST="$(pwd)/perf_${MODEL_NAME}.csv" +if [ -f "$_PERF_SRC" ]; then + cp "$_PERF_SRC" "$_PERF_DST" + echo "perf CSV -> ${_PERF_DST} ($(( $(wc -l < "$_PERF_DST") - 1 )) rows)" +else + echo "WARNING: no perf CSV at ${_PERF_SRC}; the benchmark produced no parseable results." >&2 +fi + diff --git a/scripts/vllm_multinode/run_multinode.slurm b/scripts/vllm_multinode/run_multinode.slurm index 2d6d9ba3..328b62d3 100755 --- a/scripts/vllm_multinode/run_multinode.slurm +++ b/scripts/vllm_multinode/run_multinode.slurm @@ -212,4 +212,18 @@ docker run --rm \ ' srun --nodelist="$SELECTED_NODELIST" bash -c 'docker stop $DOCKER_CONT_NAME 2>/dev/null || true; docker rm $DOCKER_CONT_NAME 2>/dev/null || true' + +# Publish the perf CSV where madengine collects it. The benchmark writes +# /run_logs/$SLURM_JOB_ID/perf.csv inside the container (= $LOG_PATH on the host), +# but madengine resolves a model's `multiple_results` against the JOB DIRECTORY it +# launched us from - so copy it there under the declared name. +_PERF_SRC="${LOG_PATH}/${SLURM_JOB_ID}/perf.csv" +_PERF_DST="$(pwd)/perf_${MODEL_NAME}.csv" +if [ -f "$_PERF_SRC" ]; then + cp "$_PERF_SRC" "$_PERF_DST" + echo "perf CSV -> ${_PERF_DST} ($(( $(wc -l < "$_PERF_DST") - 1 )) rows)" +else + echo "WARNING: no perf CSV at ${_PERF_SRC}; the benchmark produced no parseable results." >&2 +fi + echo "Colocated multi-node run complete. Logs: ${LOG_PATH}/${SLURM_JOB_ID}" From 893d4d0ea98c35ceb6e1f0d79ca4a51a16d621a6 Mon Sep 17 00:00:00 2001 From: Cemberk Date: Thu, 13 Aug 2026 17:46:45 +0000 Subject: [PATCH 05/92] docs(kimi-k3): MI300X multi-node recipes under benchmark/kimi_k3/mi300x/ Ports the writeups from PR #193 next to the existing gfx950 K3 doc, keyed to the madengine tags rather than to shell scripts: topology rationale (why MI300X needs multi-node and why the disagg pool needs TP2), the gfx942 knobs and where each one lives now, the NIAH results, and the three root-cause fixes. Corrections carried in: - the NIAH ceiling is 900K throughout. The upstream READMEs still said 300K in seven places after two later commits raised it to 500K and then 900K. - states that the NIAH harness sizes context in WORDS (~1.3x tokens) while the results table is in tokens, so the two columns are not comparable. benchmark/kimi_k3/README.md gains a pointer: all four entries there carry skip_gpu_arch gfx942, which is exactly the gap the MI300X page fills. Every tag, node count and knob value in the new page is checked against models.json and models.yaml. Co-Authored-By: Claude Opus 5 (1M context) --- benchmark/kimi_k3/README.md | 5 + benchmark/kimi_k3/mi300x/README.md | 177 +++++++++++++++++++++++++++++ 2 files changed, 182 insertions(+) create mode 100644 benchmark/kimi_k3/mi300x/README.md diff --git a/benchmark/kimi_k3/README.md b/benchmark/kimi_k3/README.md index f68f1c76..458788f8 100644 --- a/benchmark/kimi_k3/README.md +++ b/benchmark/kimi_k3/README.md @@ -12,6 +12,11 @@ MAD supports Kimi-K3 day-0 inference across **three** serving frameworks on AMD | **SGLang** | `pyt_sglang_kimi-k3` | `lmsysorg/sglang-rocm:rocm720-mi35x-k3-20260727` | [SGLang cookbook](https://docs.sglang.io/cookbook/autoregressive/Moonshotai/Kimi-K3) | | **ATOM** | `pyt_atom_kimi-k3` | `rocm/atom-dev:rocm7.2.4_ubuntu24.04_py3.12_pytorch2.10.0_20260727_kimi_k3` | ROCm ATOM | +> **On MI300X (gfx942)?** All four models above carry `skip_gpu_arch: gfx942` — +> the ~1.5 TB checkpoint does not fit a single 8-GPU MI300X node, so K3 needs +> multi-node sharding there. See [`mi300x/`](mi300x/README.md) for the 2-node +> colocated and 4-node prefill/decode-disaggregated recipes. + ## Hardware requirements - **8x MI350X or MI355X** (TP8) diff --git a/benchmark/kimi_k3/mi300x/README.md b/benchmark/kimi_k3/mi300x/README.md new file mode 100644 index 00000000..c5afc13f --- /dev/null +++ b/benchmark/kimi_k3/mi300x/README.md @@ -0,0 +1,177 @@ +# Kimi-K3 inference on AMD Instinct MI300X (gfx942) + +## Overview + +[Kimi-K3](https://huggingface.co/moonshotai/Kimi-K3) is Moonshot AI's 2.8T-parameter +Mixture-of-Experts model (896 experts, natively MXFP4/QAT, hybrid MLA + +Kimi-Delta-Attention, 1M-token context). + +The single-node recipes in [`../README.md`](../README.md) target **MI350X / MI355X +(gfx950)** at TP8 and are skipped on gfx942. MI300X has 192 GB/GPU, so the ~1.5 TB +checkpoint does **not** fit one 8-GPU node — every MI300X recipe here is +**multi-node**. That is the difference: same model, different hardware, different +sharding. + +## Models + +| MAD tag | Nodes | Parallelism | Expert all2all | Use when | +|---------|-------|-------------|----------------|----------| +| `pyt_vllm_kimi-k3_mi300x_pp2xtp8` | 2 | PP2 × TP8, no EP | — | Simplest baseline; lowest single-user latency | +| `pyt_vllm_kimi-k3_mi300x_wideep_allgather` | 2 | PP2 × TP8, EP8/node | `allgather_reducescatter` | Expert-parallel without MoRI kernels | +| `pyt_vllm_kimi-k3_mi300x_wideep_moriep` | 2 | PP2 × TP8, EP8/node | `mori_low_latency` (MoRI-EP) | MoRI-EP expert dispatch | +| `pyt_vllm_disagg_mori_kimi-k3` | 4 | 2P/2D, TP2 × DP8 → EP16 per pool | MoRI-EP + MoRIIO KV transfer | Highest concurrent throughput | + +The first three are **colocated** — one instance spanning 2 nodes, no +prefill/decode split, one request uses all 16 GPUs. The fourth is +**disaggregated**: 2 prefill + 2 decode nodes joined by the MoRIIO connector. + +**Pick by workload.** Colocated gives the lowest single-request latency. Disagg +gives **5.7× throughput at concurrency 8 (7.3× at 16)** plus decode-latency +isolation, at roughly 4× higher single-stream latency — one request runs on 2 GPUs +instead of all 16. That trade is architectural, not a tuning defect. + +> **EP scope (colocated).** The 896 experts split 8-way across each node's 8 GPUs +> (112 experts/GPU), and that EP8 group is replicated on each of the 2 pipeline +> stages. The expert all2all runs **intra-node**; the only cross-node traffic is +> the PP activation hand-off over NCCL. "16" is the GPU count, not the EP width. + +## Quick start + +```sh +# colocated, 2 nodes +madengine run --tags pyt_vllm_kimi-k3_mi300x_pp2xtp8 --keep-model-dir --live-output + +# prefill/decode disaggregated, 4 nodes +madengine run --tags pyt_vllm_disagg_mori_kimi-k3 --keep-model-dir --live-output +``` + +Both are `slurm_multi` models: madengine submits the launcher via sbatch with the +node count from each entry's `args` (`-N 2` / `-N 4`). Supply your built image via +the `DOCKER_IMAGE_NAME` env var (see [Image](#image)). + +## Hardware requirements + +- **16× MI300X** (2 nodes) colocated, or **32× MI300X** (4 nodes) disaggregated +- RDMA fabric between nodes; the defaults assume 8 NICs per node +- Checkpoint is ~1.5 TB — local NVMe strongly recommended, on every node + +## Image + +`docker/pyt_vllm_kimi_k3_mi300x.ubuntu.amd.Dockerfile` builds the whole stack from +public sources on the open ROCm vLLM CI base: MoRI 1.2.2, AITER 0.1.19 + +flydsl 0.2.4, vLLM with the K3/MoRIIO fixes, and the DP-rank/KV-notify vllm-router. +Every source is pinned to an immutable commit SHA. + +It also grafts a K3-aware AITER from the public `amdsiloai/vllm:kimi-k3-mi325x-release-v2` +image. Without that graft the K3 MoE profiling shape finds no tuned FlyDSL config, +falls back to a heuristic kernel, and aborts LLVM inside +`determine_available_memory` — the worker dies natively with no Python traceback. + +## Where the configuration lives + +Nothing K3-specific is baked into the image, matching the shared disagg image's +design: + +| What | Where | +|------|-------| +| Model serving recipe (gfx942 knobs, KV cache, cudagraph modes, MoE quant) | `scripts/vllm_dissag/models.yaml`, entry `Kimi-K3` | +| Per-variant topology (TP/PP/EP, node count, benchmark) | `models.json` `env_vars` | +| ROCm platform + MoRI/RDMA fabric env | `scripts/vllm_dissag/connectors/moriio.env` | + +### gfx942 specifics + +- **`VLLM_ROCM_USE_AITER_MLA=0` is required** — the AITER MLA kernel is gfx950-only. +- gfx942 has no scaled-MXFP4 MFMA and the a16w4 SiTUv2 heuristic FlyDSL kernel + cannot codegen there, so the MoE is requantized to packed int4 and run through + SiTUv2 (`AITER_SITUV2_A8W4=1` plus `--quantization-config + '{"moe":{"weight":"int4_per_group_32"}}'`). +- **`--max-num-batched-tokens` stays at 2048.** 8192 corrupts generation on this + stack, and the 16384 profiling shape crashes LLVM codegen in the heuristic kernel. +- **`KV_CACHE_MEMORY_BYTES=40e9`** gives a 2.84M-token GPU KV cache. The lower 8e9 + value used during bring-up was a `profile_run`-hang workaround, not a memory + limit; single requests beyond ~600K tokens need the larger cache. +- **Why TP2 in the disagg recipe:** K3's replicated (attn + shared-expert) weight is + 106.5 GiB. At TP1/DP16 that is 190.7 GiB/GPU before KV cache or the 16 GiB MoRI + heap — it does not fit 192 GB. TP2 shards it to 53.3 GiB/GPU, giving + 137.5 GiB of weights per GPU with room to spare. + +## Results — single-needle NIAH (disaggregated 2P/2D) + +Needle `HELIOTROPE-7492`, greedy (temp=0), depths 0.1 / 0.5 / 0.9. All PASS, +deterministic, across the full native context range. + +| context | result | eval time / request | +|---------|--------|---------------------| +| 10K–200K | 3/3 PASS | 5–88 s | +| 300K | 3/3 PASS | ~150 s | +| 500K | 3/3 PASS | ~301 s | +| 750K | 3/3 PASS | ~542 s | +| **900K** | **3/3 PASS** | **~717 s** | + +Scaling is sub-quadratic. Reaching the top of that range needs +`--max-model-len 1000000` (in the `Kimi-K3` models.yaml `dp:` block) and +`KV_CACHE_MEMORY_BYTES=40000000000` (in its `env:` block) — both are the defaults +here. + +> **Context units.** The NIAH harness sizes its haystack in **words** +> (`NIAH_WORDS`), and words are roughly 1.3× tokens for this filler. The table +> above is in tokens; the CSV rows emitted by a `BENCHMARK_SCRIPT=niah` run are +> labelled in words. Do not compare the two columns directly. + +**One-time warmup:** a fresh serve pays a single aiter MLA-kernel JIT compile +(`fmha_fwd_hd192x128`, ~15 min) on the first ≥200K-token request, cached +thereafter. The times above are warm. + +**Known residual:** the stricter 10-needle stress dips to ~9/10 at ≥20K — an RDMA +write-visibility race. Single-needle retrieval is unaffected. + +### The three fixes behind these numbers + +All three are committed in the vLLM the image builds (pinned by SHA); nothing is +patched at runtime. + +1. **4-KV-cache-group block routing.** K3's hybrid attention allocates **4** KV + cache groups (3 KDA/mamba + 1 MLA). The stock connector hardcoded 2-group + indices and sent MLA KV to mamba block ids, so decode read empty blocks and + generated fluent but context-free text. Each layer is now routed by its own + group index. +2. **Multi-chunk prefill transfer.** The final-chunk gate used block count, which + fires after chunk 1 when a prompt fits in ≤1 padded block — so only + `max_num_batched_tokens` of KV ever crossed, a razor cliff at 2048. The gate now + keys on compute progress from `scheduler_output`. +3. **KDA gather made sync-free.** `gather_initial_states` ran a diagnostic + `bool((indices >= n).any())` per KDA layer per prefill chunk, each forcing a + device→CPU sync — ~25k full stream drains at 750K, which presented as a hang for + contexts above ~500K. The index clamp already made the address safe, so the + diagnostic is now behind `K3_KDA_GATHER_LOG=1` (default off). Correctness is + unchanged; 750K and 900K went from hanging indefinitely to passing. + +## Benchmarks and results reporting + +Both launchers accept `BENCHMARK_SCRIPT`: + +| value | script | reports | +|-------|--------|---------| +| `sweep` | `benchmark_xPyD.sh` | throughput sweep, tok/s per (isl, osl, concurrency) | +| `long_context` | `benchmark_long_context.sh` | per-shape warmup, concurrency 1 first | +| `niah` (default for these entries) | `benchmark_niah.sh` | retrieval accuracy, needles found /10 per context size | + +All three land in madengine's `perf.csv` via `parse_to_csv.py`, so the NIAH numbers +above are a CI-visible metric rather than a table in a markdown file. A context size +whose request errors is recorded as a `FAILURE` row with performance 0, so a +pass→crash regression shows up instead of silently disappearing. + +## Relationship to PR #193 + +These recipes originate in [PR #193](https://github.com/ROCm/MAD/pull/193), which +shipped them as standalone shell scripts under `scripts/vllm/kimik3_mi300x/`. This +integration keeps the recipes and the findings and drops the duplicated machinery: + +- the 2P/2D topology is expressed as `xP=2 yD=2 TP_SIZE=2` on the existing + `scripts/vllm_dissag/` harness instead of a bespoke ssh orchestrator +- the three colocated variants share one launcher, differing only in + `ENABLE_EP` / `ALL2ALL_BACKEND` / `AITER_SITUV2_A8W4` +- the 32 runtime patchers are gone — their fixes are in the SHA-pinned vLLM +- the NIAH probes are the harness's existing `benchmark_niah.py` + +See also [`../README.md`](../README.md) for the gfx950 single-node recipes. From a7dd95193dc2f463004092adb8cf720304c50fa6 Mon Sep 17 00:00:00 2001 From: Cemberk Date: Mon, 17 Aug 2026 20:05:03 +0000 Subject: [PATCH 06/92] fix(kimi-k3): correct node sizing, NIAH model tag, and colocated run metadata MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The four MI300X Kimi-K3 entries could not run correctly on a cluster. Three separate defects, all from the colocated launcher inheriting conventions that only hold for the disaggregated one. Node sizing. The entries carried `"args": "-N 4 -n 4"` on the assumption that madengine forwards args to sbatch. It does not — args are appended to `bash .slurm`, and neither launcher parses $@, so the flags were dropped and the job was submitted with the `slurm.nodes` default of 1. Sizing now uses `distributed.nnodes`, which madengine reconciles into `#SBATCH --nodes`, and the inert args are removed. `slurm.nodes` is deliberately left unset: setting it selects the multi-node preset, whose `NCCL_SOCKET_IFNAME=eth0` is forwarded into the container by run_xPyD_models.slurm and would override the fabric interface on an RDMA cluster. `slurm.time` is set explicitly instead, since the single-node preset's 12 h default is short for a full NIAH sweep. NIAH model tag. benchmark_niah.sh requested MODEL_PATH as the model name, correct for the disagg path where vLLM defaults served_model_name to the serve argument. serve_colocated.sh passes `--served-model-name "$MODEL_NAME"`, so every request 404'd and all three colocated entries — which default to BENCHMARK_SCRIPT=niah — would have recorded a full sweep of FAILURE rows. The harness now honors NIAH_MODEL when a launcher sets it, and the colocated launcher sets it to the tag it actually serves. Colocated run metadata. parse_to_csv derived nnodes, n_gpus, deployment_type and tags from xP/yD. serve_colocated.sh exports xP=1 yD=0 only so the shared log filenames stay unique, so a 2-node 16-GPU colocated run reported itself as a 1-node 8-GPU `disagg_1P0D` with a nixl backend it never used. Node count now comes from NNODES (exported by both launchers), and a launcher whose shape is not "xP prefill + yD decode" states its own identity via PERF_DEPLOYMENT_TYPE / PERF_TAGS. The disagg path is byte-identical to before. Also corrects the README, which documented the args-to-sbatch behavior that does not exist and a quick start that cannot work now that the "" placeholder is rejected rather than pulled. Co-Authored-By: Claude Opus 5 (1M context) --- benchmark/kimi_k3/mi300x/README.md | 44 ++++++++++++++++++++--- models.json | 32 ++++++++++++----- scripts/vllm_dissag/benchmark_niah.sh | 9 +++-- scripts/vllm_dissag/parse_to_csv.py | 36 +++++++++++++++---- scripts/vllm_multinode/serve_colocated.sh | 14 ++++++++ 5 files changed, 113 insertions(+), 22 deletions(-) diff --git a/benchmark/kimi_k3/mi300x/README.md b/benchmark/kimi_k3/mi300x/README.md index c5afc13f..a62aa521 100644 --- a/benchmark/kimi_k3/mi300x/README.md +++ b/benchmark/kimi_k3/mi300x/README.md @@ -37,17 +37,39 @@ instead of all 16. That trade is architectural, not a tuning defect. ## Quick start +Run from a SLURM login node. Both steps need your image — the model cards carry +`DOCKER_IMAGE_NAME: ""` as a fill-me-in marker, and madengine +rejects it rather than trying to pull it (see [Image](#image)). + ```sh +IMG=/: + # colocated, 2 nodes -madengine run --tags pyt_vllm_kimi-k3_mi300x_pp2xtp8 --keep-model-dir --live-output +madengine build --tags pyt_vllm_kimi-k3_mi300x_pp2xtp8 --use-image "$IMG" +madengine run --manifest-file build_manifest.json --keep-model-dir --live-output # prefill/decode disaggregated, 4 nodes -madengine run --tags pyt_vllm_disagg_mori_kimi-k3 --keep-model-dir --live-output +madengine build --tags pyt_vllm_disagg_mori_kimi-k3 --use-image "$IMG" +madengine run --manifest-file build_manifest.json --keep-model-dir --live-output ``` -Both are `slurm_multi` models: madengine submits the launcher via sbatch with the -node count from each entry's `args` (`-N 2` / `-N 4`). Supply your built image via -the `DOCKER_IMAGE_NAME` env var (see [Image](#image)). +To build and distribute the image instead of supplying a pre-built one, swap +`--use-image "$IMG"` for `--registry `; the launcher then pulls it +onto every node in parallel. + +Both are `slurm_multi` models. The allocation is sized from each entry's +`distributed.nnodes` (2 or 4), which madengine turns into `#SBATCH --nodes`; the +launcher then reads `SLURM_NNODES` for node discovery. Nothing is passed to the +`.slurm` script on the command line — the topology travels entirely through +`env_vars`. + +Alternatively, run inside an allocation you already hold, in which case madengine +executes the launcher synchronously and the node count comes from the allocation: + +```sh +salloc -N 4 --ntasks-per-node=1 --gres=gpu:8 -p -t 24:00:00 +madengine run --manifest-file build_manifest.json --live-output +``` ## Hardware requirements @@ -161,6 +183,18 @@ above are a CI-visible metric rather than a table in a markdown file. A context whose request errors is recorded as a `FAILURE` row with performance 0, so a pass→crash regression shows up instead of silently disappearing. +The launcher copies that CSV to `perf_Kimi-K3.csv` beside itself, which is the name +each entry declares as `multiple_results`; madengine resolves the declared name +first and falls back to the conventional `/shared_inference/$USER/model_blog_logs/ +$SLURM_JOB_ID/perf.csv` path. + +Because the two launchers describe their topology differently, the colocated one +states its own reporting identity — `PERF_DEPLOYMENT_TYPE=colocated_pp2xtp8` and +`PERF_TAGS=vllm_multinode,colocated,…`, with node count from `NNODES`. The disagg +launcher keeps deriving `disagg_PD` from its pool sizes. Without that split +a colocated 2-node run reported itself as a 1-node `disagg_1P0D`, since it sets +`xP=1 yD=0` only to keep the shared log filenames unique. + ## Relationship to PR #193 These recipes originate in [PR #193](https://github.com/ROCm/MAD/pull/193), which diff --git a/models.json b/models.json index 83c6dcea..0392a58a 100644 --- a/models.json +++ b/models.json @@ -387,7 +387,11 @@ "timeout": -1, "skip_gpu_arch": "gfx950", "distributed": { - "launcher": "slurm_multi" + "launcher": "slurm_multi", + "nnodes": 4 + }, + "slurm": { + "time": "24:00:00" }, "env_vars": { "DOCKER_IMAGE_NAME": "", @@ -400,7 +404,7 @@ "BENCHMARK_SCRIPT": "niah", "NIAH_WORDS": "10000,50000,100000,200000" }, - "args": "-N 4 -n 4" + "args": "" }, { "name": "pyt_vllm_kimi-k3_mi300x_pp2xtp8", @@ -421,7 +425,11 @@ "timeout": -1, "skip_gpu_arch": "gfx950", "distributed": { - "launcher": "slurm_multi" + "launcher": "slurm_multi", + "nnodes": 2 + }, + "slurm": { + "time": "24:00:00" }, "env_vars": { "DOCKER_IMAGE_NAME": "", @@ -433,7 +441,7 @@ "BENCHMARK_SCRIPT": "niah", "NIAH_WORDS": "10000,50000,100000,200000" }, - "args": "-N 2 -n 2" + "args": "" }, { "name": "pyt_vllm_kimi-k3_mi300x_wideep_allgather", @@ -454,7 +462,11 @@ "timeout": -1, "skip_gpu_arch": "gfx950", "distributed": { - "launcher": "slurm_multi" + "launcher": "slurm_multi", + "nnodes": 2 + }, + "slurm": { + "time": "24:00:00" }, "env_vars": { "DOCKER_IMAGE_NAME": "", @@ -468,7 +480,7 @@ "BENCHMARK_SCRIPT": "niah", "NIAH_WORDS": "10000,50000,100000,200000" }, - "args": "-N 2 -n 2" + "args": "" }, { "name": "pyt_vllm_kimi-k3_mi300x_wideep_moriep", @@ -490,7 +502,11 @@ "timeout": -1, "skip_gpu_arch": "gfx950", "distributed": { - "launcher": "slurm_multi" + "launcher": "slurm_multi", + "nnodes": 2 + }, + "slurm": { + "time": "24:00:00" }, "env_vars": { "DOCKER_IMAGE_NAME": "", @@ -504,7 +520,7 @@ "BENCHMARK_SCRIPT": "niah", "NIAH_WORDS": "10000,50000,100000,200000" }, - "args": "-N 2 -n 2" + "args": "" }, { "name": "pyt_vllm_llama-3.1-8b", diff --git a/scripts/vllm_dissag/benchmark_niah.sh b/scripts/vllm_dissag/benchmark_niah.sh index 7e1fb381..694e86d7 100755 --- a/scripts/vllm_dissag/benchmark_niah.sh +++ b/scripts/vllm_dissag/benchmark_niah.sh @@ -17,9 +17,14 @@ echo "port=${BENCHMARK_PORT} model=${MODEL_PATH} sizes=${NIAH_WORDS:-2000,8000 # Give the router a moment to be fully ready for chat completions. sleep 10 -# The server registers the model under its path (served_model_name = MODEL_PATH). +# The model tag to request MUST equal the server's served_model_name, or every +# request 404s and the whole sweep is recorded as FAILURE. +# * vllm_dissag passes no --served-model-name, so vLLM defaults it to MODEL_PATH. +# * vllm_multinode passes --served-model-name "$MODEL_NAME" and exports NIAH_MODEL +# to match. +# Hence: honour NIAH_MODEL when the launcher sets it, else fall back to MODEL_PATH. NIAH_URL="http://127.0.0.1:${BENCHMARK_PORT}/v1/chat/completions" \ -NIAH_MODEL="${MODEL_PATH}" \ +NIAH_MODEL="${NIAH_MODEL:-${MODEL_PATH}}" \ NIAH_WORDS="${NIAH_WORDS:-2000,8000,20000,35000}" \ NIAH_MAXTOK="${NIAH_MAXTOK:-2048}" \ NIAH_TIMEOUT="${NIAH_TIMEOUT:-1800}" \ diff --git a/scripts/vllm_dissag/parse_to_csv.py b/scripts/vllm_dissag/parse_to_csv.py index d2db78ec..6292b6c4 100644 --- a/scripts/vllm_dissag/parse_to_csv.py +++ b/scripts/vllm_dissag/parse_to_csv.py @@ -113,7 +113,20 @@ def save_to_csv(results: Dict[Tuple[int, int, int], Dict], output_file: str): def _get_run_metadata(pipeline: str = "vllm"): - """Collect run metadata from environment variables.""" + """Collect run metadata from environment variables. + + Two launchers share this parser, and they describe their topology differently: + + * vllm_dissag -> disaggregated, xP prefill + yD decode nodes. + * vllm_multinode -> COLOCATED, one instance spanning NNODES nodes. It exports + xP=1 yD=0 purely so the shared benchmark log filenames stay unique. + + Deriving the topology from xP/yD is therefore only valid for the disagg path; + on the colocated path it reported a 2-node/16-GPU run as `disagg_1P0D` with + 1 node and 8 GPUs. NNODES is exported by both launchers and is authoritative, + and a launcher whose shape is not "xP prefill + yD decode" states its own + deployment_type/tags via PERF_DEPLOYMENT_TYPE / PERF_TAGS. + """ import os xP = os.environ.get('xP', '1') yD = os.environ.get('yD', '1') @@ -129,17 +142,26 @@ def _get_run_metadata(pipeline: str = "vllm"): else: backend = 'nixl' + def _as_int(value, default): + try: + return int(value) + except (TypeError, ValueError): + return default + + gpus_per_node = _as_int(gpus, 8) + nnodes = _as_int(os.environ.get('NNODES'), _as_int(xP, 1) + _as_int(yD, 1)) + return { 'pipeline': pipeline, - 'deployment_type': f'disagg_{xP}P{yD}D', - 'tags': f'{pipeline}_disagg,{backend}', - 'n_gpus': str(int(xP) * int(gpus) + int(yD) * int(gpus)), - 'nnodes': str(int(xP) + int(yD)), - 'gpus_per_node': gpus, + 'deployment_type': os.environ.get('PERF_DEPLOYMENT_TYPE') or f'disagg_{xP}P{yD}D', + 'tags': os.environ.get('PERF_TAGS') or f'{pipeline}_disagg,{backend}', + 'n_gpus': str(nnodes * gpus_per_node), + 'nnodes': str(nnodes), + 'gpus_per_node': str(gpus_per_node), 'docker_image': os.environ.get('DOCKER_IMAGE_NAME', ''), 'machine_name': os.environ.get('SLURM_JOB_NODELIST', ''), 'launcher': 'slurm_multi', - 'gpu_architecture': 'gfx942', + 'gpu_architecture': os.environ.get('PERF_GPU_ARCH', 'gfx942'), } diff --git a/scripts/vllm_multinode/serve_colocated.sh b/scripts/vllm_multinode/serve_colocated.sh index 5c808346..64542428 100755 --- a/scripts/vllm_multinode/serve_colocated.sh +++ b/scripts/vllm_multinode/serve_colocated.sh @@ -159,6 +159,20 @@ echo "[colocated] server ready after ${_elapsed}s" # in their log filenames; a colocated run reports as 1 instance, 0 decode pools. export BENCHMARK_PORT="${SERVE_PORT}" export xP="${xP:-1}" yD="${yD:-0}" + +# The shared harness assumes the disagg launcher's conventions. Override the two +# that do not hold here, so a colocated run is benchmarked and reported as itself: +# +# NIAH_MODEL we pass --served-model-name, so the served tag is MODEL_NAME, +# not MODEL_PATH. Without this every NIAH request 404s. +# PERF_DEPLOYMENT_TYPE/PERF_TAGS +# xP/yD above are filename placeholders, not a topology; left +# alone the CSV would label this 2-node run `disagg_1P0D`. +export NIAH_MODEL="${MODEL_NAME:-model}" +_ep_tag="$([ "${ENABLE_EP}" = "1" ] && echo "ep_${ALL2ALL_BACKEND:-default}" || echo "noep")" +export PERF_DEPLOYMENT_TYPE="${PERF_DEPLOYMENT_TYPE:-colocated_pp${PP_SIZE}xtp${TP_SIZE}}" +export PERF_TAGS="${PERF_TAGS:-vllm_multinode,colocated,${_ep_tag}}" + bash "${SHARED_DIR}/${BENCHMARK_SCRIPT_FILE:-benchmark_xPyD.sh}" echo "[colocated] benchmark complete; stopping server" From 8cf7ca9e3ac63300099d1b11d1639d0dc53e675b Mon Sep 17 00:00:00 2001 From: Cemberk Date: Mon, 17 Aug 2026 20:12:58 +0000 Subject: [PATCH 07/92] fix(vllm_dissag): fail fast when the allocation is smaller than xP+yD NUM_NODES is derived from the requested topology (xP+yD), not from the allocation, and the node list was then truncated with `head -n $NUM_NODES`. A short allocation therefore produced a silently wrong topology: pools were built from nodes that were never allocated, and the run failed much later as a connector handshake timeout with no indication of the real cause. This mattered little while these models were always launched from a matching `salloc`, but the sbatch path now sizes the allocation from the model card, so a mismatch between card and cluster is a reachable state. Check up front and name the three ways to fix it. The colocated launcher already fails on the equivalent mismatch via its TP*PP == NNODES*GPUS check, so it needs no counterpart. Co-Authored-By: Claude Opus 5 (1M context) --- scripts/vllm_dissag/run_xPyD_models.slurm | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/scripts/vllm_dissag/run_xPyD_models.slurm b/scripts/vllm_dissag/run_xPyD_models.slurm index 48c9dfeb..09820674 100755 --- a/scripts/vllm_dissag/run_xPyD_models.slurm +++ b/scripts/vllm_dissag/run_xPyD_models.slurm @@ -356,6 +356,19 @@ echo "SLURM_NTASKS: $SLURM_NTASKS" # Get the full nodelist and extract first NUM_NODES FULL_NODELIST=$(scontrol show hostnames "$SLURM_JOB_NODELIST") + +# NUM_NODES comes from xP+yD, not from the allocation, so a short allocation used +# to be truncated by the `head` below into a silently wrong topology: pools would +# be built from nodes that were never allocated, and the run would fail much later +# with a confusing connector timeout. Fail here instead, while the cause is obvious. +_AVAILABLE_NODES=$(echo "$FULL_NODELIST" | grep -c .) +if [ "$_AVAILABLE_NODES" -lt "$NUM_NODES" ]; then + echo "Error: topology needs ${NUM_NODES} nodes (xP=${xP} + yD=${yD}) but the allocation has ${_AVAILABLE_NODES}." >&2 + echo " Size the allocation to match: 'distributed.nnodes' in the model card," >&2 + echo " 'sbatch -N ${NUM_NODES}', or 'salloc -N ${NUM_NODES}'." >&2 + exit 1 +fi + SELECTED_NODES=$(echo "$FULL_NODELIST" | head -n $NUM_NODES) NEW_SLURM_NODELIST=$(echo "$SELECTED_NODES" | paste -sd,) From da621b79485c9979f02e8c16cd8aa950f032d0c8 Mon Sep 17 00:00:00 2001 From: Cemberk Date: Tue, 18 Aug 2026 14:07:28 +0000 Subject: [PATCH 08/92] refactor(kimi-k3): report NIAH results on the narrow madengine CSV contract MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit parse_to_csv.py hand-wrote a full 29-column perf.csv, assembling node counts, GPU counts, launcher, image and tags from the environment. Most of that is not the workload's to know: madengine already owns it, and the guesses were wrong for the colocated launcher, which sets xP=1 yD=0 only to keep log filenames unique. The NIAH writer now emits a narrow CSV — model, performance, metric, status, plus descriptive columns — and madengine merges in the run metadata via the entries' `multiple_results` declaration. The columns that genuinely describe the workload's own configuration rather than its placement (tp, pp, ep_backend, prefill_decode) move to descriptive columns, matching how scripts/vllm/run_vllm.py already reports tp/dtype/bs on the templated path. Nothing is lost; it is relocated to whichever side actually knows it. `status` stays explicit: an errored context size scores 0, and deriving status from performance would file that real failure as a SUCCESS. Scoped to leave every other workload alone. The NIAH writer is reachable only by the four Kimi entries, which all declare `multiple_results`. The throughput writer keeps its full-schema output by default, because the twelve disagg cards that use it declare no `multiple_results` and madengine reads their CSV directly with no metadata to merge — a narrow CSV there would drop every descriptive column. Their output is byte-identical; `--narrow` is the opt-in for migrating one. Co-Authored-By: Claude Opus 5 (1M context) --- benchmark/kimi_k3/mi300x/README.md | 32 +++++- scripts/vllm_dissag/parse_to_csv.py | 130 ++++++++++++++++++---- scripts/vllm_multinode/serve_colocated.sh | 3 + 3 files changed, 138 insertions(+), 27 deletions(-) diff --git a/benchmark/kimi_k3/mi300x/README.md b/benchmark/kimi_k3/mi300x/README.md index a62aa521..11453587 100644 --- a/benchmark/kimi_k3/mi300x/README.md +++ b/benchmark/kimi_k3/mi300x/README.md @@ -188,12 +188,32 @@ each entry declares as `multiple_results`; madengine resolves the declared name first and falls back to the conventional `/shared_inference/$USER/model_blog_logs/ $SLURM_JOB_ID/perf.csv` path. -Because the two launchers describe their topology differently, the colocated one -states its own reporting identity — `PERF_DEPLOYMENT_TYPE=colocated_pp2xtp8` and -`PERF_TAGS=vllm_multinode,colocated,…`, with node count from `NNODES`. The disagg -launcher keeps deriving `disagg_PD` from its pool sizes. Without that split -a colocated 2-node run reported itself as a 1-node `disagg_1P0D`, since it sets -`xP=1 yD=0` only to keep the shared log filenames unique. +### What the workload reports vs. what madengine reports + +The CSV is **narrow**: the benchmark reports only what it measured, and madengine +merges in the run metadata it already owns. This is the same contract the templated +launchers use, so these gfx942 multi-node rows and the gfx950 single-node rows in +[`../README.md`](../README.md) describe themselves identically in `perf.csv`. + +| From the benchmark | From madengine | +|---|---| +| `model`, `performance`, `metric`, `status` | `nnodes`, `n_gpus`, `gpus_per_node`, `launcher` | +| `benchmark`, `context_words` | `docker_image`, `base_docker`, `docker_sha` | +| `tp`, `pp`, `ep_backend`, `prefill_decode` | tags, pipeline, build number, machine name | + +The right-hand column used to be hand-written by `parse_to_csv.py` from `xP`/`yD`, +which is why a colocated 2-node/16-GPU run reported itself as a 1-node/8-GPU +`disagg_1P0D` on a `nixl` backend it never used — the colocated launcher sets +`xP=1 yD=0` only to keep the shared log filenames unique. A workload cannot reliably +know its own topology; madengine can, so it does. + +`status` is reported explicitly because performance alone cannot express this +benchmark's failure mode: a context size whose request errors scores 0, which is a +real measurement, and deriving status from it would file the failure as a SUCCESS. + +> `BENCHMARK_SCRIPT=sweep` still uses the older full-schema CSV (the disagg model +> cards that declare no `multiple_results` depend on it). Pass `--narrow` to +> `parse_to_csv.py` to put a sweep on the same contract. ## Relationship to PR #193 diff --git a/scripts/vllm_dissag/parse_to_csv.py b/scripts/vllm_dissag/parse_to_csv.py index 6292b6c4..d6be665e 100644 --- a/scripts/vllm_dissag/parse_to_csv.py +++ b/scripts/vllm_dissag/parse_to_csv.py @@ -112,8 +112,45 @@ def save_to_csv(results: Dict[Tuple[int, int, int], Dict], output_file: str): print(f"Saved {len(results)} benchmark configurations to {output_file}") +def _workload_config_columns(): + """Descriptive columns naming the shape the benchmark ran at. + + These are workload CONFIGURATION, not run metadata: madengine knows where a job + ran (nodes, GPUs, image, launcher) but not the parallelism the workload chose. + Path A's run_vllm.py carries the same kind of columns (tp, dtype, bs), so a + narrow CSV is the right home for them — unlike the topology fields that used to + be hand-written into deployment_type, which madengine owns. + + Only non-empty values are emitted, so a launcher that does not set them produces + no stray columns. + """ + import os + cols = {} + tp = os.environ.get('TP_SIZE') + pp = os.environ.get('PP_SIZE') + if tp: + cols['tp'] = tp + if pp: + cols['pp'] = pp + if os.environ.get('ENABLE_EP') == '1' or os.environ.get('WIDE_EP') == '1': + cols['ep_backend'] = ( + os.environ.get('ALL2ALL_BACKEND') + or os.environ.get('VLLM_ALL2ALL_BACKEND') + or 'enabled' + ) + xP, yD = os.environ.get('xP'), os.environ.get('yD') + if xP and yD and yD != '0': + cols['prefill_decode'] = f'{xP}P{yD}D' + return cols + + def _get_run_metadata(pipeline: str = "vllm"): - """Collect run metadata from environment variables. + """Collect run metadata from environment variables (LEGACY full-schema path). + + Only used by save_perf_csv(narrow=False), i.e. by model cards that do not declare + `multiple_results` and whose CSV madengine reads directly with no metadata to + merge. Cards on the narrow contract get all of this from madengine instead, which + is authoritative; prefer migrating rather than extending this function. Two launchers share this parser, and they describe their topology differently: @@ -166,12 +203,51 @@ def _as_int(value, default): def save_perf_csv(results: Dict[Tuple[int, int, int], Dict], output_file: str, - model_name: str = "", pipeline: str = "vllm"): - """Save results in madengine perf.csv format.""" + model_name: str = "", pipeline: str = "vllm", narrow: bool = False): + """Save throughput results for madengine. + + Two schemas, selected by `narrow`: + + * narrow=True -- the preferred contract. The workload reports only what it + measured and madengine merges in the run metadata it already owns, via the + model card's `multiple_results` declaration. Same contract as the templated + launchers, so rows from different launchers stay comparable. + * narrow=False -- legacy, and still the default. Writes the full 29-column + perf.csv with metadata assembled from the environment by _get_run_metadata(). + Required by the disagg model cards that do NOT declare `multiple_results`: + madengine reads their CSV directly from a conventional path, with no metadata + to merge, so a narrow CSV there would lose every descriptive column. + + To migrate a model: declare `multiple_results` on its card and pass --narrow. + """ if not results: print("No results to save to perf.csv.") return + if narrow: + config_cols = _workload_config_columns() + fieldnames = (['model', 'benchmark', 'inp', 'out', 'max_concurrency', + 'performance', 'metric'] + list(config_cols)) + with open(output_file, 'w', newline='') as f: + writer = csv.DictWriter(f, fieldnames=fieldnames) + writer.writeheader() + for (input_tokens, output_tokens, concurrency), data in sorted( + results.items(), key=lambda x: (x[0][2], x[0][0], x[0][1]) + ): + row = { + 'model': model_name, + 'benchmark': 'throughput_sweep', + 'inp': data['input_tokens'], + 'out': data['output_tokens'], + 'max_concurrency': data['concurrency'], + 'performance': f"{data['max_throughput']:.2f}", + 'metric': 'tok/s', + } + row.update(config_cols) + writer.writerow(row) + print(f"Saved {len(results)} rows (narrow schema) to {output_file}") + return + meta = _get_run_metadata(pipeline) fieldnames = [ @@ -229,26 +305,32 @@ def parse_niah_log(log_file: str): def save_niah_perf_csv(results, output_file: str, model_name: str = "", pipeline: str = "vllm"): - """Save NIAH retrieval accuracy in madengine perf.csv format. - - One row per context size; performance is needles found out of 10. A size that - errored is reported as FAILURE with performance 0 rather than dropped, so a - regression that turns a pass into a crash is visible instead of silent. + """Save NIAH retrieval accuracy as a NARROW madengine results CSV. + + Narrow means the workload reports only what it measured — model, performance, + metric and outcome — and madengine merges that with the run metadata it already + owns (node/GPU counts, image, launcher, build provenance) via the model card's + `multiple_results` declaration. This is the same contract the templated + launchers use, so a gfx942 multi-node row and a gfx950 single-node row of the + same model land in perf.csv describing themselves the same way. + + It replaces a full 29-column perf.csv that this script wrote by hand. Hand-written + metadata is how a colocated 2-node run came to report itself as `disagg_1P0D` on + 1 node: the topology was inferred from xP/yD, which the colocated launcher only + sets to keep log filenames unique. + + `status` is emitted explicitly because performance alone cannot express this + benchmark's failure mode: a context size whose request errored scores 0, which is + a real measurement, and deriving status from it would record the failure as a + SUCCESS and hide a pass->crash regression. """ if not results: print("No NIAH results to save to perf.csv.") return - meta = _get_run_metadata(pipeline) - fieldnames = [ - 'model', 'n_gpus', 'nnodes', 'gpus_per_node', 'training_precision', - 'pipeline', 'args', 'tags', 'docker_file', 'base_docker', 'docker_sha', - 'docker_image', 'git_commit', 'machine_name', 'deployment_type', 'launcher', - 'gpu_architecture', 'performance', 'metric', 'relative_change', 'status', - 'build_duration', 'test_duration', 'dataname', 'data_provider_type', - 'data_size', 'data_download_duration', 'build_number', - 'additional_docker_run_options', - ] + config_cols = _workload_config_columns() + fieldnames = (['model', 'benchmark', 'context_words', 'performance', 'metric', 'status'] + + list(config_cols)) with open(output_file, 'w', newline='') as f: writer = csv.DictWriter(f, fieldnames=fieldnames) writer.writeheader() @@ -256,13 +338,15 @@ def save_niah_perf_csv(results, output_file: str, model_name: str = "", found = results[words] row = { 'model': model_name, + 'benchmark': 'niah', + 'context_words': words, 'performance': '0' if found is None else str(found), 'metric': f'needles found /10 (NIAH ctx={words} words)', 'status': 'FAILURE' if found is None else 'SUCCESS', } - row.update(meta) + row.update(config_cols) writer.writerow(row) - print(f"Saved {len(results)} NIAH rows to perf.csv: {output_file}") + print(f"Saved {len(results)} NIAH rows (narrow schema) to {output_file}") def main(): @@ -276,6 +360,10 @@ def main(): parser.add_argument('--model-name', type=str, default='', help='Model name for perf.csv') parser.add_argument('--niah', action='store_true', help='Parse a benchmark_niah.py log (retrieval accuracy) instead of a throughput sweep') + parser.add_argument('--narrow', action='store_true', + help='Emit a narrow results CSV (model/performance/metric[/status]) for a model card ' + 'declaring multiple_results, letting madengine supply the run metadata. ' + 'Ignored with --niah, which is always narrow.') args = parser.parse_args() @@ -311,7 +399,7 @@ def main(): save_to_csv(results, output_file) if args.perf_csv: - save_perf_csv(results, args.perf_csv, args.model_name) + save_perf_csv(results, args.perf_csv, args.model_name, narrow=args.narrow) print(f"\nSummary:") print(f" Total unique configurations: {len(results)}") diff --git a/scripts/vllm_multinode/serve_colocated.sh b/scripts/vllm_multinode/serve_colocated.sh index 64542428..cddec976 100755 --- a/scripts/vllm_multinode/serve_colocated.sh +++ b/scripts/vllm_multinode/serve_colocated.sh @@ -168,6 +168,9 @@ export xP="${xP:-1}" yD="${yD:-0}" # PERF_DEPLOYMENT_TYPE/PERF_TAGS # xP/yD above are filename placeholders, not a topology; left # alone the CSV would label this 2-node run `disagg_1P0D`. +# Only consulted on the LEGACY full-schema reporting path +# (BENCHMARK_SCRIPT=sweep). With BENCHMARK_SCRIPT=niah the CSV +# is narrow and madengine supplies these fields itself. export NIAH_MODEL="${MODEL_NAME:-model}" _ep_tag="$([ "${ENABLE_EP}" = "1" ] && echo "ep_${ALL2ALL_BACKEND:-default}" || echo "noep")" export PERF_DEPLOYMENT_TYPE="${PERF_DEPLOYMENT_TYPE:-colocated_pp${PP_SIZE}xtp${TP_SIZE}}" From 09d3f99cc47d75e47a1b518b2597cb13bb2db79b Mon Sep 17 00:00:00 2001 From: Cemberk Date: Tue, 25 Aug 2026 17:02:00 -0500 Subject: [PATCH 09/92] fix(kimi-k3): size the SLURM allocation from slurm.nodes, not distributed.nnodes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit madengine reads only slurm.nodes when emitting `#SBATCH --nodes` (deployment/slurm.py: `self.nodes = slurm_config.get("nodes", 1)`), and nothing maps distributed.nnodes across — build_orchestrator copies nnodes into deployment_config.distributed for launcher detection only. The four MI300X model cards carried nnodes but no slurm.nodes, so every one of them would have been submitted as a 1-node job. The launcher's TP*PP == NNODES*GPUS_PER_NODE assertion catches it, but only after the allocation is granted. Add nodes/gpus_per_node to the four slurm blocks, and ship the site configs that carry the values a model card cannot: - results_dir, absent from the key list madengine copies out of models.json, and required because the slurm_multi collector globs results_dir for perf*.csv rather than resolving multiple_results - MODEL_DIR / LOG_PATH, which otherwise default to /shared_inference *.json is gitignored repo-wide, so the templates need explicit negations or they never reach a clone. Correct the two README claims that did not match madengine's behavior (nnodes-sizes-the-allocation, multiple_results-is-resolved-first), document that --account/--qos are unwired for slurm_multi and need SBATCH_ACCOUNT, and drop --keep-model-dir from the SLURM examples — it is a local-Docker flag that madengine ignores with a warning on SLURM. Co-Authored-By: Claude Opus 5 (1M context) --- .gitignore | 2 + benchmark/kimi_k3/mi300x/README.md | 72 +++++++++++++++---- .../mi300x/slurm-config.colocated.json | 22 ++++++ .../kimi_k3/mi300x/slurm-config.disagg.json | 21 ++++++ models.json | 8 +++ 5 files changed, 113 insertions(+), 12 deletions(-) create mode 100644 benchmark/kimi_k3/mi300x/slurm-config.colocated.json create mode 100644 benchmark/kimi_k3/mi300x/slurm-config.disagg.json diff --git a/.gitignore b/.gitignore index e2d872c7..b75cd2d0 100644 --- a/.gitignore +++ b/.gitignore @@ -73,6 +73,8 @@ venv.bak/ !perf_super.json !perf_entry_super.json !perf_entry.json +# Checked-in madengine site-config templates (not results) +!benchmark/kimi_k3/mi300x/slurm-config.*.json *.log *.out *.html diff --git a/benchmark/kimi_k3/mi300x/README.md b/benchmark/kimi_k3/mi300x/README.md index 11453587..dc8baf7b 100644 --- a/benchmark/kimi_k3/mi300x/README.md +++ b/benchmark/kimi_k3/mi300x/README.md @@ -41,27 +41,45 @@ Run from a SLURM login node. Both steps need your image — the model cards carr `DOCKER_IMAGE_NAME: ""` as a fill-me-in marker, and madengine rejects it rather than trying to pull it (see [Image](#image)). +First fill in your cluster's partition and paths in the matching site config — +[`slurm-config.colocated.json`](slurm-config.colocated.json) (2 node) or +[`slurm-config.disagg.json`](slurm-config.disagg.json) (4 node). Every field marked +`CHANGE-ME` is site-specific; see [Site configuration](#site-configuration). + ```sh IMG=/: +CFG=benchmark/kimi_k3/mi300x # colocated, 2 nodes -madengine build --tags pyt_vllm_kimi-k3_mi300x_pp2xtp8 --use-image "$IMG" -madengine run --manifest-file build_manifest.json --keep-model-dir --live-output +madengine build --tags pyt_vllm_kimi-k3_mi300x_pp2xtp8 --use-image "$IMG" \ + --additional-context-file $CFG/slurm-config.colocated.json +madengine run --manifest-file build_manifest.json --live-output # prefill/decode disaggregated, 4 nodes -madengine build --tags pyt_vllm_disagg_mori_kimi-k3 --use-image "$IMG" -madengine run --manifest-file build_manifest.json --keep-model-dir --live-output +madengine build --tags pyt_vllm_disagg_mori_kimi-k3 --use-image "$IMG" \ + --additional-context-file $CFG/slurm-config.disagg.json +madengine run --manifest-file build_manifest.json --live-output ``` +The context file is only needed on `build` — its `slurm` and `env_vars` blocks are +written into the manifest's `deployment_config` and merged back on `run`. + To build and distribute the image instead of supplying a pre-built one, swap `--use-image "$IMG"` for `--registry `; the launcher then pulls it onto every node in parallel. -Both are `slurm_multi` models. The allocation is sized from each entry's -`distributed.nnodes` (2 or 4), which madengine turns into `#SBATCH --nodes`; the -launcher then reads `SLURM_NNODES` for node discovery. Nothing is passed to the -`.slurm` script on the command line — the topology travels entirely through -`env_vars`. +Both are `slurm_multi` models: madengine generates a wrapper SBATCH script that runs +the model's `.slurm` script on the head node with `bash`, and that script manages its +own per-node containers via `srun`. The `#SBATCH` header inside the launcher is +therefore inert — every allocation knob comes from madengine. Nothing is passed to the +`.slurm` script on the command line; the topology travels entirely through `env_vars`. + +The allocation is sized from each entry's **`slurm.nodes`** (2 or 4), which madengine +emits as `#SBATCH --nodes`; the launcher then reads `SLURM_NNODES` for node discovery. +`distributed.nnodes` carries the same number for the launcher-detection path but is +**not** what sizes the allocation — madengine reads only `slurm.nodes`, defaulting to +1. Keep the two in sync when editing a model card, or override `nodes` in the site +config, where it wins over the model card. Alternatively, run inside an allocation you already hold, in which case madengine executes the launcher synchronously and the node count comes from the allocation: @@ -76,6 +94,32 @@ madengine run --manifest-file build_manifest.json --live-output - **16× MI300X** (2 nodes) colocated, or **32× MI300X** (4 nodes) disaggregated - RDMA fabric between nodes; the defaults assume 8 NICs per node - Checkpoint is ~1.5 TB — local NVMe strongly recommended, on every node +- **Docker** on the compute nodes. The launchers call `docker run` under `srun`; + podman/apptainer-only clusters will not work without editing the launcher. + +## Site configuration + +Everything cluster-specific lives in the two `slurm-config.*.json` files next to this +README, passed with `--additional-context-file`. Values set there override the model +cards (`build_orchestrator.py` only fills in a model-card key you did *not* set). + +| Field | What it is | How to find it | +|-------|-----------|----------------| +| `slurm.partition` | GPU partition to submit to | `sinfo -o '%20P %5D %14F %10G %11l'` — pick a partition whose `GRES` column shows GPUs and whose `A/I/O/T` counts show idle nodes | +| `slurm.gpus_per_node` | GPUs per node (8 on MI300X) | `GRES` column above, or `scontrol show node \| grep Gres` | +| `slurm.nodes` | 2 colocated / 4 disaggregated | Fixed by the recipe; must match `distributed.nnodes` | +| `slurm.time` | Wall clock | `TIMELIMIT` column above is the partition's cap | +| `env_vars.MODEL_DIR` | Directory holding `Kimi-K3/` | Wherever the ~1.5 TB checkpoint lives; must be readable from every node | +| `env_vars.LOG_PATH` | Run logs + per-job `perf.csv` | Any shared, writable path | + +Two knobs are **not** settable through the config file on this launcher: + +- **`--account` / `--qos`.** madengine emits these only for its templated launchers; + the hand-built `slurm_multi` header omits them. If your cluster requires an + account, export `SBATCH_ACCOUNT` (and `SBATCH_QOS`) before `madengine run` — + `sbatch` honors those environment variables and no directive conflicts with them. +- **`slurm.results_dir`.** Settable in the config file, but *not* from a model card — + it is absent from the key list madengine copies out of `models.json`. ## Image @@ -184,9 +228,13 @@ whose request errors is recorded as a `FAILURE` row with performance 0, so a pass→crash regression shows up instead of silently disappearing. The launcher copies that CSV to `perf_Kimi-K3.csv` beside itself, which is the name -each entry declares as `multiple_results`; madengine resolves the declared name -first and falls back to the conventional `/shared_inference/$USER/model_blog_logs/ -$SLURM_JOB_ID/perf.csv` path. +each entry declares as `multiple_results`. Note that madengine's `slurm_multi` +collector does **not** resolve `multiple_results` — it globs `perf*.csv` under +`slurm.results_dir`, then falls back to `/shared_inference/$USER/model_blog_logs/ +$SLURM_JOB_ID/perf.csv` and a handful of other conventional paths. That is why the +site configs set `results_dir` to the launcher's own directory +(`scripts/vllm_multinode` or `scripts/vllm_dissag`): it is what makes the declared +filename findable. `multiple_results` still carries the name for the non-SLURM paths. ### What the workload reports vs. what madengine reports diff --git a/benchmark/kimi_k3/mi300x/slurm-config.colocated.json b/benchmark/kimi_k3/mi300x/slurm-config.colocated.json new file mode 100644 index 00000000..a3ee38a5 --- /dev/null +++ b/benchmark/kimi_k3/mi300x/slurm-config.colocated.json @@ -0,0 +1,22 @@ +{ + "_comment": "Site config for the 2-node COLOCATED Kimi-K3 MI300X tags: pyt_vllm_kimi-k3_mi300x_pp2xtp8 / _wideep_allgather / _wideep_moriep", + "_usage": "madengine build --tags --use-image \"$IMG\" --additional-context-file benchmark/kimi_k3/mi300x/slurm-config.colocated.json", + "_note_nodes": "nodes/gpus_per_node also ship in the model cards; set them here only to override. Anything you set here wins over the model card.", + "_note_results_dir": "The launcher copies its CSV to