Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 8 additions & 3 deletions .github/workflows/benchmark-multinode-tmpl.yml
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,11 @@ on:
runner:
required: true
type: string
runner-node:
description: "Optional concrete self-hosted runner label used only for scheduling"
required: false
type: string
default: ''
priority:
description: "Higher-is-sooner CI priority score"
required: true
Expand Down Expand Up @@ -269,19 +274,19 @@ jobs:
inputs.skip-queue-pr != '' &&
format(
'["self-hosted",{0},{1},{2},{3}]',
toJSON(inputs.runner),
toJSON(inputs.runner-node != '' && inputs.runner-node || inputs.runner),
toJSON(format('ci-job-{0}-{1}', inputs.priority, inputs.queue-token)),
toJSON(format('ci-attempt-{0}', github.run_attempt)),
toJSON(format('ci-skip-queue-pr-{0}', inputs.skip-queue-pr))
) ||
format(
'["self-hosted",{0},{1},{2}]',
toJSON(inputs.runner),
toJSON(inputs.runner-node != '' && inputs.runner-node || inputs.runner),
toJSON(format('ci-job-{0}-{1}', inputs.priority, inputs.queue-token)),
toJSON(format('ci-attempt-{0}', github.run_attempt))
)
) ||
format('[{0}]', toJSON(inputs.runner))
format('[{0}]', toJSON(inputs.runner-node != '' && inputs.runner-node || inputs.runner))
) }}
# Full-context AgentX warmup can legitimately exceed the fixed-sequence
# eight-hour envelope at high concurrency. Keep one hour beyond the
Expand Down
1 change: 1 addition & 0 deletions .github/workflows/e2e-tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -478,6 +478,7 @@ jobs:
osl: '0'
max-model-len: '0'
runner: ${{ matrix.config.runner }}
runner-node: ${{ matrix.config['runner-node'] || '' }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
image: ${{ matrix.config.image }}
Expand Down
1 change: 1 addition & 0 deletions .github/workflows/run-sweep.yml
Original file line number Diff line number Diff line change
Expand Up @@ -652,6 +652,7 @@ jobs:
osl: '0'
max-model-len: '0'
runner: ${{ matrix.config.runner }}
runner-node: ${{ matrix.config['runner-node'] || '' }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
Expand Down
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
name: "svf-vllm-agg-gb300-tp4-mtp-agentic"

# GB300 AgentX aggregate topology: one TP4 worker occupies one four-GPU node
# and serves both prefill and decode at concurrency 4. Scheduler, CUDA-graph,
# and memory settings match the B300 vLLM TP4 MTP agentic configuration.
# and serves both prefill and decode at concurrency 8. Size max-num-seqs at
# 4x concurrency and expand the MTP CUDA-graph envelope to match.

model:
path: "deepseek-v4-pro"
Expand Down Expand Up @@ -77,7 +77,7 @@ backend:
VLLM_ENGINE_READY_TIMEOUT_S: "3600"
VLLM_RPC_TIMEOUT: "600000"
VLLM_LOG_STATS_INTERVAL: "1"
VLLM_V2_WARMUP_MAX_NUM_SEQS: "16"
VLLM_V2_WARMUP_MAX_NUM_SEQS: "32"
TILELANG_CLEANUP_TEMP_FILES: "1"
VLLM_USE_NCCL_SYMM_MEM: "0"
TORCH_SYMMMEM: "NVSHMEM"
Expand Down Expand Up @@ -110,23 +110,24 @@ backend:
tensor-parallel-size: 4
pipeline-parallel-size: 1
disable-custom-all-reduce: true
enable-cumem-allocator: true
attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}'
max-model-len: 1048576
max-num-seqs: 16
max-num-seqs: 32
max-num-batched-tokens: 8192
trust-remote-code: true
no-enable-flashinfer-autotune: true
block-size: 256
compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64],"mode":0}'
compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128],"mode":0}'
speculative-config: '{"method":"mtp","num_speculative_tokens":3,"rejection_sample_method":"synthetic","synthetic_acceptance_length":2.49}'
gpu-memory-utilization: 0.93
moe-backend: "deep_gemm_amxf4_mega_moe"
gpu-memory-utilization: 0.94
stream-interval: 10
no-disable-hybrid-kv-cache-manager: true
tokenizer-mode: "deepseek_v4"

sbatch_directives:
cpus-per-task: "72"
exclude: "im-gb300-r01-c002,im-gb300-r01-c003"
mem: "0"

srun_options:
Expand All @@ -141,6 +142,8 @@ benchmark:
PORT: "8000"
IS_MULTINODE: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
# Avoid concurrent readers observing a mismatched mmap data/index pair.
AIPERF_DATASET_MMAP_CACHE_ENABLED: "false"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"
WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126"
Original file line number Diff line number Diff line change
@@ -0,0 +1,150 @@
name: "svf-vllm-agg-gb300-tp8-c1-mtp-agentic"

# GB300 AgentX aggregate topology: one TP8 worker spans two four-GPU
# nodes and serves both prefill and decode at concurrency 1. Keep at least
# 16 sequence slots and size the MTP CUDA-graph envelope from that limit.

model:
path: "deepseek-v4-pro"
container: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f"
precision: "fp4"

identity:
model:
repo: "deepseek-ai/DeepSeek-V4-Pro"
container:
image: "vllm/vllm-openai:nightly-dev-arm64-cu13.0.1-426e59f"
frameworks:
dynamo: "1.2.1"

dynamo:
wheel: "1.2.1"
install: true

environment:
DYNAMO_WHEEL_DIRS: "/srtctl-wheels"
# The frontend shares Grace CPU capacity with the long TP8 cold start.
ETCD_LEASE_TTL: "7200"

setup_script: vllm-container-deps.sh

slurm:
time_limit: "8:00:00"

health_check:
max_attempts: 2160
interval_seconds: 10

resources:
gpu_type: "gb300"
gpus_per_node: 4
agg_nodes: 2
agg_workers: 1
gpus_per_agg: 8

infra:
etcd_nats_dedicated_node: false
nats_max_payload_mb: 32

frontend:
type: dynamo
enable_multiple_frontends: false
args:
router-mode: "kv"
router-reset-states: true
router-temperature: 0.0
router-queue-threshold: 65536
active-decode-blocks-threshold: "None"
active-prefill-tokens-threshold: "None"
active-prefill-tokens-threshold-frac: "None"
tokenizer: "fastokens"

backend:
type: vllm
connector: null
mooncake_kv_store:
store_config:
metadata_server: "P2PHANDSHAKE"
global_segment_size: "150GB"
local_buffer_size: "4GB"
protocol: "rdma"
device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3"
mode: "embedded"
enable_offload: false
aggregated_environment:
HF_HUB_CACHE: "/hf_hub_cache"
HUGGINGFACE_HUB_CACHE: "/hf_hub_cache"
TRANSFORMERS_CACHE: "/hf_hub_cache"
VLLM_ENGINE_READY_TIMEOUT_S: "3600"
VLLM_RPC_TIMEOUT: "600000"
VLLM_LOG_STATS_INTERVAL: "1"
VLLM_V2_WARMUP_MAX_NUM_SEQS: "16"
TILELANG_CLEANUP_TEMP_FILES: "1"
VLLM_USE_NCCL_SYMM_MEM: "0"
TORCH_SYMMMEM: "NVSHMEM"
NCCL_CUMEM_ENABLE: "1"
NCCL_MNNVL_ENABLE: "1"
NCCL_NVLS_ENABLE: "1"
VLLM_SERVER_DEV_MODE: "1"
VLLM_USE_V2_MODEL_RUNNER: "1"
VLLM_USE_RUST_FRONTEND: "1"
VLLM_ALLREDUCE_USE_FLASHINFER: "1"
VLLM_FLASHINFER_ALLREDUCE_BACKEND: "auto"
VLLM_MOONCAKE_LOAD_RECV_THREADS: "20"
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768"
VLLM_SPARSE_INDEXER_MAX_LOGITS_MB: "1024"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800"
UCX_MEMTYPE_CACHE: "n"
UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1"
UCX_TLS: "rc,cuda_copy"
NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3"
NCCL_P2P_LEVEL: "NVL"
MC_ENABLE_DEST_DEVICE_AFFINITY: "1"
MC_STORE_CLIENT_METRIC: "1"
MC_STORE_CLIENT_METRIC_INTERVAL: "5"
MC_TE_METRIC: "0"
DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-agg-tp8-c1-mtp-{job_id}"
vllm_config:
aggregated:
served-model-name: "deepseek-ai/DeepSeek-V4-Pro"
kv-cache-dtype: "fp8"
tensor-parallel-size: 8
pipeline-parallel-size: 1
disable-custom-all-reduce: true
attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}'
max-model-len: 1048576
max-num-seqs: 16
max-num-batched-tokens: 8192
trust-remote-code: true
no-enable-flashinfer-autotune: true
block-size: 256
compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64],"mode":0}'
speculative-config: '{"method":"mtp","num_speculative_tokens":3,"rejection_sample_method":"synthetic","synthetic_acceptance_length":2.49}'
moe-backend: "deep_gemm_amxf4_mega_moe"
gpu-memory-utilization: 0.94
stream-interval: 10
no-disable-hybrid-kv-cache-manager: true
tokenizer-mode: "deepseek_v4"

sbatch_directives:
cpus-per-task: "144"
exclude: "im-gb300-r01-c002,im-gb300-r01-c003"
mem: "0"

srun_options:
container-remap-root: ""

benchmark:
type: custom
command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh
env:
INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace"
RESULT_DIR: "/logs/agentic"
PORT: "8000"
# Keep aggregate workers in the multinode result schema so ingestion uses
# the zero decode-worker count instead of duplicating TP into P and D.
IS_MULTINODE: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"
WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126"
Original file line number Diff line number Diff line change
@@ -1,7 +1,8 @@
name: "svf-vllm-agg-gb300-tp8-mtp-agentic"

# Validated GB300 AgentX aggregate topology: one TP8 worker spans two
# four-GPU nodes and serves both prefill and decode at concurrency 1.
# GB300 AgentX aggregate topology: one TP8 worker spans two four-GPU
# nodes and serves both prefill and decode at concurrency 4. Keep at least
# 16 sequence slots and otherwise size the scheduler at 4x concurrency.

model:
path: "deepseek-v4-pro"
Expand Down Expand Up @@ -77,7 +78,7 @@ backend:
VLLM_ENGINE_READY_TIMEOUT_S: "3600"
VLLM_RPC_TIMEOUT: "600000"
VLLM_LOG_STATS_INTERVAL: "1"
VLLM_V2_WARMUP_MAX_NUM_SEQS: "32"
VLLM_V2_WARMUP_MAX_NUM_SEQS: "16"
TILELANG_CLEANUP_TEMP_FILES: "1"
VLLM_USE_NCCL_SYMM_MEM: "0"
TORCH_SYMMMEM: "NVSHMEM"
Expand Down Expand Up @@ -110,24 +111,24 @@ backend:
tensor-parallel-size: 8
pipeline-parallel-size: 1
disable-custom-all-reduce: true
enable-cumem-allocator: true
attention-config: '{"backend":"FLASHINFER_MLA_SPARSE_DSV4","use_prefill_query_quantization":true,"use_fp4_indexer_cache":true}'
max-model-len: 1048576
max-num-seqs: 32
max-num-seqs: 16
max-num-batched-tokens: 8192
trust-remote-code: true
no-enable-flashinfer-autotune: true
block-size: 256
compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","mode":0}'
max-cudagraph-capture-size: 128
compilation-config: '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64],"mode":0}'
speculative-config: '{"method":"mtp","num_speculative_tokens":3,"rejection_sample_method":"synthetic","synthetic_acceptance_length":2.49}'
gpu-memory-utilization: 0.90
moe-backend: "deep_gemm_amxf4_mega_moe"
gpu-memory-utilization: 0.94
stream-interval: 10
no-disable-hybrid-kv-cache-manager: true
tokenizer-mode: "deepseek_v4"

sbatch_directives:
cpus-per-task: "144"
exclude: "im-gb300-r01-c002,im-gb300-r01-c003"
mem: "0"

srun_options:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
name: "svf-vllm-disagg-gb300-1p1d-dep4-dep8-c128-mtp-agentic"
name: "svf-vllm-disagg-gb300-1p1d-dep4-dep8-c256-mtp-agentic"

# Validated GB300 AgentX MTP3 low-latency topology: one DEP4 prefill worker
# feeds one DEP8 decode worker at concurrency 128.
# GB300 AgentX MTP3 topology: one DEP4 prefill worker feeds one
# DEP8 decode worker at concurrency 256.

model:
path: "deepseek-v4-pro"
Expand All @@ -17,7 +17,7 @@ identity:
dynamo: "1.3.0.dev20260720"

dynamo:
version: "1.3.0.dev20260720"
wheel: "1.3.0.dev20260720"
install: true

setup_script: vllm-container-deps.sh
Expand Down Expand Up @@ -57,6 +57,8 @@ frontend:
router-session-affinity-ttl-secs: 900
env:
DYN_ROUTER_ACTIVE_REQUEST_EXPIRY_SECS: "3600"
DYN_TCP_CHANNEL_BUFFER: "128"
DYN_TCP_REQUEST_TIMEOUT: "60"

backend:
type: vllm
Expand All @@ -65,7 +67,7 @@ backend:
mooncake_kv_store:
store_config:
metadata_server: "P2PHANDSHAKE"
global_segment_size: "150GB"
global_segment_size: "180GB"
local_buffer_size: "4GB"
protocol: "rdma"
device_name: "mlx5_0,mlx5_1,mlx5_2,mlx5_3"
Expand All @@ -86,6 +88,7 @@ backend:
max-model-len: 1048576
max-num-seqs: 64
max-num-batched-tokens: 8192
long-prefill-token-threshold: 1024
trust-remote-code: true
no-enable-flashinfer-autotune: true
block-size: 256
Expand Down Expand Up @@ -149,7 +152,7 @@ backend:
VLLM_USE_BREAKABLE_CUDAGRAPH: "0"
VLLM_CONNECTOR_PREFETCH_DEPTH: "8"
VLLM_DSV4_MEGA_FP8_COMBINE: "1"
DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-c128-mtp-{job_id}"
DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep4-dep8-c256-{job_id}"
MC_ENABLE_DEST_DEVICE_AFFINITY: "1"
MC_STORE_CLIENT_METRIC: "1"
MC_STORE_CLIENT_METRIC_INTERVAL: "5"
Expand All @@ -176,14 +179,15 @@ backend:
VLLM_RANDOMIZE_DP_DUMMY_INPUTS: "1"
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "32768"
VLLM_DSV4_MEGA_FP8_COMBINE: "1"
DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-c128-mtp-{job_id}"
DG_JIT_CACHE_DIR: "/tmp/dg-cache-dsv4-gb300-1p1d-dep4-dep8-c256-{job_id}"
MC_ENABLE_DEST_DEVICE_AFFINITY: "1"
MC_STORE_CLIENT_METRIC: "1"
MC_STORE_CLIENT_METRIC_INTERVAL: "5"
MC_TE_METRIC: "0"

sbatch_directives:
cpus-per-task: "72"
exclude: "im-gb300-r01-c002,im-gb300-r01-c003"

srun_options:
container-remap-root: ""
Expand All @@ -198,6 +202,8 @@ benchmark:
IS_MULTINODE: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID: "true"
# Avoid concurrent readers observing a mismatched mmap data/index pair.
AIPERF_DATASET_MMAP_CACHE_ENABLED: "false"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"
WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126"
Loading
Loading