Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
58 commits
Select commit Hold shift + click to select a range
245dcda
feat(agentx): add Kimi K3 GB200 day-0 support
cquil11 Jul 27, 2026
3e2b960
chore: link GB200 Kimi K3 changelog
cquil11 Jul 27, 2026
4aa6b8e
Merge origin/main into GB200 Kimi K3 AgentX
cquil11 Jul 27, 2026
18b1af1
fix(agentx): align Kimi K3 DEP metadata
cquil11 Jul 27, 2026
acfd9f3
perf(gb200): load Kimi K3 from local NVMe
cquil11 Jul 27, 2026
fd088cd
chore: simplify Kimi K3 performance config
cquil11 Jul 27, 2026
6c335b5
perf(gb200): size Kimi K3 AgentX sweep from HBM
cquil11 Jul 28, 2026
dc90e33
fix(gb200): drop unsupported Kimi K3 worker parsers
cquil11 Jul 28, 2026
d3b3f05
fix(gb200): let srt-slurm select collective interfaces
cquil11 Jul 28, 2026
f513599
fix(gb200): launch Kimi K3 TP2 DP8 per node
cquil11 Jul 28, 2026
bc21aac
fix(gb200): rebalance Kimi K3 EP16 for HBM
cquil11 Jul 28, 2026
819af41
fix(gb200): use native srt-slurm TPDP support
cquil11 Jul 28, 2026
458d6f6
docs(perf): correct Kimi K3 throughput topology
cquil11 Jul 28, 2026
6d10fbc
fix(gb200): use official Kimi K3 Dynamo runtime
cquil11 Jul 28, 2026
4158041
fix(gb200): align Kimi K3 runtime metadata
cquil11 Jul 28, 2026
a32ee0b
fix(gb200): reserve Kimi K3 prefill workspace
cquil11 Jul 28, 2026
86f4e70
fix(config): report Kimi K3 TP16 as EP16
cquil11 Jul 28, 2026
ed0f177
merge: sync main into Kimi K3 GB200 support
cquil11 Jul 28, 2026
1f052d2
fix(gb200): stage srt AgentX aggregates
cquil11 Jul 28, 2026
64de559
fix(agentx): burst Kimi K3 saturation starts
cquil11 Jul 28, 2026
ff5fb06
refactor(kimik3): depend on shared phase starts
cquil11 Jul 28, 2026
2ddb6d5
merge: sync main into Kimi K3 GB200 support
cquil11 Jul 28, 2026
fc99700
Merge origin/main into agent/kimik3-gb200-agentx
cquil11 Jul 28, 2026
235aa5e
fix(agentx): extend Kimi K3 DEP warmup drain
cquil11 Jul 28, 2026
26e6697
merge: sync main into Kimi K3 GB200 support
cquil11 Jul 28, 2026
403e0c1
merge: sync latest main into Kimi K3 GB200 support
cquil11 Jul 28, 2026
1187a30
chore(runners): register fourth GB200 worker
cquil11 Jul 28, 2026
275eef2
merge: sync latest main into Kimi K3 GB200 offload
cquil11 Jul 29, 2026
d29685c
feat(agentx): add GB200 Kimi K3 CPU KV offload
cquil11 Jul 29, 2026
5bb3034
merge: sync latest main into Kimi K3 GB200 offload
cquil11 Jul 29, 2026
4321a95
Merge branch 'main' into agent/kimik3-gb200-simple-offload
cquil11 Jul 29, 2026
a1e7500
feat(kimik3): use upstream vLLM with DSpark on GB200
cquil11 Jul 29, 2026
072a242
Merge branch 'main' into agent/kimik3-gb200-simple-offload
cquil11 Jul 29, 2026
8400af0
fix(kimik3): pin K3-compatible Dynamo preview
cquil11 Jul 29, 2026
dba74e0
Merge origin/main into agent/kimik3-gb200-simple-offload
cquil11 Jul 29, 2026
74b7aec
merge: sync latest main into Kimi K3 GB200 offload
cquil11 Jul 29, 2026
9eaa372
fix(kimik3): bridge DSpark mask metadata for Dynamo
cquil11 Jul 29, 2026
389141c
fix(kimik3): use model runner v2 for DSpark
cquil11 Jul 29, 2026
d4791fc
Merge origin/main into Kimi K3 GB200 AgentX
cquil11 Jul 29, 2026
07d0a1c
Merge origin/main into Kimi K3 GB200 AgentX
cquil11 Jul 30, 2026
35c272d
fix(kimik3): preserve full context on DEP
cquil11 Jul 30, 2026
e141992
fix(kimik3): reserve DEP graph memory
cquil11 Jul 30, 2026
c6841ed
fix(kimik3): use piecewise DEP graphs
cquil11 Jul 30, 2026
07775b0
merge: sync latest main
cquil11 Jul 30, 2026
608e9c0
fix(kimik3): stabilize DSpark DEP MoE
cquil11 Jul 30, 2026
d598886
fix(agentx): reduce Kimi K3 DEP load memory
cquil11 Jul 30, 2026
4595493
fix(agentx): right-size Kimi K3 DEP graphs
cquil11 Jul 30, 2026
dce6567
fix(agentx): report pinned Dynamo version
cquil11 Jul 30, 2026
4192722
fix(agentx): reduce offload graph residency
cquil11 Jul 30, 2026
2eeac85
chore: merge main into Kimi K3 GB200 branch
cquil11 Jul 30, 2026
f002be9
fix(kimik3): extend offload sweep allocation
cquil11 Jul 30, 2026
447fcae
perf(kimik3): report prompt cache reads
cquil11 Jul 30, 2026
54c8a7c
merge: sync latest main
cquil11 Jul 30, 2026
d238ef7
fix(agentx): align GB200 replay timing
cquil11 Jul 30, 2026
4491f28
fix(agentx): let Dynamo report cached prompt tokens
cquil11 Jul 30, 2026
43f2a8b
merge: sync current main into GB200 AgentX
cquil11 Jul 30, 2026
636897b
fix(agentx): align AIPerf with Qwen frontiers
cquil11 Jul 31, 2026
2ba037b
fix(agentx): separate live and final error gates
cquil11 Jul 31, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 19 additions & 5 deletions benchmarks/benchmark_lib.sh
Original file line number Diff line number Diff line change
Expand Up @@ -1577,6 +1577,7 @@ AIPERF_CLI="${AIPERF_VENV}/bin/aiperf"
AIPERF_HF_CLI="${AIPERF_VENV}/bin/hf"
AIPERF_DEPS_READY=0
AIPERF_FAILED_REQUEST_THRESHOLD="${AIPERF_FAILED_REQUEST_THRESHOLD:-0.10}"
AIPERF_LIVE_FAILED_REQUEST_THRESHOLD="${AIPERF_LIVE_FAILED_REQUEST_THRESHOLD:-$AIPERF_FAILED_REQUEST_THRESHOLD}"

agentic_pip_install() {
local pip_install=(python3 -m pip install)
Expand Down Expand Up @@ -1743,6 +1744,7 @@ build_replay_cmd() {
local result_dir="$1"
local duration="$DURATION"
local warmup_requests_per_lane="${AIPERF_WARMUP_REQUESTS_PER_LANE:-10}"
local trace_idle_gap_cap_seconds="${AIPERF_TRACE_IDLE_GAP_CAP_SECONDS:-}"

# Fast mode minimizes setup by advancing each trajectory lane only once
# and shortens profiling to 20 minutes.
Expand Down Expand Up @@ -1775,11 +1777,23 @@ build_replay_cmd() {
REPLAY_CMD+=" --benchmark-duration $duration"
REPLAY_CMD+=" --stats-interval 30"
REPLAY_CMD+=" --random-seed 42"
# Fail runs once more than 10% of requests error. This keeps known
# transient low-rate failures from killing long sweeps while still
# catching malformed payloads or server crashes before they get aggregated
# as benchmarkable data.
REPLAY_CMD+=" --failed-request-threshold $AIPERF_FAILED_REQUEST_THRESHOLD"
# Some long AgentX traces contain recorded request-start gaps that would
# otherwise hold a trajectory lane idle for much longer than the useful
# cache-TTL window. When a recipe opts in, cap those gaps independently
# within each root trace during dataset reconstruction. This changes only
# replay timing; it is not a runtime request timeout.
if [ -n "$trace_idle_gap_cap_seconds" ]; then
if ! [[ "$trace_idle_gap_cap_seconds" =~ ^[0-9]+([.][0-9]+)?$ ]]; then
echo "ERROR: AIPERF_TRACE_IDLE_GAP_CAP_SECONDS must be a non-negative number" >&2
return 1
fi
REPLAY_CMD+=" --trace-idle-gap-cap-seconds $trace_idle_gap_cap_seconds"
fi
# Fail runs early once the live error ratio crosses the configured limit.
# Recipes with correlated low-concurrency trajectories may allow a larger
# live sample while retaining AIPERF_FAILED_REQUEST_THRESHOLD as the strict
# post-run validity gate below.
REPLAY_CMD+=" --failed-request-threshold $AIPERF_LIVE_FAILED_REQUEST_THRESHOLD"
# Sample each trajectory's warmup start position uniformly from
# [25%, 75%] of the trace's turn count, clamped by AIPerf to leave at
# least one profile turn after warmup.
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,49 @@
#!/usr/bin/env bash
set -euo pipefail

# The Kimi K3 DSpark checkpoint publishes its parallel-drafting token as
# `mask_token_id`. Dynamo's serialized draft config reaches vLLM without the
# K3 config-class alias, while vLLM's parallel drafter accepts `pard_token`.
# Build a thin local view of the upstream checkpoint and add that equivalent
# metadata alias without changing vLLM or the checkpoint weights.
python3 - <<'PY'
import json
import os
from pathlib import Path

from huggingface_hub import snapshot_download

repo_id = "Inferact/Kimi-K3-DSpark"
target = Path("/tmp/Kimi-K3-DSpark")
snapshot = Path(snapshot_download(repo_id=repo_id))
target.mkdir(parents=True, exist_ok=True)

for source in snapshot.iterdir():
if source.name == "config.json":
continue
destination = target / source.name
if destination.is_symlink():
if destination.resolve() == source.resolve():
continue
destination.unlink()
elif destination.exists():
raise RuntimeError(f"Refusing to replace non-symlink path: {destination}")
destination.symlink_to(source)

config = json.loads((snapshot / "config.json").read_text())
mask_token_id = config.get("mask_token_id")
if not isinstance(mask_token_id, int):
raise RuntimeError(f"{repo_id} config is missing integer mask_token_id")

pard_token = config.get("pard_token")
if pard_token not in (None, mask_token_id):
raise RuntimeError(
f"{repo_id} pard_token={pard_token} disagrees with mask_token_id={mask_token_id}"
)
config["pard_token"] = mask_token_id

temporary = target / "config.json.tmp"
temporary.write_text(json.dumps(config, indent=2) + "\n")
os.replace(temporary, target / "config.json")
print(f"Prepared {repo_id} compatibility view at {target}")
PY
Original file line number Diff line number Diff line change
@@ -0,0 +1,145 @@
name: "kimi-k3-vllm-agg-gb200-dep16-throughput-agentic"

# Day-0 GB200 translation of the official throughput-oriented multi_node_dep
# profile. TP4 x DP4 gives EP16 across four four-GPU GB200 nodes.
# https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_dep

model:
path: "kimi-k3"
container: "vllm/vllm-openai:kimi-k3"
precision: "fp4"

identity:
model:
repo: "moonshotai/Kimi-K3"
container:
image: "vllm/vllm-openai:kimi-k3"
frameworks:
dynamo: "1.3.0"

dynamo:
hash: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1"
install: true

setup_script: kimik3-dspark-config-compat.sh

environment:
ETCD_LEASE_TTL: "7200"

slurm:
time_limit: "8:00:00"

health_check:
max_attempts: 2160
interval_seconds: 10

resources:
gpu_type: "gb200"
gpus_per_node: 4
agg_nodes: 4
agg_workers: 1
gpus_per_agg: 16

infra:
etcd_nats_dedicated_node: false
nats_max_payload_mb: 32

frontend:
type: dynamo
enable_multiple_frontends: false
args:
dyn-chat-processor: "dynamo"
router-mode: "kv"
router-kv-events: true
router-temperature: "0"
router-min-initial-workers: 1
kv-cache-block-size: 64

backend:
type: vllm
connector: null
aggregated_environment:
HF_HUB_CACHE: "/hf_hub_cache"
HUGGINGFACE_HUB_CACHE: "/hf_hub_cache"
TRANSFORMERS_CACHE: "/hf_hub_cache"
VLLM_ENGINE_READY_TIMEOUT_S: "7200"
VLLM_RPC_TIMEOUT: "600000"
VLLM_LOG_STATS_INTERVAL: "1"
VLLM_USE_V2_MODEL_RUNNER: "1"
VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1"
VLLM_ALLREDUCE_USE_FLASHINFER: "1"
VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl"
NCCL_CUMEM_ENABLE: "1"
NCCL_MNNVL_ENABLE: "1"
NCCL_NVLS_ENABLE: "1"
UCX_MEMTYPE_CACHE: "n"
UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1"
UCX_TLS: "rc,cuda_copy"
NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3"
NCCL_P2P_LEVEL: "NVL"
NVIDIA_GDRCOPY: "1"
PYTORCH_ALLOC_CONF: "expandable_segments:True"
DG_JIT_CACHE_DIR: "/tmp/dg-cache-kimi-k3-gb200-dep16-{job_id}"
vllm_config:
aggregated:
served-model-name: "moonshotai/Kimi-K3"
tensor-parallel-size: 4
pipeline-parallel-size: 1
data-parallel-size: 4
data-parallel-rpc-port: 13345
enable-expert-parallel: true
trust-remote-code: true
# FlashInfer's larger TP4 MoE representation leaves too little transient
# HBM for fastsafetensors' GPU-staging path when DSpark is loaded.
load-format: "safetensors"
safetensors-load-strategy: "lazy"
kv-cache-dtype: "fp8"
attention-backend: "FLASHINFER_MLA"
attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}'
# DeepGEMM mega-MoE grid barriers time out under DSpark MRV2 with the
# one-sided all-to-all worker; use the supported Kimi K3 FlashInfer path.
moe-backend: "flashinfer_trtllm"
kda-prefill-backend: "flashkda"
kernel-config: '{"enable_cutedsl_warmup":true}'
all2all-backend: "flashinfer_nvlink_one_sided"
gpu-memory-utilization: 0.94
# The largest regular DEP point is c256 / DP4 = 64 sequences per engine.
# Capturing 128 sequence slots consumes 10.9 GiB and leaves too little
# runtime workspace for FlashInfer's MXFP4 MoE kernel.
max-num-seqs: 64
max-num-batched-tokens: 16384
speculative-config: '{"method":"dspark","model":"/tmp/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"synthetic","synthetic_acceptance_length":2.51}'
compilation-config: '{"cudagraph_mode":"PIECEWISE","cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42,45,48,51,54,57,60,63,66,69,72,75,78,81,84,87,90,93,96,99,102,105,108,111,114,117,120,123,126,129,132,135,138,141,144,147,150,153,156,159,162,165,168,171,174,177,180,183,186,189,192]}'
block-size: 64
language-model-only: true
disable-custom-all-reduce: true
enable-prefix-caching: true
scheduler-cls: "vllm.v1.core.sched.async_scheduler.AsyncScheduler"
dyn-tool-call-parser: "kimi_k3"
reasoning-parser: "kimi_k3"
dyn-reasoning-parser: "kimi_k3"
no-enable-flashinfer-autotune: true

sbatch_directives:
cpus-per-task: "144"
mem: "0"

srun_options:
container-remap-root: ""

benchmark:
type: custom
aiperf_server_metrics: true
command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh
env:
INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace"
AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300"
RESULT_DIR: "/logs/agentic"
PORT: "8000"
IS_MULTINODE: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400"
AGENTIC_WARMUP_GRACE_PERIOD: "3600"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"
WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126"
Original file line number Diff line number Diff line change
@@ -0,0 +1,148 @@
name: "kimi-k3-vllm-agg-gb200-dep16-throughput-vllm-simple-offload-agentic"

# High-concurrency host-DRAM KV-offload variant of the official throughput-
# oriented multi_node_dep profile. TP4 x DP4 gives EP16 across four four-GPU
# GB200 nodes. Each TP rank receives a 128 GiB CPU KV pool (512 GiB per node).
# https://recipes.vllm.ai/moonshotai/Kimi-K3?hardware=gb200&nodes=4&strategy=multi_node_dep

model:
path: "kimi-k3"
container: "vllm/vllm-openai:kimi-k3"
precision: "fp4"

identity:
model:
repo: "moonshotai/Kimi-K3"
container:
image: "vllm/vllm-openai:kimi-k3"
frameworks:
dynamo: "1.3.0"

dynamo:
hash: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1"
install: true

setup_script: kimik3-dspark-config-compat.sh

environment:
ETCD_LEASE_TTL: "7200"

slurm:
time_limit: "12:00:00"

health_check:
max_attempts: 2160
interval_seconds: 10

resources:
gpu_type: "gb200"
gpus_per_node: 4
agg_nodes: 4
agg_workers: 1
gpus_per_agg: 16

infra:
etcd_nats_dedicated_node: false
nats_max_payload_mb: 32

frontend:
type: dynamo
enable_multiple_frontends: false
args:
dyn-chat-processor: "dynamo"
router-mode: "kv"
router-kv-events: true
router-temperature: "0"
router-min-initial-workers: 1
kv-cache-block-size: 64

backend:
type: vllm
connector: null
aggregated_environment:
HF_HUB_CACHE: "/hf_hub_cache"
HUGGINGFACE_HUB_CACHE: "/hf_hub_cache"
TRANSFORMERS_CACHE: "/hf_hub_cache"
VLLM_ENGINE_READY_TIMEOUT_S: "7200"
VLLM_RPC_TIMEOUT: "600000"
VLLM_LOG_STATS_INTERVAL: "1"
VLLM_USE_V2_MODEL_RUNNER: "1"
VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1"
VLLM_ALLREDUCE_USE_FLASHINFER: "1"
VLLM_FLASHINFER_ALLREDUCE_BACKEND: "mnnvl"
NCCL_CUMEM_ENABLE: "1"
NCCL_MNNVL_ENABLE: "1"
NCCL_NVLS_ENABLE: "1"
UCX_MEMTYPE_CACHE: "n"
UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1"
UCX_TLS: "rc,cuda_copy"
NCCL_IB_HCA: "mlx5_0,mlx5_1,mlx5_2,mlx5_3"
NCCL_P2P_LEVEL: "NVL"
NVIDIA_GDRCOPY: "1"
PYTHONHASHSEED: "42"
PYTORCH_ALLOC_CONF: "expandable_segments:True"
DG_JIT_CACHE_DIR: "/tmp/dg-cache-kimi-k3-gb200-dep16-offload-{job_id}"
vllm_config:
aggregated:
served-model-name: "moonshotai/Kimi-K3"
tensor-parallel-size: 4
pipeline-parallel-size: 1
data-parallel-size: 4
data-parallel-rpc-port: 13345
enable-expert-parallel: true
trust-remote-code: true
# FlashInfer's larger TP4 MoE representation leaves too little transient
# HBM for fastsafetensors' GPU-staging path when DSpark is loaded.
load-format: "safetensors"
safetensors-load-strategy: "lazy"
kv-cache-dtype: "fp8"
attention-backend: "FLASHINFER_MLA"
attention-config: '{"mla_prefill_backend":"TRTLLM_RAGGED","use_prefill_query_quantization":true}'
# DeepGEMM mega-MoE grid barriers time out under DSpark MRV2 with the
# one-sided all-to-all worker; use the supported Kimi K3 FlashInfer path.
moe-backend: "flashinfer_trtllm"
kda-prefill-backend: "flashkda"
kernel-config: '{"enable_cutedsl_warmup":true}'
all2all-backend: "flashinfer_nvlink_one_sided"
gpu-memory-utilization: 0.94
# The largest offload point is c384 / DP4 = 96 sequences per engine.
# Capture even sequence counts: all configured DP4 steady-state batch
# sizes are exact hits, while odd loads pad by at most one sequence.
max-num-seqs: 96
max-num-batched-tokens: 16384
speculative-config: '{"method":"dspark","model":"/tmp/Kimi-K3-DSpark","num_speculative_tokens":2,"attention_backend":"FLASHINFER_MLA","draft_sample_method":"probabilistic","rejection_sample_method":"synthetic","synthetic_acceptance_length":2.51}'
compilation-config: '{"cudagraph_mode":"PIECEWISE","cudagraph_capture_sizes":[6,12,18,24,30,36,42,48,54,60,66,72,78,84,90,96,102,108,114,120,126,132,138,144,150,156,162,168,174,180,186,192,198,204,210,216,222,228,234,240,246,252,258,264,270,276,282,288]}'
block-size: 64
language-model-only: true
disable-custom-all-reduce: true
enable-prefix-caching: true
kv-transfer-config: '{"kv_connector":"SimpleCPUOffloadConnector","kv_role":"kv_both","kv_connector_extra_config":{"cpu_bytes_to_use":549755813888,"cpu_bytes_to_use_per_rank":137438953472,"lazy_offload":false}}'
scheduler-cls: "vllm.v1.core.sched.async_scheduler.AsyncScheduler"
dyn-tool-call-parser: "kimi_k3"
reasoning-parser: "kimi_k3"
dyn-reasoning-parser: "kimi_k3"
no-enable-flashinfer-autotune: true

sbatch_directives:
cpus-per-task: "144"
mem: "0"

srun_options:
container-remap-root: ""

benchmark:
type: custom
aiperf_server_metrics: true
command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh
env:
INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace"
AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300"
RESULT_DIR: "/logs/agentic"
PORT: "8000"
IS_MULTINODE: "true"
AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING: "0"
AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400"
AGENTIC_WARMUP_GRACE_PERIOD: "3600"
AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache"
HF_HUB_CACHE: "/hf_hub_cache"
WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126"
Loading
Loading