diff --git a/MODELS.md b/MODELS.md index 3fc822420e..00936552c9 100644 --- a/MODELS.md +++ b/MODELS.md @@ -18,7 +18,7 @@ This document tracks every model benchmarked by InferenceX-e2e: when it was adde | Model architecture class | Prefix | Date added | Active scenarios | Deprecated scenarios | |---|---|---|---|---| | Qwen3.8 2.4T | `qwen3.8` | TBD | Agentic coding | | -| Kimi-K3 | `kimik3` | 2026-07-27 | Agentic coding | | +| Kimi-K3 | `kimik3` | 2026-07-27 ([#2357](https://github.com/SemiAnalysisAI/InferenceX/pull/2357)) | Agentic coding | | | GLM-5.2 | `glm5.2` | 2026-07-18 ([#2268](https://github.com/SemiAnalysisAI/InferenceX/pull/2268)) | Agentic coding | | | MiniMax-M3 | `minimaxm3` | 2026-06-12 ([#1724](https://github.com/SemiAnalysisAI/InferenceX/pull/1724)) | Single-turn 8k1k, Agentic coding | Single-turn 1k1k | | DeepSeek-V4-Pro | `dsv4` | 2026-04-24 ([#1130](https://github.com/SemiAnalysisAI/InferenceX/pull/1130)) | Single-turn 8k1k, Agentic coding | Single-turn 1k1k | diff --git a/MODELS_zh.md b/MODELS_zh.md index 177a4b4ba3..7be80603bd 100644 --- a/MODELS_zh.md +++ b/MODELS_zh.md @@ -18,7 +18,7 @@ | 模型架构类别 | 前缀 | 加入日期 | 启用场景 | 已弃用场景 | |---|---|---|---|---| | Qwen3.8 2.4T | `qwen3.8` | 待定 | 智能体编码 | | -| Kimi-K3 | `kimik3` | 2026-07-27 | 智能体编码 | | +| Kimi-K3 | `kimik3` | 2026-07-27 ([#2357](https://github.com/SemiAnalysisAI/InferenceX/pull/2357)) | 智能体编码 | | | GLM-5.2 | `glm5.2` | 2026-07-18([#2268](https://github.com/SemiAnalysisAI/InferenceX/pull/2268)) | 智能体编码 | | | MiniMax-M3 | `minimaxm3` | 2026-06-12([#1724](https://github.com/SemiAnalysisAI/InferenceX/pull/1724)) | 单轮 8k1k、智能体编码 | 单轮 1k1k | | DeepSeek-V4-Pro | `dsv4` | 2026-04-24([#1130](https://github.com/SemiAnalysisAI/InferenceX/pull/1130)) | 单轮 8k1k、智能体编码 | 单轮 1k1k | diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml new file mode 100644 index 0000000000..7109b14097 --- /dev/null +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml @@ -0,0 +1,117 @@ +name: "kimik3-vllm-agg-b200-tp8pp2-agentic" + +# Kimi-K3 MXFP4 B200 AGGREGATED TP8 x PP2 agentic recipe (2 nodes / 16 GPUs). +# The native MXFP4 checkpoint (2.8T total params, ~1.4TB of weights) does not +# fit one 8xB200 node, so TP8 shards attention/dense (/8) and PP2 splits the +# 93 layers (/2) across 16 GPUs. Plain TP (NOT TEP): expert parallelism is +# deliberately off, so the 896 routed experts are TP-sharded inside each +# pipeline stage. Node allocation = tp*pp/gpus_per_node = 8*2/8 = 2 nodes. +# Aggregated (single worker, decode num-worker 0) — no P/D split, no NIXL. +# VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION fuses the K3 LatentMoE tail path in +# the kimi-k3 bring-up image. +model: + path: "kimik3" + container: "vllm/vllm-openai:kimi-k3" + precision: "fp4" + +identity: + model: + repo: "moonshotai/Kimi-K3" + container: + image: "vllm/vllm-openai:kimi-k3" + frameworks: + dynamo: "1.4.0-kimi-k3-dev.1" + +dynamo: + install: true + # Day-zero Kimi-K3 dynamo ("feat: Added support for Kimi-K3", tag + # v1.4.0-kimi-k3-dev.1 == this commit): adds the kimi_k3 tiktoken tokenizer + # to the rust frontend (dynamo <=1.2.1 only knows kimi/kimi_k2/kimi_k25/ + # deepseek_v3, so the model never registers and every request 404s) and the + # kimi_k3 tool-call/reasoning parser worker args. + hash: "ba83080ecd31c1ce918559e576d3c5bc9e092ff1" + +slurm: + time_limit: "8:00:00" + +health_check: + interval_seconds: 10 + max_attempts: 1440 + +resources: + gpu_type: "b200" + gpus_per_node: 8 + agg_nodes: 2 + agg_workers: 1 + gpus_per_agg: 16 + +infra: + etcd_nats_dedicated_node: false + nats_max_payload_mb: 32 + +frontend: + type: dynamo + enable_multiple_frontends: false + +backend: + type: vllm + connector: null + aggregated_environment: + VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION: "1" + VLLM_SERVER_DEV_MODE: "1" + # ~1.4TB of MXFP4 weights off shared Lustre: keep the engine-ready window + # generous, and let one long AgentX request hold a PP stage beyond vLLM's + # 300-second model-execution default. + VLLM_ENGINE_READY_TIMEOUT_S: "3600" + VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "1800" + NCCL_CUMEM_ENABLE: "1" + TILELANG_CLEANUP_TEMP_FILES: "1" + UCX_MEMTYPE_CACHE: "n" + UCX_MEMTYPE_REG_WHOLE: "n" + UCX_NET_DEVICES: "mlx5_0:1,mlx5_1:1,mlx5_2:1,mlx5_3:1,mlx5_4:1,mlx5_5:1,mlx5_10:1,mlx5_11:1" + HF_HUB_CACHE: "/hf_hub_cache" + HUGGINGFACE_HUB_CACHE: "/hf_hub_cache" + vllm_config: + aggregated: + served-model-name: "moonshotai/Kimi-K3" + tensor-parallel-size: 8 + pipeline-parallel-size: 2 + trust-remote-code: true + load-format: fastsafetensors + moe-backend: auto + gpu-memory-utilization: 0.95 + no-enable-flashinfer-autotune: true + # kimi_k3 parsers via the plain vLLM spellings (parser-flag experiment + # variant B — sibling PRs try the --dyn-* namespaced args and no parser + # flags). Note dynamo <=1.2.1's worker rejected --tool-call-parser as + # unrecognized; the day-zero K3 dynamo pinned above may accept it. + tool-call-parser: kimi_k3 + reasoning-parser: kimi_k3 + # No explicit max-model-len: let vLLM derive the native 1M window from + # the model config (agentic trajectories blow past any small cap, and + # K3's KDA layers keep per-token KV small — only the 24 gated-MLA + # layers hold cache). Prefix caching stays on (default) for trajectory + # reuse. Cap prefill chunks so a single long request cannot OOM a + # pipeline stage; let vLLM pick max-num-seqs. + max-num-batched-tokens: 8192 + +sbatch_directives: + segment: "1" + +srun_options: + container-remap-root: "" + +benchmark: + type: custom + command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh + env: + INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" + RESULT_DIR: "/logs/agentic" + PORT: "8000" + # Keep the aggregate worker in the multinode result schema so ingestion + # uses the zero decode-worker count instead of duplicating TP into P and D. + IS_MULTINODE: "true" + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS: "14400" + AIPERF_DATASET_MMAP_CACHE_DIR: "/aiperf_mmap_cache" + HF_HUB_CACHE: "/hf_hub_cache" + WEKA_LOADER_OVERRIDE: "semianalysis_cc_traces_weka_062126" diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index 36ae816809..89bf2fad2b 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8325,3 +8325,48 @@ qwen3.5-fp8-gb200-dynamo-sglang-mtp: tp: 16 ep: 16 dp-attn: true + +# Kimi-K3 MXFP4 B200 aggregated vLLM via Dynamo (TP8 x PP2, 2 nodes / 16 +# GPUs), agentic bring-up. The native MXFP4 checkpoint (2.8T total params, +# ~1.4TB weights) does not fit one 8xB200 node, so TP8 shards attention/dense +# and PP2 splits layers. Plain TP (NOT TEP): ep 1, no expert parallelism — +# the 896 routed experts are TP-sharded within each pipeline stage. Node +# count = tp*pp/gpus_per_node = 8*2/8 = 2. Aggregated (prefill num-worker 1 + +# decode num-worker 0, RECIPES.md section 5) — the single worker serves both +# phases, so no P/D KV transfer. Dedicated kimi-k3 vLLM bring-up image with +# VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION=1 and the kimi_k3 tool-call/reasoning +# parsers. +# Recipe: benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml +kimik3-fp4-b200-dynamo-vllm-agentic: + image: vllm/vllm-openai:kimi-k3 + model: moonshotai/Kimi-K3 + model-prefix: kimik3 + runner: cluster:b200-dgxc + precision: fp4 + framework: dynamo-vllm + router: { name: dynamo-router, version: "1.4.0-kimi-k3-dev.1" } + multinode: true + disagg: false + scenarios: + agentic-coding: + - search-space: + # Single-concurrency smoke test for the bring-up; widen the conc curve + # once the topology is proven green. + - spec-decoding: none + conc-list: [8] + prefill: + num-worker: 1 + tp: 8 + pp: 2 + ep: 1 + dp-attn: false + additional-settings: + - "CONFIG_FILE=recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml" + # The aggregate worker also performs decode; keep the decode worker + # count at zero so result aggregation counts the 16 GPUs only once. + decode: + num-worker: 0 + tp: 8 + pp: 2 + ep: 1 + dp-attn: false diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 0524d0c0c9..4980393654 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5116,3 +5116,15 @@ description: - "Bump image from lmsysorg/sglang:v0.5.14-rocm720-mi35x to lmsysorg/sglang-rocm:v0.5.16-rocm720-mi35x-20260726" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2349 + +- config-keys: + - kimik3-fp4-b200-dynamo-vllm-agentic + description: + - "Add Kimi-K3 MXFP4 B200 aggregated multinode Dynamo-vLLM agentic-coding bring-up (new model on B200; first kimik3 benchmark config)" + - "Aggregated TP8 x PP2 across 2 B200 nodes (16 GPUs), plain TP (NOT TEP: ep 1, no enable-expert-parallel) — the native MXFP4 checkpoint (2.8T total params, ~1.4TB weights) does not fit one 8xB200 node, so TP8 shards attention/dense and PP2 splits the 93 layers. Aggregated mode (prefill num-worker 1 + decode num-worker 0, RECIPES.md section 5): one worker serves prefill and decode, no P/D KV transfer" + - "Dedicated bring-up image vllm/vllm-openai:kimi-k3 with VLLM_ENABLE_K3_LATENT_MOE_TAIL_FUSION=1 (fuses the K3 LatentMoE tail path), --load-format fastsafetensors, --moe-backend auto, --gpu-memory-utilization 0.95, --no-enable-flashinfer-autotune, --trust-remote-code, and kimi_k3 parsers via the plain vLLM spellings --tool-call-parser kimi_k3 --reasoning-parser kimi_k3 (parser-flag experiment variant B; sibling agentic-experiment PRs try dynamo's --dyn-* namespaced args and no parser flags)" + - "No explicit max-model-len (vLLM derives the native 1M window from the model config; K3's KDA layers keep per-token KV small — only the 24 gated-MLA layers hold cache), prefix caching on for trajectory reuse, max-num-batched-tokens 8192 so a single long prefill cannot OOM a pipeline stage; single-concurrency smoke test at conc 8 (widen the curve once the topology is proven green)" + - "Dynamo hash-pinned to ba83080ecd31c1ce918559e576d3c5bc9e092ff1 ('feat: Added support for Kimi-K3', tag v1.4.0-kimi-k3-dev.1, published 2026-07-27) via srt-slurm's hash-cached source install. Required because dynamo 1.2.1's rust frontend tokenizer rejects Kimi-K3's tiktoken model_type kimi_k3 (only kimi/kimi_k2/kimi_k25/deepseek_v3 supported), so the model never registered with the frontend and every chat completion 404'd, aborting the AgentX warmup (third sweep attempt); the TP8xPP2 engine itself loaded and served (health-ready in ~14 min via fastsafetensors)" + - "Model pre-staged at /lustre/fsw/models/Kimi-K3 (moonshotai/Kimi-K3); launcher launch_b200-dgxc.sh gains the kimik3/fp4 model-path mapping, pins the agentic srt-slurm base to upstream NVIDIA/srt-slurm v1.0.36 (validated in #2302/#2341; replaces the cquil11/srt-slurm-nv fork branch whose older srtctl schema rejects newer recipe fields such as benchmark.aiperf_server_metrics), overlays the kimi-k3 agentic recipes onto the clone, and adds the agentic default_mounts (/aiperf_mmap_cache, /hf_hub_cache) already used by the GB200/GB300 agentic paths" + - "Recipe: benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-agentic.yaml on the cluster:b200-dgxc pool" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2357 diff --git a/runners/launch_b200-dgxc.sh b/runners/launch_b200-dgxc.sh index a276644575..1dd4e19d14 100644 --- a/runners/launch_b200-dgxc.sh +++ b/runners/launch_b200-dgxc.sh @@ -72,6 +72,10 @@ elif [[ $MODEL_PREFIX == "minimaxm3" && $PRECISION == "fp4" ]]; then # NVFP4 checkpoint, pre-staged on the b200-dgxc scratch tree. export MODEL_PATH="/scratch/fsw/models/MiniMax-M3-NVFP4" export SRT_SLURM_MODEL_PREFIX="minimax-m3-nvfp4" +elif [[ $MODEL_PREFIX == "kimik3" && $PRECISION == "fp4" ]]; then + # Native MXFP4 checkpoint, pre-staged on the SRE-managed Lustre tree. + export MODEL_PATH="/lustre/fsw/models/Kimi-K3" + export SRT_SLURM_MODEL_PREFIX="kimik3" else echo "Unsupported model prefix/precision: $MODEL_PREFIX/$PRECISION" echo "Available models under /lustre/fsw/models:" @@ -105,8 +109,22 @@ if [[ "$IS_MULTINODE" == "true" ]]; then # TODO(CJQ): make first class upon srt-slurm upstream refactor if [[ "$IS_AGENTIC" == "1" ]]; then - git clone --branch cam/sa-submission-q2-2026 --single-branch https://github.com/cquil11/srt-slurm-nv.git "$SRT_REPO_DIR" + # Agentic recipes use NVIDIA/srt-slurm v1.0.36, the upstream version + # validated in InferenceX PR #2302/#2341 for the vLLM agentic path + # (BenchmarkType.CUSTOM + benchmark.command/env, DynamoConfig.wheel, + # srun_options propagation, per-node DP, matching Dynamo health + # counts). Keep it pinned so sweeps are reproducible. Note the older + # cquil11/srt-slurm-nv cam/sa-submission-q2-2026 fork previously + # cloned here rejects newer recipe schema fields. + git clone --branch v1.0.36 --single-branch https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" || exit 1 cd "$SRT_REPO_DIR" || exit 1 + # Overlay InferenceX-staged agentic recipes onto the clone (cp -rT so + # an upstream stub directory is merged rather than nested). + if [[ $MODEL_PREFIX == "kimik3" ]]; then + mkdir -p recipes/vllm/kimi-k3/agentic || exit 1 + cp -rT "$GITHUB_WORKSPACE/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic" \ + recipes/vllm/kimi-k3/agentic || exit 1 + fi elif [[ $FRAMEWORK == "dynamo-vllm" && $MODEL_PREFIX == "dsv4" ]]; then git clone https://github.com/NVIDIA/srt-slurm.git "$SRT_REPO_DIR" cd "$SRT_REPO_DIR" || exit 1 @@ -207,6 +225,22 @@ if [[ "$IS_MULTINODE" == "true" ]]; then export OSL="$OSL" export EVAL_ONLY="${EVAL_ONLY:-false}" + # Agentic runs bind-mount two persistent caches into every worker + # container (Lustre, shared across nodes): aiperf's content-addressed + # dataset mmap cache and the HF hub cache holding the trace dataset + # download. The container-side paths are referenced by the agentic + # recipes' benchmark.env (AIPERF_DATASET_MMAP_CACHE_DIR=/aiperf_mmap_cache, + # HF_HUB_CACHE=/hf_hub_cache). + DEFAULT_MOUNTS_BLOCK="" + if [[ "$IS_AGENTIC" == "1" ]]; then + HF_HUB_CACHE_HOST_PATH="/lustre/fsw/gharunners/hf-hub-cache" + mkdir -p "$AIPERF_MMAP_CACHE_HOST_PATH" "$HF_HUB_CACHE_HOST_PATH" + chmod 777 "$AIPERF_MMAP_CACHE_HOST_PATH" "$HF_HUB_CACHE_HOST_PATH" 2>/dev/null || true + DEFAULT_MOUNTS_BLOCK="default_mounts: + ${AIPERF_MMAP_CACHE_HOST_PATH}: /aiperf_mmap_cache + ${HF_HUB_CACHE_HOST_PATH}: /hf_hub_cache" + fi + # Create srtslurm.yaml for srtctl (used by both frameworks) SRTCTL_ROOT="${GITHUB_WORKSPACE}/${SRT_REPO_DIR}" echo "Creating srtslurm.yaml configuration..." @@ -234,6 +268,7 @@ containers: "${IMAGE}": "${SQUASH_FILE}" nginx-sqsh: "${NGINX_SQUASH_FILE}" use_exclusive_sbatch_directive: true +${DEFAULT_MOUNTS_BLOCK} EOF echo "Generated srtslurm.yaml:"