diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-agentic.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-agentic.yaml index 7dc3d4979..0c6bb858b 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-agentic.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-agentic.yaml @@ -95,12 +95,12 @@ srun_options: benchmark: type: custom - aiperf_server_metrics: true command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh env: INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" + AIPERF_SERVER_METRICS_URLS: "http://localhost:8000/metrics" AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" RESULT_DIR: "/logs/agentic" PORT: "8000" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-eval-agentic.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-eval-agentic.yaml index f56268252..05bf13002 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-eval-agentic.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-eval-agentic.yaml @@ -93,12 +93,12 @@ srun_options: benchmark: type: custom - aiperf_server_metrics: true command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh env: INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" + AIPERF_SERVER_METRICS_URLS: "http://localhost:8000/metrics" AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" RESULT_DIR: "/logs/agentic" PORT: "8000" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-vllm-simple-offload-dspark-agentic.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-vllm-simple-offload-dspark-agentic.yaml index 0119d7921..97d4d71af 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-vllm-simple-offload-dspark-agentic.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-vllm-simple-offload-dspark-agentic.yaml @@ -96,12 +96,12 @@ srun_options: benchmark: type: custom - aiperf_server_metrics: true command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh env: INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" + AIPERF_SERVER_METRICS_URLS: "http://localhost:8000/metrics" AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" RESULT_DIR: "/logs/agentic" PORT: "8000" diff --git a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-vllm-simple-offload-dspark-eval-agentic.yaml b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-vllm-simple-offload-dspark-eval-agentic.yaml index a3ae8342b..0d50ff7ec 100644 --- a/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-vllm-simple-offload-dspark-eval-agentic.yaml +++ b/benchmarks/multi_node/srt-slurm-recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-vllm-simple-offload-dspark-eval-agentic.yaml @@ -94,12 +94,12 @@ srun_options: benchmark: type: custom - aiperf_server_metrics: true command: bash /infmax-workspace/benchmarks/multi_node/agentic_srt.sh env: INFMAX_CONTAINER_WORKSPACE: "/infmax-workspace" AIPERF_TRACE_IDLE_GAP_CAP_SECONDS: "300" AIPERF_LIVE_FAILED_REQUEST_THRESHOLD: "0.25" + AIPERF_SERVER_METRICS_URLS: "http://localhost:8000/metrics" AIPERF_REQUIRED_SERVER_METRIC_PREFIX: "vllm:" RESULT_DIR: "/logs/agentic" PORT: "8000" diff --git a/configs/nvidia-master.yaml b/configs/nvidia-master.yaml index c752b04f6..596bb8205 100644 --- a/configs/nvidia-master.yaml +++ b/configs/nvidia-master.yaml @@ -8051,9 +8051,9 @@ kimik3-fp4-b200-dynamo-vllm-agentic-dspark: agentic-coding: - dram-utilization: 0.63 search-space: - # Low-latency and high-interactivity points. + # Retained resident latency curve from the completed broad fast sweep. - spec-decoding: mtp - conc-list: [1, 2, 4] + conc-list: [2, 4, 8, 12] prefill: num-worker: 1 tp: 8 @@ -8067,44 +8067,12 @@ kimik3-fp4-b200-dynamo-vllm-agentic-dspark: tp: 8 ep: 16 dp-attn: true - # Balanced medium-concurrency points. - - spec-decoding: mtp - conc-list: [8, 12, 16] - prefill: - num-worker: 1 - tp: 8 - ep: 16 - dp-attn: true - additional-settings: - - "CONFIG_FILE=recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-agentic.yaml" - - "EVAL_CONFIG_FILE=recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-eval-agentic.yaml" - decode: - num-worker: 0 - tp: 8 - ep: 16 - dp-attn: true - # GPU-resident throughput points around the prior c16-c32 KV cliff. - - spec-decoding: mtp - conc-list: [20, 24, 28, 32] - prefill: - num-worker: 1 - tp: 8 - ep: 16 - dp-attn: true - additional-settings: - - "CONFIG_FILE=recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-agentic.yaml" - - "EVAL_CONFIG_FILE=recipes/vllm/kimi-k3/agentic/agg-b200-tp8dp2-latency-dspark-eval-agentic.yaml" - decode: - num-worker: 0 - tp: 8 - ep: 16 - dp-attn: true - # CPU KV-offload crossover and capacity points. Keep the resident points - # above so the same concurrency can be compared with one variable changed. + # Retain the measured SimpleCPUOffloadConnector knee, immediate boundary, + # and high-capacity controls from the completed broad fast sweep. - spec-decoding: mtp kv-offloading: dram kv-offload-backend: { name: vllm-simple, version: "13c59a3" } - conc-list: [8, 12, 16, 20, 24, 28, 32, 48, 64] + conc-list: [8, 12, 16, 28, 32, 48] prefill: num-worker: 1 tp: 8 diff --git a/perf-changelog.yaml b/perf-changelog.yaml index 0bce8d894..7ba903ef4 100644 --- a/perf-changelog.yaml +++ b/perf-changelog.yaml @@ -5818,3 +5818,10 @@ description: - "Extend the SimpleCPUOffloadConnector grid to c8/c12/c16/c20/c24/c28/c32/c48/c64 to locate its crossover against the resident curve" pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2475 + +- config-keys: + - kimik3-fp4-b200-dynamo-vllm-agentic-dspark + description: + - "Fix AIPerf server-metrics configuration for the pinned Kimi K3 srt-slurm renderer by passing the aggregate vLLM endpoint through the supported custom-benchmark environment contract" + - "After the complete 19-point AgentX-fast search, retain resident c2/c4/c8/c12 and SimpleCPUOffloadConnector c8/c12/c16/c28/c32/c48 for the full-duration latency curve, knee, cliff boundary, and high-capacity controls" + pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2569