group: LM Eval
depends_on: 
  - image-build
steps:
- label: ":nvidia: (H200 MIG 35GB) LM Eval Small Models"
  device: h200_35gb
  key: lm-eval-small-models
  timeout_in_minutes: 45
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  autorun_on_main: true
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small.txt
  mirror:
    amd:
      label: ":amd: (MI300) LM Eval Small Models"
      dind: false
      device: mi300_1
      timeout_in_minutes: 45
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - csrc/
      - vllm/model_executor/layers/quantization
      - vllm/model_executor/models/
      - vllm/model_executor/model_loader/
      - vllm/v1/attention/backends/
      - vllm/v1/attention/selector.py
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py

- label: ":amd: (MI300) LM Eval Small Models Harness"
  key: amd-lm-eval-small-models-harness
  dind: false
  device: mi300_1
  timeout_in_minutes: 35
  depends_on:
  - image-build-amd
  working_dir: /vllm-workspace/.buildkite/lm-eval-harness
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - .buildkite/lm-eval-harness/
  commands:
  - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-small-rocm.txt

- label: ":nvidia: (H200 MIG 35GB) LM Eval Watermarking"
  device: h200_35gb
  key: lm-eval-watermarking
  # Recorded runs took 80–140 seconds
  timeout_in_minutes: 10
  source_file_dependencies:
  - .buildkite/test_areas/lm_eval.yaml
  - tests/evals/gsm8k/
  - vllm/config/watermarking.py
  - vllm/sampling_params.py
  - vllm/v1/watermarking/
  - vllm/v1/worker/gpu/
  autorun_on_main: true
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-watermark.txt
  mirror:
    amd:
      label: ":amd: (MI355 DPX) LM Eval Watermarking"
      dind: false
      device: mi355_dpx
      num_devices: 1
      timeout_in_minutes: 45
      working_dir: /vllm-workspace/tests
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - .buildkite/test_areas/lm_eval.yaml
      - tests/evals/gsm8k/
      - vllm/config/watermarking.py
      - vllm/sampling_params.py
      - vllm/v1/watermarking/
      - vllm/v1/worker/gpu/
      - vllm/platforms/rocm.py
      env:
        VLLM_WORKER_MULTIPROC_METHOD: "spawn"
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-watermark.txt

# - label: LM Eval Large Models (4xA100)
#   key: lm-eval-large-models-4xa100
#   device: a100
#   optional: true
#   num_devices: 4
#   working_dir: "/vllm-workspace/.buildkite/lm-eval-harness"
#   source_file_dependencies:
#   - csrc/
#   - vllm/model_executor/layers/quantization
#   commands:
#   - export VLLM_WORKER_MULTIPROC_METHOD=spawn
#   - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large.txt --tp-size=4

- label: ":nvidia: (H100) LM Eval Large Models"
  key: lm-eval-large-models-4xh100
  device: h100
  optional: true
  num_devices: 4
  working_dir: "/vllm-workspace/.buildkite/lm-eval-harness"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  commands:
    - export VLLM_USE_DEEP_GEMM=0  # We found Triton is faster than DeepGEMM for H100
    - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large-hopper.txt --tp-size=4
  mirror:
    amd:
      label: ":amd: (MI355) LM Eval Large Models FP8"
      dind: false
      device: mi355_4
      timeout_in_minutes: 45
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - csrc/
      - vllm/model_executor/layers/quantization
      - vllm/model_executor/models/
      - vllm/model_executor/model_loader/
      - vllm/v1/attention/backends/
      - vllm/v1/attention/selector.py
      - .buildkite/lm-eval-harness/
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      commands:
      - export VLLM_USE_DEEP_GEMM=0
      - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large-rocm-tp4.txt --tp-size=4

- label: ":nvidia: (B200) LM Eval Small Models"
  key: lm-eval-small-models-1xb200
  timeout_in_minutes: 50
  device: b200-k8s
  optional: true
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-blackwell.txt
  mirror:
    amd:
      label: ":amd: (MI355 DPX) LM Eval Small Models"
      dind: false
      device: mi355_dpx
      num_devices: 1
      timeout_in_minutes: 45
      working_dir: /vllm-workspace/tests
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - .buildkite/test_areas/lm_eval.yaml
      - csrc/
      - tests/evals/gsm8k/
      - vllm/model_executor/layers/quantization
      - vllm/model_executor/kernels/linear/
      - vllm/model_executor/layers/fused_moe/
      - vllm/model_executor/models/
      - vllm/model_executor/model_loader/
      - vllm/v1/attention/backends/
      - vllm/v1/attention/selector.py
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-gfx950-small.txt

- label: ":nvidia: (B200) LM Eval Small Models Distributed"
  key: lm-eval-small-models-distributed-2xb200
  timeout_in_minutes: 120
  device: b200-k8s
  num_devices: 2
  optional: true
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  autorun_on_main: true
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small-tp.txt
  mirror:
    amd:
      label: ":amd: (MI355) LM Eval Small Models Distributed"
      dind: false
      device: mi355_2
      timeout_in_minutes: 145
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - csrc/
      - vllm/model_executor/layers/quantization
      - vllm/model_executor/models/
      - vllm/model_executor/model_loader/
      - vllm/v1/attention/backends/
      - vllm/v1/attention/selector.py
      - vllm/v1/worker/
      - vllm/v1/core/
      - vllm/config/
      - tests/evals/gsm8k/configs/models-mi3xx-fp8-and-mixed.txt
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      - tests/evals/gsm8k/
      - tests/utils.py
      - vllm/distributed/
      - vllm/transformers_utils/configs/diffusion_gemma.py
      - vllm/v1/attention/ops/
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-mi3xx-fp8-and-mixed.txt
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small-tp.txt
      # Dense MLA PCP-only accuracy; DP1/DCP1.
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small-tp-mi355.txt

- label: ":nvidia: (B200) LM Eval HiSparse Nightly"
  key: lm-eval-hisparse-nightly-4xb200
  timeout_in_minutes: 60
  device: b200-k8s
  num_devices: 4
  optional: true
  source_file_dependencies:
  - .buildkite/test_areas/lm_eval.yaml
  - tests/evals/gsm8k/
  - vllm/v1/hisparse/
  - vllm/distributed/kv_transfer/kv_connector/v1/hisparse/
  - vllm/v1/attention/backends/mla/
  - vllm/model_executor/layers/attention/sparse_mla_attention.py
  - vllm/v1/worker/gpu/
  - csrc/libtorch_stable/hisparse_kernels.cu
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-hisparse.txt

- label: ":nvidia: (B200) LM Eval PCP"
  key: lm-eval-pcp-4xb200
  timeout_in_minutes: 360
  device: b200-k8s
  num_devices: 4
  optional: true
  source_file_dependencies:
  - csrc/
  - tests/evals/gsm8k/configs/GLM-5.2-NVFP4-TP2-PCP2-EP.yaml
  - tests/evals/gsm8k/configs/GLM-5.2-NVFP4-TP1-PCP4-EP.yaml
  - tests/evals/gsm8k/configs/GLM-5.2-NVFP4-TP1-PCP4-DCP4-EP.yaml
  - tests/evals/gsm8k/configs/models-pcp.txt
  - vllm/model_executor/layers/quantization
  - vllm/config/parallel.py
  - vllm/distributed/parallel_state.py
  - vllm/model_executor/layers/attention/mla_attention.py
  - vllm/model_executor/layers/attention/pcp.py
  - vllm/model_executor/layers/attention/sparse_mla_attention.py
  - vllm/model_executor/layers/sparse_attn_indexer.py
  - vllm/v1/attention/backends/mla/flashmla_sparse.py
  - vllm/v1/attention/backends/mla/indexer.py
  - vllm/v1/worker/gpu/model_runner.py
  - vllm/v1/worker/gpu/pcp_manager.py
  autorun_on_main: true
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-pcp.txt

- label: ":nvidia: (H100) LM Eval Watermark Feature Combination"
  key: lm-eval-dspark-watermark-2xh100
  device: h100
  num_devices: 2
  # Local TP2 runs took 84–379 seconds, including startup.
  timeout_in_minutes: 15
  source_file_dependencies:
  - .buildkite/test_areas/lm_eval.yaml
  - tests/evals/gsm8k/
  - vllm/config/speculative.py
  - vllm/config/vllm.py
  - vllm/config/watermarking.py
  - vllm/sampling_params.py
  - vllm/model_executor/models/qwen3_dspark.py
  - vllm/model_executor/models/qwen3_dflash.py
  - vllm/model_executor/warmup/
  - vllm/v1/watermarking/
  - vllm/v1/worker/gpu/
  autorun_on_main: true
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-watermark-dspark-tp2.txt
  mirror:
    amd:
      label: ":amd: (MI355) LM Eval Watermark Feature Combination"
      dind: false
      device: mi355_2
      num_devices: 2
      timeout_in_minutes: 60
      working_dir: /vllm-workspace/tests
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - .buildkite/test_areas/lm_eval.yaml
      - tests/evals/gsm8k/
      - vllm/config/speculative.py
      - vllm/config/vllm.py
      - vllm/config/watermarking.py
      - vllm/sampling_params.py
      - vllm/model_executor/models/qwen3_dspark.py
      - vllm/model_executor/models/qwen3_dflash.py
      - vllm/model_executor/warmup/
      - vllm/v1/watermarking/
      - vllm/v1/worker/gpu/
      - vllm/platforms/rocm.py
      env:
        VLLM_WORKER_MULTIPROC_METHOD: "spawn"
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-watermark-dspark-tp2.txt

- label: ":nvidia: (B200) LM Eval Spec Decode"
  key: lm-eval-spec-decode-4xb200
  timeout_in_minutes: 120
  device: b200-k8s
  num_devices: 4
  optional: true
  source_file_dependencies:
  - tests/evals/gsm8k/configs/DeepSeek-V4-Flash-DSpark-confidence-TEP4.yaml
  - tests/evals/gsm8k/configs/Kimi-K3-pruned75-DSpark-TP4.yaml
  - tests/evals/gsm8k/configs/models-spec-decode.txt
  - vllm/models/deepseek_v4/
  - vllm/models/kimi_k3/
  - vllm/model_executor/models/qwen3_dspark.py
  - vllm/v1/spec_decode/metrics.py
  - vllm/v1/worker/gpu/spec_decode/adaptive_verification.py
  - vllm/v1/worker/gpu/spec_decode/dflash/
  - vllm/v1/worker/gpu/spec_decode/dspark/
  - vllm/v1/worker/gpu/spec_decode/rejection_sampler.py
  - vllm/v1/worker/gpu/spec_decode/speculator.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-spec-decode.txt
  mirror:
    amd:
      label: ":amd: (MI355) LM Eval Spec Decode (Fixed-length FP8)"
      dind: false
      device: mi355_4
      timeout_in_minutes: 145
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - csrc/
      - tests/evals/gsm8k/configs/Kimi-K3-pruned75-DSpark-AITER-TP4.yaml
      - tests/evals/gsm8k/configs/DeepSeek-V4-Flash-DSpark-AITER-TEP4.yaml
      - tests/evals/gsm8k/configs/DeepSeek-V4-Flash-DSpark-FP8-TP4-ROCm.yaml
      - tests/evals/gsm8k/configs/models-spec-decode-rocm.txt
      - tests/evals/gsm8k/test_gsm8k_correctness.py
      - tests/evals/gsm8k/gsm8k_eval.py
      - tests/evals/gsm8k/conftest.py
      - vllm/config/
      - vllm/envs.py
      - vllm/models/kimi_k3/
      - vllm/models/deepseek_v4/
      - vllm/model_executor/kernels/mhc/
      - vllm/model_executor/layers/mhc.py
      - vllm/model_executor/layers/quantization/
      - vllm/model_executor/layers/sparse_attn_indexer.py
      - vllm/model_executor/models/config.py
      - vllm/model_executor/models/registry.py
      - vllm/model_executor/models/qwen3_dspark.py
      - vllm/model_executor/model_loader/
      - vllm/model_executor/warmup/
      - vllm/model_executor/layers/fused_moe/
      - vllm/tokenizers/
      - vllm/v1/attention/
      - vllm/v1/core/kv_cache_utils.py
      - vllm/v1/spec_decode/metrics.py
      - vllm/v1/worker/
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      - tests/evals/gsm8k/
      - tests/utils.py
      - vllm/distributed/
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-spec-decode-rocm.txt

- label: ":nvidia: (B200) LM Eval Large Models EP"
  key: lm-eval-large-models-ep-2xb200
  timeout_in_minutes: 60
  device: b200-k8s
  optional: true
  num_devices: 2
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-blackwell-ep.txt
  mirror:
    amd:
      label: ":amd: (MI355) LM Eval Large Models EP"
      dind: false
      device: mi355_2
      num_devices: 2
      timeout_in_minutes: 120
      working_dir: "/vllm-workspace/tests"
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - .buildkite/test_areas/lm_eval.yaml
      - tests/evals/gsm8k/
      - tests/utils.py
      - tests/conftest.py
      - csrc/
      - vllm/config/
      - vllm/distributed/
      - vllm/model_executor/models/config.py
      - vllm/model_executor/models/registry.py
      - vllm/model_executor/models/utils.py
      - vllm/model_executor/models/qwen3_next.py
      - vllm/model_executor/models/qwen3_next_mtp.py
      - vllm/model_executor/models/nemotron_h.py
      - vllm/model_executor/models/nemotron_h_mtp.py
      - vllm/transformers_utils/configs/qwen3_next.py
      - vllm/transformers_utils/configs/nemotron_h.py
      - vllm/model_executor/model_loader/
      - vllm/model_executor/layers/quantization/
      - vllm/model_executor/layers/fused_moe/
      - vllm/model_executor/layers/mamba/
      - vllm/model_executor/layers/layernorm.py
      - vllm/model_executor/layers/linear.py
      - vllm/model_executor/kernels/linear/
      - vllm/third_party/flash_linear_attention/ops/
      - vllm/v1/attention/
      - vllm/v1/spec_decode/
      - vllm/v1/worker/
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      commands:
      - export VLLM_WORKER_MULTIPROC_METHOD=spawn
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-blackwell-ep-mi355.txt

- label: ":nvidia: (B200) LM Eval Qwen3.5 Models"
  key: lm-eval-qwen3-5-models-2xb200
  timeout_in_minutes: 45
  device: b200-k8s
  optional: true
  num_devices: 2
  source_file_dependencies:
  - vllm/model_executor/models/qwen3_5.py
  - vllm/model_executor/models/qwen3_5_mtp.py
  - vllm/transformers_utils/configs/qwen3_5.py
  - vllm/transformers_utils/configs/qwen3_5_moe.py
  - vllm/model_executor/models/qwen3_next.py
  - vllm/model_executor/models/qwen3_next_mtp.py
  - vllm/third_party/flash_linear_attention/ops/
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-qwen35-blackwell.txt
  mirror:
    amd:
      label: ":amd: (MI355) LM Eval Qwen3.5 Models"
      dind: false
      device: mi355_2
      timeout_in_minutes: 75
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - vllm/model_executor/models/qwen3_5.py
      - vllm/model_executor/models/qwen3_5_mtp.py
      - vllm/transformers_utils/configs/qwen3_5.py
      - vllm/transformers_utils/configs/qwen3_5_moe.py
      - vllm/model_executor/models/qwen2.py
      - vllm/model_executor/models/qwen3.py
      - vllm/model_executor/models/qwen3_next.py
      - vllm/model_executor/models/qwen3_next_mtp.py
      - vllm/third_party/flash_linear_attention/ops/
      - vllm/distributed/
      - vllm/model_executor/layers/fused_moe/
      - vllm/model_executor/layers/quantization/modelopt.py
      - vllm/model_executor/kernels/linear/nvfp4/
      - vllm/v1/attention/
      - vllm/v1/worker/
      - tests/evals/gsm8k/configs/models-qwen35-mi355.txt
      - tests/evals/gsm8k/configs/Qwen3.5-35B-A3B-DEP2.yaml
      - tests/evals/gsm8k/configs/Qwen3.5-35B-A3B-FP8-DEP2.yaml
      - tests/evals/gsm8k/configs/Qwen3.5-35B-A3B-MXFP4-AITER-TP2.yaml
      - tests/evals/gsm8k/configs/Qwen3.5-35B-A3B-MXFP4-EMU-TP2.yaml
      - tests/evals/gsm8k/configs/Qwen3.5-397B-A17B-NVFP4-DEP2.yaml
      - tests/evals/gsm8k/test_gsm8k_correctness.py
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-qwen35-mi355.txt

- label: ":nvidia: (H200) LM Eval DeepSeek R1 TP"
  key: lm-eval-deepseek-r1-tp-8xh200
  timeout_in_minutes: 35
  device: h200
  optional: true
  num_devices: 8
  commands:
    - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-h200-deepseek-r1-tp.txt

- label: ":nvidia: (H200) LM Eval DeepSeek R1 DP"
  key: lm-eval-deepseek-r1-dp-8xh200
  timeout_in_minutes: 35
  device: h200
  optional: true
  num_devices: 8
  commands:
    - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-h200-deepseek-r1-dp.txt

- label: ":nvidia: (H200) LM Eval DeepSeek V3.2 TP"
  key: lm-eval-deepseek-v32-tp-8xh200
  timeout_in_minutes: 35
  device: h200
  optional: true
  num_devices: 8
  commands:
    - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-h200-deepseek-v32-tp.txt

- label: ":nvidia: (H200) LM Eval DeepSeek V3.2 DP"
  key: lm-eval-deepseek-v32-dp-8xh200
  timeout_in_minutes: 35
  device: h200
  optional: true
  num_devices: 8
  commands:
    - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-h200-deepseek-v32-dp.txt

- label: ":nvidia: (H200) LM Eval Nemotron 3 Super"
  key: lm-eval-nemotron-3-super-8xh200
  timeout_in_minutes: 35
  device: h200
  optional: true
  num_devices: 8
  commands:
    - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-h200-nemotron-3-super.txt

- label: ":amd: (MI300) LM Eval Large Models"
  key: amd-lm-eval-large-models
  dind: false
  device: mi300_8
  optional: true
  timeout_in_minutes: 65
  depends_on:
  - image-build-amd
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - export PYTORCH_ROCM_ARCH=gfx942 # Limit Quark compilation to save time
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-mi3xx.txt
  source_file_dependencies:
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py

- label: ":nvidia: (H100) MoE Refactor Integration TEMPORARY"
  key: moe-refactor-integration-test-h100-temporary
  device: h100
  optional: true
  num_devices: 2
  parallelism: 4
  commands:
    # Timing-balanced static split of the former config-h100.txt (measured on
    # build 83851): shards ~12.4/13.6/12.4/14.5m of eval time. New configs must
    # be added to exactly one config-h100-shard-*.txt.
    - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/moe-refactor/config-h100-shard-$$BUILDKITE_PARALLEL_JOB.txt
  mirror:
    amd:
      label: ":amd: (MI355) MoE Refactor Integration Shard %N"
      dind: false
      device: mi355_2
      num_devices: 2
      timeout_in_minutes: 105
      working_dir: /vllm-workspace/tests
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - vllm/platforms/rocm.py
      - tests/evals/gsm8k/
      - vllm/model_executor/layers/fused_moe/
      - vllm/model_executor/layers/quantization/
      - vllm/_aiter_ops.py
      env:
        VLLM_WORKER_MULTIPROC_METHOD: "spawn"
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/moe-refactor/config-rocm-shard-$$BUILDKITE_PARALLEL_JOB.txt

- label: ":nvidia: (B200) MoE Refactor Integration TEMPORARY Shard %N"
  key: moe-refactor-integration-test-b200-temporary
  device: b200-k8s
  optional: true
  num_devices: 2
  parallelism: 4
  commands:
    - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/moe-refactor/config-b200-shard-$$BUILDKITE_PARALLEL_JOB.txt

- label: ":nvidia: (B200) MoE Refactor DP Integration TEMPORARY"
  key: moe-refactor-integration-test-b200-dp-temporary
  device: b200-k8s
  optional: true
  num_devices: 2
  commands:
    - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/moe-refactor-dp-ep/config-b200.txt

- label: ":nvidia: (L4) LM Eval Humming TEMPORARY"
  key: lm-eval-humming-l4
  optional: true
  timeout_in_minutes: 90
  device: l4
  num_devices: 1
  source_file_dependencies:
  - .buildkite/test_areas/lm_eval.yaml
  - tests/evals/gsm8k/configs/humming/
  - requirements/cuda.txt
  - vllm/utils/humming.py
  - vllm/model_executor/layers/quantization/humming.py
  - vllm/model_executor/layers/quantization/utils/humming/
  - vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
  - vllm/model_executor/layers/fused_moe/oracle/
  - vllm/model_executor/kernels/linear/
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-l4.txt

- label: ":nvidia: (A100) LM Eval Humming Default TEMPORARY Shard %N"
  key: lm-eval-humming-a100-default
  optional: true
  timeout_in_minutes: 75
  device: a100
  num_devices: 1
  parallelism: 3
  source_file_dependencies:
  - vllm/model_executor/layers/quantization/humming.py
  - vllm/model_executor/layers/quantization/utils/humming/
  - vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
  - vllm/model_executor/layers/fused_moe/oracle/
  - vllm/model_executor/kernels/linear/
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-a100-shard-$$BUILDKITE_PARALLEL_JOB.txt

- label: ":nvidia: (A100) LM Eval Humming Activation INT8 TEMPORARY Shard %N"
  key: lm-eval-humming-a100-act
  optional: true
  timeout_in_minutes: 45
  device: a100
  num_devices: 1
  parallelism: 2
  source_file_dependencies:
  - vllm/model_executor/layers/quantization/humming.py
  - vllm/model_executor/layers/quantization/utils/humming/
  - vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
  - vllm/model_executor/layers/fused_moe/oracle/
  - vllm/model_executor/kernels/linear/
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-a100-act-int8-shard-$$BUILDKITE_PARALLEL_JOB.txt

- label: ":nvidia: (H100) LM Eval Humming Default TEMPORARY Shard %N"
  key: lm-eval-humming-h100-default
  optional: true
  timeout_in_minutes: 70
  device: h100
  num_devices: 1
  parallelism: 3
  source_file_dependencies:
  - vllm/model_executor/layers/quantization/humming.py
  - vllm/model_executor/layers/quantization/utils/humming/
  - vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
  - vllm/model_executor/layers/fused_moe/oracle/
  - vllm/model_executor/kernels/linear/
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-h100-shard-$$BUILDKITE_PARALLEL_JOB.txt

- label: ":nvidia: (H100) LM Eval Humming Activation FP8/INT8 TEMPORARY"
  key: lm-eval-humming-h100-act
  optional: true
  timeout_in_minutes: 90
  device: h100
  num_devices: 1
  source_file_dependencies:
  - vllm/model_executor/layers/quantization/humming.py
  - vllm/model_executor/layers/quantization/utils/humming/
  - vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
  - vllm/model_executor/layers/fused_moe/oracle/
  - vllm/model_executor/kernels/linear/
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-act-fp8.txt
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-act-int8.txt

- label: ":nvidia: (H100) LM Eval Humming DeepEP TEMPORARY"
  key: lm-eval-humming-h100-deepep
  optional: true
  timeout_in_minutes: 60
  device: h100
  num_devices: 2
  source_file_dependencies:
  - .buildkite/test_areas/lm_eval.yaml
  - tests/evals/gsm8k/configs/humming/
  - requirements/cuda.txt
  - vllm/utils/humming.py
  - vllm/model_executor/layers/quantization/humming.py
  - vllm/model_executor/layers/quantization/utils/humming/
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/kernels/linear/
  - vllm/distributed/
  - csrc/moe/
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-deepep.txt

- label: ":nvidia: (B200) LM Eval Humming Default TEMPORARY"
  key: lm-eval-humming-b200-default
  optional: true
  timeout_in_minutes: 50
  device: b200-k8s
  num_devices: 1
  source_file_dependencies:
  - vllm/model_executor/layers/quantization/humming.py
  - vllm/model_executor/layers/quantization/utils/humming/
  - vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
  - vllm/model_executor/layers/fused_moe/oracle/
  - vllm/model_executor/kernels/linear/
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config.txt

- label: ":nvidia: (B200) LM Eval Humming Activation FP8/INT8 TEMPORARY"
  key: lm-eval-humming-b200-act
  optional: true
  timeout_in_minutes: 50
  device: b200-k8s
  num_devices: 1
  source_file_dependencies:
  - vllm/model_executor/layers/quantization/humming.py
  - vllm/model_executor/layers/quantization/utils/humming/
  - vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
  - vllm/model_executor/layers/fused_moe/oracle/
  - vllm/model_executor/kernels/linear/
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-act-fp8.txt
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-act-int8.txt

- label: ":nvidia: (H200 MIG 18GB) LM Eval TurboQuant k8v4"
  key: lm-eval-turboquant-k8v4
  timeout_in_minutes: 30
  device: h200_18gb
  source_file_dependencies:
  - vllm/model_executor/layers/quantization/turboquant/
  - vllm/v1/attention/backends/turboquant_attn.py
  - vllm/v1/attention/ops/triton_turboquant_decode.py
  - vllm/v1/attention/ops/triton_turboquant_store.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-k8v4.txt
  mirror:
    amd:
      label: ":amd: (MI355 DPX) LM Eval TurboQuant k8v4"
      dind: false
      device: mi355_dpx
      num_devices: 1
      timeout_in_minutes: 30
      working_dir: /vllm-workspace/tests
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - .buildkite/test_areas/lm_eval.yaml
      - tests/evals/gsm8k/
      - tests/utils.py
      - tests/conftest.py
      - vllm/model_executor/layers/quantization/turboquant/
      - vllm/v1/attention/backends/turboquant_attn.py
      - vllm/v1/attention/backends/triton_attn.py
      - vllm/v1/attention/ops/triton_unified_attention.py
      - vllm/v1/attention/selector.py
      - vllm/v1/attention/ops/triton_turboquant_decode.py
      - vllm/v1/attention/ops/triton_turboquant_store.py
      - vllm/v1/attention/ops/flydsl_turboquant_decode.py
      - vllm/v1/attention/ops/flydsl_kernels/
      - vllm/v1/attention/ops/turboquant_soa/
      - vllm/platforms/rocm.py
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-k8v4-rocm.txt

- label: ":nvidia: (H200 MIG 18GB) LM Eval TurboQuant t4nc"
  key: lm-eval-turboquant-t4nc
  timeout_in_minutes: 30
  device: h200_18gb
  source_file_dependencies:
  - vllm/model_executor/layers/quantization/turboquant/
  - vllm/v1/attention/backends/turboquant_attn.py
  - vllm/v1/attention/ops/triton_turboquant_decode.py
  - vllm/v1/attention/ops/triton_turboquant_store.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-t4nc.txt
  mirror:
    amd:
      label: ":amd: (MI355 DPX) LM Eval TurboQuant t4nc"
      dind: false
      device: mi355_dpx
      num_devices: 1
      timeout_in_minutes: 30
      working_dir: /vllm-workspace/tests
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - .buildkite/test_areas/lm_eval.yaml
      - tests/evals/gsm8k/
      - tests/utils.py
      - tests/conftest.py
      - vllm/model_executor/layers/quantization/turboquant/
      - vllm/v1/attention/backends/turboquant_attn.py
      - vllm/v1/attention/backends/triton_attn.py
      - vllm/v1/attention/ops/triton_unified_attention.py
      - vllm/v1/attention/selector.py
      - vllm/v1/attention/ops/triton_turboquant_decode.py
      - vllm/v1/attention/ops/triton_turboquant_store.py
      - vllm/v1/attention/ops/flydsl_turboquant_decode.py
      - vllm/v1/attention/ops/flydsl_kernels/
      - vllm/v1/attention/ops/turboquant_soa/
      - vllm/platforms/rocm.py
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-t4nc-rocm.txt

- label: ":nvidia: (H200 MIG 18GB) LM Eval TurboQuant k3v4nc"
  key: lm-eval-turboquant-k3v4nc
  timeout_in_minutes: 30
  device: h200_18gb
  source_file_dependencies:
  - vllm/model_executor/layers/quantization/turboquant/
  - vllm/v1/attention/backends/turboquant_attn.py
  - vllm/v1/attention/ops/triton_turboquant_decode.py
  - vllm/v1/attention/ops/triton_turboquant_store.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-k3v4nc.txt
  mirror:
    amd:
      label: ":amd: (MI355 DPX) LM Eval TurboQuant k3v4nc"
      dind: false
      device: mi355_dpx
      num_devices: 1
      timeout_in_minutes: 30
      working_dir: /vllm-workspace/tests
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - .buildkite/test_areas/lm_eval.yaml
      - tests/evals/gsm8k/
      - tests/utils.py
      - tests/conftest.py
      - vllm/model_executor/layers/quantization/turboquant/
      - vllm/v1/attention/backends/turboquant_attn.py
      - vllm/v1/attention/backends/triton_attn.py
      - vllm/v1/attention/ops/triton_unified_attention.py
      - vllm/v1/attention/selector.py
      - vllm/v1/attention/ops/triton_turboquant_decode.py
      - vllm/v1/attention/ops/triton_turboquant_store.py
      - vllm/v1/attention/ops/flydsl_turboquant_decode.py
      - vllm/v1/attention/ops/flydsl_kernels/
      - vllm/v1/attention/ops/turboquant_soa/
      - vllm/platforms/rocm.py
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-k3v4nc-rocm.txt

- label: ":nvidia: (H200 MIG 18GB) LM Eval TurboQuant t3nc"
  key: lm-eval-turboquant-t3nc
  timeout_in_minutes: 30
  device: h200_18gb
  source_file_dependencies:
  - vllm/model_executor/layers/quantization/turboquant/
  - vllm/v1/attention/backends/turboquant_attn.py
  - vllm/v1/attention/ops/triton_turboquant_decode.py
  - vllm/v1/attention/ops/triton_turboquant_store.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-t3nc.txt
  mirror:
    amd:
      label: ":amd: (MI355 DPX) LM Eval TurboQuant t3nc"
      dind: false
      device: mi355_dpx
      num_devices: 1
      timeout_in_minutes: 30
      working_dir: /vllm-workspace/tests
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - .buildkite/test_areas/lm_eval.yaml
      - tests/evals/gsm8k/
      - tests/utils.py
      - tests/conftest.py
      - vllm/model_executor/layers/quantization/turboquant/
      - vllm/v1/attention/backends/turboquant_attn.py
      - vllm/v1/attention/backends/triton_attn.py
      - vllm/v1/attention/ops/triton_unified_attention.py
      - vllm/v1/attention/selector.py
      - vllm/v1/attention/ops/triton_turboquant_decode.py
      - vllm/v1/attention/ops/triton_turboquant_store.py
      - vllm/v1/attention/ops/flydsl_turboquant_decode.py
      - vllm/v1/attention/ops/flydsl_kernels/
      - vllm/v1/attention/ops/turboquant_soa/
      - vllm/platforms/rocm.py
      commands:
      - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-t3nc-rocm.txt

- label: ":nvidia: (H100) GPQA Eval (GPT-OSS)"
  key: gpqa-eval-gpt-oss-2xh100
  timeout_in_minutes: 35
  device: h100
  optional: true
  num_devices: 2
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - tests/evals/gpt_oss/
  commands:
    - pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-h100.txt

- label: ":nvidia: (B200) GPQA Eval (GPT-OSS)"
  key: gpqa-eval-gpt-oss-2xb200
  timeout_in_minutes: 30
  device: b200-k8s
  optional: true
  num_devices: 2
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - tests/evals/gpt_oss/
  commands:
    - pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-b200.txt
  mirror:
    amd:
      label: ":amd: (MI355) GPQA Eval (GPT-OSS)"
      dind: false
      device: mi355_2
      timeout_in_minutes: 40
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - csrc/
      - vllm/model_executor/layers/quantization
      - vllm/model_executor/models/
      - vllm/model_executor/model_loader/
      - vllm/v1/attention/backends/
      - vllm/v1/attention/selector.py
      - vllm/model_executor/layers/fused_moe/
      - tests/evals/gpt_oss/
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      commands:
      - pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-gfx950.txt

- label: ":nvidia: (B200) Qwen3.8-Flash-Next-FP8 Accuracy Eval"
  key: accuracy-eval-qwen3-8-flash-next-fp8-b200
  timeout_in_minutes: 60
  device: b200-k8s
  optional: true
  num_devices: 4
  source_file_dependencies:
  - csrc/libtorch_stable/gdn/
  - tests/evals/qwen4_exp/
  - vllm/model_executor/layers/mamba/
  - vllm/models/qwen4_exp/
  - vllm/transformers_utils/configs/qwen4_exp.py
  - vllm/v1/attention/backends/short_conv_attn.py
  - vllm/v1/spec_decode/qwen4_exp.py
  commands:
    - uv pip install --system 'evalscope==1.10.0'
    - pytest -s -v evals/qwen4_exp/test_accuracy.py --config-list-file=configs/models-b200.txt

- label: ":nvidia: (H200) Qwen3.8-Flash-Next-FP8 Accuracy Eval"
  key: accuracy-eval-qwen3-8-flash-next-fp8-h200
  timeout_in_minutes: 60
  device: h200
  optional: true
  num_devices: 4
  source_file_dependencies:
  - csrc/libtorch_stable/gdn/
  - tests/evals/qwen4_exp/
  - vllm/model_executor/layers/mamba/
  - vllm/models/qwen4_exp/
  - vllm/transformers_utils/configs/qwen4_exp.py
  - vllm/v1/attention/backends/short_conv_attn.py
  - vllm/v1/spec_decode/qwen4_exp.py
  commands:
    - uv pip install --system 'evalscope==1.10.0'
    - pytest -s -v evals/qwen4_exp/test_accuracy.py --config-list-file=configs/models-h200.txt
  mirror:
    amd:
      label: ":amd: (MI355) Qwen3.8-Flash-Next-FP8 Accuracy Eval"
      dind: false
      device: mi355_4
      num_devices: 4
      timeout_in_minutes: 105
      working_dir: /vllm-workspace/tests
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - csrc/libtorch_stable/gdn/
      - tests/evals/qwen4_exp/
      - vllm/model_executor/layers/mamba/
      - vllm/models/qwen4_exp/
      - vllm/transformers_utils/configs/qwen4_exp.py
      - vllm/v1/attention/backends/short_conv_attn.py
      - vllm/platforms/rocm.py
      env:
        VLLM_WORKER_MULTIPROC_METHOD: "spawn"
      commands:
      - uv pip install --system 'evalscope==1.10.0'
      - pytest -s -v evals/qwen4_exp/test_accuracy.py --config-list-file=configs/models-rocm.txt

- label: ":nvidia: (DGX) Spark GPQA Eval (GPT-OSS)"
  key: gpqa-eval-gpt-oss-spark
  timeout_in_minutes: 35
  device: dgx-spark
  optional: true
  num_devices: 1
  depends_on:
  - arm64-image-build
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - tests/evals/gpt_oss/
  commands:
    - pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-spark.txt

- label: ":nvidia: (H200 MIG 35GB) LM Eval KV-Offload"
  key: kv-offload-small
  timeout_in_minutes: 30
  device: h200_35gb
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
  - vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
  - vllm/v1/kv_offload/
  - vllm/v1/simple_kv_offload/
  - tests/evals/gsm8k/test_gsm8k_offloading.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "nemotron-h-8b or gemma-4-e4b-it"
  mirror:
    amd:
      label: ":amd: (MI300) LM Eval KV-Offload"
      dind: false
      device: mi300_1
      timeout_in_minutes: 55
      depends_on:
      - image-build-amd
      working_dir: /vllm-workspace/tests
      source_file_dependencies:
      - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
      - vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
      - vllm/v1/kv_offload/
      - vllm/v1/simple_kv_offload/
      - tests/evals/gsm8k/test_gsm8k_offloading.py
      - vllm/platforms/rocm.py

- label: ":nvidia: (H100) LM Eval KV-Offload Medium"
  key: kv-offload-medium
  timeout_in_minutes: 45
  device: h100
  num_devices: 2
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
  - vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
  - vllm/v1/kv_offload/
  - vllm/v1/simple_kv_offload/
  - tests/evals/gsm8k/test_gsm8k_offloading.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "qwen3.5-35b or deepseek-v2-lite"
  mirror:
    amd:
      label: ":amd: (MI300) LM Eval KV-Offload Medium"
      dind: false
      device: mi300_2
      timeout_in_minutes: 70
      depends_on:
      - image-build-amd
      working_dir: /vllm-workspace/tests
      source_file_dependencies:
      - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
      - vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
      - vllm/v1/kv_offload/
      - vllm/v1/simple_kv_offload/
      - tests/evals/gsm8k/test_gsm8k_offloading.py
      - vllm/platforms/rocm.py

- label: ":nvidia: (H100) LM Eval KV-Offload Large"
  key: kv-offload-large
  timeout_in_minutes: 40
  device: h100
  num_devices: 4
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
  - vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
  - vllm/v1/kv_offload/
  - vllm/v1/simple_kv_offload/
  - tests/evals/gsm8k/test_gsm8k_offloading.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "deepseek-v4-flash"
  mirror:
    amd:
      label: ":amd: (MI300) LM Eval KV-Offload Large"
      dind: false
      device: mi300_4
      timeout_in_minutes: 65
      depends_on:
      - image-build-amd
      working_dir: /vllm-workspace/tests
      source_file_dependencies:
      - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
      - vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
      - vllm/v1/kv_offload/
      - vllm/v1/simple_kv_offload/
      - tests/evals/gsm8k/test_gsm8k_offloading.py
      - vllm/platforms/rocm.py
      commands:
      - VLLM_ROCM_USE_AITER=1 pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "deepseek-v4-flash"

- label: ":nvidia: (H200 MIG 35GB) MRCR Eval Small Models"
  key: mrcr-eval-small-models
  device: h200_35gb
  timeout_in_minutes: 25
  source_file_dependencies:
  - tests/evals/mrcr/
  commands:
    - pytest -s -v evals/mrcr/test_mrcr_correctness.py --config-list-file=evals/mrcr/configs/models-small.txt
  mirror:
    amd:
      label: ":amd: (MI355 DPX) MRCR Eval Small Models"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 40
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - vllm/model_executor/models/
      - vllm/model_executor/model_loader/
      - vllm/v1/attention/backends/
      - vllm/v1/attention/selector.py
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      - tests/evals/mrcr/
