group: Quantization
depends_on:
  - image-build
steps:
- label: ":nvidia: (H200 MIG 35GB) Quantization Shard %N"
  device: h200_35gb
  key: quantization
  timeout_in_minutes: 40
  parallelism: 4
  env:
    VLLM_USE_V2_MODEL_RUNNER: "0"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - tests/quantization
  commands:
  # Pin torchao to the version compatible with the CI PyTorch and CUDA stack.
  - uv pip install --system torchao==0.17.0 --index-url https://download.pytorch.org/whl/cu130
  - uv pip install --system conch-triton-kernels
  # The SM90-only checkpoint currently contains a removed weight_chan_scale
  # parameter. It was not exercised by the previous L4 job.
  - VLLM_TEST_FORCE_LOAD_FORMAT=auto pytest -v -s quantization/ --ignore quantization/test_blackwell_moe.py --ignore quantization/test_rocm_moe.py -k 'not test_compressed_tensors_w4a8_fp8' --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
  mirror:
    amd:
      label: ":amd: (MI300) Quantization Shard %N"
      dind: false
      device: mi300_1
      timeout_in_minutes: 115
      working_dir: "/vllm-workspace/tests"
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - csrc/
      - vllm/model_executor/layers/quantization
      - tests/quantization
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      - vllm/model_executor/layers/fused_moe/oracle/unquantized.py
      - vllm/model_executor/layers/fused_moe/unquantized_fused_moe_method.py
      - tests/rocm/test_moe_weight_replay.py
      commands:
      # Use the ROCm default instead of the NVIDIA runner override.
      - unset VLLM_USE_V2_MODEL_RUNNER
      - uv pip install --system torchao==0.17.0
      - uv pip install --system conch-triton-kernels
      - VLLM_TEST_FORCE_LOAD_FORMAT=auto pytest -v -s quantization/ --ignore quantization/test_blackwell_moe.py --ignore quantization/test_rocm_moe.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
      - if [ "$$BUILDKITE_PARALLEL_JOB" = "0" ]; then pytest -v -s rocm/test_moe_weight_replay.py; fi

- label: ":nvidia: (H200 MIG 35GB) Quantized Fusions"
  device: h200_35gb
  key: quantized-fusions
  timeout_in_minutes: 20
  source_file_dependencies:
  - tests/fusion
  - vllm/model_executor/layers/fusion
  - vllm/model_executor/kernels/linear
  - vllm/model_executor/layers/quantization/compressed_tensors
  - vllm/model_executor/layers/quantization/modelopt.py
  commands:
    - pytest -v -s fusion/
  mirror:
    amd:
      label: ":amd: (MI300) Quantized Fusions"
      dind: false
      device: mi300_1
      timeout_in_minutes: 45
      working_dir: "/vllm-workspace/tests"
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - tests/fusion
      - vllm/model_executor/layers/fusion
      - vllm/model_executor/kernels/linear
      - vllm/model_executor/layers/quantization/compressed_tensors
      - vllm/model_executor/layers/quantization/modelopt.py
      - vllm/platforms/rocm.py

- label: ":nvidia: (B200) Quantized MoE"
  key: quantized-moe-test-b200
  timeout_in_minutes: 120
  working_dir: "/vllm-workspace/"
  device: b200-k8s
  source_file_dependencies:
  - tests/quantization/test_blackwell_moe.py
  - vllm/model_executor/models/deepseek_v2.py
  - vllm/model_executor/models/gpt_oss.py
  - vllm/model_executor/models/llama4.py
  - vllm/model_executor/layers/fused_moe
  - vllm/model_executor/layers/quantization/compressed_tensors
  - vllm/model_executor/layers/quantization/modelopt.py
  - vllm/model_executor/layers/quantization/mxfp4.py
  - vllm/v1/attention/backends/flashinfer.py
  commands:
    - pytest -s -v tests/quantization/test_blackwell_moe.py
  mirror:
    amd:
      label: ":amd: (MI355) Quantized MoE"
      dind: false
      device: mi355_1
      timeout_in_minutes: 45
      working_dir: "/vllm-workspace/"
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - tests/quantization/test_rocm_moe.py
      - tests/utils.py
      - vllm/model_executor/models/deepseek_v2.py
      - vllm/model_executor/models/qwen3_moe.py
      - vllm/model_executor/models/gpt_oss.py
      - vllm/model_executor/model_loader/
      - vllm/model_executor/layers/fused_moe/
      - vllm/model_executor/layers/quantization/
      - vllm/model_executor/kernels/linear/
      - vllm/v1/attention/
      - vllm/v1/worker/gpu/
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      - vllm/envs.py
      commands:
      - pytest -v -s tests/quantization/test_rocm_moe.py

- label: ":nvidia: (H200 MIG 35GB) Quantized Models"
  device: h200_35gb
  key: quantized-models-test
  timeout_in_minutes: 65
  env:
    VLLM_USE_V2_MODEL_RUNNER: "0"
  source_file_dependencies:
  - vllm/model_executor/layers/quantization
  - tests/models/quantization
  commands:
    - pytest -v -s models/quantization
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Quantized Models"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 80
      working_dir: "/vllm-workspace/tests"
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - vllm/model_executor/layers/quantization
      - tests/models/quantization
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      - vllm/model_executor/model_loader/
      commands:
      # Use the ROCm default instead of the NVIDIA runner override.
      - unset VLLM_USE_V2_MODEL_RUNNER
      - pytest -v -s models/quantization
