# In this file, you can add more tests to run either by adding a new step or
# adding a new command to an existing step. See different options here for examples.

# This script will be feed into Jinja template in `test-template-amd.j2` at
# https://github.com/vllm-project/buildkite-ci/blob/main/scripts/test-template-amd.j2
# to generate the final pipeline yaml file.

# Documentation
# label(str): the name of the test. emojis allowed.
# fast_check(bool): whether to run this on each commit on the fastcheck pipeline.
# fast_check_only(bool): run this test on the fastcheck pipeline only
# optional(bool): never run this test by default (i.e. need to unblock manually) unless it's a scheduled nightly run.
# soft_fail(bool): allow this step to fail without failing the entire pipeline (useful for flaky or experimental tests).
# command(str): the single command to run for tests. incompatible with commands.
# commands(list): the list of commands to run for the test. incompatible with command.
# mirror_hardwares(list): selector tags for AMD pipelines that should include the test.
# dind(bool): when false, run the job directly in the AMD Kubernetes pod.
#     When true or omitted, use the legacy Docker-in-Docker path.
# num_gpus(int): override the number of GPUs for the test. defaults to 1 GPU. currently supports 2,4.
# num_nodes(int): whether to simulate multi-node setup by launching multiple containers on one host,
#     in this case, commands must be specified. the first command runs on the first host, the second
#     command runs on the second host.
# timeout_in_minutes(int): sets a timeout for the step in minutes. if not specified, uses the default timeout.
# parallelism(int): number of parallel jobs to run for this step. enables test sharding using $$BUILDKITE_PARALLEL_JOB
#     and $$BUILDKITE_PARALLEL_JOB_COUNT environment variables.
# working_dir(str): specify the place where the command should execute, default to /vllm-workspace/tests
# source_file_dependencies(list): the list of prefixes to opt-in the test for, if empty, the test will always run.

# When adding a test
# - If the test belongs to an existing group, add it there
# - If the test is short, add to any existing step
# - If the test takes more than 10min, then it is okay to create a new step.
#   Note that all steps execute in parallel.


#####################################################################################################################################
#                                                                                                                                   #
#                                                             README                                                                #
#                                                                                                                                   #
#####################################################################################################################################
#                                                                                                                                   #
# IMPORTANT:                                                                                                                        #
#   * Currently AMD CI has MI250 agents, MI300 agents, and MI355 agents. All upcoming feature improvements are                      #
#         tracked in: https://github.com/vllm-project/vllm/issues/34994                                                             #
#                                                                                                                                   #
#-----------------------------------------------------------------------------------------------------------------------------------#
#                                                                                                                                   #
# NOTES:                                                                                                                            #
#   * [Pytorch Nightly Dependency Override Check]: if this test fails, it means the nightly torch version is not compatible with    #
#                                                  some of the dependencies. Please check the error message and add the package to  #
#                                                  whitelist in `/vllm/tools/pre_commit/generate_nightly_torch_test.py`.            #
#   * [Entrypoints Integration (LLM)]:                                                                                              #
#     - {`pytest -v -s entrypoints/llm/test_generate.py`}: It needs a clean process                                                 #
#     - {`pytest -v -s entrypoints/offline_mode`}: Needs to avoid interference with other tests                                     #
#   * [Engine / Engine (1 GPU) / e2e Scheduling / e2e Core / V1 e2e / Spec Decode / V1 Sample + Logits / V1 Core + KV + Metrics]:   #
#     - Previously a single "V1 Test e2e + engine" step, now split across multiple groups.                                          #
#     - V1 e2e (2/4 GPUs) uses 4 GPUs but is scheduled on 8-GPU machines for stability. See:                                        #
#       https://github.com/vllm-project/vllm/pull/31040                                                                             #
#   * [V1 Sample + Logits / V1 Core + KV + Metrics / V1 others (CPU)]:                                                              #
#     - Previously a single "V1 others" step, now split to avoid interference.                                                      #
#     - Integration test for streaming correctness (requires special branch for __harness__ lib).                                   #
#   * [V1 others (CPU)]: Split the tests to avoid interference                                                                      #
#   * [PyTorch Compilation Unit Tests]: Run unit tests defined directly under `compile/`, not including subdirectories, which       #
#                                       are usually heavier tests covered elsewhere. Use `find` to launch multiple instances        #
#                                       of pytest so that they do not suffer from:                                                  #
#                                       https://github.com/vllm-project/vllm/issues/28965                                           #
#   * [PyTorch Fullgraph Smoke Test]: Run smoke tests under fullgraph directory, except `test_full_graph.py` as it is a heavy       #
#                                     test that is covered in other steps. Use `find` to launch multiple instances of pytest        #
#                                     so that they do not suffer from: https://github.com/vllm-project/vllm/issues/28965            #
#   * [PyTorch Fullgraph]:                                                                                                          #
#     - Limit to no custom ops to reduce running time. Wrap with quotes to escape yaml and avoid starting `-k` string               #
#       with a `-`                                                                                                                  #
#     - Old E2E tests such as:                                                                                                      #
#           ```bash                                                                                                                 #
#           pytest -v -s compile/distributed/test_fusions_e2e.py -k 'TRITON and not +quant_fp8 and not Llama-4'                     #
#           ```                                                                                                                     #
#       were removed in https://github.com/vllm-project/vllm/pull/33293 in favor of new tests in `fusions_e2e`. We                  #
#       avoid replicating the new jobs in this file as it's deprecated.                                                             #
#   * [Basic Models Tests (Extra Initialization) %N]: Only when vLLM model source is modified - test initialization of a            #
#                                                     large subset of supported models (the complement of the small subset in       #
#                                                     the above test.) Also run if model initialization test file is modified.      #
#   * [Language Models Tests (Extra Standard) %N]: Shard slow subset of standard language models tests. Only run when model         #
#                                                  source is modified, or when specified test files are modified.                   #
#   * [Language Models Tests (Hybrid) %N]: Install fast path packages for testing against transformers (mamba, conv1d).             #
#   * [Language Models Test (Extended Generation)]: Install fast path packages for testing against transformers (mamba, conv1d).    #
#   * [Multi-Modal Models (Standard) 1-4]:                                                                                          #
#     - Do NOT remove `VLLM_WORKER_MULTIPROC_METHOD=spawn` setting as ROCm requires this for certain models to function.            #
#   * [Transformers Nightly Models]: Whisper needs `VLLM_WORKER_MULTIPROC_METHOD=spawn` to avoid deadlock.                          #
#   * [Plugin Tests (2 GPUs)]:                                                                                                      #
#     - {`pytest -v -s plugins_tests/test_oot_registration_online.py`}: It needs a clean process                                    #
#     - {`pytest -v -s plugins_tests/test_oot_registration_offline.py`}: It needs a clean process                                   #
#     - {`pytest -v -s plugins_tests/lora_resolvers`}: Unit tests for in-tree lora resolver plugins                                 #
#   * [LoRA TP (Distributed)]:                                                                                                      #
#     - There is some Tensor Parallelism related processing logic in LoRA that requires multi-GPU testing for validation.           #
#     - {`pytest -v -s -x lora/test_gptoss_tp.py`}: Disabled for now because MXFP4 backend on non-cuda platform doesn't support     #
#                                                   LoRA yet.                                                                       #
#   * [Distributed Tests (NxGPUs)(HW-TAG)]: Don't test llama model here, it seems hf implementation is buggy. See:                  #
#                                           https://github.com/vllm-project/vllm/pull/5689                                          #
#   * [Distributed Tests (NxGPUs)(HW-TAG)]: Some old E2E tests were removed in https://github.com/vllm-project/vllm/pull/33293      #
#                                           in favor of new tests in fusions_e2e. We avoid replicating the new jobs in              #
#                                           this file as it's deprecated.                                                           #
#                                                                                                                                   #
#####################################################################################################################################


steps:

#########################################################################################################################################
#                                                                                                                                       #
#                                                           AMD CPU tests                                                               #
#                                                                                                                                       #
#########################################################################################################################################

- label: ":computer: (CPU) Basic Models Other" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  no_gpu: true
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/test_utils.py
  - tests/models/test_vision.py
  - tests/models/test_adapters.py
  - tests/models/transformers/fusers/
  - tests/model_executor/layers/test_activation.py
  - tests/models/test_qwen3_5_mtp_config.py
  commands:
  - pytest -v -s models/test_utils.py models/test_vision.py models/test_adapters.py models/test_qwen3_5_mtp_config.py models/transformers/fusers/ model_executor/layers/test_activation.py

- label: ":computer: (CPU) Multimodal Processor Shard %N" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  no_gpu: true
  parallelism: 6
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal
  - tests/models/registry.py
  commands:
  - pytest -v -s models/multimodal/processing --ignore models/multimodal/processing/test_tensor_schema.py --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":computer: (CPU) V1 Others" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  no_gpu: true
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/v1
  commands:
  - pytest -v -s -m 'cpu_test' v1/core
  - pytest -v -s v1/structured_output
  - pytest -v -s v1/test_serial_utils.py
  - pytest -v -s v1/test_kv_cache_spec_registry.py
  - pytest -v -s v1/cudagraph/test_cudagraph_manager.py
  - pytest -v -s -m 'cpu_test' v1/kv_connector/unit
  - pytest -v -s -m 'cpu_test' v1/ec_connector/unit
  - pytest -v -s -m 'cpu_test' v1/metrics

- label: ":computer: (CPU) Async Engine, Inputs, Utils, Worker, Config" # TBD
  timeout_in_minutes: 80
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  no_gpu: true
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/test_envs.py
  - tests/test_outputs.py
  - tests/test_pooling_params.py
  - tests/test_ray_env.py
  - tests/test_sampling_params.py
  - tests/multimodal
  - tests/renderers
  - tests/standalone_tests/lazy_imports.py
  - tests/tokenizers_
  - tests/reasoning
  - tests/tool_parsers
  - tests/parser
  - tests/transformers_utils
  - tests/config
  commands:
  - python3 standalone_tests/lazy_imports.py
  - pytest -v -s test_envs.py
  - pytest -v -s test_outputs.py
  - pytest -v -s test_pooling_params.py
  - pytest -v -s test_ray_env.py
  - pytest -v -s test_sampling_params.py
  - pytest -v -s -m 'cpu_test' multimodal
  - pytest -v -s renderers
  - pytest -v -s reasoning
  - pytest -v -s tool_parsers
  - pytest -v -s tokenizers_
  - pytest -v -s parser
  - pytest -v -s transformers_utils
  - pytest -v -s config --ignore=config/test_mp_reducer.py

- label: ":computer: (CPU) Rust Frontend Cargo Style + Clippy" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  no_gpu: true
  optional: true
  working_dir: "/vllm-workspace"
  source_file_dependencies:
  - rust/
  - rust-toolchain.toml
  - .buildkite/test_areas/rust_frontend_cargo.yaml
  - .buildkite/scripts/run-rust-frontend-cargo-ci.sh
  commands:
  - .buildkite/scripts/run-rust-frontend-cargo-ci.sh style-clippy

- label: ":computer: (CPU) Rust Frontend Cargo" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  no_gpu: true
  optional: true
  working_dir: "/vllm-workspace"
  source_file_dependencies:
  - rust/
  - rust-toolchain.toml
  - .buildkite/test_areas/rust_frontend_cargo.yaml
  - .buildkite/scripts/run-rust-frontend-cargo-ci.sh
  commands:
  - .buildkite/scripts/run-rust-frontend-cargo-ci.sh test

- label: ":computer: (CPU) Docker Build Metadata" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  no_gpu: true
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - .buildkite/scripts/docker-build-metadata-args.sh
  - .buildkite/scripts/ci-bake-rocm.sh
  - .buildkite/release-pipeline.yaml
  - docker/Dockerfile
  - docker/Dockerfile.cpu
  - docker/Dockerfile.rocm
  - docker/Dockerfile.rocm_base
  - docker/ci-rocm.hcl
  - docker/docker-bake.hcl
  - docker/docker-bake-rocm.hcl
  - tests/tools/test_docker_build_metadata_args.py
  commands:
  - pytest -v -s tools/test_docker_build_metadata_args.py

- label: ":amd: (MI250) Torch Stable ABI Audit" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  no_gpu: true
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - .buildkite/check-torch-abi.py
  - csrc/
  - cmake/
  - setup.py
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system torch-abi-audit==0.0.1
  - python3 /vllm-workspace/.buildkite/check-torch-abi.py

#########################################################################################################################################
#                                                                                                                                       #
#                                                         MI250 (gfx90a) tests                                                          #
#                                                                                                                                       #
#########################################################################################################################################

#----------------------------------------------------------  mi250 · compile  ----------------------------------------------------------#

- label: ":amd: (MI250) PyTorch Fullgraph Smoke" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/compilation/
  - vllm/model_executor/
  - vllm/v1/attention/
  - vllm/config/compilation.py
  - csrc/
  - tests/compile
  - vllm/platforms/rocm.py
  commands:
  - "find compile/fullgraph/ -name 'test_*.py' -not -name 'test_full_graph.py' -exec pytest -s -v {} \\\\;"

#--------------------------------------------------------  mi250 · distributed  --------------------------------------------------------#

- label: ":amd: (MI250) Pipeline + Context Parallelism" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_4
  num_gpus: 4
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/
  - vllm/engine/
  - vllm/executor/
  - vllm/model_executor/models/
  - vllm/model_executor/layers/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - tests/distributed/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/v1/distributed/test_pp_dp_v2.py
  optional: true
  commands:
  - pytest -v -s distributed/test_pp_cudagraph.py
  - pytest -v -s distributed/test_pipeline_parallel.py
  - pytest -v -s v1/distributed/test_pp_dp_v2.py

#----------------------------------------------------------  mi250 · kernels  ----------------------------------------------------------#

- label: ":amd: (MI250) Helion Kernels Shard %N"
  timeout_in_minutes: 45
  parallelism: 2
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/utils/import_utils.py
  - tools/benchmark_helion_kernels.py
  - tests/kernels/helion/
  - vllm/platforms/rocm.py
  commands:
  - pip install helion==1.4.0
  - pytest -v -s kernels/helion/ --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

#-----------------------------------------------------  mi250 · models / language  -----------------------------------------------------#

- label: ":amd: (MI250) Language Models (PPL)" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/language/generation_ppl_test
  - vllm/model_executor/models/qwen3_5.py
  - vllm/model_executor/models/qwen3_5_mtp.py
  - vllm/transformers_utils/configs/qwen3_5.py
  - vllm/transformers_utils/configs/qwen3_5_moe.py
  - vllm/model_executor/models/qwen2.py
  - vllm/model_executor/models/qwen3.py
  - vllm/model_executor/models/qwen3_next.py
  - vllm/model_executor/models/qwen3_next_mtp.py
  - vllm/model_executor/layers/fla/ops/
  - vllm/_aiter_ops.py
  - vllm/v1/attention/backends/triton_attn.py
  - vllm/v1/attention/backends/rocm_attn.py
  - vllm/v1/attention/backends/rocm_aiter_unified_attn.py
  - vllm/v1/attention/backends/rocm_aiter_fa.py
  - vllm/v1/attention/backends/flex_attention.py
  - vllm/v1/attention/ops/
  - vllm/platforms/rocm.py
  optional: true
  commands:
  - pytest -v -s models/language/generation_ppl_test

#----------------------------------------------------  mi250 · models / multimodal  ----------------------------------------------------#

- label: ":amd: (MI250) Multimodal Models (Standard) 3: llava + qwen2_vl" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal
  commands:
  - pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "not qwen2 and not qwen3 and not gemma"
  - pytest -v -s models/multimodal/generation/test_qwen2_vl.py -m core_model

#------------------------------------------------------------  mi250 · v1  -------------------------------------------------------------#

- label: ":amd: (MI250) Batch Invariance" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/attention
  - vllm/model_executor/layers
  - tests/v1/determinism/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pip install pytest-timeout pytest-forked
  - pytest -v -s v1/determinism/test_batch_invariance.py
  - pytest -v -s v1/determinism/test_rms_norm_batch_invariant.py

- label: ":amd: (MI250) CUDAGraph" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - tests/v1/cudagraph
  - vllm/v1/cudagraph_dispatcher.py
  - vllm/config/compilation.py
  - vllm/compilation
  - vllm/v1/worker/encoder_cudagraph.py
  - vllm/v1/worker/encoder_cudagraph_defs.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/cudagraph/test_cudagraph_dispatch.py
  - pytest -v -s v1/cudagraph/test_cudagraph_mode.py
  - pytest -v -s v1/cudagraph/test_breakable_cudagraph.py
  - pytest -v -s v1/cudagraph/test_encoder_cudagraph.py

- label: ":amd: (MI250) E2E Core" # TBD
  timeout_in_minutes: 50
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/
  - tests/v1/e2e/general/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/e2e/general --ignore v1/e2e/general/test_async_scheduling.py --ignore v1/e2e/general/test_kv_sharing_fast_prefill.py -k "not test_mamba_prefix_cache_mrv2"

- label: ":amd: (MI250) E2E Scheduling" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/
  - tests/v1/e2e/general/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/e2e/general/test_async_scheduling.py

- label: ":amd: (MI250) V1 Engine" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/
  - tests/v1/engine/
  - tests/v1/test_tensor_ipc_queue.py
  - tests/config/test_mp_reducer.py
  - vllm/config/
  - vllm/engine/arg_utils.py
  - vllm/transformers_utils/config.py
  - vllm/platforms/rocm.py
  - vllm/v1/engine/
  commands:
  - pytest -v -s v1/engine/test_preprocess_error_handling.py
  - pytest -v -s v1/engine --ignore v1/engine/test_preprocess_error_handling.py
  - pytest -v -s v1/test_tensor_ipc_queue.py
  - pytest -v -s config/test_mp_reducer.py

- label: ":amd: (MI250) Spec Decode Draft Model" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/spec_decode/
  - vllm/v1/worker/gpu/spec_decode/
  - vllm/model_executor/model_loader/
  - vllm/v1/sample/
  - vllm/model_executor/layers/
  - tests/v1/e2e/spec_decode/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/e2e/spec_decode/draft_model/

- label: ":amd: (MI250) Spec Decode Speculators + MTP" # TBD
  timeout_in_minutes: 75
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx90anightly, amdmi250]
  dind: false
  agent_pool: mi250_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/spec_decode/
  - vllm/v1/worker/gpu/spec_decode/
  - vllm/model_executor/model_loader/
  - vllm/v1/sample/
  - vllm/model_executor/layers/
  - vllm/transformers_utils/configs/speculators/
  - tests/v1/e2e/spec_decode/
  - vllm/platforms/rocm.py
  - vllm/v1/attention/backends/
  commands:
  - pytest -v -s v1/e2e/spec_decode/speculators/
  - pytest -v -s v1/e2e/spec_decode/mtp/

#-------------------------------------------------------------  mi250 · misc  ------------------------------------------------------------#

- label: ":amd: (MI300) Python-only Installation" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - tests/standalone_tests/python_only_compile.sh
  - setup.py
  - vllm/platforms/rocm.py
  commands:
  - bash standalone_tests/python_only_compile.sh

#########################################################################################################################################
#                                                                                                                                       #
#                                                         MI300 (gfx942) tests                                                          #
#                                                                                                                                       #
#########################################################################################################################################

#-----------------------------------------------------  mi300 · basic_correctness  -----------------------------------------------------#

- label: ":amd: (MI300) Distributed Models Shard %N"
  timeout_in_minutes: 80
  parallelism: 3
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/models/hy_v4/
  - vllm/config/
  - vllm/v1/worker/gpu/
  - vllm/v1/attention/ops/
  - vllm/v1/attention/backend.py
  - vllm/model_executor/model_loader/sharded_state_loader.py
  - vllm/model_executor/models/
  - vllm/model_executor/layers/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/basic_correctness/
  - tests/model_executor/model_loader/test_sharded_state_loader.py
  - tests/models/
  - vllm/v1/worker/
  - tests/conftest.py
  - tests/utils.py
  - tests/models/test_vision.py
  commands:
  - case "$$BUILDKITE_PARALLEL_JOB" in 0) TARGET_TEST_SUITE=MI300 pytest basic_correctness/ -v -s -m 'distributed(num_gpus=2)' && HIP_VISIBLE_DEVICES=0,1 pytest -v -s model_executor/model_loader/test_sharded_state_loader.py -m '(not slow_test)' ;; 1) pytest models/transformers/test_backend.py -v -s -m 'distributed(num_gpus=2)' && pytest models/language -v -s -m 'distributed(num_gpus=2)' && VLLM_ROCM_USE_AITER=1 pytest -v -s models/test_hyv4_rocm.py -m 'distributed(num_gpus=2)' ;; 2) pytest models/multimodal -v -s -m 'distributed(num_gpus=2)' --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_phi4siglip.py && pytest models/multimodal/generation/test_phi4siglip.py -v -s -m 'distributed(num_gpus=2)' && VLLM_WORKER_MULTIPROC_METHOD=spawn pytest models/multimodal/generation/test_whisper.py -v -s -m 'distributed(num_gpus=2)' && pytest -v -s models/test_vision.py -m 'distributed(num_gpus=2)' ;; *) exit 2 ;; esac

#----------------------------------------------------------  mi300 · compile  ----------------------------------------------------------#

- label: ":amd: (MI300) PyTorch Compilation" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/compilation/
  - vllm/model_executor/layers/
  - vllm/v1/worker/
  - vllm/v1/attention/
  - vllm/v1/cudagraph_dispatcher.py
  - vllm/config/compilation.py
  - csrc/
  - tests/compile
  - vllm/platforms/rocm.py
  - vllm/__init__.py
  - vllm/_aiter_ops.py
  - vllm/_custom_ops.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/env_override.py
  - vllm/envs.py
  - vllm/forward_context.py
  - vllm/inputs/
  - vllm/ir/
  - vllm/kernels/
  - vllm/logger.py
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/platforms/
  - vllm/plugins/
  - vllm/sampling_params.py
  - vllm/sequence.py
  - vllm/transformers_utils/
  - vllm/triton_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - "find compile/ -maxdepth 1 -name 'test_*.py' -print0 | xargs -0 -n1 -I{} pytest -s -v '{}'"
  - pytest -s -v compile/dynamic_shapes/

- label: ":amd: (MI300) Fusion E2E Config Sweep" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  num_gpus: 1
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - csrc/quantization/
  - vllm/compilation/
  - vllm/model_executor/layers/layernorm.py
  - vllm/model_executor/layers/activation.py
  - vllm/model_executor/layers/attention/attention.py
  - vllm/model_executor/layers/quantization/input_quant_fp8.py
  - tests/compile/fusions_e2e/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  optional: true
  commands:
  - rocm-smi
  - pytest -v -s tests/compile/fusions_e2e/test_tp1_quant.py -k "llama-3"

- label: ":amd: (MI300) Fusion E2E Quick" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  num_gpus: 1
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - csrc/quantization/
  - vllm/model_executor/
  - vllm/v1/attention/
  - vllm/compilation/
  - tests/compile/fusions_e2e/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  optional: true
  commands:
  - rocm-smi
  # Run all models and attn backends but only Inductor partition and native custom ops
  - "pytest -v -s tests/compile/fusions_e2e/test_tp1_quant.py -k 'inductor_partition and not +rms_norm and not +quant_fp8'"
  # Different from CUDA, Qwen requires +rms_norm and +quant_fp8 as rms+quant fusion is only supported on AITER
  - "pytest -v -s tests/compile/fusions_e2e/test_tp1_quant.py -k 'inductor_partition and +rms_norm and +quant_fp8 and qwen3'"
  # DeepSeek-Coder uses the portable static-FP8 MLA fusion path on ROCm.
  - "pytest -v -s tests/compile/fusions_e2e/test_tp1_quant.py -k 'inductor_partition and not +rms_norm and +quant_fp8 and DeepSeek-Coder and TRITON_MLA and not True'"

- label: ":amd: (MI300) PyTorch Compilation Passes" # TBD
  timeout_in_minutes: 80
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/compile/passes
  - vllm/platforms/rocm.py
  - vllm/__init__.py
  - vllm/_aiter_ops.py
  - vllm/_custom_ops.py
  - vllm/compilation/
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/env_override.py
  - vllm/envs.py
  - vllm/forward_context.py
  - vllm/inputs/
  - vllm/ir/
  - vllm/kernels/
  - vllm/logger.py
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/platforms/
  - vllm/plugins/
  - vllm/sampling_params.py
  - vllm/sequence.py
  - vllm/transformers_utils/
  - vllm/triton_utils/
  - vllm/utils/
  - vllm/v1/
  optional: true
  commands:
  - pytest -s -v compile/passes --ignore compile/passes/distributed

- label: ":amd: (MI300) PyTorch Fullgraph" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/compilation/
  - vllm/model_executor/
  - vllm/v1/attention/
  - vllm/config/compilation.py
  - csrc/
  - tests/compile
  - vllm/platforms/rocm.py
  - vllm/__init__.py
  - vllm/_aiter_ops.py
  - vllm/_custom_ops.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/env_override.py
  - vllm/envs.py
  - vllm/forward_context.py
  - vllm/inputs/
  - vllm/ir/
  - vllm/kernels/
  - vllm/logger.py
  - vllm/multimodal/
  - vllm/platforms/
  - vllm/plugins/
  - vllm/sampling_params.py
  - vllm/sequence.py
  - vllm/transformers_utils/
  - vllm/triton_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - find compile/fullgraph/ -name 'test_*.py' -not -name 'test_full_cudagraph.py' -print0 | xargs -0 -n1 -I{} pytest -s -v '{}'

- label: ":amd: (MI300) PyTorch Nightly Dependency Override Check" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - requirements/test/nightly-torch.txt
  - vllm/platforms/rocm.py
  commands:
  - bash standalone_tests/pytorch_nightly_dependency.sh

- label: ":amd: (MI300) Distributed Compile (Inductor Partition)" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - vllm/compilation/
  - vllm/model_executor/layers
  - tests/compile/passes/distributed/
  - tests/compile/fusions_e2e/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_TEST_CLEAN_GPU_MEMORY=1
  - pytest -s -v tests/compile/passes/distributed
  - pytest -v -s tests/compile/fusions_e2e/test_tp2_ar_rms.py::test_tp2_ar_rms_fusions -k "inductor_partition"

- label: ":amd: (MI300) Distributed Compile (Dynamo Partition)" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - vllm/compilation/
  - vllm/model_executor/layers
  - tests/compile/passes/distributed/
  - tests/compile/fusions_e2e/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_TEST_CLEAN_GPU_MEMORY=1
  - VLLM_TEST_CLEAN_GPU_MEMORY=1 pytest -v -s tests/compile/passes/distributed/test_async_tp.py
  - pytest -v -s tests/compile/fusions_e2e/test_tp2_ar_rms.py::test_tp2_ar_rms_fusions -k "dynamo_partition"

# - label: ":amd: (MI300) Sequence Parallel Correctness" # TBD
#   timeout_in_minutes: 180
#   mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
#   dind: false
#   agent_pool: mi300_2
#   num_gpus: 2
#   optional: true
#   working_dir: "/vllm-workspace/"
#   source_file_dependencies:
#   - vllm/model_executor/layers/
#   - vllm/compilation/
#   - vllm/v1/worker/
#   - vllm/v1/cudagraph_dispatcher.py
#   - tests/compile/correctness_e2e/test_sequence_parallel.py
#   - vllm/platforms/rocm.py
#   commands:
#   - export VLLM_TEST_CLEAN_GPU_MEMORY=1
#   - pytest -v -s tests/compile/correctness_e2e/test_sequence_parallel.py

- label: ":amd: (MI300) Distributed Compile + RPC" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/compilation/
  - vllm/distributed/
  - vllm/engine/
  - vllm/executor/
  - vllm/worker/worker_base.py
  - vllm/v1/engine/
  - vllm/v1/worker/
  - tests/compile/fullgraph/test_basic_correctness.py
  - tests/compile/test_wrapper.py
  - tests/entrypoints/llm/test_collective_rpc.py
  - vllm/platforms/rocm.py
  optional: true
  commands:
  - pytest -v -s entrypoints/llm/test_collective_rpc.py
  - pytest -v -s ./compile/test_wrapper.py

#-----------------------------------------------------------  mi300 · cuda  ------------------------------------------------------------#

- label: ":amd: (MI300) CUDA Platform" # TBD
  timeout_in_minutes: 25
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/cuda
  - vllm/envs.py
  - vllm/logger.py
  - vllm/platforms/
  - vllm/plugins/
  - vllm/utils/
  optional: true
  commands:
  - pytest -v -s cuda/test_cuda_context.py
  - pytest -v -s cuda/test_platform_no_cuda_init.py
  - pytest -v -s cuda/test_cuda_compatibility_path.py

#--------------------------------------------------------  mi300 · detokenizer  --------------------------------------------------------#

- label: ":amd: (MI300) Async Engine, Inputs, Utils, Worker" # TBD
  timeout_in_minutes: 50
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/detokenizer
  - tests/multimodal
  - tests/utils_
  - vllm/assets/
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/inputs/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - pytest -v -s detokenizer
  - pytest -v -s -m 'not cpu_test' multimodal
  - pytest -v -s multimodal/test_cache.py::test_sleep_wake_preserves_mm_cache_consistency
  - pytest -v -s utils_

#--------------------------------------------------------  mi300 · distributed  --------------------------------------------------------#

- label: ":amd: (MI300) Distributed Comm Ops" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed
  - tests/distributed
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s distributed/test_comm_ops.py
  - pytest -v -s distributed/test_shm_broadcast.py
  - pytest -v -s distributed/test_shm_buffer.py
  - pytest -v -s distributed/test_shm_storage.py

- label: ":amd: (MI300) EPLB Execution" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/eplb
  - tests/distributed/test_eplb_execute.py
  - tests/distributed/test_eplb_spec_decode.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s distributed/test_eplb_execute.py
  - pytest -v -s distributed/test_eplb_spec_decode.py

- label: ":amd: (MI300) Distributed Features" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - vllm/distributed/
  - vllm/v1/distributed/
  - vllm/model_executor/layers/fused_moe/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - tests/v1/distributed/test_dbo.py
  - tests/distributed/test_context_parallel.py
  - examples/features/data_parallel/data_parallel_offline.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - csrc/custom_quickreduce.cu
  - csrc/ops.h
  - csrc/torch_bindings.cpp
  - vllm/model_executor/layers/
  - vllm/entrypoints/llm.py
  - vllm/config/parallel.py
  - vllm/v1/engine/
  - vllm/v1/executor/
  - vllm/v1/worker/
  - vllm/_custom_ops.py
  - vllm/envs.py
  - tests/distributed/test_rocm_aiter_custom_ar.py
  - tests/distributed/test_rocm_quick_reduce.py
  - tests/distributed/test_quick_all_reduce.py
  - tests/v1/e2e/general/test_rocm_aiter_custom_ar.py
  - tests/utils.py
  - examples/rl/rlhf_async_new_apis.py
  - tests/distributed/test_weight_transfer.py
  - tests/distributed/test_packed_tensor.py
  optional: true
  commands:
  - pytest -v -s tests/distributed/test_context_parallel.py
  - VLLM_ALLOW_INSECURE_SERIALIZATION=1 python3 examples/rl/rlhf_async_new_apis.py
  - VLLM_LOGGING_LEVEL=DEBUG python3 examples/features/data_parallel/data_parallel_offline.py --model=Qwen/Qwen1.5-MoE-A2.7B -tp=1 -dp=2 --max-model-len=2048 --all2all-backend=deepep_high_throughput
  - VLLM_LOGGING_LEVEL=DEBUG python3 examples/features/data_parallel/data_parallel_offline.py --model=Qwen/Qwen1.5-MoE-A2.7B -tp=1 -dp=2 --max-model-len=2048 --all2all-backend=allgather_reducescatter --disable-nccl-for-dp-synchronization
  - pytest -v -s tests/v1/distributed/test_dbo.py
  - VLLM_ALLOW_INSECURE_SERIALIZATION=1 pytest -v -s tests/distributed/test_weight_transfer.py
  - pytest -v -s tests/distributed/test_packed_tensor.py
  - pytest -v -s tests/distributed/test_rocm_aiter_custom_ar.py
  - pytest -v -s tests/v1/e2e/general/test_rocm_aiter_custom_ar.py
  - pytest -v -s tests/distributed/test_rocm_quick_reduce.py
  - pytest -v -s tests/distributed/test_quick_all_reduce.py

- label: ":amd: (MI300) Distributed" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/test_vision.py
  commands:
  - pytest -v -s distributed/test_custom_all_reduce.py
  - torchrun --nproc_per_node=2 distributed/test_ca_buffer_sharing.py
  - TARGET_TEST_SUITE=MI300 pytest basic_correctness/ -v -s -m 'distributed(num_gpus=2)'
  - pytest -v -s -x lora/test_mixtral.py
  - pytest -v -s models/test_vision.py -m 'distributed(num_gpus=4)'

- label: ":amd: (MI300) Distributed Torchrun + Examples" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/
  - tests/distributed/test_torchrun_example.py
  - tests/distributed/test_torchrun_example_moe.py
  - examples/rl/
  - examples/features/data_parallel/data_parallel_offline.py
  - vllm/platforms/rocm.py
  commands:
  - torchrun --nproc-per-node=4 distributed/test_torchrun_example.py
  - PP_SIZE=2 torchrun --nproc-per-node=4 distributed/test_torchrun_example.py
  - TP_SIZE=4 torchrun --nproc-per-node=4 distributed/test_torchrun_example_moe.py
  - PP_SIZE=2 TP_SIZE=2 torchrun --nproc-per-node=4 distributed/test_torchrun_example_moe.py
  - DP_SIZE=4 ENABLE_EP=1 torchrun --nproc-per-node=4 distributed/test_torchrun_example_moe.py
  - TP_SIZE=2 DP_SIZE=2 ENABLE_EP=1 torchrun --nproc-per-node=4 distributed/test_torchrun_example_moe.py
  - python3 ../examples/features/data_parallel/data_parallel_offline.py --enforce-eager
  # rlhf examples
  - VLLM_ALLOW_INSECURE_SERIALIZATION=1 python3 ../examples/rl/rlhf_http_nccl.py
  - VLLM_ALLOW_INSECURE_SERIALIZATION=1 python3 ../examples/rl/rlhf_http_ipc.py

- label: ":amd: (MI300) Elastic EP Scaling" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/
  - vllm/engine/
  - vllm/executor/
  - vllm/compilation/
  - tests/distributed/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - pytest -v -s distributed/test_elastic_ep.py

- label: ":amd: (MI300) RayExecutorV2" # TBD
  timeout_in_minutes: 75
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/executor/ray_executor_v2.py
  - vllm/v1/executor/abstract.py
  - vllm/v1/executor/multiproc_executor.py
  - tests/distributed/test_ray_v2_executor.py
  - tests/distributed/test_ray_v2_executor_e2e.py
  - tests/distributed/test_pipeline_parallel.py
  - tests/basic_correctness/models/test_basic_correctness.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_USE_RAY_V2_EXECUTOR_BACKEND=1
  - pytest -v -s distributed/test_ray_v2_executor.py
  - pytest -v -s distributed/test_ray_v2_executor_e2e.py
  - pytest -v -s distributed/test_pipeline_parallel.py -k "ray"
  - TARGET_TEST_SUITE=L4 pytest -v -s basic_correctness/models/test_basic_correctness.py -k "ray"

- label: ":amd: (MI300) Distributed DP + EP" # TBD
  timeout_in_minutes: 50
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_8
  num_gpus: 8
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - examples/features/torchrun/torchrun_dp_example_offline.py
  - vllm/config/parallel.py
  - vllm/distributed/
  - vllm/v1/engine/llm_engine.py
  - vllm/v1/executor/uniproc_executor.py
  - vllm/v1/worker/gpu_worker.py
  - vllm/platforms/rocm.py
  commands:
  - torchrun --nproc-per-node=8 ../examples/features/torchrun/torchrun_dp_example_offline.py --tp-size=2 --pp-size=1 --dp-size=4 --enable-ep

- label: ":amd: (MI300) Distributed Torchrun + Shutdown" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/
  - vllm/engine/
  - vllm/executor/
  - vllm/worker/worker_base.py
  - vllm/v1/engine/
  - vllm/v1/worker/
  - tests/distributed/
  - tests/v1/shutdown
  - tests/v1/worker/test_worker_memory_snapshot.py
  - vllm/platforms/rocm.py
  commands:
  - VLLM_TEST_SAME_HOST=1 torchrun --nproc-per-node=4 distributed/test_same_node.py | grep 'Same node test passed'
  - VLLM_TEST_SAME_HOST=1 VLLM_TEST_WITH_DEFAULT_DEVICE_SET=1 torchrun --nproc-per-node=4 distributed/test_same_node.py | grep 'Same node test passed'
  - HIP_VISIBLE_DEVICES=0,1 pytest -v -s v1/shutdown
  - pytest -v -s v1/worker/test_worker_memory_snapshot.py

- label: ":amd: (MI300) Distributed Compile + Comm" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/
  - tests/distributed/test_pynccl
  - tests/distributed/test_events
  - tests/compile/fullgraph/test_basic_correctness.py
  - tests/distributed/test_symm_mem_allreduce.py
  - tests/distributed/test_multiproc_executor.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s compile/fullgraph/test_basic_correctness.py
  - pytest -v -s distributed/test_pynccl.py
  - pytest -v -s distributed/test_events.py
  - pytest -v -s distributed/test_symm_mem_allreduce.py
  - pytest -v -s distributed/test_multiproc_executor.py::test_multiproc_executor_multi_node

#--------------------------------------------------------  mi300 · entrypoints  --------------------------------------------------------#

- label: ":amd: (MI300) Entrypoints Unit" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  fast_check: true
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/entrypoints
  - tests/entrypoints/unit_tests
  - tests/entrypoints/weight_transfer
  - tests/entrypoints/launchers
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s entrypoints/unit_tests
  - pytest -v -s entrypoints/weight_transfer
  - pytest -v -s entrypoints/launchers

- label: ":amd: (MI300) Entrypoints Integration (Responses API)" # TBD
  timeout_in_minutes: 50
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/openai/responses
  commands:
  - pytest -v -s entrypoints/openai/responses

- label: ":amd: (MI300) Entrypoints Integration (Speech to Text)" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/speech_to_text
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/speech_to_text

- label: ":amd: (MI300) Entrypoints Integration (Multimodal)"
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/multimodal
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/multimodal

- label: ":amd: (MI300) Scale-out EC Connector E2E"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/ec_transfer/
  - vllm/entrypoints/scale_out/
  - tests/entrypoints/scale_out/ec_integration/
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - bash entrypoints/scale_out/ec_integration/run_scale_out_ec_e2e_test.sh

#-----------------------------------------------------------  mi300 · evals  -----------------------------------------------------------#

- label: ":amd: (MI300) LM Eval Small Models" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  optional: true
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small.txt

- label: ":amd: (MI300) DeepSeek V2-Lite Prefetch Offload Accuracy" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  num_gpus: 1
  working_dir: "/vllm-workspace"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/backends/mla/
  - vllm/v1/attention/selector.py
  - .buildkite/scripts/scheduled_integration_test/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  optional: true
  commands:
  - bash .buildkite/scripts/scheduled_integration_test/deepseek_v2_lite_prefetch_offload.sh 0.25 200 8030

- label: ":amd: (MI300) MRCR Eval Small Models" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/evals/mrcr/
  commands:
  - pytest -s -v evals/mrcr/test_mrcr_correctness.py --config-list-file=evals/mrcr/configs/models-small.txt

- label: ":amd: (MI300) LM Eval KV-Offload" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
  - vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
  - vllm/v1/kv_offload/
  - vllm/v1/simple_kv_offload/
  - tests/evals/gsm8k/test_gsm8k_offloading.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "nemotron-h-8b or gemma-4-e4b-it"

- label: ":amd: (MI300) LM Eval KV-Offload Medium" # TBD
  timeout_in_minutes: 70
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
  - vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
  - vllm/v1/kv_offload/
  - vllm/v1/simple_kv_offload/
  - tests/evals/gsm8k/test_gsm8k_offloading.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "qwen3.5-35b or deepseek-v2-lite"

- label: ":amd: (MI300) LM Eval KV-Offload Large" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
  - vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
  - vllm/v1/kv_offload/
  - vllm/v1/simple_kv_offload/
  - tests/evals/gsm8k/test_gsm8k_offloading.py
  - vllm/platforms/rocm.py
  commands:
  - VLLM_ROCM_USE_AITER=1 pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "deepseek-v4-flash"

- label: ":amd: (MI300) LM Eval Small Models Harness" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/.buildkite/lm-eval-harness"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - .buildkite/lm-eval-harness/
  commands:
  - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-small-rocm.txt

- label: ":amd: (MI300) GPQA Eval (GPT-OSS)" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/model_executor/layers/fused_moe/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/evals/gpt_oss/
  commands:
    - pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-gfx942.txt

- label: ":amd: (MI300) LM Eval Small Models Distributed"
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/evals/gsm8k/configs/models-mi3xx-fp8-and-mixed.txt
  optional: true
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-mi3xx-fp8-and-mixed.txt

- label: ":amd: (MI300) DeepSeek V2-Lite Sync EPLB Accuracy" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/distributed/eplb
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/backends/mla/
  - vllm/v1/attention/selector.py
  - .buildkite/scripts/scheduled_integration_test/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - VLLM_ENGINE_READY_TIMEOUT_S=1800 bash .buildkite/scripts/scheduled_integration_test/deepseek_v2_lite_ep_eplb.sh 0.25 200 8010

- label: ":amd: (MI300) LM Eval Large Models Harness" # TBD
  timeout_in_minutes: 85
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/.buildkite/lm-eval-harness"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large.txt --tp-size=4

- label: ":amd: (MI300) Qwen3-30B-A3B-FP8-block Sync EPLB Accuracy" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/quantization/
  - vllm/distributed/eplb
  - vllm/model_executor/layers/fused_moe/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - .buildkite/scripts/scheduled_integration_test/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - bash .buildkite/scripts/scheduled_integration_test/qwen30b_a3b_fp8_block_ep_eplb.sh 0.8 200 8020

- label: ":amd: (MI300) Qwen3-30B-A3B-FP8 DP4 Async EPLB Accuracy" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/quantization/
  - vllm/distributed/eplb
  - vllm/model_executor/layers/fused_moe/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - .buildkite/scripts/scheduled_integration_test/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - bash .buildkite/scripts/scheduled_integration_test/qwen30b_a3b_fp8_dp4_async_eplb.sh 0.8 200 8050

# TODO(akaratza): Retire this standalone ROCm coverage after
# tests/distributed/test_eplb_spec_decode.py is enabled and validated on ROCm.
# The upstream parity job currently skips that entire module on ROCm.
- label: ":amd: (MI300) Qwen3-Next-80B-A3B-Instruct MTP Async EPLB Accuracy" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/spec_decode/
  - vllm/distributed/eplb
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - .buildkite/scripts/scheduled_integration_test/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - VLLM_ENGINE_READY_TIMEOUT_S=1800 bash .buildkite/scripts/scheduled_integration_test/qwen3_next_mtp_async_eplb.sh 0.8 1319 8040

- label: ":amd: (MI300) LM Eval Large Models" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_8
  optional: true
  num_gpus: 8
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/quantization/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/model_executor/layers/layernorm.py
  - csrc/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/evals/
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - export PYTORCH_ROCM_ARCH=gfx942
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-mi3xx.txt

- label: ":amd: (MI300) LM Eval Large Models ROCm Harness" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_8
  optional: true
  num_gpus: 8
  working_dir: "/vllm-workspace/.buildkite/lm-eval-harness"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/quantization/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/model_executor/layers/layernorm.py
  - csrc/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large-rocm.txt --tp-size=8 --timeout=5400

- label: ":amd: (MI300) LM Eval Large Models FP8" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/.buildkite/lm-eval-harness"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_USE_DEEP_GEMM=0
  - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large-rocm-tp4.txt --tp-size=4

#----------------------------------------------------------  mi300 · kernels  ----------------------------------------------------------#

- label: ":amd: (MI300) Core Operation Kernels Shard %N" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  parallelism: 3
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/
  - tests/kernels/core
  - tests/kernels/test_top_k_per_row.py
  - tests/kernels/test_concat_mla_q.py
  - tests/kernels/test_rocm_fp32_router_gemm.py
  - vllm/model_executor/layers/fused_moe/router/gate_linear.py
  - vllm/model_executor/layers/fused_moe/router/rocm_fp32_router_gemm.py
  - vllm/model_executor/layers/rotary_embedding/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/kernels/test_fused_qk_norm_rope_gate.py
  optional: true
  commands:
  - pytest -v -s kernels/core --ignore=kernels/core/test_minimax_reduce_rms.py kernels/test_fused_qk_norm_rope_gate.py kernels/test_concat_mla_q.py kernels/test_top_k_per_row.py kernels/test_rocm_fp32_router_gemm.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

- label: ":amd: (MI300) Quantization Kernels Shard %N"
  timeout_in_minutes: 120
  parallelism: 6
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/quantization/
  - csrc/rocm/
  - vllm/model_executor/layers/quantization
  - tests/kernels/quantization
  - tests/kernels/quant_utils.py
  - tests/kernels/utils.py
  - vllm/_aiter_ops.py
  - vllm/kernels/aiter_ops.py
  - vllm/_custom_ops.py
  - vllm/envs.py
  - vllm/platforms/rocm.py
  - vllm/model_executor/kernels/
  - vllm/v1/attention/backends/rocm_aiter_fa.py
  - vllm/config/
  - tests/kernels/quantization/test_rocm_skinny_gemms.py
  commands:
  - pytest -v -s kernels/quantization -k "not test_nvfp4_dense_emulation" --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

- label: ":amd: (MI300) MoE Kernels Shard %N" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  parallelism: 5
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/quantization/cutlass_w8a8/moe/
  - csrc/moe/
  - csrc/rocm/
  - tests/kernels/moe
  - tests/kernels/utils.py
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/mxfp4.py
  - vllm/model_executor/layers/quantization/utils/quant_utils.py
  - vllm/distributed/device_communicators/
  - vllm/envs.py
  - vllm/config
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s kernels/moe --ignore=kernels/moe/test_modular_oai_triton_moe.py --ignore=kernels/moe/test_modular_kernel_combinations.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
  - pytest -v -s kernels/moe/test_modular_oai_triton_moe.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
  - VLLM_ROCM_USE_AITER=1 VLLM_ROCM_USE_AITER_MOE=1 pytest -v -s kernels/moe/test_modular_kernel_combinations.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

- label: ":amd: (MI300) vLLM IR" # TBD
  timeout_in_minutes: 50
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - vllm/ir
  - vllm/kernels
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/ir/
  - tests/kernels/ir/
  commands:
  - pytest -v -s tests/ir
  - pytest -v -s tests/kernels/ir

- label: ":amd: (MI300) Attention Kernels Shard %N" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  parallelism: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/attention/
  - vllm/v1/attention
  - vllm/model_executor/layers/attention
  - tests/kernels/attention
  - vllm/_aiter_ops.py
  - vllm/envs.py
  - vllm/platforms/rocm.py
  - vllm/utils/flashinfer.py
  commands:
  - pytest -v -s kernels/attention --ignore=kernels/attention/test_triton_unified_attention_diffkv.py --ignore=kernels/attention/test_rocm_aiter_mla_decode.py --ignore=kernels/attention/test_rocm_aiter_mla_decode_metadata.py --ignore=kernels/attention/test_rocm_aiter_mla_fp8_support.py --ignore=kernels/attention/test_rocm_aiter_mla_op_registration.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

- label: ":amd: (MI300) MiniMax Reduce RMS Kernels" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/layers/minimax_rms_norm/
  - vllm/distributed/
  - vllm/_aiter_ops.py
  - vllm/envs.py
  - vllm/platforms/rocm.py
  - tests/kernels/core/test_minimax_reduce_rms.py
  - tests/utils.py
  - csrc/libtorch_stable/minimax_reduce_rms_kernel.cu
  - csrc/libtorch_stable/minimax_reduce_rms_kernel.h
  - csrc/libtorch_stable/ops.h
  - csrc/libtorch_stable/torch_bindings.cpp
  - vllm/model_executor/layers/mamba/linear/minimax_linear_attn.py
  optional: true
  commands:
  - pytest -v -s kernels/core/test_minimax_reduce_rms.py

- label: ":amd: (MI300) Kimi K3" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/third_party/flash_linear_attention/ops/kda.py
  - vllm/third_party/flash_linear_attention/ops/chunk_delta_h.py
  - vllm/third_party/flash_linear_attention/ops/l2norm.py
  - vllm/models/kimi_k3/nvidia/kda.py
  - vllm/models/kimi_k3/nvidia/kda_metadata.py
  - vllm/models/kimi_k3/nvidia/ops/third_party/kda/
  - vllm/models/kimi_k3/amd/
  - CMakeLists.txt
  - csrc/libtorch_stable/kimi_k3/fused_kda_decode_kernel_rocm.cu
  - csrc/libtorch_stable/ops.h
  - csrc/libtorch_stable/torch_bindings.cpp
  - tests/models/kimi_k3/test_amd_attn_res.py
  - tests/models/kimi_k3/test_amd_kda_decode.py
  - tests/models/kimi_k3/test_kda.py
  - tests/models/kimi_k3/test_kda_metadata.py
  - vllm/_custom_ops.py
  - vllm/platforms/rocm.py
  commands:
  # Run the ROCm AttnRes/KDA decode counterparts plus existing KDA coverage.
  - pytest -v -s models/kimi_k3/test_amd_attn_res.py models/kimi_k3/test_amd_kda_decode.py models/kimi_k3/test_kda.py models/kimi_k3/test_kda_metadata.py

- label: ":amd: (MI355) Kimi K3" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_1
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/models/kimi_k3/
  - csrc/libtorch_stable/kimi_k3/
  - tests/models/kimi_k3/
  - vllm/platforms/rocm.py
  commands:
  # Mirrors the ":amd: (MI355) Kimi K3" step generated from
  # .buildkite/test_areas/models_basic.yaml. The ROCm KDA/AttnRes kernels are
  # gated on gfx950, so this has to be MI355. The two kernels/ files the NVIDIA
  # step runs are CUDA-only (they import cuda.bindings.driver), so they are
  # deliberately left out here.
  - pytest -v -s models/kimi_k3

- label: ":amd: (MI355) Kimi K3 Multi-GPU" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/models/kimi_k3/
  - tests/models/kimi_k3/
  - vllm/platforms/rocm.py
  commands:
  # The single-GPU step above skips every @multi_gpu_test case. `-m "distributed"`
  # selects exactly those: multi_gpu_test() applies pytest.mark.distributed
  # alongside its skipif (tests/utils.py:2080), so this picks up a new
  # multi-GPU test anywhere in the directory without editing this list.
  # Four GPUs run the two tp4 cases in test_amd_latent_moe_runner.py; the tp8
  # case and the 8/16-GPU cases in test_latent_moe_tail.py still skip.
  - pytest -v -s -m "distributed" models/kimi_k3

- label: ":amd: (MI300) Mamba Kernels" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/mamba/
  - tests/kernels/mamba
  - vllm/model_executor/layers/mamba/gdn
  - vllm/model_executor/layers/mamba/ops
  - vllm/third_party/flash_linear_attention/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s kernels/mamba

- label: ":amd: (MI300) DeepEP FP8 MoE Kernels" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/moe/
  - csrc/quantization/w8a8/cutlass/moe/
  - vllm/model_executor/layers/fused_moe/
  - tests/kernels/moe/test_deepep_moe.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/envs.py
  commands:
    - pytest -v -s kernels/moe/test_deepep_moe.py

#-----------------------------------------------------------  mi300 · lora  ------------------------------------------------------------#

- label: ":amd: (MI300) LoRA TP (Distributed) %N"
  timeout_in_minutes: 65
  parallelism: 4
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/lora
  - vllm/model_executor/layers/fused_moe/
  - tests/lora
  - vllm/platforms/rocm.py
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - case "$$BUILDKITE_PARALLEL_JOB" in 0|1|2|3) ;; *) echo "unexpected BUILDKITE_PARALLEL_JOB=$$BUILDKITE_PARALLEL_JOB" && exit 1;; esac
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "3" ]; then pytest -v -s -x lora/test_chatglm3_tp.py; fi
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "1" ]; then pytest -v -s -x lora/test_llama_tp.py; fi
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "3" ]; then pytest -v -s -x lora/test_qwen3_with_multi_loras.py; fi
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "2" ]; then pytest -v -s -x lora/test_olmoe_tp.py; fi
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "0" ]; then pytest -v -s -x lora/test_gptoss_tp.py; fi
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "1" ]; then pytest -v -s -x lora/test_qwen35_densemodel_lora.py; fi
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "2" ]; then pytest -v -s -x lora/test_gemma4_tp.py; fi

#------------------------------------------------------  mi300 · model_executor  -------------------------------------------------------#

- label: ":amd: (MI300) Model Executor" # TBD
  timeout_in_minutes: 75
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/engine/arg_utils.py
  - vllm/config/model.py
  - vllm/model_executor
  - tests/model_executor
  - tests/entrypoints/openai/completion/test_tensorizer_entrypoint.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/model_executor/warmup
  - tests/model_executor/test_jit_warmup.py
  - tests/model_executor/test_jit_warmup_cutedsl_launcher.py
  - tests/model_executor/test_jit_warmup_triton_launcher.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - export PYTHONFAULTHANDLER=1
  - pytest -v -s model_executor -m '(not slow_test)' --timeout=900 --timeout-method=thread
  - pytest -v -s entrypoints/openai/completion/test_tensorizer_entrypoint.py --timeout=900 --timeout-method=thread

#----------------------------------------------------  mi300 · model_runner_v2  -------------------------------------------------------#

- label: ":amd: (MI300) Model Runner V2 Core" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/worker/gpu/
  - vllm/v1/worker/gpu_worker.py
  - vllm/v1/core/sched/
  - vllm/v1/attention/
  - tests/v1/engine/test_llm_engine.py
  - tests/v1/e2e/
  - tests/entrypoints/llm/test_struct_output_generate.py
  - vllm/platforms/rocm.py
  commands:
  - set -x
  - export VLLM_USE_V2_MODEL_RUNNER=1
  - pytest -v -s v1/engine/test_llm_engine.py -k "not test_engine_metrics"
  - pytest -v -s v1/e2e/general/test_async_scheduling.py -k "not ngram"
  - pytest -v -s v1/e2e/general/test_context_length.py
  - pytest -v -s v1/e2e/general/test_min_tokens.py
  - pytest -v -s entrypoints/llm/test_struct_output_generate.py -k "xgrammar and not speculative_config6 and not speculative_config7 and not speculative_config8 and not speculative_config0"

- label: ":amd: (MI300) Model Runner V2 Examples" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/examples"
  source_file_dependencies:
  - vllm/v1/worker/gpu/
  - vllm/v1/core/sched/
  - vllm/v1/worker/gpu_worker.py
  - examples/basic/offline_inference/
  - examples/generate/multimodal/
  - examples/features/
  - examples/pooling/embed/vision_embedding_offline.py
  - examples/features/tensorize_vllm_model.py
  - examples/deployment/llm_engine_example.py
  - vllm/platforms/rocm.py
  commands:
  - set -x
  - export VLLM_USE_V2_MODEL_RUNNER=1
  - pip install tensorizer
  - python3 basic/offline_inference/chat.py
  - python3 basic/offline_inference/generate.py --model facebook/opt-125m
  - python3 generate/multimodal/audio_language_offline.py --seed 0
  - python3 generate/multimodal/vision_language_offline.py --seed 0
  - python3 generate/multimodal/vision_language_multi_image_offline.py --seed 0
  - python3 generate/multimodal/encoder_decoder_multimodal_offline.py --model-type whisper --seed 0
  - python3 pooling/embed/vision_embedding_offline.py --seed 0
  - python3 features/automatic_prefix_caching/prefix_caching_offline.py
  - python3 deployment/llm_engine_example.py
  - python3 features/tensorize_vllm_model.py --model facebook/opt-125m serialize --serialized-directory /tmp/ --suffix v1 && python3 features/tensorize_vllm_model.py --model facebook/opt-125m deserialize --path-to-tensors /tmp/vllm/facebook/opt-125m/v1/model.tensors
  - python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 2048
  - python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle3 --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 1536

- label: ":amd: (MI300) Model Runner V2 Distributed" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/worker/gpu/
  - vllm/v1/worker/gpu_worker.py
  - tests/basic_correctness/test_basic_correctness.py
  - tests/v1/distributed/test_async_llm_dp.py
  - tests/v1/distributed/test_eagle_dp.py
  - vllm/platforms/rocm.py
  commands:
  - set -x
  - export VLLM_USE_V2_MODEL_RUNNER=1
  - TARGET_TEST_SUITE=MI300 pytest -v -s basic_correctness/test_basic_correctness.py -m 'distributed(num_gpus=2)' -k "not ray and not True"
  - TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_async_llm_dp.py -k "not ray"
  - TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_eagle_dp.py

- label: ":amd: (MI300) Model Runner V2 Pipeline Parallelism" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/worker/gpu/
  - vllm/v1/worker/gpu_worker.py
  - tests/distributed/test_pipeline_parallel.py
  - tests/distributed/test_pp_cudagraph.py
  - vllm/platforms/rocm.py
  commands:
  - set -x
  - export VLLM_USE_V2_MODEL_RUNNER=1
  - pytest -v -s distributed/test_pipeline_parallel.py -k "not ray"
  - pytest -v -s distributed/test_pp_cudagraph.py -k "not ray"

- label: ":amd: (MI300) Model Runner V2 Spec Decode" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/worker/gpu/
  - vllm/v1/worker/gpu_worker.py
  - tests/v1/spec_decode/test_max_len.py
  - tests/v1/spec_decode/test_rejection_sampler_utils.py
  - tests/v1/spec_decode/test_synthetic_rejection_sampler_utils.py
  - tests/v1/e2e/spec_decode/
  - vllm/platforms/rocm.py
  commands:
  - set -x
  - export VLLM_USE_V2_MODEL_RUNNER=1
  - pytest -v -s v1/spec_decode/test_max_len.py -k "eagle or mtp"
  - pytest -v -s v1/spec_decode/test_rejection_sampler_utils.py
  - pytest -v -s v1/spec_decode/test_synthetic_rejection_sampler_utils.py
  - pytest -v -s v1/e2e/spec_decode/eagle/
  - pytest -v -s v1/e2e/spec_decode/speculators/
  - pytest -v -s v1/e2e/spec_decode/mtp/

#------------------------------------------------------  mi300 · models / basic  -------------------------------------------------------#

- label: ":amd: (MI300) Basic Models (Initialization)" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/test_initialization.py
  - tests/models/registry.py
  optional: true
  commands:
  - pytest -v -s models/test_initialization.py::test_can_initialize_small_subset

#-----------------------------------------------------  mi300 · models / language  -----------------------------------------------------#

- label: ":amd: (MI300) Language Models (MTEB)" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/language/pooling_mteb_test
  commands:
  - pytest -v -s models/language/pooling_mteb_test

- label: ":amd: (MI300) Language Models (Extended Pooling) Shard %N"
  timeout_in_minutes: 95
  parallelism: 4
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/language/pooling
  commands:
  - pytest -v -s models/language/pooling -m 'not core_model' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":amd: (MI300) Language Models (Extended Generation)" # TBD
  timeout_in_minutes: 95
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/language/generation
  commands:
  - MAMBA_FORCE_BUILD=TRUE uv pip install --system --no-build-isolation 'git+https://github.com/AndreasKaratzas/mamba@fix-rocm-7.0-warp-size-constexpr'
  - CAUSAL_CONV1D_FORCE_BUILD=TRUE uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
  - pytest -v -s models/language/generation -m '(not core_model) and (not hybrid_model)'

#----------------------------------------------------  mi300 · models / multimodal  ----------------------------------------------------#

- label: ":amd: (MI300) Multimodal Models (Extended Generation 1)" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal/generation
  - tests/models/multimodal/test_mapping.py
  commands:
  - pytest -v -s models/multimodal/generation -m 'not core_model' --ignore models/multimodal/generation/test_common.py
  - pytest -v -s models/multimodal/test_mapping.py

- label: ":amd: (MI300) Multimodal Models (Extended Generation 2) Shard %N"
  timeout_in_minutes: 100
  parallelism: 4
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal/generation
  commands:
  - pytest -v -s models/multimodal/generation/test_common.py -m 'split(group=0) and not core_model' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB


- label: ":amd: (MI300) Multimodal Models (Extended Generation 3)" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal/generation
  commands:
  - pytest -v -s models/multimodal/generation/test_common.py -m 'split(group=1) and not core_model'

- label: ":amd: (MI300) Multimodal Models (Standard) 3: llava + qwen2_vl" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal/generation
  - tests/models/multimodal/test_mapping.py
  - tests/models/multimodal
  commands:
  - pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "not qwen2 and not qwen3 and not gemma"
  - pytest -v -s models/multimodal/generation/test_qwen2_vl.py -m core_model

- label: ":amd: (MI300) Multimodal Processor Shard %N" # TBD
  timeout_in_minutes: 115
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  parallelism: 4
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal
  - tests/models/registry.py
  commands:
  - pytest -v -s models/multimodal/processing/test_tensor_schema.py --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":amd: (MI300) Multimodal Models (Extended Pooling)" # TBD
  timeout_in_minutes: 75
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal/pooling
  commands:
  - pytest -v -s models/multimodal/pooling -m 'not core_model'

#-----------------------------------------------------  mi300 · models / quantized  -----------------------------------------------------#

- label: ":amd: (MI300) Quantized Models" # TBD
  timeout_in_minutes: 80
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/layers/quantization
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/models/quantization
  - vllm/model_executor/model_loader/
  optional: true
  commands:
  - unset VLLM_USE_V2_MODEL_RUNNER
  - pytest -v -s models/quantization

#--------------------------------------------------  mi300 · models / transformers  ---------------------------------------------------#

# - label: ":amd: (MI300) Transformers Nightly Models (Initialization) Shard %N" # TBD
#   timeout_in_minutes: 65
#   mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
#   dind: false
#   agent_pool: mi300_1
#   parallelism: 6
#   optional: true
#   working_dir: "/vllm-workspace/"
#   source_file_dependencies:
#   - vllm/model_executor/models/
#   - vllm/model_executor/model_loader/
#   - vllm/multimodal/
#   - vllm/model_executor/layers/
#   - vllm/v1/attention/backends/
#   - vllm/v1/attention/selector.py
#   - vllm/_aiter_ops.py
#   - vllm/platforms/rocm.py
#   - tests/models/
#   commands:
#   - pip install --upgrade git+https://github.com/huggingface/transformers
#   - pytest -v -s tests/models/test_initialization.py --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

# - label: ":amd: (MI300) Transformers Nightly Models (Processing) Shard %N" # TBD
#   timeout_in_minutes: 145
#   mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
#   dind: false
#   agent_pool: mi300_1
#   parallelism: 8
#   optional: true
#   working_dir: "/vllm-workspace/"
#   source_file_dependencies:
#   - vllm/model_executor/models/
#   - vllm/model_executor/model_loader/
#   - vllm/multimodal/
#   - vllm/model_executor/layers/
#   - vllm/v1/attention/backends/
#   - vllm/v1/attention/selector.py
#   - vllm/_aiter_ops.py
#   - vllm/platforms/rocm.py
#   - tests/models/
#   commands:
#   - pip install --upgrade git+https://github.com/huggingface/transformers
#   - pytest -v -s tests/models/multimodal/processing/ --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

# - label: ":amd: (MI300) Transformers Nightly Models (Single)" # TBD
#   timeout_in_minutes: 50
#   mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
#   dind: false
#   agent_pool: mi300_1
#   optional: true
#   working_dir: "/vllm-workspace/"
#   source_file_dependencies:
#   - vllm/model_executor/models/
#   - vllm/model_executor/model_loader/
#   - vllm/multimodal/
#   - vllm/model_executor/layers/
#   - vllm/v1/attention/backends/
#   - vllm/v1/attention/selector.py
#   - vllm/_aiter_ops.py
#   - vllm/platforms/rocm.py
#   - tests/models/
#   - examples/
#   commands:
#   - pip install --upgrade git+https://github.com/huggingface/transformers
#   - pytest -v -s tests/models/transformers/test_backend.py
#   - pytest -v -s tests/models/multimodal/test_mapping.py
#   - python3 examples/basic/offline_inference/chat.py
#   - python3 examples/generate/multimodal/vision_language_offline.py --model-type qwen2_5_vl
#   - VLLM_WORKER_MULTIPROC_METHOD=spawn python3 examples/generate/multimodal/audio_language_offline.py --model-type whisper

#----------------------------------------------------------  mi300 · plugins  ----------------------------------------------------------#

- label: ":amd: (MI300) Plugin Integration" # TBD
  timeout_in_minutes: 50
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/plugins/
  - tests/plugins/
  - vllm/platforms/rocm.py
  commands:
  # BEGIN: platform plugin and general plugin tests, all the code in-between runs on dummy platform
  # END: platform plugin tests
  # BEGIN: `io_processor` plugins test, all the code in between uses the `prithvi_io_processor` plugin
  # END: `io_processor` plugins test
  # BEGIN: `bge_m3_sparse io_processor` test
  # END: `bge_m3_sparse io_processor` test
  # BEGIN: `colbert_query io_processor` test
  # END: `colbert_query io_processor` test
  # BEGIN: `stat_logger` plugins test
  # END: `stat_logger` plugins test
  # BEGIN: `endpoint` plugins test
  # END: `endpoint` plugins test
  # BEGIN: other tests
  - pip install -e ./plugins/vllm_add_dummy_platform
  - pytest -v -s plugins_tests/test_platform_plugins.py
  - pip uninstall vllm_add_dummy_platform -y
  - pytest -v -s ./plugins_tests/test_io_processor_plugins.py
  - pip install -e ./plugins/prithvi_io_processor_plugin
  - pytest -v -s plugins_tests/test_terratorch_io_processor_plugins.py
  - pip uninstall prithvi_io_processor_plugin -y
  - pip install -e ./plugins/bge_m3_sparse_plugin
  - pytest -v -s plugins_tests/test_bge_m3_sparse_io_processor_plugins.py
  - pip uninstall bge_m3_sparse_plugin -y
  - pip install -e ./plugins/colbert_query_plugin
  - pytest -v -s plugins_tests/test_colbert_query_io_processor_plugins.py
  - pip uninstall colbert_query_plugin -y
  - pip install -e ./plugins/vllm_add_dummy_stat_logger
  - pytest -v -s plugins_tests/test_stats_logger_plugins.py
  - pip uninstall dummy_stat_logger -y
  - pip install -e ./plugins/vllm_add_dummy_endpoint_plugin
  - pytest -v -s plugins_tests/test_endpoint_plugins.py
  - pip uninstall vllm_add_dummy_endpoint_plugin -y
  - pytest -v -s plugins_tests/test_scheduler_plugins.py
  - pip install -e ./plugins/vllm_add_dummy_model
  - pytest -v -s distributed/test_distributed_oot.py
  - pytest -v -s plugins_tests/test_oot_registration_online.py
  - pytest -v -s plugins_tests/test_oot_registration_offline.py
  - pytest -v -s plugins_tests/lora_resolvers

- label: ":amd: (MI300) GGUF Plugin" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/layers/quantization
  - tests/plugins_tests/test_gguf_plugin.py
  - tests/plugins_tests/gguf
  - vllm/platforms/rocm.py
  commands:
  - pip install "vllm-gguf-plugin >= 0.0.2"
  - pytest -v -s plugins_tests/gguf

#-------------------------------------------------------  mi300 · rust_frontend  -------------------------------------------------------#

- label: ":amd: (MI300) Rust Frontend OpenAI Coverage" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - rust/
  - vllm/benchmarks/
  - vllm/entrypoints/openai/
  - vllm/entrypoints/serve/
  - vllm/v1/sample/
  - tests/utils.py
  - tests/benchmarks/test_serve_cli.py
  - tests/entrypoints/openai/chat_completion/test_chat_completion.py
  - tests/entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py
  - tests/entrypoints/launchers/test_shutdown.py
  - tests/entrypoints/openai/test_return_token_ids.py
  - tests/entrypoints/serve/instrumentator/test_uds.py
  - tests/v1/sample/test_logprobs_e2e.py
  - vllm/platforms/rocm.py
  - tests/entrypoints/openai/chat_completion/test_include_reasoning.py
  - tests/entrypoints/openai/chat_completion/test_serving_chat.py
  - tests/entrypoints/openai/completion/test_shutdown.py
  - tests/entrypoints/openai/test_uds.py
  commands:
  - export VLLM_USE_RUST_FRONTEND=1
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s benchmarks/test_serve_cli.py -k "not insecure and not (test_bench_serve and not test_bench_serve_chat)"
  - pytest -v -s entrypoints/openai/chat_completion/test_chat_completion.py -k "not test_invalid_json_schema and not test_invalid_regex and not test_kv_transfer_prompt_token_ids_round_trip and not test_kv_transfer_prompt_token_ids_streaming"
  - pytest -v -s entrypoints/openai/chat_completion/test_include_reasoning.py
  - pytest -v -s entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py -k "not multiple"
  - pytest -v -s entrypoints/openai/chat_completion/test_serving_chat.py -k "test_chat_per_request_metrics_follow_server_flag or test_streaming_reasoning_usage_counts_across_deltas or test_completion_tokens_details"
  - pytest -v -s entrypoints/launchers/test_shutdown.py -k "not engine_failure and not test_abort_timeout_exits_quickly"
  - pytest -v -s entrypoints/openai/test_return_token_ids.py -k "not test_comparison"
  - pytest -v -s entrypoints/serve/instrumentator/test_uds.py
  - pytest -v -s v1/sample/test_logprobs_e2e.py -k "test_prompt_logprobs_e2e_server"

- label: ":amd: (MI300) Rust Frontend Serve-Admin Coverage"
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - rust/
  - vllm/entrypoints/openai/
  - vllm/entrypoints/serve/
  - vllm/v1/engine/
  - tests/utils.py
  - tests/entrypoints/serve/dev/rpc/test_collective_rpc.py
  - tests/entrypoints/scale_out/token_in_token_out/test_serving_tokens.py
  - tests/entrypoints/serve/instrumentator/test_basic.py
  - tests/entrypoints/serve/instrumentator/test_metrics.py
  - tests/entrypoints/serve/tokenize/test_tokenization.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_USE_RUST_FRONTEND=1
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - PYTHONPATH=/vllm-workspace pytest -v -s entrypoints/serve/dev/rpc/test_collective_rpc.py
  - pytest -v -s entrypoints/serve/instrumentator/test_basic.py -k "not server_load"
  - pytest -v -s entrypoints/scale_out/token_in_token_out/test_serving_tokens.py -k "not stream and not lora and not test_generate_logprobs and not stop_string_workflow"
  - pytest -v -s entrypoints/serve/instrumentator/test_metrics.py -k "text and not show and not run_batch and not test_metrics_counts and not test_metrics_exist"
  - pytest -v -s entrypoints/serve/tokenize/test_tokenization.py -k "not tokenizer_info"

- label: ":amd: (MI300) Rust Frontend Core Correctness" # TBD
  timeout_in_minutes: 25
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - rust/
  - vllm/entrypoints/openai/
  - tests/utils.py
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_USE_RUST_FRONTEND=1
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine

- label: ":amd: (MI300) Rust Frontend Tool Use" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - rust/
  - vllm/entrypoints/openai/
  - vllm/tool_parsers/
  - tests/utils.py
  - tests/tool_use/
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_USE_RUST_FRONTEND=1
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s tool_use --ignore=tool_use/mistral --models llama3.2 -k "not test_response_format_with_tool_choice_required and not test_parallel_tool_calls_false and not test_tool_call_and_choice"

- label: ":amd: (MI300) Rust Frontend Distributed" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - rust/
  - vllm/distributed/
  - vllm/engine/
  - vllm/executor/
  - vllm/v1/engine/
  - vllm/v1/worker/
  - tests/utils.py
  - tests/v1/distributed/test_dense_dp_world_size.py
  - tests/v1/distributed/test_external_lb_dp.py
  - tests/v1/distributed/test_hybrid_lb_dp.py
  - tests/v1/distributed/test_internal_lb_dp.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_USE_RUST_FRONTEND=1
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - TP_SIZE=1 DP_SIZE=4 pytest -v -s v1/distributed/test_dense_dp_world_size.py
  - VLLM_ENGINE_READY_TIMEOUT_S=1800 TP_SIZE=1 DP_SIZE=4 pytest -v -s v1/distributed/test_internal_lb_dp.py -k "not 4 and not server_info"
  - TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_external_lb_dp.py -k "not 4 and not server_info"
  - TP_SIZE=1 DP_SIZE=4 pytest -v -s v1/distributed/test_hybrid_lb_dp.py -k "not 4 and not server_info"

#-------------------------------------------------------  mi300 · quantization  --------------------------------------------------------#

- label: ":amd: (MI300) Quantization Shard %N"
  timeout_in_minutes: 115
  parallelism: 4
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/model_executor/layers/fused_moe/oracle/unquantized.py
  - vllm/model_executor/layers/fused_moe/unquantized_fused_moe_method.py
  - tests/quantization
  - tests/rocm/test_moe_weight_replay.py
  optional: true
  commands:
  # temporary install here since we need nightly, will move to requirements/test.in
  # after torchao 0.12 release, and pin a working version of torchao nightly here
  # since torchao nightly is only compatible with torch nightly currently
  # https://github.com/pytorch/ao/issues/2919, we'll have to skip new torchao tests for now
  # we can only upgrade after this is resolved
  # TODO(jerryzh168): resolve the above comment
  - unset VLLM_USE_V2_MODEL_RUNNER
  - uv pip install --system torchao==0.18.0
  - uv pip install --system conch-triton-kernels
  - VLLM_TEST_FORCE_LOAD_FORMAT=auto pytest -v -s quantization/ --ignore quantization/test_blackwell_moe.py --ignore quantization/test_rocm_moe.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "0" ]; then pytest -v -s rocm/test_moe_weight_replay.py; fi

- label: ":amd: (MI300) Quantized Fusions" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - tests/fusion
  - vllm/model_executor/layers/fusion
  - vllm/model_executor/kernels/linear
  - vllm/model_executor/layers/quantization/compressed_tensors
  - vllm/model_executor/layers/quantization/modelopt.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s fusion/

#---------------------------------------------------------  mi300 · samplers  ----------------------------------------------------------#

- label: ":amd: (MI300) Samplers" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/layers
  - vllm/sampling_metadata.py
  - vllm/v1/sample/
  - vllm/entrypoints/generate/beam_search/
  - tests/samplers
  - tests/conftest.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/entrypoints/generate/beam_search
  commands:
  - pytest -v -s samplers -k "not test_beam_search_passes_multimodal_data"

#------------------------------------------------------------  mi300 · misc  ------------------------------------------------------------#

- label: ":amd: (MI300) Regression" # TBD
  timeout_in_minutes: 25
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/test_regression
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/inputs/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  optional: true
  commands:
  - pip install 'modelscope<1.38'
  - pytest -v -s test_regression.py

#---------------------------------------------------------  mi300 · ray_compat  ---------------------------------------------------------#

- label: ":amd: (MI300) Ray Dependency Compatibility Check" # TBD
  timeout_in_minutes: 25
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  working_dir: "/"
  source_file_dependencies:
  - requirements/
  - setup.py
  - vllm/platforms/rocm.py
  - .buildkite/scripts/check-ray-compatibility.sh
  optional: true
  commands:
  - bash /vllm-workspace/.buildkite/scripts/check-ray-compatibility.sh

#------------------------------------------------------------  mi300 · v1  -------------------------------------------------------------#

- label: ":amd: (MI300) Spec Decode Speculators + MTP" # TBD
  timeout_in_minutes: 75
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/spec_decode/
  - vllm/v1/worker/gpu/spec_decode/
  - vllm/model_executor/model_loader/
  - vllm/v1/sample/
  - vllm/model_executor/layers/
  - vllm/transformers_utils/configs/speculators/
  - tests/v1/e2e/spec_decode/
  - vllm/platforms/rocm.py
  - vllm/v1/attention/backends/
  commands:
    - pytest -v -s v1/e2e/spec_decode/speculators/
    - pytest -v -s v1/e2e/spec_decode/mtp/

- label: ":amd: (MI300) V1 Spec Decode" # TBD
  timeout_in_minutes: 50
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/v1/spec_decode
  - vllm/config/
  - vllm/distributed/
  - vllm/inputs/
  - vllm/model_executor/
  - vllm/models/qwen4_exp/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s -m 'not slow_test' v1/spec_decode

- label: ":amd: (MI300) Acceptance Length (Large Models)" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/spec_decode/
  - vllm/model_executor/models/mlp_speculator.py
  - tests/v1/spec_decode/test_acceptance_length.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_ALLOW_INSECURE_SERIALIZATION=1
  - pytest -v -s v1/spec_decode/test_acceptance_length.py -m slow_test

- label: ":amd: (MI300) E2E Scheduling" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/
  - tests/v1/e2e/general/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/e2e/general/test_async_scheduling.py

- label: ":amd: (MI300) V1 Engine" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/engine/
  - tests/v1/engine/
  - tests/v1/test_tensor_ipc_queue.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/engine/test_preprocess_error_handling.py
  - pytest -v -s v1/engine --ignore v1/engine/test_preprocess_error_handling.py
  - pytest -v -s v1/test_tensor_ipc_queue.py

- label: ":amd: (MI300) Speculators Correctness Nightly"
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/config/speculative.py
  - vllm/engine/arg_utils.py
  - vllm/transformers_utils/config.py
  - vllm/transformers_utils/configs/speculators/
  - vllm/v1/spec_decode/
  - vllm/v1/worker/gpu/spec_decode/
  - vllm/v1/worker/gpu_model_runner.py
  - vllm/v1/sample/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/
  - vllm/model_executor/models/llama_eagle3.py
  - vllm/model_executor/models/qwen3.py
  - vllm/model_executor/models/qwen3_dflash.py
  - vllm/model_executor/models/registry.py
  - vllm/_aiter_ops.py
  - tests/evals/gsm8k/
  - tests/v1/spec_decode/test_speculators_correctness.py
  - vllm/platforms/rocm.py
  - vllm/v1/worker/gpu/spec_decode/dflash/
  commands:
  - export VLLM_ALLOW_INSECURE_SERIALIZATION=1
  - pytest -v -s v1/spec_decode/test_speculators_correctness.py -m slow_test

- label: ":amd: (MI300) Extract Hidden States Integration" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  working_dir: /vllm-workspace
  source_file_dependencies:
  - vllm/config/speculative.py
  - vllm/distributed/kv_transfer/kv_connector/
  - vllm/model_executor/layers/attention/
  - vllm/model_executor/layers/mamba/
  - vllm/model_executor/model_loader/
  - vllm/model_executor/models/extract_hidden_states.py
  - vllm/model_executor/models/llama.py
  - vllm/model_executor/models/qwen3_5.py
  - vllm/model_executor/models/qwen3_next.py
  - vllm/model_executor/models/registry.py
  - vllm/transformers_utils/configs/extract_hidden_states.py
  - vllm/transformers_utils/configs/qwen3_5.py
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/v1/kv_cache_interface.py
  - vllm/v1/spec_decode/extract_hidden_states.py
  - vllm/v1/worker/gpu_model_runner.py
  - vllm/_aiter_ops.py
  - tests/v1/kv_connector/extract_hidden_states_integration/
  - vllm/platforms/rocm.py
  - vllm/distributed/kv_transfer/kv_connector/v1/example_hidden_states_connector.py
  - tests/v1/kv_connector/extract_hidden_states_integration
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s -m 'distributed' tests/v1/kv_connector/extract_hidden_states_integration

- label: ":amd: (MI300) V1 Attention Shard %N" # TBD
  timeout_in_minutes: 125
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_1
  parallelism: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/config/attention.py
  - vllm/model_executor/layers/attention
  - vllm/v1/attention
  - tests/v1/attention
  - vllm/_aiter_ops.py
  - vllm/envs.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/attention --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

- label: ":amd: (MI300) Metrics, Tracing" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  optional: true
  num_gpus: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/v1/tracing
  - tests/tracing/
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/inputs/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/tracing/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - "pip install \
      'opentelemetry-sdk>=1.26.0' \
      'opentelemetry-api>=1.26.0' \
      'opentelemetry-exporter-otlp>=1.26.0' \
      'opentelemetry-semantic-conventions-ai>=0.4.1'"
  - pytest -v -s v1/tracing
  - pytest -v -s tracing

- label: ":amd: (MI300) V1 E2E" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/v1/e2e
  - vllm/compilation/
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/envs.py
  - vllm/forward_context.py
  - vllm/inputs/
  - vllm/logger.py
  - vllm/logging_utils/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/transformers_utils/
  - vllm/triton_utils/
  - vllm/utils/
  - vllm/v1/
  - tests/v1/e2e/spec_decode/
  commands:
    - >-
      pytest -v -s
      v1/e2e/spec_decode/draft_model/test_draft_model.py::test_draft_model_tensor_parallelism
      v1/e2e/spec_decode/draft_model/test_draft_model.py::test_draft_model_engine_args_tensor_parallelism

- label: ":amd: (MI300) Mooncake EC TCP E2E"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace"
  source_file_dependencies:
  - vllm/distributed/ec_transfer/
  - vllm/config/ec_transfer.py
  - vllm/config/multimodal.py
  - vllm/config/vllm.py
  - vllm/multimodal/
  - vllm/v1/core/encoder_cache_manager.py
  - vllm/v1/core/sched/
  - vllm/v1/engine/
  - vllm/v1/worker/ec_connector_model_runner_mixin.py
  - vllm/v1/worker/gpu_model_runner.py
  - vllm/v1/worker/gpu_worker.py
  - vllm/v1/worker/gpu/ec_connector.py
  - vllm/v1/worker/gpu/model_runner.py
  - vllm/v1/worker/mm_encoder_model_runner.py
  - vllm/v1/worker/encoder_cudagraph.py
  - vllm/v1/worker/encoder_cudagraph_defs.py
  - vllm/v1/attention/ops/vit_attn_wrappers.py
  - vllm/v1/worker/gpu/model_states/
  - vllm/v1/worker/gpu/mm/
  - tests/v1/cudagraph/test_encoder_cudagraph.py
  - tests/models/multimodal/generation/test_vit_cudagraph.py
  - examples/disaggregated/disaggregated_encoder/
  - tests/v1/ec_connector/
  - requirements/kv_connectors.txt
  - vllm/platforms/rocm.py
  - requirements/rocm.txt
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - export PYTHON_BIN='python'
  - export PYTHONUNBUFFERED='1'
  - export VLLM_HOST_IP='127.0.0.1'
  - export VLLM_USE_V2_MODEL_RUNNER='1'
  - export E_CUDAGRAPH_MM_ENCODER='1'
  - export E_MM_ENCODER_ATTN_BACKEND='TRITON_ATTN'
  - export MOONCAKE_EC_PROTOCOL='tcp'
  - export USE_MM_PROMPTS='1'
  - export SKIP_BASELINE='0'
  - export CONCURRENCY='3'
  - export REPEAT='2'
  - export LOG_PATH='/tmp/mooncake-ec-e2e'
  - export BASELINE_FILE='/tmp/mooncake-ec-e2e/baseline.json'
  - bash tests/v1/ec_connector/integration/run_epd_mooncake_ec_full_pipeline.sh

- label: ":amd: (MI300) NixlConnector PD edge case" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - vllm/v1/core/sched/
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - PREFILL_GPU_ID=0 DECODE_GPU_ID=1 ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/run_edge_case_test.sh

- label: ":amd: (MI300) CrossLayer KV layout Distributed NixlConnector PD accuracy" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - CROSS_LAYERS_BLOCKS=True ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh

- label: ":amd: (MI300) Distributed DP Extended" # TBD
  timeout_in_minutes: 75
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/
  - tests/v1/distributed
  - tests/v1/engine/test_engine_core_client.py
  - tests/distributed/test_utils
  - vllm/platforms/rocm.py
  commands:
  - export NCCL_CUMEM_HOST_ENABLE=0
  - TP_SIZE=2 DP_SIZE=2 pytest -v -s v1/distributed/test_async_llm_dp.py
  - TP_SIZE=2 DP_SIZE=2 pytest -v -s v1/distributed/test_eagle_dp.py
  - TP_SIZE=2 DP_SIZE=2 pytest -v -s v1/distributed/test_external_lb_dp.py
  - TP_SIZE=1 DP_SIZE=4 pytest -v -s v1/distributed/test_dense_dp_world_size.py
  - TP_SIZE=1 DP_SIZE=4 pytest -v -s v1/distributed/test_internal_lb_dp.py
  - TP_SIZE=1 DP_SIZE=4 pytest -v -s v1/distributed/test_hybrid_lb_dp.py
  - pytest -v -s v1/engine/test_engine_core_client.py::test_kv_cache_events_dp
  - pytest -v -s distributed/test_utils.py

- label: ":amd: (MI300) Distributed NixlConnector PD accuracy" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh

- label: ":amd: (MI300) Push NixlConnector PP prefill PD accuracy" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - tests/v1/kv_connector/nixl_integration/
  - tests/v1/kv_connector/nixl_push_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_push_integration/config_sweep_accuracy_test.sh

- label: ":amd: (MI300) DP EP Distributed NixlConnector PD accuracy" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - DP_EP=1 ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh

- label: ":amd: (MI300) Hybrid SSM NixlConnector PD accuracy" # TBD
  timeout_in_minutes: 80
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - HYBRID_SSM=1 ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh

- label: ":amd: (MI300) Hybrid SSM NixlConnector PD prefix cache" # TBD
  timeout_in_minutes: 25
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - vllm/v1/core/sched/
  - vllm/v1/core/kv_cache_coordinator.py
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/run_mamba_prefix_cache_test.sh

- label: ":amd: (MI300) MultiConnector (Nixl+Offloading) PD accuracy" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py
  - vllm/distributed/kv_transfer/kv_connector/v1/offloading_connector.py
  - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/run_multi_connector_accuracy_test.sh

- label: ":amd: (MI300) MultiConnector (Nixl+Offloading) PD edge cases" # TBD
  timeout_in_minutes: 25
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py
  - vllm/distributed/kv_transfer/kv_connector/v1/offloading_connector.py
  - vllm/distributed/kv_transfer/kv_connector/v1/offloading/
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/run_multi_connector_edge_case_test.sh

- label: ":amd: (MI300) V1 E2E Hybrid Chunked Prefill" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx942nightly, amdmi300]
  dind: false
  agent_pool: mi300_4
  num_gpus: 4
  optional: true
  source_file_dependencies:
  - vllm/v1/attention/backends/utils.py
  - vllm/v1/worker/gpu_model_runner.py
  - tests/v1/e2e/test_hybrid_chunked_prefill.py
  - vllm/v1/core/
  working_dir: /vllm-workspace/tests
  commands:
    - pytest -v -s v1/e2e/test_hybrid_chunked_prefill.py

#########################################################################################################################################
#                                                                                                                                       #
#                                                         MI355 (gfx950) tests                                                          #
#                                                                                                                                       #
#########################################################################################################################################

#-----------------------------------------------------  mi355 · basic_correctness  -----------------------------------------------------#

- label: ":amd: (MI355 DPX) Basic Correctness Models %N" # TBD
  timeout_in_minutes: 60
  parallelism: 2
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/basic_correctness/
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - VLLM_TARGET_TEST_SUITE=MI355 pytest -v -s basic_correctness/models --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":amd: (MI355 DPX) Basic Correctness CuMem" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/basic_correctness/
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s basic_correctness/memory/cumem

- label: ":amd: (MI355 DPX) Basic Correctness Sleep Mode" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/basic_correctness/
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s basic_correctness/memory/sleep_mode

- label: ":amd: (MI355 DPX) Basic Correctness CPU Offload" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/basic_correctness/
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s basic_correctness/cpu_offload

- label: ":amd: (MI355 DPX) Basic Correctness Prefetch Offload" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/basic_correctness/
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s basic_correctness/prefetch_offload

#--------------------------------------------------------  mi355 · benchmarks  ---------------------------------------------------------#

- label: ":amd: (MI355 DPX) Benchmarks CLI" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/benchmarks/
  - rust/src/bench/tests/python_serve_flags.txt
  commands:
  - pytest -v -s benchmarks/

- label: ":amd: (MI355) Attention Benchmark Smoke" # TBD
  timeout_in_minutes: 25
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - benchmarks/attention_benchmarks/
  - vllm/v1/attention/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - python3 benchmarks/attention_benchmarks/benchmark.py --backends ROCM_ATTN ROCM_AITER_FA ROCM_AITER_UNIFIED_ATTN --batch-specs "8q1s1k"

#----------------------------------------------------------  mi355 · compile  ----------------------------------------------------------#

- label: ":amd: (MI355 DPX) PyTorch Compilation H100 Cases"
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/__init__.py
  - vllm/_aiter_ops.py
  - vllm/_custom_ops.py
  - vllm/compilation/
  - vllm/config/
  - vllm/distributed/
  - "!vllm/distributed/kv_transfer/"
  - vllm/engine/
  - vllm/env_override.py
  - vllm/envs.py
  - vllm/forward_context.py
  - vllm/inputs/
  - vllm/ir/
  - vllm/kernels/
  - vllm/logger.py
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/platforms/
  - vllm/plugins/
  - vllm/sampling_params.py
  - vllm/sequence.py
  - vllm/transformers_utils/
  - vllm/triton_utils/
  - vllm/utils/
  - vllm/v1/
  - tests/compile/h100/
  - tests/utils.py
  - csrc/
  commands:
  - pytest -v -s compile/h100/test_startup.py

- label: ":amd: (MI355) Fusion and Compile FP8"
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - vllm/compilation/
  - vllm/model_executor/layers/
  - vllm/v1/attention/
  - vllm/distributed/
  - vllm/_aiter_ops.py
  - tests/compile/passes/
  - tests/compile/correctness_e2e/test_attn_quant.py
  - vllm/model_executor/kernels/linear/
  - vllm/model_executor/models/llama.py
  commands:
  - amd-smi
  - pytest -v -s tests/compile/passes/test_fusion_attn.py tests/compile/passes/test_silu_mul_quant_fusion.py tests/compile/passes/test_silu_mul_quant_manual_fusion.py
  - pytest -v -s tests/compile/correctness_e2e/test_attn_quant.py

- label: ":amd: (MI355) Sequence Parallel Correctness"
  timeout_in_minutes: 180
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - .buildkite/test_areas/compile.yaml
  - vllm/compilation/
  - vllm/config/
  - vllm/distributed/
  - vllm/model_executor/kernels/linear/
  - vllm/model_executor/layers/
  - vllm/model_executor/models/llama.py
  - vllm/v1/attention/
  - vllm/v1/worker/
  - vllm/v1/cudagraph_dispatcher.py
  - vllm/platforms/rocm.py
  - tests/compile/correctness_e2e/test_sequence_parallel.py
  - tests/models/registry.py
  - tests/utils.py
  commands:
  - export VLLM_TEST_CLEAN_GPU_MEMORY=1
  - pytest -v -s tests/compile/correctness_e2e/test_sequence_parallel.py

- label: ":amd: (MI355) AsyncTP Correctness"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - .buildkite/test_areas/compile.yaml
  - vllm/compilation/
  - vllm/config/
  - vllm/distributed/
  - vllm/model_executor/layers/
  - vllm/model_executor/kernels/linear/
  - vllm/model_executor/models/llama.py
  - vllm/model_executor/models/qwen3.py
  - vllm/v1/attention/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/transformers_utils/repo_utils.py
  - tests/compile/correctness_e2e/test_async_tp.py
  - tests/compile/fusions_e2e/common.py
  - tests/models/registry.py
  - tests/utils.py
  commands:
  - export VLLM_TEST_CLEAN_GPU_MEMORY=1
  - pytest -v -s tests/compile/correctness_e2e/test_async_tp.py::test_rocm_async_tp_bf16_output_correctness
  - pytest -v -s tests/compile/correctness_e2e/test_async_tp.py::test_async_tp_pass_correctness -k RedHatAI
  - pytest -v -s tests/compile/correctness_e2e/test_async_tp.py::test_async_tp_pass_nvfp4_correctness

- label: ":amd: (MI355 DPX) Fusion E2E FP8 Config Sweep"
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - .buildkite/test_areas/compile.yaml
  - csrc/quantization/
  - vllm/compilation/
  - vllm/model_executor/
  - vllm/v1/attention/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/compile/fusions_e2e/
  commands:
  - rocm-smi
  - pytest -v -s tests/compile/fusions_e2e/test_tp1_quant.py -k "test_tp1_fp8_fusions and inductor_partition and not +rms_norm and not True and ((Llama-3 and TRITON_ATTN) or (Qwen3-30B and +quant_fp8 and TRITON_ATTN) or (DeepSeek-Coder and +quant_fp8 and TRITON_MLA))"

#--------------------------------------------------------  mi355 · distributed  --------------------------------------------------------#

- label: ":amd: (MI355 DPX) EPLB Algorithm" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/eplb
  - tests/distributed/test_eplb_algo.py
  - tests/distributed/test_eplb_utils.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s distributed/test_eplb_algo.py
  - pytest -v -s distributed/test_eplb_utils.py

- label: ":amd: (MI355 DPX) Sharded RDT Weight Transfer"
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/weight_transfer/
  - tests/distributed/test_sharded_rdt_plan.py
  - tests/distributed/test_sharded_rdt_producer.py
  - tests/distributed/test_sharded_rdt_trainer.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - uv pip install --system --no-deps 'ray==2.56.1'
  - pytest -v -s distributed/test_sharded_rdt_plan.py
  - pytest -v -s distributed/test_sharded_rdt_producer.py
  - pytest -v -s distributed/test_sharded_rdt_trainer.py

- label: ":amd: (MI355) Distributed DP Basic" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  optional: true
  num_gpus: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/
  - vllm/engine/
  - vllm/executor/
  - vllm/worker/worker_base.py
  - vllm/v1/engine/
  - vllm/v1/worker/
  - tests/v1/distributed
  - tests/entrypoints/launchers/api_server/test_multi_api_servers.py
  - vllm/platforms/rocm.py
  - tests/v1/e2e/general/test_sharded_sampling.py
  - tests/v1/e2e/spec_decode/test_sharded_sampling.py
  - tests/v1/e2e/spec_decode/utils.py
  commands:
  - export NCCL_CUMEM_HOST_ENABLE=0
  - TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_async_llm_dp.py
  - TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_eagle_dp.py
  - TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_external_lb_dp.py
  - DP_SIZE=2 pytest -v -s entrypoints/launchers/api_server/test_multi_api_servers.py
  - pytest -v -s v1/e2e/general/test_sharded_sampling.py
  - pytest -v -s v1/e2e/spec_decode/test_sharded_sampling.py

- label: ":amd: (MI355) MoRI EP Healthy Baseline"
  timeout_in_minutes: 10
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/moe/
  - csrc/rocm/
  - tests/kernels/moe/
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/
  - vllm/distributed/
  - vllm/config/
  - vllm/forward_context.py
  - vllm/v1/worker/workspace.py
  - vllm/utils/import_utils.py
  - vllm/utils/math_utils.py
  - vllm/utils/torch_utils.py
  - vllm/platforms/
  - vllm/_aiter_ops.py
  optional: true
  # TODO: Add fault injection, detection and recovery once MoRI supports them.
  # This job currently checks healthy EP execution only.
  commands:
  - VLLM_WORKER_MULTIPROC_METHOD=spawn pytest -v -s -rs kernels/moe/test_moe_layer.py::test_moe_layer_mori_graph
  - VLLM_TEST_ENABLE_MORI_MOE_LAYER=1 VLLM_WORKER_MULTIPROC_METHOD=spawn pytest -v -s -rs 'kernels/moe/test_moe_layer.py::test_moe_layer[False-mori_high_throughput-2-1-True]' --subtests='[32-512-512-8-2-bfloat16-None-False-False-False-False-mori_high_throughput-2-2-1]'

- label: ":amd: (MI355) NixlConnector PD + Spec Decode acceptance" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - vllm/v1/worker/kv_connector_model_runner_mixin.py
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - KV_CACHE_MEMORY_BYTES=8G ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_spec_decode_test.sh

- label: ":amd: (MI355) Distributed NixlConnector PD accuracy" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh

- label: ":amd: (MI355) Distributed AITER NixlConnector PD accuracy"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - tests/v1/kv_connector/nixl_integration/
  - tests/v1/kv_connector/rocm_pd_accuracy_utils.py
  - tests/utils.py
  - vllm/v1/attention/backends/rocm_aiter_unified_attn.py
  - vllm/platforms/rocm.py
  - requirements/common.txt
  - requirements/kv_connectors_rocm.txt
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - pytest -v -s v1/kv_connector/nixl_integration/test_accuracy_rocm.py

- label: ":amd: (MI355) Distributed MooncakeConnector HIP PD accuracy"
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/mooncake/
  - tests/v1/kv_connector/mooncake_integration/
  - tests/v1/kv_connector/rocm_pd_accuracy_utils.py
  - tests/utils.py
  - examples/disaggregated/mooncake_connector/
  - vllm/v1/attention/backends/rocm_aiter_unified_attn.py
  - vllm/platforms/rocm.py
  - requirements/common.txt
  - requirements/kv_connectors_rocm.txt
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - pytest -v -s v1/kv_connector/mooncake_integration/test_accuracy_rocm.py

- label: ":amd: (MI355) DP EP Distributed NixlConnector PD accuracy" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - tests/v1/kv_connector/nixl_integration/
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - DP_EP=1 ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh

- label: ":amd: (MI355) Kimi-Linear-48B-A3B Disaggregated DP EP"
  timeout_in_minutes: 105
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/platforms/rocm.py
  - vllm/models/kimi_k3/
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - tests/v1/kv_connector/nixl_integration/
  - requirements/kv_connectors_rocm.txt
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - export ENABLE_HMA_FLAG='1'
  - export DP_EP='1'
  - export GPU_MEMORY_UTILIZATION='0.9'
  - export VLLM_SSM_CONV_STATE_LAYOUT='DS'
  - export PREFILLER_TP_SIZE='2'
  - export DECODER_TP_SIZE='2'
  - export MODEL_NAMES='moonshotai/Kimi-Linear-48B-A3B-Instruct'
  - export PREFILL_BLOCK_SIZE='2048'
  - export DECODE_BLOCK_SIZE='2048'
  - export VLLM_SERVE_EXTRA_ARGS='--trust-remote-code,--mamba-cache-mode align,--max-model-len 32768'
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  - bash v1/kv_connector/nixl_integration/run_accuracy_test.sh

- label: ":amd: (MI355) Distributed Features" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - csrc/custom_quickreduce.cu
  - csrc/ops.h
  - csrc/torch_bindings.cpp
  - vllm/distributed/
  - vllm/model_executor/layers/
  - vllm/entrypoints/llm.py
  - vllm/config/parallel.py
  - vllm/model_executor/layers/fused_moe/
  - vllm/v1/engine/
  - vllm/v1/executor/
  - vllm/v1/worker/
  - vllm/v1/distributed/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/_custom_ops.py
  - vllm/platforms/rocm.py
  - vllm/envs.py
  - examples/offline_inference/data_parallel.py
  - tests/distributed/test_context_parallel.py
  - tests/distributed/test_rocm_aiter_custom_ar.py
  - tests/distributed/test_rocm_quick_reduce.py
  - tests/distributed/test_quick_all_reduce.py
  - tests/v1/e2e/general/test_rocm_aiter_custom_ar.py
  - tests/v1/distributed/test_dbo.py
  - tests/utils.py
  - examples/features/data_parallel/data_parallel_offline.py
  - examples/rl/rlhf_async_new_apis.py
  - tests/distributed/test_weight_transfer.py
  - tests/distributed/test_packed_tensor.py
  commands:
  - pytest -v -s tests/distributed/test_context_parallel.py
  - VLLM_ALLOW_INSECURE_SERIALIZATION=1 python3 examples/rl/rlhf_async_new_apis.py
  - VLLM_LOGGING_LEVEL=DEBUG python3 examples/features/data_parallel/data_parallel_offline.py --model=Qwen/Qwen1.5-MoE-A2.7B -tp=1 -dp=2 --max-model-len=2048 --all2all-backend=deepep_high_throughput
  - VLLM_LOGGING_LEVEL=DEBUG python3 examples/features/data_parallel/data_parallel_offline.py --model=Qwen/Qwen1.5-MoE-A2.7B -tp=1 -dp=2 --max-model-len=2048 --all2all-backend=allgather_reducescatter --disable-nccl-for-dp-synchronization
  - pytest -v -s tests/v1/distributed/test_dbo.py
  - VLLM_ALLOW_INSECURE_SERIALIZATION=1 pytest -v -s tests/distributed/test_weight_transfer.py
  - pytest -v -s tests/distributed/test_packed_tensor.py
  - pytest -v -s tests/distributed/test_rocm_aiter_custom_ar.py
  - pytest -v -s tests/v1/e2e/general/test_rocm_aiter_custom_ar.py
  - pytest -v -s tests/distributed/test_rocm_quick_reduce.py
  - pytest -v -s tests/distributed/test_quick_all_reduce.py

- label: ":amd: (MI355) Distributed AgRs All2All"
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - .buildkite/test_areas/distributed.yaml
  - vllm/distributed/
  - vllm/config/parallel.py
  - vllm/config/vllm.py
  - vllm/forward_context.py
  - vllm/platforms/rocm.py
  - tests/distributed/test_mnnvl_alltoall.py
  - tests/utils.py
  commands:
  - pytest -v -s tests/distributed/test_mnnvl_alltoall.py::test_args_dispatch_combine

- label: ":amd: (MI355) Fusion E2E TP2 Quick" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - csrc/quantization/
  - vllm/compilation/
  - vllm/distributed/
  - vllm/model_executor/
  - vllm/v1/attention/
  - tests/compile/fusions_e2e/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_TEST_CLEAN_GPU_MEMORY=1
  - >-
    VLLM_ROCM_USE_AITER=1
    VLLM_ROCM_USE_AITER_CUSTOM_AR=1
    pytest -v -s tests/compile/fusions_e2e/test_tp2_ar_rms.py
    -k 'test_tp2_ar_rms_fp8_fusions and inductor_partition and
    not +quant_fp8 and not +rms_norm and Llama-3 and
    ROCM_AITER_UNIFIED_ATTN'
  - >-
    VLLM_ROCM_USE_AITER=1
    VLLM_ROCM_USE_AITER_CUSTOM_AR=1
    pytest -v -s tests/compile/fusions_e2e/test_tp2_ar_rms.py
    -k 'test_tp2_ar_rms_fp8_fusions and inductor_partition and
    +quant_fp8 and +rms_norm and qwen3 and
    ROCM_AITER_UNIFIED_ATTN'
  - >-
    VLLM_ROCM_USE_AITER=1
    VLLM_ROCM_USE_AITER_CUSTOM_AR=1
    pytest -v -s tests/compile/fusions_e2e/test_tp2_ar_rms.py
    -k 'test_tp2_ar_rms_fp8_fusions and inductor_partition and
    not +quant_fp8 and not +rms_norm and DeepSeek-Coder and
    ROCM_AITER_MLA'

- label: ":amd: (MI355) DSv4-Flash Disaggregated DP EP" # TBD
  timeout_in_minutes: 85
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_8
  num_gpus: 8
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - requirements/kv_connectors_rocm.txt
  - vllm/distributed/kv_transfer/kv_connector/v1/nixl/
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/
  - vllm/models/deepseek_v4/
  - vllm/v1/attention/
  - tests/v1/kv_connector/nixl_integration/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
  # A mirror does not inherit the parent env, so it is repeated here in
  # the parent's order. The layouts must stay empty: unset, the script
  # defaults to HND, which the V4 indexer rejects.
  - >-
    VLLM_ROCM_USE_AITER=1
    ENABLE_HMA_FLAG=1
    DP_EP=1
    GPU_MEMORY_UTILIZATION=0.85
    PREFILLER_TP_SIZE=4
    DECODER_TP_SIZE=4
    PREFILL_BLOCK_SIZE=256
    DECODE_BLOCK_SIZE=256
    MODEL_NAMES=deepseek-ai/DeepSeek-V4-Flash
    VLLM_ENGINE_READY_TIMEOUT_S=1800
    VLLM_SERVE_EXTRA_ARGS=--trust-remote-code,--kv-cache-dtype,fp8
    PREFILLER_KV_LAYOUT=
    DECODER_KV_LAYOUT=
    bash v1/kv_connector/nixl_integration/run_accuracy_test.sh

#----------------------------------------------------------  mi355 · engine  -----------------------------------------------------------#

- label: ":amd: (MI355 DPX) Engine Core" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/engine
  - tests/test_sequence
  - tests/test_config
  - tests/test_logger
  - tests/test_vllm_port
  - tests/jit_monitor/test_hooks.py
  - tests/jit_monitor/test_hooks_gpu.py
  - vllm/compilation/
  - vllm/config/
  - vllm/engine/
  - vllm/entrypoints/logger.py
  - vllm/envs.py
  - vllm/logger.py
  - vllm/logging_utils/
  - vllm/platforms/
  - vllm/sequence.py
  - vllm/triton_utils/
  - vllm/utils/
  commands:
  - pytest -v -s engine test_sequence.py test_config.py test_logger.py test_vllm_port.py jit_monitor/test_hooks.py jit_monitor/test_hooks_gpu.py

#--------------------------------------------------------  mi355 · entrypoints  --------------------------------------------------------#

- label: ":amd: (MI355 DPX) Entrypoints Integration (LLM)" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/llm
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/llm --ignore=entrypoints/llm/test_generate.py --ignore=entrypoints/llm/test_collective_rpc.py --ignore=entrypoints/llm/offline_mode
  - pytest -v -s entrypoints/llm/test_generate.py # it needs a clean process
  - pytest -v -s entrypoints/llm/offline_mode # Needs to avoid interference with other tests

- label: ":amd: (MI355 DPX) Entrypoints Integration (API Server) %N"
  timeout_in_minutes: 65
  parallelism: 4
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/serve
  - tests/entrypoints/scale_out
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/serve --ignore=entrypoints/serve/dev/rpc --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "1" ]; then PYTHONPATH=/vllm-workspace pytest -v -s entrypoints/serve/dev/rpc; fi
  - pytest -v -s entrypoints/scale_out --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":amd: (MI355 DPX) Entrypoints Integration (OpenAI API completion)" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/openai
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/openai/completion --ignore=entrypoints/openai/completion/test_tensorizer_entrypoint.py
  - pytest -v -s entrypoints/openai --ignore=entrypoints/openai/completion --ignore=entrypoints/openai/chat_completion --ignore=entrypoints/openai/responses --ignore=entrypoints/openai/correctness

- label: ":amd: (MI355 DPX) Entrypoints Integration (OpenAI API chat_completion)" # TBD
  timeout_in_minutes: 70
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/openai
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/openai/chat_completion

- label: ":amd: (MI355 DPX) Entrypoints Integration (API Server Generate)" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/tool_use
  - tests/entrypoints/tool_parsers
  - tests/entrypoints/anthropic
  - tests/entrypoints/cohere
  - tests/entrypoints/generate
  commands:
  - pytest -v -s tool_use
  - pytest -v -s entrypoints/tool_parsers
  - pytest -v -s entrypoints/generate
  - pytest -v -s entrypoints/anthropic
  - pytest -v -s entrypoints/cohere

- label: ":amd: (MI355 DPX) OpenAI API Correctness" # TBD
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/
  - vllm/entrypoints/openai/
  - vllm/model_executor/layers/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/model_executor/model_loader/
  commands:
  - bash ../tools/install_torchcodec_rocm.sh || exit 1
  - pytest -s entrypoints/openai/correctness/

- label: ":amd: (MI355 DPX) Entrypoints Integration (API Server) %N (MI355 suite)"
  timeout_in_minutes: 65
  parallelism: 4
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/serve
  - tests/entrypoints/scale_out
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/serve --ignore=entrypoints/serve/dev/rpc --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "1" ]; then PYTHONPATH=/vllm-workspace pytest -v -s entrypoints/serve/dev/rpc; fi
  - pytest -v -s entrypoints/scale_out --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":amd: (MI355 DPX) Entrypoints Integration (OpenAI API completion) (MI355 suite)" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/openai
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/openai/completion --ignore=entrypoints/openai/completion/test_tensorizer_entrypoint.py
  - pytest -v -s entrypoints/openai --ignore=entrypoints/openai/completion --ignore=entrypoints/openai/chat_completion --ignore=entrypoints/openai/responses --ignore=entrypoints/openai/correctness

- label: ":amd: (MI355 DPX) Entrypoints Integration (OpenAI API chat_completion) (MI355 suite)" # TBD
  timeout_in_minutes: 70
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/openai
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/openai/chat_completion

- label: ":amd: (MI355 DPX) Entrypoints Integration (API Server Generate) (MI355 suite)" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/tool_use
  - tests/entrypoints/tool_parsers
  - tests/entrypoints/anthropic
  - tests/entrypoints/generate
  - tests/entrypoints/cohere
  commands:
  - pytest -v -s tool_use
  - pytest -v -s entrypoints/tool_parsers
  - pytest -v -s entrypoints/generate
  - pytest -v -s entrypoints/anthropic
  - pytest -v -s entrypoints/cohere

- label: ":amd: (MI355 DPX) Entrypoints Integration (Speech to Text)" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/speech_to_text
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/speech_to_text

- label: ":amd: (MI355 DPX) Entrypoints Integration (Multimodal)"
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/multimodal
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/multimodal

- label: ":amd: (MI355 DPX) Entrypoints Integration (Pooling) Shard %N" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  fast_check: true
  parallelism: 4
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/entrypoints/pooling
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/pooling --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

#-----------------------------------------------------------  mi355 · evals  -----------------------------------------------------------#

- label: ":amd: (MI355 DPX) Multimodal Accuracy Eval (Small Models)" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/.buildkite/lm-eval-harness"
  source_file_dependencies:
  - vllm/multimodal/
  - vllm/inputs/
  - vllm/v1/core/
  - vllm/platforms/rocm.py
  - vllm/model_executor/model_loader/
  commands:
  - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-mm-small.txt --tp-size=1

- label: ":amd: (MI355) GPQA Eval (GPT-OSS)" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/model_executor/layers/fused_moe/
  - tests/evals/gpt_oss/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
    - pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-gfx950.txt

- label: ":amd: (MI355 DPX) LM Eval Small Models"
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - .buildkite/test_areas/lm_eval.yaml
  - csrc/
  - tests/evals/gsm8k/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/kernels/linear/
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-gfx950-small.txt

- label: ":amd: (MI355 DPX) LM Eval Watermarking"
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - .buildkite/test_areas/lm_eval.yaml
  - tests/evals/gsm8k/
  - vllm/config/watermarking.py
  - vllm/sampling_params.py
  - vllm/v1/watermarking/
  - vllm/v1/worker/gpu/
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-watermark.txt

- label: ":amd: (MI355) LM Eval Watermark Feature Combination"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - .buildkite/test_areas/lm_eval.yaml
  - tests/evals/gsm8k/
  - vllm/config/speculative.py
  - vllm/config/vllm.py
  - vllm/config/watermarking.py
  - vllm/sampling_params.py
  - vllm/model_executor/models/qwen3_dspark.py
  - vllm/model_executor/models/qwen3_dflash.py
  - vllm/model_executor/warmup/
  - vllm/v1/watermarking/
  - vllm/v1/worker/gpu/
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-watermark-dspark-tp2.txt

- label: ":amd: (MI355) LM Eval Qwen3-5 Models" # TBD
  timeout_in_minutes: 75
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/models/qwen3_5.py
  - vllm/model_executor/models/qwen3_5_mtp.py
  - vllm/transformers_utils/configs/qwen3_5.py
  - vllm/transformers_utils/configs/qwen3_5_moe.py
  - vllm/model_executor/models/qwen2.py
  - vllm/model_executor/models/qwen3.py
  - vllm/model_executor/models/qwen3_next.py
  - vllm/model_executor/models/qwen3_next_mtp.py
  - vllm/distributed/
  - vllm/model_executor/layers/fused_moe/
  - vllm/third_party/flash_linear_attention/ops/
  - vllm/model_executor/layers/quantization/modelopt.py
  - vllm/model_executor/kernels/linear/nvfp4/
  - vllm/v1/attention/
  - vllm/v1/worker/
  - tests/evals/gsm8k/
  - tests/utils.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-qwen35-mi355.txt

- label: ":amd: (MI355) LM Eval Large Models EP"
  timeout_in_minutes: 120
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - .buildkite/test_areas/lm_eval.yaml
  - tests/evals/gsm8k/
  - tests/utils.py
  - tests/conftest.py
  - csrc/
  - vllm/config/
  - vllm/distributed/
  - vllm/model_executor/models/config.py
  - vllm/model_executor/models/registry.py
  - vllm/model_executor/models/utils.py
  - vllm/model_executor/models/qwen3_next.py
  - vllm/model_executor/models/qwen3_next_mtp.py
  - vllm/model_executor/models/nemotron_h.py
  - vllm/model_executor/models/nemotron_h_mtp.py
  - vllm/transformers_utils/configs/qwen3_next.py
  - vllm/transformers_utils/configs/nemotron_h.py
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/quantization/
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/mamba/
  - vllm/model_executor/layers/layernorm.py
  - vllm/model_executor/layers/linear.py
  - vllm/model_executor/kernels/linear/
  - vllm/third_party/flash_linear_attention/ops/
  - vllm/v1/attention/
  - vllm/v1/spec_decode/
  - vllm/v1/worker/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-blackwell-ep-mi355.txt

- label: ":amd: (MI355) Qwen3-8-Flash-Next-FP8 Accuracy Eval"
  timeout_in_minutes: 105
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/libtorch_stable/gdn/
  - tests/evals/qwen4_exp/
  - vllm/model_executor/layers/mamba/
  - vllm/models/qwen4_exp/
  - vllm/transformers_utils/configs/qwen4_exp.py
  - vllm/v1/attention/backends/short_conv_attn.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - uv pip install --system 'evalscope==1.10.0'
  - pytest -s -v evals/qwen4_exp/test_accuracy.py --config-list-file=configs/models-rocm.txt

- label: ":amd: (MI355 DPX) LM Eval TurboQuant k8v4"
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - tests/evals/gsm8k/
  - tests/utils.py
  - tests/conftest.py
  - vllm/model_executor/layers/quantization/turboquant/
  - vllm/v1/attention/backends/turboquant_attn.py
  - vllm/v1/attention/backends/triton_attn.py
  - vllm/v1/attention/ops/triton_unified_attention.py
  - vllm/v1/attention/selector.py
  - vllm/v1/attention/ops/triton_turboquant_decode.py
  - vllm/v1/attention/ops/triton_turboquant_store.py
  - vllm/v1/attention/ops/flydsl_turboquant_decode.py
  - vllm/v1/attention/ops/flydsl_kernels/
  - vllm/v1/attention/ops/turboquant_soa/
  - vllm/platforms/rocm.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-k8v4-rocm.txt

- label: ":amd: (MI355 DPX) LM Eval TurboQuant t4nc"
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - tests/evals/gsm8k/
  - tests/utils.py
  - tests/conftest.py
  - vllm/model_executor/layers/quantization/turboquant/
  - vllm/v1/attention/backends/turboquant_attn.py
  - vllm/v1/attention/backends/triton_attn.py
  - vllm/v1/attention/ops/triton_unified_attention.py
  - vllm/v1/attention/selector.py
  - vllm/v1/attention/ops/triton_turboquant_decode.py
  - vllm/v1/attention/ops/triton_turboquant_store.py
  - vllm/v1/attention/ops/flydsl_turboquant_decode.py
  - vllm/v1/attention/ops/flydsl_kernels/
  - vllm/v1/attention/ops/turboquant_soa/
  - vllm/platforms/rocm.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-t4nc-rocm.txt

- label: ":amd: (MI355 DPX) LM Eval TurboQuant k3v4nc"
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - tests/evals/gsm8k/
  - tests/utils.py
  - tests/conftest.py
  - vllm/model_executor/layers/quantization/turboquant/
  - vllm/v1/attention/backends/turboquant_attn.py
  - vllm/v1/attention/backends/triton_attn.py
  - vllm/v1/attention/ops/triton_unified_attention.py
  - vllm/v1/attention/selector.py
  - vllm/v1/attention/ops/triton_turboquant_decode.py
  - vllm/v1/attention/ops/triton_turboquant_store.py
  - vllm/v1/attention/ops/flydsl_turboquant_decode.py
  - vllm/v1/attention/ops/flydsl_kernels/
  - vllm/v1/attention/ops/turboquant_soa/
  - vllm/platforms/rocm.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-k3v4nc-rocm.txt

- label: ":amd: (MI355 DPX) LM Eval TurboQuant t3nc"
  timeout_in_minutes: 30
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - .buildkite/test-amd.yaml
  - tests/evals/gsm8k/
  - tests/utils.py
  - tests/conftest.py
  - vllm/model_executor/layers/quantization/turboquant/
  - vllm/v1/attention/backends/turboquant_attn.py
  - vllm/v1/attention/backends/triton_attn.py
  - vllm/v1/attention/ops/triton_unified_attention.py
  - vllm/v1/attention/selector.py
  - vllm/v1/attention/ops/triton_turboquant_decode.py
  - vllm/v1/attention/ops/triton_turboquant_store.py
  - vllm/v1/attention/ops/flydsl_turboquant_decode.py
  - vllm/v1/attention/ops/flydsl_kernels/
  - vllm/v1/attention/ops/turboquant_soa/
  - vllm/platforms/rocm.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant-t3nc-rocm.txt

- label: ":amd: (MI355) LM Eval Spec Decode (Fixed-length FP8)" # TBD
  timeout_in_minutes: 145
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - tests/evals/gsm8k/
  - tests/utils.py
  - vllm/distributed/
  - vllm/models/deepseek_v4/
  - vllm/model_executor/models/qwen3_dspark.py
  - vllm/model_executor/layers/fused_moe/
  - vllm/v1/spec_decode/metrics.py
  - vllm/v1/worker/gpu/spec_decode/dspark/
  - vllm/v1/worker/gpu/spec_decode/rejection_sampler.py
  - vllm/v1/worker/gpu/spec_decode/speculator.py
  - vllm/model_executor/layers/quantization/
  - vllm/v1/attention/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-spec-decode-rocm.txt

- label: ':amd: (MI355) LM Eval Small Models Distributed'
  timeout_in_minutes: 145
  mirror_hardwares:
  - amdexperimental
  - amdproduction
  - amdgfx950nightly
  - amdmi355
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/v1/worker/
  - vllm/v1/core/
  - vllm/config/
  - tests/evals/gsm8k/configs/models-mi3xx-fp8-and-mixed.txt
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/evals/gsm8k/
  - tests/utils.py
  - vllm/distributed/
  - vllm/transformers_utils/configs/diffusion_gemma.py
  - vllm/v1/attention/ops/
  - tests/evals/gsm8k/configs/models-small-tp-mi355.txt
  commands:
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-mi3xx-fp8-and-mixed.txt
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small-tp.txt
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small-tp-mi355.txt

- label: ":amd: (MI355) MoE Refactor Integration Shard %N"
  timeout_in_minutes: 105
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  parallelism: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/platforms/rocm.py
  - tests/evals/gsm8k/
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/
  - vllm/_aiter_ops.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/moe-refactor/config-rocm-shard-$$BUILDKITE_PARALLEL_JOB.txt


- label: ":amd: (MI355) Qwen3-30B-A3B-FP8-block Sync EPLB Accuracy" # TBD
  timeout_in_minutes: 25
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  working_dir: "/vllm-workspace"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/quantization/
  - vllm/model_executor/layers/fused_moe/
  - vllm/distributed/eplb
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - .buildkite/scripts/scheduled_integration_test/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - bash .buildkite/scripts/scheduled_integration_test/qwen30b_a3b_fp8_block_ep_eplb.sh 0.8 200 8020 2 1

- label: ":amd: (MI355) LM Eval Large Models FP8" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_4
  num_gpus: 4
  optional: true
  working_dir: "/vllm-workspace/.buildkite/lm-eval-harness"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_USE_DEEP_GEMM=0
  - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large-rocm-tp4.txt --tp-size=4

- label: ":amd: (MI355) LM Eval Large Models" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_8
  optional: true
  num_gpus: 8
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/quantization/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - vllm/model_executor/layers/layernorm.py
  - csrc/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - tests/evals/
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-gfx950-large.txt

#---------------------------------------------------------  mi355 · examples  ----------------------------------------------------------#

- label: ":amd: (MI355 DPX) Examples" # TBD
  timeout_in_minutes: 75
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/examples"
  source_file_dependencies:
  - vllm/entrypoints
  - vllm/multimodal
  - examples/
  - vllm/platforms/rocm.py
  commands:
    # Basic
    # Multi-modal models
    # Pooling models
    # Features demo
  - pip install --no-deps tensorizer
  - python3 basic/offline_inference/chat.py
  - python3 basic/offline_inference/generate.py --model facebook/opt-125m
  - python3 basic/offline_inference/generate.py --model meta-llama/Llama-2-13b-chat-hf --cpu-offload-gb 10
  - python3 basic/offline_inference/classify.py
  - python3 basic/offline_inference/embed.py
  - python3 basic/offline_inference/score.py
  - python3 generate/multimodal/audio_language_offline.py --seed 0
  - python3 generate/multimodal/vision_language_offline.py --seed 0
  - python3 generate/multimodal/vision_language_multi_image_offline.py --seed 0
  - python3 generate/multimodal/encoder_decoder_multimodal_offline.py --model-type whisper --seed 0
  - python3 pooling/embed/vision_embedding_offline.py --seed 0
  - python3 features/automatic_prefix_caching/prefix_caching_offline.py
  - python3 deployment/llm_engine_example.py
  - python3 features/tensorize_vllm_model.py --model facebook/opt-125m serialize --serialized-directory /tmp/ --suffix v1 && python3 features/tensorize_vllm_model.py --model facebook/opt-125m deserialize --path-to-tensors /tmp/vllm/facebook/opt-125m/v1/model.tensors
  - python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 2048
  - python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle3 --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 1536

#----------------------------------------------------------  mi355 · kernels  ----------------------------------------------------------#

- label: ":amd: (MI355 DPX) Attention DiffKV Kernels"
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/attention/ops/triton_unified_attention_diffkv.py
  - vllm/v1/attention/backends/triton_attn_diffkv.py
  - tests/kernels/attention/test_triton_unified_attention.py
  - tests/kernels/attention/test_triton_unified_attention_diffkv.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s kernels/attention/test_triton_unified_attention_diffkv.py

- label: ":amd: (MI355) MiniMax Reduce RMS Kernels" # TBD
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/layers/minimax_rms_norm/
  - vllm/distributed/
  - vllm/_aiter_ops.py
  - vllm/envs.py
  - vllm/platforms/rocm.py
  - tests/kernels/core/test_minimax_reduce_rms.py
  - tests/utils.py
  - csrc/libtorch_stable/minimax_reduce_rms_kernel.cu
  - csrc/libtorch_stable/minimax_reduce_rms_kernel.h
  - csrc/libtorch_stable/ops.h
  - csrc/libtorch_stable/torch_bindings.cpp
  - vllm/model_executor/layers/mamba/linear/minimax_linear_attn.py
  optional: true
  commands:
  - pytest -v -s kernels/core/test_minimax_reduce_rms.py

- label: ":amd: (MI355 DPX) Kernels" # TBD
  timeout_in_minutes: 25
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - csrc/quantization/fp4/
  - csrc/attention/mla/
  - csrc/quantization/cutlass_w8a8/moe/
  - vllm/model_executor/layers/fused_moe/cutlass_moe.py
  - vllm/v1/attention/backends/triton_attn.py
  - vllm/v1/attention/backends/rocm_attn.py
  - vllm/v1/attention/backends/rocm_aiter_fa.py
  - vllm/v1/attention/backends/rocm_aiter_unified_attn.py
  - vllm/v1/attention/backends/mla/aiter_triton_mla.py
  - vllm/v1/attention/backends/mla/rocm_aiter_mla.py
  - vllm/v1/attention/selector.py
  - vllm/platforms/rocm.py
  - vllm/_aiter_ops.py
  - examples/basic/offline_inference/chat.py
  - tests/kernels/attention/test_attention_selector.py
  optional: true
  commands:
  - amd-smi
  - python3 examples/basic/offline_inference/chat.py --attention-backend TRITON_ATTN
  - pytest -v -s tests/kernels/attention/test_attention_selector.py

- label: ":amd: (MI355 DPX) MLA Kernels"
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - vllm/v1/attention/backends/mla/rocm_aiter_mla.py
  - vllm/v1/attention/selector.py
  - vllm/platforms/rocm.py
  - vllm/_aiter_ops.py
  - tests/kernels/attention/test_rocm_aiter_mla_decode.py
  - tests/kernels/attention/test_rocm_aiter_mla_decode_metadata.py
  - tests/kernels/attention/test_rocm_aiter_mla_fp8_support.py
  - tests/kernels/attention/test_rocm_aiter_mla_op_registration.py
  commands:
  - amd-smi
  - pytest -v -s tests/kernels/attention/test_rocm_aiter_mla_decode_metadata.py
  - pytest -v -s tests/kernels/attention/test_rocm_aiter_mla_decode.py
  - pytest -v -s tests/kernels/attention/test_rocm_aiter_mla_fp8_support.py
  - pytest -v -s tests/kernels/attention/test_rocm_aiter_mla_op_registration.py

- label: ":amd: (MI355 DPX) DeepSeek V4 Fused Kernels"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/libtorch_stable/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu
  - vllm/models/deepseek_v4/common/ops/
  - tests/kernels/test_fused_deepseek_v4_qnorm_rope_kv_insert.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s kernels/test_fused_deepseek_v4_qnorm_rope_kv_insert.py

- label: ":amd: (MI355 DPX) Attention Kernels Shard %N" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  parallelism: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/attention/
  - vllm/v1/attention
  - vllm/model_executor/layers/attention
  - tests/kernels/attention
  - vllm/_aiter_ops.py
  - vllm/envs.py
  - vllm/platforms/rocm.py
  - vllm/utils/flashinfer.py
  optional: true
  commands:
  - pytest -v -s kernels/attention --ignore=kernels/attention/test_triton_unified_attention_diffkv.py --ignore=kernels/attention/test_rocm_aiter_mla_decode.py --ignore=kernels/attention/test_rocm_aiter_mla_decode_metadata.py --ignore=kernels/attention/test_rocm_aiter_mla_fp8_support.py --ignore=kernels/attention/test_rocm_aiter_mla_op_registration.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

- label: ":amd: (MI355 DPX) MoE Kernels Shard %N (MI355 suite)" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  parallelism: 5
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/quantization/cutlass_w8a8/moe/
  - csrc/moe/
  - csrc/rocm/
  - tests/kernels/moe
  - tests/kernels/utils.py
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/mxfp4.py
  - vllm/model_executor/layers/quantization/utils/quant_utils.py
  - vllm/distributed/device_communicators/
  - vllm/envs.py
  - vllm/config
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  optional: true
  commands:
  - pytest -v -s kernels/moe
      --ignore=kernels/moe/test_modular_oai_triton_moe.py
      --ignore=kernels/moe/test_gpt_oss_triton_kernels.py
      --ignore=kernels/moe/test_moe.py
      --ignore=kernels/moe/test_block_int8.py
      --ignore=kernels/moe/test_triton_moe_no_act_mul.py
      --ignore=kernels/moe/test_triton_moe_ptpc_fp8.py
      --ignore=kernels/moe/test_deepep_moe.py
      --ignore=kernels/moe/test_moe_layer.py
      --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
  - pytest -v -s kernels/moe/test_modular_oai_triton_moe.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

- label: ":amd: (MI355) FusedMoE Layer Kernels" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/moe/
  - csrc/rocm/
  - tests/kernels/moe
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/
  - vllm/distributed/
  - vllm/config/
  - vllm/forward_context.py
  - vllm/v1/worker/workspace.py
  - vllm/utils/import_utils.py
  - vllm/utils/math_utils.py
  - vllm/utils/torch_utils.py
  - vllm/platforms/
  - vllm/_aiter_ops.py
  - csrc/quantization/cutlass_w8a8/moe/
  - vllm/distributed/device_communicators/
  - vllm/config
  commands:
  # MoRI graph cases run in the Fault Tolerance group's healthy EP baseline.
  - pytest -v -s kernels/moe/test_moe_layer.py -k "not test_moe_layer_mori_graph"

- label: ":amd: (MI355) MoRI EP Numerics"
  timeout_in_minutes: 20
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - tests/kernels/moe/test_mori_moe.py
  - tests/kernels/moe/utils.py
  - tests/kernels/moe/modular_kernel_tools/parallel_utils.py
  - tests/kernels/utils.py
  - tests/utils.py
  - tests/conftest.py
  - vllm/utils/torch_utils.py
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/
  - vllm/distributed/
  - vllm/config/
  - vllm/forward_context.py
  - vllm/v1/worker/workspace.py
  - vllm/platforms/
  - vllm/_aiter_ops.py
  - requirements/rocm-test.txt
  optional: true
  commands:
  - amd-smi
  - VLLM_ROCM_USE_AITER=1 VLLM_ROCM_USE_AITER_MOE=1 VLLM_ROCM_USE_AITER_FUSION_SHARED_EXPERTS=0 pytest -v -s kernels/moe/test_mori_moe.py

- label: ":amd: (MI355) Quantization Kernels Shard %N"
  timeout_in_minutes: 120
  parallelism: 6
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/quantization/
  - csrc/rocm/
  - vllm/model_executor/layers/quantization
  - tests/kernels/quantization
  - tests/kernels/quant_utils.py
  - tests/kernels/utils.py
  - vllm/_aiter_ops.py
  - vllm/kernels/aiter_ops.py
  - vllm/_custom_ops.py
  - vllm/envs.py
  - vllm/platforms/rocm.py
  - vllm/model_executor/kernels/
  - vllm/v1/attention/backends/rocm_aiter_fa.py
  - vllm/config/
  - tests/kernels/quantization/test_rocm_skinny_gemms.py
  optional: true
  commands:
  - pytest -v -s kernels/quantization -k "not test_nvfp4_dense_emulation" --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

- label: ":amd: (MI355) NVFP4 Dense Emulation Shard %N"
  timeout_in_minutes: 10
  parallelism: 3
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_1
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - tests/kernels/quantization/test_nvfp4_emulation.py
  - tests/kernels/quantization/nvfp4_utils.py
  - tests/conftest.py
  - vllm/model_executor/kernels/linear/
  - vllm/model_executor/layers/quantization/utils/nvfp4_emulation_utils.py
  - vllm/config/kernel.py
  - vllm/platforms/
  - vllm/envs.py
  - vllm/utils/torch_utils.py
  - requirements/rocm-test.txt
  optional: true
  commands:
  - amd-smi
  - pytest -v -s kernels/quantization/test_nvfp4_emulation.py -k test_nvfp4_dense_emulation --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

- label: ":amd: (MI355 DPX) FP8 MoE Kernels" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/moe/
  - vllm/model_executor/layers/fused_moe/
  - tests/kernels/moe/test_deepep_moe.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/envs.py
  - tests/kernels/moe/test_gpt_oss_triton_kernels.py
  - tests/kernels/moe/test_modular_oai_triton_moe.py
  - tests/kernels/moe/test_moe.py
  - tests/kernels/moe/test_block_int8.py
  - tests/kernels/moe/test_triton_moe_no_act_mul.py
  - tests/kernels/moe/test_triton_moe_ptpc_fp8.py
  optional: true
  commands:
    - pytest -v -s kernels/moe/test_gpt_oss_triton_kernels.py
    - pytest -v -s kernels/moe/test_modular_oai_triton_moe.py
    - pytest -v -s kernels/moe/test_moe.py
    - pytest -v -s kernels/moe/test_block_int8.py
    - pytest -v -s kernels/moe/test_triton_moe_no_act_mul.py
    - pytest -v -s kernels/moe/test_triton_moe_ptpc_fp8.py

- label: ":amd: (MI355) DeepEP FP8 MoE Kernels" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/moe/
  - csrc/quantization/w8a8/cutlass/moe/
  - vllm/model_executor/layers/fused_moe/
  - tests/kernels/moe/test_deepep_moe.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/envs.py
  commands:
    - pytest -v -s kernels/moe/test_deepep_moe.py

#-----------------------------------------------------------  mi355 · lora  ------------------------------------------------------------#

- label: ":amd: (MI355 DPX) LoRA Shard %N" # TBD
  timeout_in_minutes: 85
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  parallelism: 4
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/lora
  - tests/lora
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s lora --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --ignore=lora/test_chatglm3_tp.py --ignore=lora/test_llama_tp.py --ignore=lora/test_qwen3_with_multi_loras.py --ignore=lora/test_olmoe_tp.py --ignore=lora/test_deepseekv2_tp.py --ignore=lora/test_gptoss_tp.py --ignore=lora/test_qwen3moe_tp.py --ignore=lora/test_qwen35_densemodel_lora.py

#------------------------------------------------------  mi355 · models / basic  -------------------------------------------------------#

- label: ":amd: (MI355 DPX) Basic Models (Other)" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/test_terratorch.py
  - tests/models/transformers/test_backend.py
  - tests/models/test_registry.py
  - tests/models/test_hyv4_rocm.py
  - tests/models/test_vision.py
  - tests/models/test_deepseek_v4_vl_rocm.py
  commands:
  - pytest -v -s models/test_terratorch.py models/transformers/test_backend.py models/test_registry.py models/test_deepseek_v4_vl_rocm.py
  - VLLM_ROCM_USE_AITER=1 pytest -v -s models/test_hyv4_rocm.py -m 'not distributed'
  - pytest -v -s models/test_vision.py::test_simple_mrope_vision_model_spatial_merge

- label: ":amd: (MI355 DPX) GLM5Next"
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/models/glm5next/
  - tests/models/glm5next/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s models/glm5next

- label: ":amd: (MI355 DPX) Basic Models (Extra Initialization) Shard %N" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  parallelism: 6
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/layers/
  - tests/models/test_initialization.py
  - tests/models/registry.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  optional: true
  commands:
  - pytest -v -s models/test_initialization.py -k 'not test_can_initialize_small_subset' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":amd: (MI355) Inkling" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/models/inkling/
  - vllm/cute_utils/
  - cmake/external_projects/tml_fa4.cmake
  - tests/models/inkling/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s models/inkling/rocm

#-----------------------------------------------------  mi355 · models / language  -----------------------------------------------------#

- label: ":amd: (MI355 DPX) Language Models (Standard)" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/language
  commands:
  - pip freeze | grep -E 'torch'
  - pytest -v -s models/language -m 'core_model and (not slow_test)'

- label: ":amd: (MI355 DPX) Language Models (Extra Standard) Shard %N" # TBD
  timeout_in_minutes: 40
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  parallelism: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/models/
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/
  - vllm/v1/attention/backends/
  - vllm/v1/attention/selector.py
  - tests/models/language/pooling/test_embedding.py
  - tests/models/language/generation/test_common.py
  - tests/models/language/pooling/test_classification.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - pip freeze | grep -E 'torch'
  - pytest -v -s models/language -m 'core_model and slow_test' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":amd: (MI355 DPX) Language Models (Hybrid) Shard %N" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  parallelism: 2
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/language/generation
  commands:
  - export GIT_CONFIG_COUNT=1
  - export GIT_CONFIG_KEY_0=http.version
  - export GIT_CONFIG_VALUE_0=HTTP/1.1
  - MAMBA_FORCE_BUILD=TRUE uv pip install --system --no-build-isolation 'git+https://github.com/AndreasKaratzas/mamba@fix-rocm-7.0-warp-size-constexpr'
  - CAUSAL_CONV1D_FORCE_BUILD=TRUE uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
  - pytest -v -s models/language/generation -m hybrid_model -k 'not granite-4.0-tiny-preview' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":amd: (MI355 DPX) Granite Language Model Compatibility"
  timeout_in_minutes: 80
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - tests/models/language/generation
  commands:
  - export GIT_CONFIG_COUNT=1
  - export GIT_CONFIG_KEY_0=http.version
  - export GIT_CONFIG_VALUE_0=HTTP/1.1
  - MAMBA_FORCE_BUILD=TRUE uv pip install --system --no-build-isolation 'git+https://github.com/AndreasKaratzas/mamba@fix-rocm-7.0-warp-size-constexpr'
  - CAUSAL_CONV1D_FORCE_BUILD=TRUE uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
  - pytest -v -s models/language/generation -m hybrid_model -k 'granite-4.0-tiny-preview'

- label: ":amd: (MI355 DPX) Language Models (Extended Generation)" # TBD
  timeout_in_minutes: 95
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/language/generation
  optional: true
  commands:
  - MAMBA_FORCE_BUILD=TRUE uv pip install --system --no-build-isolation 'git+https://github.com/AndreasKaratzas/mamba@rocm-7.0-v2.3.0'
  - CAUSAL_CONV1D_FORCE_BUILD=TRUE uv pip install --system --no-build-isolation 'git+https://github.com/Dao-AILab/causal-conv1d@v1.6.0'
  - pytest -v -s models/language/generation -m '(not core_model) and (not hybrid_model)'

- label: ":amd: (MI355 DPX) Language Models (Extended Pooling) Shard %N"
  timeout_in_minutes: 95
  parallelism: 4
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/language/pooling
  commands:
  - pytest -v -s models/language/pooling -m 'not core_model' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":amd: (MI355 DPX) Language Models (PPL)" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/models/qwen3_5.py
  - vllm/model_executor/models/qwen3_5_mtp.py
  - vllm/transformers_utils/configs/qwen3_5.py
  - vllm/transformers_utils/configs/qwen3_5_moe.py
  - vllm/model_executor/models/qwen2.py
  - vllm/model_executor/models/qwen3.py
  - vllm/model_executor/models/qwen3_next.py
  - vllm/model_executor/models/qwen3_next_mtp.py
  - vllm/third_party/flash_linear_attention/ops/
  - vllm/_aiter_ops.py
  - vllm/v1/attention/backends/triton_attn.py
  - vllm/v1/attention/backends/rocm_attn.py
  - vllm/v1/attention/backends/rocm_aiter_unified_attn.py
  - vllm/v1/attention/backends/rocm_aiter_fa.py
  - vllm/v1/attention/backends/flex_attention.py
  - vllm/v1/attention/ops/
  - vllm/platforms/rocm.py
  - tests/models/language/generation_ppl_test
  - vllm/model_executor/layers/fla/ops/
  - vllm/
  commands:
  - pytest -v -s models/language/generation_ppl_test

- label: ":amd: (MI355 DPX) Language Models (Standard) (MI355 suite)" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/language
  optional: true
  commands:
  - pip freeze | grep -E 'torch'
  - pytest -v -s models/language -m 'core_model and (not slow_test)'

#----------------------------------------------------  mi355 · models / multimodal  ----------------------------------------------------#

- label: ":amd: (MI355 DPX) Multimodal Models (Standard) 1: qwen2" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal
  commands:
  - pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "qwen2"
  - pytest -v -s models/multimodal/generation/test_ultravox.py -m core_model

- label: ":amd: (MI355 DPX) Multimodal Models (Standard) 2: qwen3 + gemma" # TBD
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal
  commands:
  - pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "qwen3 or gemma"
  - pytest -v -s models/multimodal/generation/test_mm_prefix_lm.py -m core_model
  - pytest -v -s models/multimodal/generation/test_qwen2_5_vl.py -m core_model

- label: ":amd: (MI355 DPX) Multimodal Models (Extended Generation 1)" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal/generation
  - tests/models/multimodal/test_mapping.py
  commands:
  - pytest -v -s models/multimodal/generation -m 'not core_model' --ignore models/multimodal/generation/test_common.py
  - pytest -v -s models/multimodal/test_mapping.py

- label: ":amd: (MI355 DPX) Multimodal Models (Extended Generation 3)" # TBD
  timeout_in_minutes: 90
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal/generation
  commands:
  - pytest -v -s models/multimodal/generation/test_common.py -m 'split(group=1) and not core_model'

- label: ":amd: (MI355 DPX) Multimodal Models (Extended Pooling)" # TBD
  timeout_in_minutes: 75
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal/pooling
  commands:
  - pytest -v -s models/multimodal/pooling -m 'not core_model'

- label: ":amd: (MI355 DPX) Multimodal Models (Standard) 1: qwen2 (MI355 suite)" # TBD
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal
  commands:
  - pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "qwen2"
  - pytest -v -s models/multimodal/generation/test_ultravox.py -m core_model

- label: ":amd: (MI355 DPX) Multimodal Models (Standard) 4: other + whisper" # TBD
  timeout_in_minutes: 70
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/models/multimodal
  commands:
  - pytest -v -s models/multimodal -m core_model --ignore models/multimodal/generation/test_common.py --ignore models/multimodal/generation/test_ultravox.py --ignore models/multimodal/generation/test_qwen2_5_vl.py --ignore models/multimodal/generation/test_qwen2_vl.py --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_mm_prefix_lm.py --ignore models/multimodal/generation/test_memory_leak.py --ignore models/multimodal/generation/test_vit_cudagraph.py --ignore models/multimodal/processing
  - pytest -v -s models/multimodal/generation/test_vit_cudagraph.py -m core_model
  - pytest models/multimodal/generation/test_memory_leak.py -m core_model
  - cd .. && VLLM_WORKER_MULTIPROC_METHOD=spawn pytest -v -s tests/models/multimodal/generation/test_whisper.py -m core_model

#-----------------------------------------------------  mi355 · models / quantized  -----------------------------------------------------#

- label: ":amd: (MI355 DPX) Quantized Models" # TBD
  timeout_in_minutes: 80
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/layers/quantization
  - tests/models/quantization
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/model_executor/model_loader/
  optional: true
  commands:
  - unset VLLM_USE_V2_MODEL_RUNNER
  - pytest -v -s models/quantization

#-------------------------------------------------------  mi355 · quantization  --------------------------------------------------------#

- label: ":amd: (MI355 DPX) Quantization Shard %N"
  timeout_in_minutes: 115
  parallelism: 4
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - csrc/
  - vllm/model_executor/layers/quantization
  - tests/quantization
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/model_executor/layers/fused_moe/oracle/unquantized.py
  - vllm/model_executor/layers/fused_moe/unquantized_fused_moe_method.py
  - tests/rocm/test_moe_weight_replay.py
  optional: true
  commands:
  - unset VLLM_USE_V2_MODEL_RUNNER
  - uv pip install --system torchao==0.17.0
  - uv pip install --system conch-triton-kernels
  - VLLM_TEST_FORCE_LOAD_FORMAT=auto pytest -v -s quantization/ --ignore quantization/test_blackwell_moe.py --ignore quantization/test_rocm_moe.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "0" ]; then pytest -v -s rocm/test_moe_weight_replay.py; fi

# - label: Quantized MoE Test (B200-MI355) # TBD
#   timeout_in_minutes: 180
#   mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
#   dind: false
#   agent_pool: mi355_1
#   working_dir: "/vllm-workspace/"
#   source_file_dependencies:
#   - tests/quantization/test_gfx950_moe.py
#   - vllm/model_executor/models/deepseek_v2.py
#   - vllm/model_executor/models/gpt_oss.py
#   - vllm/model_executor/models/llama4.py
#   - vllm/model_executor/layers/fused_moe
#   - vllm/model_executor/layers/quantization/compressed_tensors
#   - vllm/model_executor/layers/quantization/modelopt.py
#   - vllm/model_executor/layers/quantization/mxfp4.py
#   - vllm/v1/attention/backends/triton_attn.py
#   - vllm/v1/attention/backends/rocm_attn.py
#   - vllm/v1/attention/backends/rocm_aiter_fa.py
#   - vllm/v1/attention/backends/mla/
#   - vllm/v1/attention/selector.py
#   - vllm/model_executor/layers/layernorm.py
#   - vllm/_aiter_ops.py
#   - vllm/platforms/rocm.py
#   - vllm/model_executor/model_loader/
#   commands:
#   - pytest -s -v tests/quantization/test_gfx950_moe.py

#---------------------------------------------------------  mi355 · samplers  ----------------------------------------------------------#

- label: ":amd: (MI355 DPX) Samplers Multimodal Beam Search"
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/model_executor/layers
  - vllm/sampling_metadata.py
  - vllm/v1/sample/
  - vllm/entrypoints/generate/beam_search/
  - tests/samplers
  - tests/conftest.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s samplers/test_beam_search.py::test_beam_search_passes_multimodal_data

#------------------------------------------------------------  mi355 · v1  -------------------------------------------------------------#

- label: ":amd: (MI355) Spec Decode AL DFlash2 Nightly"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_1
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - .buildkite/test_areas/spec_decode.yaml
  - tests/v1/e2e/spec_decode/
  - tests/evals/gsm8k/
  - tests/utils.py
  - vllm/v1/spec_decode/
  - vllm/v1/worker/gpu/spec_decode/
  - vllm/v1/worker/gpu/sample/
  - vllm/v1/worker/gpu/model_runner.py
  - vllm/v1/attention/
  - vllm/model_executor/models/qwen3_dflash.py
  - vllm/model_executor/models/qwen3_dflash2.py
  - vllm/model_executor/models/qwen3_5.py
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/quantization/
  - vllm/model_executor/kernels/linear/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/e2e/spec_decode/acceptance_rates/dflash/ -k "dflash2"

- label: ":amd: (MI355 DPX) Spec Decode Draft Model" # TBD
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/spec_decode/
  - vllm/v1/worker/gpu/spec_decode/
  - vllm/model_executor/model_loader/
  - vllm/v1/sample/
  - vllm/model_executor/layers/
  - tests/v1/e2e/spec_decode/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/e2e/spec_decode/draft_model/

- label: ":amd: (MI355 DPX) Spec Decode Eagle 1: DeepSeek + Qwen"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/spec_decode/
  - vllm/v1/worker/gpu/spec_decode/
  - vllm/model_executor/model_loader/
  - vllm/v1/sample/
  - vllm/model_executor/layers/
  - tests/v1/e2e/spec_decode/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/e2e/spec_decode/eagle/ -k "deepseek_eagle or qwen3_eagle3"

- label: ":amd: (MI355 DPX) Spec Decode Eagle 2: Llama 3 + Qwen VL + Other"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/v1/spec_decode/
  - vllm/v1/worker/gpu/spec_decode/
  - vllm/model_executor/model_loader/
  - vllm/v1/sample/
  - vllm/model_executor/layers/
  - tests/v1/e2e/spec_decode/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/e2e/spec_decode/eagle/ -k "not deepseek_eagle and not qwen3_eagle3"


- label: ":amd: (MI355 DPX) Spec Decode N-Gram + Suffix" # TBD
  timeout_in_minutes: 35
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/spec_decode/
  - vllm/v1/worker/gpu/spec_decode/
  - vllm/model_executor/model_loader/
  - vllm/v1/sample/
  - vllm/model_executor/layers/
  - tests/v1/e2e/spec_decode/
  - tests/spec_decode/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/e2e/spec_decode/ngram_suffix/
  - python3 spec_decode/test_custom_proposer.py

- label: ":amd: (MI355 DPX) V1 Core"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/simple_kv_offload
  - tests/v1/worker
  - tests/v1/streaming_input
  - tests/v1/kv_connector/unit
  - tests/v1/ec_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/entrypoints/pooling/
  - vllm/inputs/
  - vllm/lora/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/outputs.py
  - vllm/platforms/
  - vllm/pooling_params.py
  - vllm/profiler/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  optional: true
  commands:
  # - export HSA_NO_SCRATCH_RECLAIM=1
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s -m 'not cpu_test' v1/core

- label: ":amd: (MI355 DPX) V1 Executor + Worker"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/simple_kv_offload
  - tests/v1/worker
  - tests/v1/streaming_input
  - tests/v1/kv_connector/unit
  - tests/v1/ec_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/entrypoints/pooling/
  - vllm/inputs/
  - vllm/lora/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/outputs.py
  - vllm/platforms/
  - vllm/pooling_params.py
  - vllm/profiler/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s v1/executor
  - pytest -v -s v1/worker
  - pytest -v -s v1/streaming_input
  optional: true

- label: ":amd: (MI355 DPX) V1 KV Offload"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/simple_kv_offload
  - tests/v1/worker
  - tests/v1/streaming_input
  - tests/v1/kv_connector/unit
  - tests/v1/ec_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/entrypoints/pooling/
  - vllm/inputs/
  - vllm/lora/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/outputs.py
  - vllm/platforms/
  - vllm/pooling_params.py
  - vllm/profiler/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - bash /vllm-workspace/.buildkite/scripts/install-kv-offload.sh
  - pytest -v -s v1/kv_offload
  - pytest -v -s v1/simple_kv_offload
  optional: true

- label: ":amd: (MI355 DPX) V1 KV Connectors Shard %N"
  timeout_in_minutes: 60
  parallelism: 4
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/simple_kv_offload
  - tests/v1/worker
  - tests/v1/streaming_input
  - tests/v1/kv_connector/unit
  - tests/v1/ec_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/entrypoints/pooling/
  - vllm/inputs/
  - vllm/lora/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/outputs.py
  - vllm/platforms/
  - vllm/pooling_params.py
  - vllm/profiler/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s -m 'not cpu_test' v1/kv_connector/unit --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
  - pytest -v -s -m 'not cpu_test' v1/ec_connector/unit --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
  optional: true

- label: ":amd: (MI355 DPX) V1 Metrics + LM Eval"
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/simple_kv_offload
  - tests/v1/worker
  - tests/v1/streaming_input
  - tests/v1/kv_connector/unit
  - tests/v1/ec_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/entrypoints/pooling/
  - vllm/inputs/
  - vllm/lora/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/outputs.py
  - vllm/platforms/
  - vllm/pooling_params.py
  - vllm/profiler/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - export GIT_CONFIG_COUNT=1
  - export GIT_CONFIG_KEY_0=http.version
  - export GIT_CONFIG_VALUE_0=HTTP/1.1
  - export GIT_TERMINAL_PROMPT=0
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s -m 'not cpu_test' v1/metrics
  - pip install -U git+https://github.com/vllm-project/lm-evaluation-harness.git@streaming-api
  - pytest -v -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
  optional: true


- label: ":amd: (MI355 DPX) V1 Sample"
  timeout_in_minutes: 70
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/v1/sample
  - tests/v1/logits_processors
  - tests/v1/test_oracle.py
  - tests/v1/test_request.py
  - tests/v1/test_outputs.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/inputs/
  - vllm/logger.py
  - vllm/model_executor/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s v1/sample

- label: ":amd: (MI355 DPX) V1 Logits + Oracle"
  timeout_in_minutes: 70
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/sample
  - tests/v1/logits_processors
  - tests/v1/test_oracle.py
  - tests/v1/test_request.py
  - tests/v1/test_outputs.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/inputs/
  - vllm/logger.py
  - vllm/model_executor/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s v1/logits_processors
  - pytest -v -s v1/test_oracle.py
  - pytest -v -s v1/test_request.py
  - pytest -v -s v1/test_outputs.py


- label: ":amd: (MI355 DPX) E2E Core Large Memory"
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/
  - tests/v1/e2e/general/
  - vllm/platforms/rocm.py
  commands:
  - pytest -v -s v1/e2e/general/test_kv_sharing_fast_prefill.py
  - pytest -v -s v1/e2e/general/test_mamba_prefix_cache.py::test_mamba_prefix_cache_mrv2 v1/e2e/general/test_mamba_prefix_cache.py::test_mamba_prefix_cache_mrv2_async

- label: ":amd: (MI355 DPX) Extract Hidden States Integration Single GPU"
  timeout_in_minutes: 45
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace"
  source_file_dependencies:
  - vllm/v1/spec_decode/extract_hidden_states.py
  - vllm/model_executor/models/extract_hidden_states.py
  - vllm/transformers_utils/configs/extract_hidden_states.py
  - vllm/distributed/kv_transfer/kv_connector/v1/example_hidden_states_connector.py
  - tests/v1/kv_connector/extract_hidden_states_integration
  - vllm/platforms/rocm.py
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s -m 'not distributed' tests/v1/kv_connector/extract_hidden_states_integration

- label: ":amd: (MI355 DPX) V1 Attention Shard %N" # TBD
  timeout_in_minutes: 125
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  parallelism: 2
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/config/attention.py
  - vllm/model_executor/layers/attention
  - vllm/v1/attention
  - tests/v1/attention
  - vllm/_aiter_ops.py
  - vllm/envs.py
  - vllm/platforms/rocm.py
  optional: true
  commands:
  - pytest -v -s v1/attention --ignore=v1/attention/test_mla_backends.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
  - VLLM_ROCM_USE_AITER=1 pytest -v -s v1/attention/test_mla_backends.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT

- label: ":amd: (MI355 DPX) V1 Core (MI355 suite)"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/worker
  - tests/v1/cudagraph
  - tests/v1/kv_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/entrypoints/pooling/
  - vllm/inputs/
  - vllm/lora/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/outputs.py
  - vllm/platforms/
  - vllm/pooling_params.py
  - vllm/profiler/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  - tests/v1/simple_kv_offload
  - tests/v1/streaming_input
  - tests/v1/ec_connector/unit
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s -m 'not cpu_test' v1/core

- label: ":amd: (MI355 DPX) V1 Executor + Worker (MI355 suite)"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/worker
  - tests/v1/cudagraph
  - tests/v1/kv_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/entrypoints/pooling/
  - vllm/inputs/
  - vllm/lora/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/outputs.py
  - vllm/platforms/
  - vllm/pooling_params.py
  - vllm/profiler/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  - tests/v1/simple_kv_offload
  - tests/v1/streaming_input
  - tests/v1/ec_connector/unit
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s v1/executor
  - pytest -v -s v1/worker
  - pytest -v -s v1/streaming_input

- label: ":amd: (MI355 DPX) V1 KV Offload (MI355 suite)"
  timeout_in_minutes: 60
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/worker
  - tests/v1/cudagraph
  - tests/v1/kv_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/entrypoints/pooling/
  - vllm/inputs/
  - vllm/lora/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/outputs.py
  - vllm/platforms/
  - vllm/pooling_params.py
  - vllm/profiler/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  - tests/v1/simple_kv_offload
  - tests/v1/streaming_input
  - tests/v1/ec_connector/unit
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - bash /vllm-workspace/.buildkite/scripts/install-kv-offload.sh
  - pytest -v -s v1/kv_offload
  - pytest -v -s v1/simple_kv_offload

- label: ":amd: (MI355 DPX) V1 KV Connectors Shard %N (MI355 suite)"
  timeout_in_minutes: 60
  parallelism: 4
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/worker
  - tests/v1/cudagraph
  - tests/v1/kv_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/entrypoints/pooling/
  - vllm/inputs/
  - vllm/lora/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/outputs.py
  - vllm/platforms/
  - vllm/pooling_params.py
  - vllm/profiler/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  - tests/v1/simple_kv_offload
  - tests/v1/streaming_input
  - tests/v1/ec_connector/unit
  commands:
  - bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s -m 'not cpu_test' v1/kv_connector/unit --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
  - pytest -v -s -m 'not cpu_test' v1/ec_connector/unit --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB

- label: ":amd: (MI355 DPX) V1 Metrics + LM Eval (MI355 suite)"
  timeout_in_minutes: 65
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/worker
  - tests/v1/cudagraph
  - tests/v1/kv_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/entrypoints/pooling/
  - vllm/inputs/
  - vllm/lora/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/outputs.py
  - vllm/platforms/
  - vllm/pooling_params.py
  - vllm/profiler/
  - vllm/sampling_params.py
  - vllm/tokenizers/
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  - tests/v1/simple_kv_offload
  - tests/v1/streaming_input
  - tests/v1/ec_connector/unit
  commands:
  - export GIT_CONFIG_COUNT=1
  - export GIT_CONFIG_KEY_0=http.version
  - export GIT_CONFIG_VALUE_0=HTTP/1.1
  - export GIT_TERMINAL_PROMPT=0
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s -m 'not cpu_test' v1/metrics
  - pip install -U git+https://github.com/vllm-project/lm-evaluation-harness.git@streaming-api
  - pytest -v -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine

- label: ":amd: (MI355 DPX) CUDAGraph"
  timeout_in_minutes: 55
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/core
  - tests/v1/executor
  - tests/v1/kv_offload
  - tests/v1/worker
  - tests/v1/cudagraph
  - tests/v1/kv_connector/unit
  - tests/v1/metrics
  - tests/entrypoints/openai/correctness/test_lmeval.py
  - vllm/v1/cudagraph_dispatcher.py
  - vllm/config/compilation.py
  - vllm/compilation
  - vllm/platforms/rocm.py
  - vllm/v1/worker/encoder_cudagraph.py
  - vllm/v1/worker/encoder_cudagraph_defs.py
  commands:
  - pytest -v -s v1/cudagraph/test_cudagraph_dispatch.py
  - pytest -v -s v1/cudagraph/test_cudagraph_mode.py
  - pytest -v -s v1/cudagraph/test_breakable_cudagraph.py
  - pytest -v -s v1/cudagraph/test_encoder_cudagraph.py


- label: ":amd: (MI355 DPX) V1 Sample (MI355 suite)"
  timeout_in_minutes: 70
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/v1/sample
  - tests/v1/logits_processors
  - tests/v1/test_oracle.py
  - tests/v1/test_request.py
  - tests/v1/test_outputs.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/inputs/
  - vllm/logger.py
  - vllm/model_executor/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s v1/sample

- label: ":amd: (MI355 DPX) V1 Logits + Oracle (MI355 suite)"
  timeout_in_minutes: 70
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: /vllm-workspace/tests
  source_file_dependencies:
  - vllm/
  - tests/v1/sample
  - tests/v1/logits_processors
  - tests/v1/test_oracle.py
  - tests/v1/test_request.py
  - tests/v1/test_outputs.py
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/inputs/
  - vllm/logger.py
  - vllm/model_executor/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s v1/logits_processors
  - pytest -v -s v1/test_oracle.py
  - pytest -v -s v1/test_request.py
  - pytest -v -s v1/test_outputs.py


- label: ":amd: (MI355 DPX) V1 Spec Decode" # TBD
  timeout_in_minutes: 50
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/v1/spec_decode
  - vllm/config/
  - vllm/distributed/
  - vllm/inputs/
  - vllm/model_executor/
  - vllm/models/qwen4_exp/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  optional: true
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s -m 'not slow_test' v1/spec_decode

- label: ":amd: (MI355 DPX) Spec Decode AL MTP + Other Acceptance Nightly"
  timeout_in_minutes: 55
  parallelism: 3
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/v1/spec_decode/
  - vllm/v1/sample/
  - vllm/v1/worker/gpu/
  - vllm/v1/worker/gpu_model_runner.py
  - vllm/v1/worker/gpu_worker.py
  - vllm/model_executor/model_loader/
  - vllm/model_executor/models/gemma4.py
  - vllm/model_executor/models/gemma4_mm.py
  - vllm/model_executor/models/gemma4_mtp.py
  - vllm/model_executor/models/llama.py
  - vllm/model_executor/models/llama_eagle3.py
  - vllm/model_executor/models/medusa.py
  - vllm/model_executor/models/registry.py
  - tests/evals/gsm8k/
  - tests/utils.py
  - tests/v1/e2e/spec_decode/
  - vllm/platforms/rocm.py
  - vllm/v1/worker/gpu/spec_decode/
  - tests/v1/e2e/spec_decode/conftest.py
  commands:
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "0" ]; then pytest -v -s v1/e2e/spec_decode/acceptance_rates/mtp_other/ --deselect "tests/v1/e2e/spec_decode/acceptance_rates/mtp_other/test_mtp.py::test_gemma4_mtp_acceptance_lengths[False]" --deselect "tests/v1/e2e/spec_decode/acceptance_rates/mtp_other/test_mtp.py::test_gemma4_mtp_acceptance_lengths[True]"; fi
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "1" ]; then pytest -v -s "v1/e2e/spec_decode/acceptance_rates/mtp_other/test_mtp.py::test_gemma4_mtp_acceptance_lengths[False]"; fi
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "2" ]; then pytest -v -s "v1/e2e/spec_decode/acceptance_rates/mtp_other/test_mtp.py::test_gemma4_mtp_acceptance_lengths[True]"; fi

#-----------------------------------------------------------  mi355 · misc  ------------------------------------------------------------#

- label: ":amd: (MI355 DPX) Regression" # TBD
  timeout_in_minutes: 25
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  optional: true
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - tests/test_regression
  - vllm/config/
  - vllm/distributed/
  - vllm/engine/
  - vllm/inputs/
  - vllm/model_executor/
  - vllm/multimodal/
  - vllm/platforms/
  - vllm/sampling_params.py
  - vllm/transformers_utils/
  - vllm/utils/
  - vllm/v1/
  commands:
  - pip install 'modelscope<1.38'
  - pytest -v -s test_regression.py

- label: ':amd: (MI355) DeepSeek V4 MoE Integration'
  timeout_in_minutes: 20
  source_file_dependencies:
  - tests/models/test_deepseek_v4_vl_rocm.py
  - vllm/models/deepseek_v4/amd/
  - vllm/models/deepseek_v4/common/
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/activation.py
  - vllm/model_executor/layers/linear.py
  - vllm/config/
  - vllm/forward_context.py
  - vllm/v1/worker/workspace.py
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/envs.py
  commands:
  - VLLM_ROCM_USE_AITER_MOE=0 pytest -v -s models/test_deepseek_v4_vl_rocm.py
  dind: false
  working_dir: /vllm-workspace/tests
  agent_pool: mi355_1
  num_gpus: 1
  mirror_hardwares:
  - amdexperimental
  - amdproduction
  - amdgfx950nightly
  - amdmi355
  optional: true

- label: ':amd: (MI355) Quantized MoE'
  timeout_in_minutes: 45
  working_dir: /vllm-workspace/
  source_file_dependencies:
  - tests/quantization/test_rocm_moe.py
  - tests/utils.py
  - vllm/model_executor/models/deepseek_v2.py
  - vllm/model_executor/models/qwen3_moe.py
  - vllm/model_executor/models/gpt_oss.py
  - vllm/model_executor/model_loader/
  - vllm/model_executor/layers/fused_moe/
  - vllm/model_executor/layers/quantization/
  - vllm/model_executor/kernels/linear/
  - vllm/v1/attention/
  - vllm/v1/worker/gpu/
  - vllm/_aiter_ops.py
  - vllm/platforms/rocm.py
  - vllm/envs.py
  commands:
  - pytest -v -s tests/quantization/test_rocm_moe.py
  dind: false
  agent_pool: mi355_1
  num_gpus: 1
  mirror_hardwares:
  - amdexperimental
  - amdproduction
  - amdgfx950nightly
  - amdmi355
  optional: true

- label: ':amd: (MI355) Fusion E2E TP2 AsyncTP BF16 Config Sweep'
  timeout_in_minutes: 30
  working_dir: /vllm-workspace/
  source_file_dependencies:
  - vllm/compilation/
  - vllm/model_executor/layers/linear.py
  - vllm/model_executor/layers/utils.py
  - vllm/distributed/
  - vllm/v1/attention/backends/triton_attn.py
  - tests/compile/fusions_e2e/
  - tests/compile/passes/distributed/test_async_tp.py
  - tests/compile/rocm/test_async_tp.py
  - tests/compile/correctness_e2e/test_async_tp.py
  - vllm/model_executor/kernels/linear/
  - vllm/model_executor/models/qwen3.py
  commands:
  - rocm-smi
  - pytest -v -s tests/compile/rocm/test_async_tp.py
  - pytest -v -s tests/compile/fusions_e2e/test_tp2_rocm_async_tp.py
  - pytest -v -s tests/compile/correctness_e2e/test_async_tp.py::test_rocm_async_tp_bf16_output_correctness
  dind: false
  agent_pool: mi355_2
  num_gpus: 2
  mirror_hardwares:
  - amdexperimental
  - amdproduction
  - amdgfx950nightly
  - amdmi355
  optional: true

- label: ":amd: (MI355 DPX) MRV2 Sampler JIT Warmup"
  timeout_in_minutes: 10
  mirror_hardwares: [amdexperimental, amdproduction, amdgfx950nightly, amdmi355]
  dind: false
  agent_pool: mi355_dpx
  num_gpus: 1
  optional: true
  working_dir: "/vllm-workspace/"
  source_file_dependencies:
  - tests/jit_monitor/
  - tests/models/utils.py
  - tests/models/registry.py
  - tests/utils.py
  - vllm/v1/worker/gpu/
  - vllm/v1/worker/gpu_worker.py
  - vllm/model_executor/warmup/
  - vllm/v1/sample/ops/topk_topp_sampler.py
  - vllm/utils/jit_monitor.py
  - vllm/config/observability.py
  - vllm/platforms/rocm.py
  - vllm/envs.py
  commands:
  - pytest -v -s tests/jit_monitor/test_no_runtime_jit_rocm.py --timeout=300
