group: Benchmarks
depends_on: 
  - image-build
steps:
- label: ":nvidia: (H200 MIG 18GB) Benchmarks CLI"
  key: benchmarks-cli-test
  timeout_in_minutes: 45
  device: h200_18gb
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - rust/src/bench/tests/python_serve_flags.txt
  - tests/benchmarks/
  commands:
  - pytest -v -s benchmarks/
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Benchmarks CLI"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 40
      depends_on:
      - image-build-amd

- label: ":nvidia: (B200) Attention Benchmark Smoke"
  key: attention-benchmarks-smoke-test-b200
  device: b200-k8s
  num_gpus: 2
  optional: true
  working_dir: "/vllm-workspace/"
  timeout_in_minutes: 20
  source_file_dependencies:
  - benchmarks/attention_benchmarks/
  - vllm/v1/attention/
  commands:
  - python3 benchmarks/attention_benchmarks/benchmark.py --backends flash flashinfer --batch-specs "8q1s1k"
  mirror:
    amd:
      label: ":amd: (MI355) Attention Benchmark Smoke"
      dind: false
      device: mi355_2
      timeout_in_minutes: 25
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - benchmarks/attention_benchmarks/
      - vllm/v1/attention/
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      commands:
      - python3 benchmarks/attention_benchmarks/benchmark.py --backends ROCM_ATTN ROCM_AITER_FA ROCM_AITER_UNIFIED_ATTN --batch-specs "8q1s1k"
