group: LoRA
depends_on: 
  - image-build
steps:
- label: ":nvidia: (H200 MIG 35GB) LoRA Shard %N"
  device: h200_35gb
  key: lora
  timeout_in_minutes: 40
  source_file_dependencies:
  - vllm/lora
  - tests/lora
  commands:
    - pytest -v -s lora --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --ignore=lora/test_chatglm3_tp.py --ignore=lora/test_llama_tp.py --ignore=lora/test_qwen3_with_multi_loras.py --ignore=lora/test_olmoe_tp.py --ignore=lora/test_deepseekv2_tp.py --ignore=lora/test_gptoss_tp.py --ignore=lora/test_qwen3moe_tp.py --ignore=lora/test_qwen35_densemodel_lora.py 
  parallelism: 4
  mirror:
    amd:
      label: ":amd: (MI355 DPX) LoRA Shard %N"
      dind: false
      device: mi355_dpx
      working_dir: "/vllm-workspace/tests"
      timeout_in_minutes: 85
      source_file_dependencies:
      - vllm/lora
      - tests/lora
      - vllm/platforms/rocm.py
      depends_on:
      - image-build-amd


- label: ":nvidia: (L4) LoRA TP (Distributed) %N"
  key: lora-tp-distributed
  timeout_in_minutes: 25
  device: l4
  num_devices: 4
  parallelism: 4
  source_file_dependencies:
  - vllm/lora
  - vllm/model_executor/layers/fused_moe/
  - tests/lora
  commands:
    # FIXIT: find out which code initialize cuda before running the test
    # before the fix, we need to use spawn to test it
    - export VLLM_WORKER_MULTIPROC_METHOD=spawn
    # Alot of these tests are on the edge of OOMing
    - export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
    # There is some Tensor Parallelism related processing logic in LoRA that
    # requires multi-GPU testing for validation.
    # Timing-balanced static split from per-file walls in builds 83921/83936.
    - case "$$BUILDKITE_PARALLEL_JOB" in 0|1|2|3) ;; *) echo "unexpected BUILDKITE_PARALLEL_JOB=$$BUILDKITE_PARALLEL_JOB" && exit 1;; esac
    - test "$$BUILDKITE_PARALLEL_JOB" != "3" || pytest -v -s -x lora/test_chatglm3_tp.py
    - test "$$BUILDKITE_PARALLEL_JOB" != "1" || pytest -v -s -x lora/test_llama_tp.py
    - test "$$BUILDKITE_PARALLEL_JOB" != "3" || pytest -v -s -x lora/test_qwen3_with_multi_loras.py
    - test "$$BUILDKITE_PARALLEL_JOB" != "2" || pytest -v -s -x lora/test_olmoe_tp.py
    - test "$$BUILDKITE_PARALLEL_JOB" != "0" || pytest -v -s -x lora/test_gptoss_tp.py
    - test "$$BUILDKITE_PARALLEL_JOB" != "1" || pytest -v -s -x lora/test_qwen35_densemodel_lora.py
    - test "$$BUILDKITE_PARALLEL_JOB" != "2" || pytest -v -s -x lora/test_gemma4_tp.py
    - test "$$BUILDKITE_PARALLEL_JOB" != "0" || pytest -v -s -x lora/test_sequence_classification.py::test_batched_loras_tp
  mirror:
    amd:
      label: ":amd: (MI300) LoRA TP (Distributed) %N"
      dind: false
      device: mi300_4
      working_dir: "/vllm-workspace/tests"
      timeout_in_minutes: 65
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - vllm/lora
      - vllm/model_executor/layers/fused_moe/
      - vllm/platforms/rocm.py
      - tests/lora
      commands:
      - export VLLM_WORKER_MULTIPROC_METHOD=spawn
      # expandable_segments breaks custom all-reduce IPC on ROCm.
      - case "$$BUILDKITE_PARALLEL_JOB" in 0|1|2|3) ;; *) echo "unexpected BUILDKITE_PARALLEL_JOB=$$BUILDKITE_PARALLEL_JOB" && exit 1;; esac
      - if [ "$$BUILDKITE_PARALLEL_JOB" = "3" ]; then pytest -v -s -x lora/test_chatglm3_tp.py; fi
      - if [ "$$BUILDKITE_PARALLEL_JOB" = "1" ]; then pytest -v -s -x lora/test_llama_tp.py; fi
      - if [ "$$BUILDKITE_PARALLEL_JOB" = "3" ]; then pytest -v -s -x lora/test_qwen3_with_multi_loras.py; fi
      - if [ "$$BUILDKITE_PARALLEL_JOB" = "2" ]; then pytest -v -s -x lora/test_olmoe_tp.py; fi
      - if [ "$$BUILDKITE_PARALLEL_JOB" = "0" ]; then pytest -v -s -x lora/test_gptoss_tp.py; fi
      - if [ "$$BUILDKITE_PARALLEL_JOB" = "1" ]; then pytest -v -s -x lora/test_qwen35_densemodel_lora.py; fi
      - if [ "$$BUILDKITE_PARALLEL_JOB" = "2" ]; then pytest -v -s -x lora/test_gemma4_tp.py; fi
      - if [ "$$BUILDKITE_PARALLEL_JOB" = "0" ]; then pytest -v -s -x lora/test_sequence_classification.py::test_batched_loras_tp; fi
