group: Entrypoints
depends_on: 
  - image-build
steps:
- label: ":nvidia: (H200) Initialized Snapshot E2E"
  key: initialized-snapshot-e2e
  device: h200
  num_devices: 1
  no_plugin: true
  optional: true
  timeout_in_minutes: 60
  working_dir: "."
  env:
    SNAPSHOT_E2E_RUN_LIMIT_S: "1200"
    NVIDIA_VISIBLE_DEVICES: "0"
  source_file_dependencies:
  - .buildkite/scripts/initialized-snapshot-e2e.sh
  - .buildkite/test_areas/entrypoints.yaml
  - docker/Dockerfile
  - tools/install_snapshot_runtime.sh
  - vllm/entrypoints/cli/
  - vllm/snapshot/
  commands:
  - bash .buildkite/scripts/initialized-snapshot-e2e.sh "$IMAGE_TAG" "$BUILDKITE_COMMIT"

- label: ":nvidia: (H200 MIG 35GB) Entrypoints Unit"
  device: h200_35gb
  key: entrypoints-unit-tests
  timeout_in_minutes: 40
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/entrypoints
  - tests/entrypoints/unit_tests
  - tests/entrypoints/weight_transfer
  - tests/entrypoints/launchers
  commands:
  - pytest -v -s entrypoints/unit_tests
  - pytest -v -s entrypoints/weight_transfer
  - pytest -v -s entrypoints/launchers
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Entrypoints Unit"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 50
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - vllm/platforms/rocm.py

- label: ":nvidia: (H200 MIG 35GB) Entrypoints Integration (LLM)"
  device: h200_35gb
  key: entrypoints-integration-llm
  timeout_in_minutes: 60
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - tests/entrypoints/llm
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/llm --ignore=entrypoints/llm/test_generate.py --ignore=entrypoints/llm/test_collective_rpc.py --ignore=entrypoints/llm/offline_mode
  - pytest -v -s entrypoints/llm/test_generate.py # it needs a clean process
  - pytest -v -s entrypoints/llm/offline_mode # Needs to avoid interference with other tests
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Entrypoints Integration (LLM)"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 90
      depends_on:
      - image-build-amd

- label: ":nvidia: (H200 MIG 35GB) Entrypoints Integration (API Server) %N"
  key: entrypoints-integration-api-server
  device: h200_35gb
  timeout_in_minutes: 75
  parallelism: 4
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - tests/entrypoints/serve
  - tests/entrypoints/scale_out
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/serve --ignore=entrypoints/serve/dev/rpc --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
  - if [ "$$BUILDKITE_PARALLEL_JOB" = "1" ]; then PYTHONPATH=/vllm-workspace pytest -v -s entrypoints/serve/dev/rpc; fi
  - pytest -v -s entrypoints/scale_out --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Entrypoints Integration (API Server) %N"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 65
      depends_on:
      - image-build-amd

- label: ":nvidia: (H200 MIG 35GB) Entrypoints Integration (OpenAI API completion)"
  device: h200_35gb
  key: entrypoints-integration-api-server-openai-completion
  timeout_in_minutes: 68
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - tests/entrypoints/openai
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/openai/completion --ignore=entrypoints/openai/completion/test_tensorizer_entrypoint.py
  - pytest -v -s entrypoints/openai --ignore=entrypoints/openai/completion --ignore=entrypoints/openai/chat_completion --ignore=entrypoints/openai/responses --ignore=entrypoints/openai/correctness
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Entrypoints Integration (OpenAI API completion)"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 65
      depends_on:
      - image-build-amd

- label: ":nvidia: (H200 MIG 35GB) Entrypoints Integration (OpenAI API chat_completion)"
  device: h200_35gb
  key: entrypoints-integration-api-server-openai-chat_completion
  timeout_in_minutes: 83
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - tests/entrypoints/openai
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/openai/chat_completion
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Entrypoints Integration (OpenAI API chat_completion)"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 70
      depends_on:
      - image-build-amd

- label: ":nvidia: (H200 MIG 35GB) Entrypoints Integration (API Server Generate)"
  device: h200_35gb
  key: entrypoints-integration-api-server-generate
  timeout_in_minutes: 50
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - tests/tool_use
  - tests/entrypoints/tool_parsers
  - tests/entrypoints/generate
  - tests/entrypoints/anthropic
  - tests/entrypoints/cohere
  commands:
  - pytest -v -s tool_use
  - pytest -v -s entrypoints/tool_parsers
  - pytest -v -s entrypoints/generate
  - pytest -v -s entrypoints/anthropic
  - pytest -v -s entrypoints/cohere
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Entrypoints Integration (API Server Generate)"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 65
      depends_on:
      - image-build-amd

- label: ":nvidia: (H200 MIG 35GB) Entrypoints Integration (Responses API)"
  device: h200_35gb
  key: entrypoints-integration-responses-api
  timeout_in_minutes: 50
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - tests/entrypoints/openai/responses
  commands:
  - pytest -v -s entrypoints/openai/responses
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Entrypoints Integration (Responses API)"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 50
      depends_on:
      - image-build-amd

- label: ":nvidia: (H200 MIG 35GB) Entrypoints Integration (Speech to Text)"
  device: h200_35gb
  key: entrypoints-integration-speech_to_text
  timeout_in_minutes: 45
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - tests/entrypoints/speech_to_text
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/speech_to_text
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Entrypoints Integration (Speech to Text)"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 60
      working_dir: "/vllm-workspace/tests"
      depends_on:
      - image-build-amd

- label: ":nvidia: (H200 MIG 35GB) Entrypoints Integration (Multimodal)"
  device: h200_35gb
  key: entrypoints-integration-multimodal
  timeout_in_minutes: 45
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - tests/entrypoints/multimodal
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/multimodal
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Entrypoints Integration (Multimodal)"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 55
      depends_on:
      - image-build-amd

- label: ":nvidia: (H200 MIG 35GB) Entrypoints Integration (Pooling)"
  device: h200_35gb
  key: entrypoints-integration-pooling
  timeout_in_minutes: 75
  working_dir: "/vllm-workspace/tests"
  source_file_dependencies:
  - vllm/
  - "!vllm/distributed/kv_transfer/"
  - tests/entrypoints/pooling
  commands:
  - export VLLM_WORKER_MULTIPROC_METHOD=spawn
  - pytest -v -s entrypoints/pooling
  mirror:
    amd:
      label: ":amd: (MI355 DPX) Entrypoints Integration (Pooling)"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 65
      working_dir: "/vllm-workspace/tests"
      depends_on:
      - image-build-amd

- label: ":nvidia: (H200 MIG 18GB) OpenAI API Correctness"
  key: openai-api-correctness
  timeout_in_minutes: 20
  device: h200_18gb
  source_file_dependencies:
  - csrc/
  - vllm/entrypoints/openai/
  commands: # LMEval
  - pytest -s entrypoints/openai/correctness/
  mirror:
    amd:
      label: ":amd: (MI355 DPX) OpenAI API Correctness"
      dind: false
      device: mi355_dpx
      timeout_in_minutes: 30
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - csrc/
      - vllm/entrypoints/openai/
      - vllm/model_executor/layers/
      - vllm/v1/attention/backends/
      - vllm/v1/attention/selector.py
      - vllm/_aiter_ops.py
      - vllm/platforms/rocm.py
      - vllm/model_executor/model_loader/
      commands:
      - bash ../tools/install_torchcodec_rocm.sh || exit 1
      - pytest -s entrypoints/openai/correctness/
