group: LM Eval
depends_on:
  - image-build-xpu
steps:
- label: "XPU GPQA Eval (GPT-OSS)"
  key: xpu-gpqa-eval-gpt-oss
  timeout_in_minutes: 60
  device: intel_gpu
  agent_tags:
    label: production
    gpu: 1+
    mem: 24+
  no_plugin: true
  working_dir: "."
  env:
    REGISTRY: "public.ecr.aws/q9t5s3a7"
    REPO: "vllm-ci-test-repo"
    VLLM_TEST_DEVICE: "xpu"
  source_file_dependencies:
    - csrc/
    - vllm/model_executor/layers/quantization
    - tests/evals/gpt_oss/
  commands:
    - >-
      bash .buildkite/scripts/hardware_ci/run-intel-test.sh
      'cd tests &&
      pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-xpu.txt'
- label: LM Eval Large Models(2 GPUs)
  key: lm-eval-large-models-2-gpus
  timeout_in_minutes: 45
  device: intel_gpu
  agent_tags:
    label: production
    gpu: 2+
    mem: 24+
  no_plugin: true
  optional: true
  working_dir: "."
  env:
    REGISTRY: "public.ecr.aws/q9t5s3a7"
    REPO: "vllm-ci-test-repo"
    VLLM_TEST_DEVICE: "xpu"
  source_file_dependencies:
    - vllm/
    - .buildkite/lm-eval-harness/
  commands:
    - >-
      bash .buildkite/scripts/hardware_ci/run-intel-test.sh
      'cd .buildkite/lm-eval-harness &&
       export VLLM_USE_DEEP_GEMM=0 &&
      pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large-xpu-fp8.txt --tp-size=2'
