group: JIT Monitor
depends_on:
  - image-build
steps:
- label: ":nvidia: (H200 MIG 35GB) No Runtime JITs E2E"
  key: jit-monitor-no-runtime-jit
  device: h200_35gb
  timeout_in_minutes: 45
  source_file_dependencies:
  - vllm/utils/jit_monitor.py
  - vllm/v1/worker/gpu_worker.py
  - vllm/model_executor/warmup/
  - vllm/config/observability.py
  - tests/jit_monitor/test_no_runtime_jit.py
  - tests/models/registry.py
  commands:
    # Boot a curated JIT-heavy model set with the JIT monitor in "error" mode
    # and run generation; any post-warmup JIT compilation fails the test.
    # Per-test watchdog so a wedged engine/CUDA init fails with a traceback
    # instead of running until the build timeout.
    - export PYTHONFAULTHANDLER=1
    - pytest -v -s jit_monitor/test_no_runtime_jit.py --timeout=900 --timeout-method=thread
  mirror:
    amd:
      # Focused MRV2 regression; the broader model/backend matrix is pending.
      label: ':amd: (MI355 DPX) MRV2 Sampler JIT Warmup'
      dind: false
      device: mi355_dpx
      num_devices: 1
      timeout_in_minutes: 10
      working_dir: /vllm-workspace/
      depends_on:
      - image-build-amd
      source_file_dependencies:
      - tests/jit_monitor/
      - tests/models/utils.py
      - tests/utils.py
      - vllm/v1/worker/gpu/
      - vllm/v1/sample/ops/topk_topp_sampler.py
      - vllm/platforms/rocm.py
      - vllm/envs.py
      commands:
      - pytest -v -s tests/jit_monitor/test_no_runtime_jit_rocm.py --timeout=300
