name: RL CPU Unit Tests

on:
  push:
    branches: [ main ]
    tags:
      - ciflow/rl/*
  schedule:
    # Runs every 12 hours
    - cron: '0 */12 * * *'

concurrency:
  group: unit-test-${{ github.workflow }}-${{ github.ref == 'refs/heads/main' && github.run_number || github.ref }}
  cancel-in-progress: true

permissions:
  id-token: write
  contents: read

jobs:
  set-matrix:
    # Skip scheduled runs on forks, where they would only fail and email the fork owner
    if: github.repository_owner == 'pytorch' || github.event_name != 'schedule'
    uses: ./.github/workflows/set-matrix.yaml
    with:
      runner-cuda: mt-l-x86aavx2-11-41-a10g
      gpu-arch: cuda

  rl-cpu-unit-tests:
    name: RL CPU Unit Tests
    needs: set-matrix
    if: ${{ needs.set-matrix.outputs.docker-hash != '' }}
    uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
    with:
      runner: mt-l-x86iavx512-8-64
      docker-image: 308535385114.dkr.ecr.us-east-1.amazonaws.com/torchtitan/torchtitan-ubuntu-22.04-clang12:rl-${{ needs.set-matrix.outputs.docker-hash }}
      repository: pytorch/torchtitan
      timeout: 60
      script: |
        set -eux

        CONDA_ENV=$(conda env list --json | jq -r ".envs | .[-1]")
        conda activate "${CONDA_ENV}"

        pip config --user set global.progress_bar off

        python -m pip install uv

        # The DAPO math example's rubric needs math-verify.
        uv pip install -r torchtitan/rl/examples/dapo_math/requirements.txt

        # Install matching torch, torchvision, and vLLM CUDA nightlies here.
        uv pip install --upgrade torch torchvision vllm xformers --pre \
          --extra-index-url https://download.pytorch.org/whl/nightly/cu132 \
          --index-strategy unsafe-best-match \
          --constraint .ci/docker/requirements.txt \
          --constraint torchtitan/rl/examples/dapo_math/requirements.txt
        python -c '
        from importlib.metadata import version
        wheels = {name: version(name) for name in ("torch", "torchvision", "vllm")}
        print(wheels)
        assert all(".dev" in wheel and "+cu132" in wheel for wheel in wheels.values()), wheels
        '

        # TorchStore is not on PyPI; do not let its dependencies change torch.
        uv pip install --no-deps "git+https://github.com/meta-pytorch/torchstore.git@main"

        # The shared HF cache is read-only and does not contain this tokenizer.
        # Fetch only tokenizer files into a writable cache (no model weights).
        export MUSE_GLIMMER_TOKENIZER="$(python -c '
        import os
        from huggingface_hub import snapshot_download

        print(snapshot_download(
            "meta-models/Muse-Glimmer-30B",
            cache_dir=os.path.join(os.environ["RUNNER_TEMP"], "rl-hf-cache"),
            allow_patterns=["config.json", "tokenizer*.json", "chat_template.jinja"],
        ))
        ')"

        export PYTHONPATH="$PWD:${PYTHONPATH:-}"
        pytest tests/rl/unit_tests/cpu --strict-markers --durations=20 -vv

        # Cover the optional Verifiers example last, without installing it for all
        # RL jobs. Verifiers 0.3.1 requires mcp<2 while current vLLM nightlies
        # require mcp>=2, so resolving them together falls back to an old vLLM
        # nightly that pins an old torch. Keep the nightlies installed above and
        # let Verifiers downgrade its own dependencies (e.g. mcp) for these tests.
        python -c '
        from importlib.metadata import version
        for name in ("torch", "torchvision", "vllm"):
            print(f"{name}=={version(name)}")
        ' > "${RUNNER_TEMP}/nightly-constraints.txt"
        uv pip install -r torchtitan/rl/examples/verifiers/dapo_math/requirements.txt \
          --constraint "${RUNNER_TEMP}/nightly-constraints.txt"
        pytest tests/rl/unit_tests/cpu/test_verifiers.py \
          tests/rl/unit_tests/cpu/test_verifiers_example.py \
          --strict-markers --durations=20 -vv
