name: Transformers Modeling Backend 8 GPU Integration Tests

on:
  push:
    branches: [ main ]
    paths:
      - 'torchtitan/experiments/transformers_modeling_backend/**'
      - '.github/workflows/integration_test_8gpu_transformers_modeling_backend.yaml'
  pull_request:
    types: [opened, synchronize, reopened, ready_for_review]
    paths:
      - 'torchtitan/experiments/transformers_modeling_backend/**'
      - '.github/workflows/integration_test_8gpu_transformers_modeling_backend.yaml'
  schedule:
    # Runs every 12 hours
    - cron: '0 */12 * * *'

concurrency:
  group: unit-test${{ github.workflow }}-${{ github.ref == 'refs/heads/main' && github.run_number || github.ref }}
  cancel-in-progress: true

defaults:
  run:
    shell: bash -l -eo pipefail {0}

permissions:
      id-token: write
      contents: read

jobs:
  # Step 1: Dynamically compute the matrix based on conditions
  set-matrix:
    # Skip scheduled runs on forks, where they would only fail and email the fork owner
    if: github.repository_owner == 'pytorch' || github.event_name != 'schedule'
    uses: ./.github/workflows/set-matrix.yaml
    with:
      is-experimental: true

  # Step 2: Use the dynamic matrix in the build-test job
  build-test:
    needs: set-matrix
    uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
    strategy:
      fail-fast: false
      matrix: ${{ fromJSON(needs.set-matrix.outputs.matrix) }}
    with:
      runner: ${{ matrix.runner }}
      gpu-arch-type: ${{ matrix.gpu-arch-type }}
      gpu-arch-version: ${{ matrix.gpu-arch-version }}
      docker-image: 308535385114.dkr.ecr.us-east-1.amazonaws.com/torchtitan/${{ matrix.docker-image }}:${{ needs.set-matrix.outputs.docker-hash }}
      repository: pytorch/torchtitan
      upload-artifact: outputs
      timeout: 45
      script: |
        set -eux

        # The generic Linux job chooses to use base env, not the one setup by the image
        CONDA_ENV=$(conda env list --json | jq -r ".envs | .[-1]")
        conda activate "${CONDA_ENV}"

        # Log CUDA driver version for debugging.
        DRIVER_VERSION=$(nvidia-smi --query-gpu=driver_version --format=csv,noheader | head -n 1 || true)
        echo "CUDA driver version: ${DRIVER_VERSION}"

        pip config --user set global.progress_bar off

        TORCH_SPEC="torch"
        if [ -n "${{ matrix.torch-version }}" ]; then
          TORCH_SPEC="torch==${{ matrix.torch-version }}"
        fi
        # torchvision must be installed from the nightly channel alongside torch
        # so its C++ extensions (e.g. torchvision::nms) match the torch ABI.
        # transformers pulls in torchvision transitively via image_utils.
        python -m pip install --force-reinstall --pre \
          "${TORCH_SPEC}" torchvision --index-url ${{ matrix.index-url }}

        if [[ "${{ matrix.gpu-arch-type }}" == "rocm" ]]; then
          export HIPBLASLT_TENSILE_LIBPATH="$(python -c 'import os, torch; print(os.path.join(os.path.dirname(torch.__file__), "lib", "hipblaslt", "library"))')"
          echo "HIPBLASLT_TENSILE_LIBPATH=${HIPBLASLT_TENSILE_LIBPATH}"
        fi

        USE_CPP=0 python -m pip install --pre torchao --index-url ${{ matrix.index-url }}

        # MoE parallelism requires transformers >= 5.0.0 (grouped experts format).
        # Install here rather than in the shared Docker image to avoid affecting
        # other CI workflows (e.g. unit_test_cpu tokenizer tests).
        python -m pip install transformers==5.9.0

        sudo mkdir -p "$RUNNER_TEMP/artifacts-to-be-uploaded"
        sudo chown -R $(id -u):$(id -g) "$RUNNER_TEMP/artifacts-to-be-uploaded"

        python -m torchtitan.experiments.transformers_modeling_backend.tests.integration_tests $RUNNER_TEMP/artifacts-to-be-uploaded --ngpu 8
