name: CI (CUDA, windows)

# TODO: this workflow is only triggered manually because it is very heavy on the CI
#       when we provision dedicated windows runners, we can enable it for pushes too
# note: running this workflow manually will populate the ccache for the release builds
#       this can be used before merging a PR to speed up the release workflow
on:
  workflow_dispatch: # allows manual triggering

cache-mode: write
permissions:
  actions: write
  contents: read

# note: this will run in queue with the release workflow
concurrency:
  group: release
  queue: max

env:
  GH_TOKEN: ${{ github.token }}
  GGML_NLOOP: 3
  GGML_N_THREADS: 1
  LLAMA_ARG_LOG_COLORS: 1
  LLAMA_ARG_LOG_PREFIX: 1
  LLAMA_ARG_LOG_TIMESTAMPS: 1

jobs:
  cuda:
    name: windows-cuda (${{ matrix.cuda }}, ${{ matrix.arch }})
    runs-on: windows-2022

    permissions:
      actions: write

    strategy:
      matrix:
        include:
          # CTK >= 13.5 bundles CCCL >= 3.5; omit GGML_CUDA_CCCL_VERSION for those versions.
          - cuda: '12.4'
            arch: x64
            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
          - cuda: '13.4'
            arch: x64
            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
          - cuda: '13.4'
            arch: arm64
            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3 -DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'

    steps:
      - name: Clone
        id: checkout
        uses: actions/checkout@v6

      - name: ccache
        uses: ggml-org/ccache-action@v1.2.24
        with:
          key: release-windows-2022-${{ matrix.arch }}-cuda-${{ matrix.cuda }}

      - name: Install Cuda Toolkit
        uses: ./.github/actions/windows-setup-cuda
        with:
          cuda_version: ${{ matrix.cuda }}
          cuda_arch: ${{ matrix.arch }}

      - name: Install Ninja
        id: install_ninja
        run: |
          choco install ninja

      - name: Build
        id: cmake_build
        shell: cmd
        run: |
          call "C:\Program Files\Microsoft Visual Studio\2022\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" ${{ matrix.arch == 'x64' && 'x64' || 'amd64_arm64' }}
          cmake -S . -B build -G "Ninja Multi-Config" ^
            -DGGML_BACKEND_DL=ON ^
            -DGGML_NATIVE=OFF ^
            -DGGML_CPU=OFF ^
            -DGGML_CUDA=ON ^
            -DLLAMA_BUILD_BORINGSSL=ON ${{ matrix.defines }}
          set /A NINJA_JOBS=%NUMBER_OF_PROCESSORS%-1
          cmake --build build --config Release -j %NINJA_JOBS% --target ggml-cuda

      - name: ccache-clear
        uses: ./.github/actions/ccache-clear
        with:
          key: release-windows-2022-${{ matrix.arch }}-cuda-${{ matrix.cuda }}

  hip:
    runs-on: windows-2022

    permissions:
      actions: write

    env:
      # Make sure this is in sync with build-cache.yml
      ROCM_VERSION: "7.14.0"

    strategy:
      matrix:
        include:
          # sync with release.yml
          - name: "radeon"
            gpu_targets: "gfx1150;gfx1151;gfx1200;gfx1201;gfx1100;gfx1101;gfx1102;gfx1030;gfx1031;gfx1032"

    steps:
      - name: Clone
        id: checkout
        uses: actions/checkout@v6

      # - name: Cache ROCm Installation
      #   uses: actions/cache@v5
      #   id: cache-rocm
      #   with:
      #     path: C:\TheRock\build
      #     key: rocm-wheels-${{ env.ROCM_VERSION }}-multi-arch-${{ runner.os }}

      - name: Setup ROCm
        # if: steps.cache-rocm.outputs.cache-hit != 'true'
        uses: ./.github/actions/windows-setup-rocm
        with:
          version: ${{ env.ROCM_VERSION }}

      - name: Setup ROCm Environment
        run: |
          $ErrorActionPreference = "Stop"

          # Activate venv from cache or fresh install
          & C:\TheRock\build\.venv\Scripts\Activate.ps1

          # Expand the devel tree (idempotent; no-op if already done during install)
          rocm-sdk init
          if ($LASTEXITCODE -ne 0) { throw "rocm-sdk init failed with exit code $LASTEXITCODE" }

          # Get ROCm installation paths using the rocm-sdk CLI tool
          $rocmPath = (rocm-sdk path --root)
          if (-not $rocmPath) { throw "rocm-sdk path --root returned empty - devel package may not be installed" }
          $rocmPath = $rocmPath.Trim()
          $cmakePath = (rocm-sdk path --cmake).Trim()
          $binPath = (rocm-sdk path --bin).Trim()
          write-host "ROCm root: $rocmPath"

          echo "HIP_PATH=$rocmPath" >> $env:GITHUB_ENV
          echo "CMAKE_PREFIX_PATH=$cmakePath" >> $env:GITHUB_ENV
          echo "HIP_DEVICE_LIB_PATH=$rocmPath\lib\llvm\amdgcn\bitcode" >> $env:GITHUB_ENV
          echo "HIP_PLATFORM=amd" >> $env:GITHUB_ENV
          echo "LLVM_PATH=$rocmPath\lib\llvm" >> $env:GITHUB_ENV
          echo "$binPath" >> $env:GITHUB_PATH

          # Keep venv in PATH for subsequent steps
          echo "C:\TheRock\build\.venv\Scripts" >> $env:GITHUB_PATH

      - name: Verify ROCm
        id: verify
        run: |
          # Test the ROCm clang shipped in the installed wheel
          & "${env:HIP_PATH}\lib\llvm\bin\clang.exe" --version

      - name: ccache
        uses: ggml-org/ccache-action@v1.2.24
        with:
          # TODO: this build does not match the build in release.yml, so we use a different cache key
          #       ideally, the builds should match, similar to the CUDA build above so that we would be able
          #       to populate the ccache for the release with manual runs of this workflow
          #key: release-windows-2022-x64-hip-${{ env.ROCM_VERSION }}-${{ matrix.name }}
          key: cuda-windows-2022-x64-hip-${{ env.ROCM_VERSION }}-${{ matrix.name }}

      - name: Build
        id: cmake_build
        run: |
          cmake -G "Unix Makefiles" -B build -S . `
            -DCMAKE_PREFIX_PATH="${env:HIP_PATH}" `
            -DCMAKE_C_COMPILER="${env:HIP_PATH}\lib\llvm\bin\clang.exe" `
            -DCMAKE_CXX_COMPILER="${env:HIP_PATH}\lib\llvm\bin\clang++.exe" `
            -DCMAKE_HIP_COMPILER="${env:HIP_PATH}\lib\llvm\bin\clang.exe" `
            -DCMAKE_BUILD_TYPE=Release `
            -DLLAMA_BUILD_BORINGSSL=ON `
            -DHIP_PATH="${env:HIP_PATH}" `
            -DGGML_HIP=ON `
            -DGPU_TARGETS="gfx1100" `
            -DGGML_RPC=ON
          cmake --build build -j ${env:NUMBER_OF_PROCESSORS}

      - name: ccache-clear
        uses: ./.github/actions/ccache-clear
        with:
          #key: release-windows-2022-x64-hip-${{ env.ROCM_VERSION }}-${{ matrix.name }}
          key: cuda-windows-2022-x64-hip-${{ env.ROCM_VERSION }}-${{ matrix.name }}
