name: CI (self-hosted CUDA backend)

on:
  workflow_dispatch: # allows manual triggering
  push:
    branches:
      - master
    paths: [
      '.github/workflows/ci-self-hosted-cuda.yml',
      'ci/run.sh',
      '**/CMakeLists.txt',
      '**/.cmake',
      '**/*.h',
      '**/*.hpp',
      '**/*.c',
      '**/*.cpp',
      '**/*.cu',
      '**/*.cuh'
    ]

  pull_request:
    types: [opened, synchronize, reopened]
    paths: [
      '.github/workflows/ci-self-hosted-cuda.yml',
      'ci/run.sh',
      '**/CMakeLists.txt',
      '**/.cmake',
      'ggml/src/*',
      'ggml/src/ggml-cpu/**',
      'ggml/src/ggml-cuda/**'
    ]

cache-mode: none
permissions:
  contents: read

concurrency:
  group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
  cancel-in-progress: true

env:
  # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
  HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
  GGML_NLOOP: 3
  GGML_N_THREADS: 1
  LLAMA_ARG_LOG_COLORS: 1
  LLAMA_ARG_LOG_PREFIX: 1
  LLAMA_ARG_LOG_TIMESTAMPS: 1

jobs:
  gpu-cuda:
    runs-on: "hf-jobs-t4-medium:cuda13"

    steps:
      - name: Clone
        id: checkout
        uses: actions/checkout@v6

      - name: Install dependencies
        run: |
          sudo apt update
          sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip

      - name: ccache
        uses: ggml-org/ccache-action@v1.2.24
        with:
          restore: false
          save: false

      - name: ccache-buckets-restore
        uses: ./.github/actions/ccache-buckets
        with:
          key: self-hosted-gpu-cuda
          folder: llama.cpp
          hf_bucket: ggml-org/cache

      - name: Test
        id: ggml-ci
        run: |
          nvidia-smi
          GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp

      - name: ccache-buckets-save
        if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
        uses: ./.github/actions/ccache-buckets
        env:
          HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
        with:
          key: self-hosted-gpu-cuda
          folder: llama.cpp
          evict-old-files: 1d
          hf_bucket: ggml-org/cache
          save: true

  gpu-rocm:
    runs-on: [self-hosted, Linux, AMD]

    steps:
      - name: Clone
        id: checkout
        uses: actions/checkout@v6

      - name: Test
        id: ggml-ci
        # HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness
        # issue on integrated RDNA3.5 (gfx1151) where batched inference returns
        # incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches
        # restores correctness. Remove once the underlying ROCm/HIP issue is fixed.
        env:
          HIP_LAUNCH_BLOCKING: "1"
        run: |
          rocminfo
          GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp

  # TODO: provision AMD GPU machine
  # amd-rocm:
  #   runs-on: [self-hosted, Linux, AMD]

  #   steps:
  #     - name: Clone
  #       id: checkout
  #       uses: actions/checkout@v6

  #     - name: Test
  #       id: ggml-ci
  #       run: |
  #         amd-smi static
  #         GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
