name: Benchmarks
env:
  # TODO: this rescheduling makes gpt2, mixtral and llama unjitted slower
  # TODO: very slow for llama 70B and resnet training 6 GPU
  CAPTURE_PROCESS_REPLAY: "1"
  ASSERT_PROCESS_REPLAY: "0"
  PYTHONPATH: .
  GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}

on:
  push:
    branches:
      - master
      - update_benchmark
      - update_benchmark_staging
  workflow_dispatch:

jobs:
  # the goal of this test is to replicate a normal person on a laptop running the test
  # no process replay, no benchmarks, no CI, just a normal laptop person
  # the 3 minute timeout should not be raised
  testmacpytest:
    name: Mac pytest
    env:
      CI: ""
      CAPTURE_PROCESS_REPLAY: "0"
    runs-on: [self-hosted, macOS]
    timeout-minutes: 4
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    # brew install uv
    - name: setup python environment
      run: |
        rm -rf /tmp/tinygrad_pytest_ci
        uv venv /tmp/tinygrad_pytest_ci
        source /tmp/tinygrad_pytest_ci/bin/activate
        uv pip install .[testing]
    - name: setup staging db
      run: |
        echo "CACHEDB=/tmp/pytest-db-ci.db" >> $GITHUB_ENV
        rm -f /tmp/pytest-db-ci*
    - name: Run pytest -nauto
      run: |
        source /tmp/tinygrad_pytest_ci/bin/activate
        pytest -nauto --durations=20

  # TODO: reenable when not flaky
  #testframeworkpytest:
  #  name: framework pytest
  #  env:
  #    CI: ""
  #    CAPTURE_PROCESS_REPLAY: "0"
  #  runs-on: [self-hosted, framework]
  #  timeout-minutes: 10
  #  defaults:
  #    run:
  #      shell: bash -e -o pipefail {0}
  #  if: github.repository_owner == 'tinygrad'
  #  steps:
  #  - name: Checkout Code
  #    uses: actions/checkout@v6
  #  - name: setup python environment
  #    run: |
  #      rm -rf /tmp/tinygrad_pytest_ci
  #      uv venv /tmp/tinygrad_pytest_ci
  #      source /tmp/tinygrad_pytest_ci/bin/activate
  #      uv pip install .[testing]
  #  - name: setup staging db
  #    run: |
  #      echo "CACHEDB=/tmp/pytest-db-ci.db" >> $GITHUB_ENV
  #      rm -f /tmp/pytest-db-ci*
  #  - name: Run pytest -nauto
  #    run: |
  #      source /tmp/tinygrad_pytest_ci/bin/activate
  #      pytest -nauto --durations=20

  llmbenchmark:
    name: Benchmark ${{ matrix.model }} (DEV=${{ matrix.dev }})
    runs-on: [self-hosted, "${{ matrix.dev == 'METAL' && 'macOS' || matrix.dev == 'AMD' && 'tinybox' || 'tinyboxgreen' }}"]
    strategy:
      fail-fast: false
      matrix:
        dev: ['METAL', 'AMD', 'NV']
        model: ['llama3.2:3b-f16', 'qwen3.8:27b', 'olmoe']
        # qwen3.8:27b doesn't fit on mac
        exclude: [{ dev: 'METAL', model: 'qwen3.8:27b' }, { dev: 'AMD', model: 'olmoe' }, { dev: 'NV', model: 'olmoe' }]
    timeout-minutes: 15
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    env:
      DEV: ${{ matrix.dev }}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: Setup (AMD)
      if: ${{ matrix.dev == 'AMD' }}
      run: |
        ./extra/hcq/hcq_smi.py amd rmmod --expect
        ./extra/hcq/hcq_smi.py amd kill_pids --sudoless
    - name: Setup (NV)
      if: ${{ matrix.dev == 'NV' }}
      run: ./extra/hcq/hcq_smi.py nv kill_pids --sudoless
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: reset process replay
      run: python3 test/external/process_replay/reset.py
    - name: Run ${{ matrix.model }}
      run: |
        MODEL=${{ matrix.model }}
        BENCHMARK_LOG=${MODEL//./} JITBEAM=2 IGNORE_BEAM_CACHE=1 python3 -m tinygrad.llm -m $MODEL --benchmark --warmup
    - name: Run process replay tests
      uses: ./.github/actions/process-replay

  cifarbenchmark:
    name: HLB-CIFAR10 (DEV=${{ matrix.dev }})
    runs-on: [self-hosted, "${{ matrix.dev == 'METAL' && 'macOS' || matrix.dev == 'AMD' && 'tinybox' || 'tinyboxgreen' }}"]
    strategy:
      fail-fast: false
      matrix:
        dev: ['METAL', 'AMD', 'NV']
    timeout-minutes: 10
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    env:
      DEV: ${{ matrix.dev }}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: Setup (AMD)
      if: ${{ matrix.dev == 'AMD' }}
      run: |
        ./extra/hcq/hcq_smi.py amd rmmod --expect
        ./extra/hcq/hcq_smi.py amd kill_pids --sudoless
    - name: Setup (NV)
      if: ${{ matrix.dev == 'NV' }}
      run: ./extra/hcq/hcq_smi.py nv kill_pids --sudoless
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: reset process replay
      run: python3 test/external/process_replay/reset.py
    - name: Run 10 CIFAR training steps
      env:
        ASSERT_MIN_STEP_TIME: ${{ matrix.dev == 'NV' && '130' || matrix.dev == 'AMD' && '200' || '3000' }}
      run: BENCHMARK_LOG=cifar_10steps STEPS=10 python3 examples/hlb_cifar10.py
    - name: Run 10 CIFAR training steps w HALF
      env:
        ASSERT_MIN_STEP_TIME: ${{ matrix.dev == 'NV' && '120' || matrix.dev == 'AMD' && '235' || '3000' }}
      run: BENCHMARK_LOG=cifar_10steps_half STEPS=10 DEFAULT_FLOAT=HALF python3 examples/hlb_cifar10.py
    - name: Run full CIFAR training w 1 GPU
      # slow on metal
      if: ${{ matrix.dev != 'METAL' }}
      run: time BENCHMARK_LOG=cifar DEFAULT_FLOAT=HALF STEPS=1000 TARGET_EVAL_ACC_PCT=93.0 python3 examples/hlb_cifar10.py
    - name: Run process replay tests
      uses: ./.github/actions/process-replay

  mlperfbenchmark:
    name: MLPerf (${{ matrix.dev }})
    runs-on: [self-hosted, Linux, "${{ matrix.dev == 'AMD' && 'tinybox' || 'tinyboxgreen' }}"]
    strategy:
      fail-fast: false
      matrix:
        dev: ['AMD', 'NV']
    timeout-minutes: 5
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    env:
      DEV: ${{ matrix.dev }}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: Setup (AMD)
      if: ${{ matrix.dev == 'AMD' }}
      run: |
        ./extra/hcq/hcq_smi.py amd rmmod --expect
        ./extra/hcq/hcq_smi.py amd kill_pids --sudoless
    - name: Setup (NV)
      if: ${{ matrix.dev == 'NV' }}
      run: ./extra/hcq/hcq_smi.py nv kill_pids --sudoless
    - name: Symlink models and datasets
      run: |
        mkdir -p extra/datasets
        ln -s /raid/datasets/imagenet extra/datasets/imagenet
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: reset process replay
      run: test/external/process_replay/reset.py
    - name: Run 10 MLPerf ResNet50 training steps (1 gpu)
      run: BENCHMARK_LOG=resnet_10steps DEFAULT_FLOAT=HALF BENCHMARK=10 BS=256 GPUS=1 MODEL=resnet python3 examples/mlperf/model_train.py
    - name: Run process replay tests
      uses: ./.github/actions/process-replay

  sdbenchmark:
    name: Stable Diffusion (DEV=${{ matrix.dev }})
    runs-on: [self-hosted, "${{ matrix.dev == 'METAL' && 'macOS' || matrix.dev == 'AMD' && 'tinybox' || 'tinyboxgreen' }}"]
    strategy:
      fail-fast: false
      matrix:
        dev: ['METAL', 'AMD', 'NV']
    timeout-minutes: 15
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    env:
      DEV: ${{ matrix.dev }}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: Setup (AMD)
      if: ${{ matrix.dev == 'AMD' }}
      run: |
        ./extra/hcq/hcq_smi.py amd rmmod --expect
        ./extra/hcq/hcq_smi.py amd kill_pids --sudoless
    - name: Setup (NV)
      if: ${{ matrix.dev == 'NV' }}
      run: ./extra/hcq/hcq_smi.py nv kill_pids --sudoless
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: reset process replay
      run: python3 test/external/process_replay/reset.py
    - name: Run Stable Diffusion
      env:
        ASSERT_MIN_STEP_TIME: ${{ matrix.dev == 'METAL' && '720' || matrix.dev == 'AMD' && '550' || '0' }}
      run: BENCHMARK_LOG=stable_diffusion python3 examples/stable_diffusion.py --fp16 --seed 0 --noshow --timing
    - name: Run SDXL
      if: ${{ matrix.dev != 'NV' }}
      env:
        ASSERT_MIN_STEP_TIME: ${{ matrix.dev == 'METAL' && '5000' || matrix.dev == 'AMD' && '3200' || '2000' }}
      run: BENCHMARK_LOG=stable_diffusion_xl CAPTURE_PROCESS_REPLAY=0 python3 examples/sdxl.py --seed 0 --noshow --timing
    - name: Run process replay tests
      uses: ./.github/actions/process-replay

  multigpubenchmark:
    name: Multi-GPU Benchmarks (DEV=${{ matrix.dev }})
    runs-on: [self-hosted, "${{ matrix.dev == 'AMD' && 'tinybox' || 'tinyboxgreen' }}"]
    strategy:
      fail-fast: false
      matrix:
        dev: ['AMD', 'NV']
    timeout-minutes: 20
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    env:
      DEV: ${{ matrix.dev }}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: Setup (AMD)
      if: ${{ matrix.dev == 'AMD' }}
      run: |
        ./extra/hcq/hcq_smi.py amd rmmod --expect
        ./extra/hcq/hcq_smi.py amd kill_pids --sudoless
    - name: Setup (NV)
      if: ${{ matrix.dev == 'NV' }}
      run: ./extra/hcq/hcq_smi.py nv kill_pids --sudoless
    - name: Symlink models and datasets
      run: |
        mkdir -p weights
        mkdir -p extra/datasets
        ln -s /raid/weights/LLaMA-3 weights/LLaMA-3
        ln -s /raid/datasets/imagenet extra/datasets/imagenet
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: reset process replay
      run: python3 test/external/process_replay/reset.py
    - name: Run LLaMA-3 8B on 4 GPUs with BEAM
      run: BENCHMARK_LOG=llama3_beam_4gpu JITBEAM=2 IGNORE_BEAM_CACHE=1 CAPTURE_PROCESS_REPLAY=0 python3 examples/llama3.py --size 8B --shard 4 --model weights/LLaMA-3/8B-SF-DPO/ --benchmark --temperature 0
    - name: Run full CIFAR training steps w 6 GPUS
      run: time BENCHMARK_LOG=cifar_6gpu CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF STEPS=350 BS=1536 GPUS=6 TARGET_EVAL_ACC_PCT=93.0 python3 examples/hlb_cifar10.py
    - name: Run MLPerf resnet eval on training data
      run: time BENCHMARK_LOG=resnet_eval MODEL=resnet python3 examples/mlperf/model_eval.py
    - name: Run 10 MLPerf ResNet50 training steps (6 gpu)
      run: BENCHMARK_LOG=resnet_10steps_6gpu CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=1536 GPUS=6 MODEL=resnet python3 examples/mlperf/model_train.py
    - name: Run 10 MLPerf Bert training steps (6 gpu)
      # TODO: remove BERT_LAYERS once scheduler is fast
      run: BENCHMARK_LOG=bert_10steps_6gpu CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=72 GPUS=6 BERT_LAYERS=2 MODEL=bert python3 examples/mlperf/model_train.py
    - name: Run process replay tests
      uses: ./.github/actions/process-replay

  tests:
    name: Tests (DEV=${{ matrix.dev }})
    runs-on: [self-hosted, "${{ matrix.dev == 'METAL' && 'macOS' || matrix.dev == 'AMD' && 'tinybox' || 'tinyboxgreen' }}"]
    strategy:
      fail-fast: false
      matrix:
        dev: ['METAL', 'AMD', 'NV']
    timeout-minutes: 11
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    env:
      DEV: ${{ matrix.dev }}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: Setup (AMD)
      if: ${{ matrix.dev == 'AMD' }}
      run: |
        ./extra/hcq/hcq_smi.py amd rmmod --expect
        ./extra/hcq/hcq_smi.py amd kill_pids --sudoless
    - name: Setup (NV)
      if: ${{ matrix.dev == 'NV' }}
      run: ./extra/hcq/hcq_smi.py nv kill_pids --sudoless
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: reset process replay
      run: python3 test/external/process_replay/reset.py
    - name: Test tiny
      run: |
        DEBUG=2 python -m pytest -rA test/test_tiny.py
        if [[ "${{ matrix.dev }}" == "NV" ]]; then
          DEBUG=2 DEV=CUDA python -m pytest -rA test/test_tiny.py
        fi
    - name: Test tensor cores
      run: |
        if [[ "${{ matrix.dev }}" == "METAL" ]]; then
          python3 test/runtime/test_tensor_cores.py
          DEBUG=2 SHOULD_USE_TC=1 python3 extra/gemm/simple_matmul.py
          DEBUG=2 SHOULD_USE_TC=1 HALF=1 python3 extra/gemm/simple_matmul.py
          DEBUG=2 SHOULD_USE_TC=1 BFLOAT16=1 python3 extra/gemm/simple_matmul.py
          M_START=6 M_STOP=10 M_STEP=1 N_START=6 N_STOP=10 N_STEP=1 K_START=6 K_STOP=24 K_STEP=1 TC_OPT=2 DEBUG=2 python3 ./extra/gemm/fuzz_matmul.py
        elif [[ "${{ matrix.dev }}" == "NV" ]]; then
          ALLOW_TF32=1 python3 test/runtime/test_tensor_cores.py
          DEV=NV:PTX ALLOW_TF32=1 python3 test/runtime/test_tensor_cores.py
          DEV=CUDA SHOULD_USE_TC=1 HALF=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
          DEV=CUDA SHOULD_USE_TC=1 BFLOAT16=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
          DEV=CUDA SHOULD_USE_TC=1 ALLOW_TF32=1 DEBUG=2 ATOL=2e-2 python3 extra/gemm/simple_matmul.py
          DEV=CUDA SHOULD_USE_TC=1 FP8E4M3=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
          DEV=NV:PTX SHOULD_USE_TC=1 HALF=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
          SHOULD_USE_TC=1 HALF=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
          # TODO: too slow
          # M_START=12 M_STOP=20 M_STEP=1 N_START=6 N_STOP=10 N_STEP=1 K_START=28 K_STOP=36 K_STEP=1 HALF=1 TC_OPT=2 python3 ./extra/gemm/fuzz_matmul.py
          # DEV=NV:PTX M_START=12 M_STOP=20 M_STEP=1 N_START=6 N_STOP=10 N_STEP=1 K_START=28 K_STOP=36 K_STEP=1 HALF=1 TC_OPT=2 python3 ./extra/gemm/fuzz_matmul.py
        else
          python3 test/runtime/test_tensor_cores.py
          # TODO: this is flaky
          # DEV=AMD:LLVM python3 test/runtime/test_tensor_cores.py
          SHOULD_USE_TC=1 BFLOAT16=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
          SHOULD_USE_TC=1 HALF=1 DEBUG=2 ATOL=2e-2 python3 extra/gemm/simple_matmul.py
          # TODO: AMD compiler bug causes this to fail
          # HSA=1 M_START=12 M_STOP=20 M_STEP=1 N_START=12 N_STOP=20 N_STEP=1 K_START=28 K_STOP=36 K_STEP=1 HALF=1 TC_OPT=2 DEBUG=2 python3 ./extra/gemm/fuzz_matmul.py
        fi
    - name: Run model inference benchmark
      # TODO: unstable on AMD
      if: ${{ matrix.dev != 'AMD' }}
      run: CAPTURE_PROCESS_REPLAY=0 NOCLANG=1 python3 test/external/external_model_benchmark.py
    - name: Test speed vs torch
      # TODO: unstable on AMD
      if: ${{ matrix.dev != 'AMD' }}
      env:
        HALF: ${{ matrix.dev == 'NV' && '1' || '0' }}
      run: CAPTURE_PROCESS_REPLAY=0 BIG=2 ${{ matrix.dev == 'METAL' && 'MPS=1' || 'TORCHCUDA=1' }} python3 test/speed/external_test_speed_v_torch.py
    - name: Test speed vs theoretical
      # no targets for METAL
      if: ${{ matrix.dev != 'METAL' }}
      run: IGNORE_BEAM_CACHE=1 CCACHE=0 BEAM_DEBUG=1 DEBUG=1 python -m pytest -rA test/external/speed_v_theoretical.py --durations=20
    - name: Train MNIST
      run: time TARGET_EVAL_ACC_PCT=96.0 python3 examples/beautiful_mnist.py
    - name: Test benchmark allreduce
      if: ${{ matrix.dev == 'NV' }}
      run: python test/external/external_benchmark_multitensor_allreduce.py
    # TODO: HEVC decode timing test
    # - name: HEVC Decode Benchmark
    #   if: ${{ matrix.dev == 'NV' }}
    #   run: IGNORE_BEAM_CACHE=1 VALIDATE=1 MAX_FRAMES=100 ASSERT_FPS=1400 JITBEAM=1 PYTHONPATH=. python3 extra/hevc/decode.py
    - uses: actions/upload-artifact@v7
      if: ${{ matrix.dev != 'AMD' }}
      with:
        name: Speed (${{ matrix.dev }})
        path: |
          onnx_inference_speed.csv
    - name: Run process replay tests
      uses: ./.github/actions/process-replay

  testusbgpu:
    name: UsbGPU Benchmark
    runs-on: [self-hosted, macOS]
    timeout-minutes: 3
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: Kill stale pids
      run: |
        ./extra/hcq/hcq_smi.py amd kill_pids --sudoless
        ./extra/hcq/hcq_smi.py nv kill_pids --sudoless
    - name: reset chestnut
      run: python3 extra/usbgpu/debug.py -rnw
    - name: UsbGPU boot time
      run: GMMU=0 DEBUG=2 AM_RESET=1 DEV=USB+AMD time python3.11 test/test_tiny.py TestTiny.test_plus
    - name: UsbGPU tiny tests
      run: GMMU=0 DEV=USB+AMD python3.11 test/test_tiny.py
    - name: UsbGPU copy speeds
      run: SIZE=64000000 PYTHONPATH=. GMMU=0 DEV=USB+AMD python3.11 test/external/external_test_usb_asm24.py

  testcomma:
    strategy:
      matrix:
        dev: ['QCOM', 'QCOM:IR3']
        model: [driving_supercombo, dmonitoring_model]
        include:
          - model: driving_supercombo
            timing: 20
          - model: dmonitoring_model
            timing: 12
      fail-fast: false
    name: openpilot ${{ matrix.model }} (DEV=${{ matrix.dev }})
    runs-on: [self-hosted, Linux, comma4]
    timeout-minutes: 5
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    if: github.repository_owner == 'tinygrad'
    env:
      DEV: ${{ matrix.dev }}
      # sha for https://huggingface.co/commaai/openpilot-lfs
      LFS_SHA: "51b12d20a25b76fa8a5717c2e5af01f148e469e2"
      BENCHMARK_LOG: ${{ matrix.dev == 'QCOM:IR3' && 'ir3_' || '' }}openpilot_${{ matrix.model }}
      ASSERT_MIN_STEP_TIME: ${{ matrix.timing }}
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: reset process replay
      run: test/external/process_replay/reset.py
    - name: compile
      run: FLOAT16=1 IMAGE=1 taskset -c 4-7 python3 examples/openpilot/compile_onnx.py "https://huggingface.co/commaai/openpilot-lfs/resolve/$LFS_SHA/openpilot/selfdrive/modeld/models/${{ matrix.model }}.onnx" openpilot.pkl
    - name: run pickle
      run: BENCHMARK_LOG="${BENCHMARK_LOG}_run_pickle" taskset -c 4-7 python3 examples/openpilot/load_pickle.py --run openpilot.pkl
    - name: Compile driving warp
      run: |
        taskset -c 4-7 python examples/openpilot/compile_warp.py --output warp.pkl --benchmark-runs 1 \
          --frame "1928,1208,2048,1216,608,${{ matrix.model == 'driving_supercombo' && '3735552' || '4804608' }}" \
          --warp-to ${{ matrix.model == 'driving_supercombo' && '512x256' || '1440x960' }} \
          --layout ${{ matrix.model == 'driving_supercombo' && 'yuv420' || 'luma' }} \
          ${{ matrix.model == 'driving_supercombo' && '--frames 2' || '--border-fill 16 --transform-device NPY' }}
    - name: run warp pickle
      run: ASSERT_MIN_STEP_TIME=0 BENCHMARK_LOG="${BENCHMARK_LOG}_warp" taskset -c 4-7 python3 examples/openpilot/load_pickle.py --run warp.pkl
    - name: Run process replay tests
      uses: ./.github/actions/process-replay

  testqualcommdsp:
    name: DSP Benchmark
    runs-on: [self-hosted, Linux, comma4]
    timeout-minutes: 5
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: reset process replay
      run: test/external/process_replay/reset.py
    - name: benchmark MobileNetV2 on DSP
      run: |
        # generate quantized weights
        ln -s ~/tinygrad/extra/datasets/imagenet extra/datasets/imagenet
        ln -s ~/tinygrad/testsig-*.so .
        PYTHONPATH=. DEV=CPU QUANT=1 CNT=0 python3 examples/test_onnx_imagenet.py https://github.com/xamcat/mobcat-samples/raw/refs/heads/master/onnx_runtime/InferencingSample/InferencingSample/mobilenetv2-7.onnx /tmp/model.quant.onnx
        # benchmark on DSP with NOOPT=1, the devectorizer has issues
        PYTHONPATH=. DEV=DSP NOOPT=1 CNT=2 DEBUG=2 python3 examples/test_onnx_imagenet.py /tmp/model.quant.onnx
    - name: Run process replay tests
      uses: ./.github/actions/process-replay

  testcommausbgpubenchmark:
    name: UsbGPU Benchmark (comma)
    env:
      LFS_SHA: "51b12d20a25b76fa8a5717c2e5af01f148e469e2"
    runs-on: [self-hosted, Linux, comma4]
    timeout-minutes: 14
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: openpilot compile big_driving_supercombo
      run: |
        GMMU=0 DEV=USB+AMD:LLVM taskset -c 4-7 python3 examples/openpilot/compile_onnx.py "https://huggingface.co/commaai/openpilot-lfs/resolve/$LFS_SHA/openpilot/selfdrive/modeld/models/big_driving_supercombo.onnx" openpilot.pkl --out-of-band --retargetable
        BENCHMARK_LOG=usbgpu_openpilot_big_driving_supercombo GMMU=0 DEV=USB+AMD:LLVM ASSERT_MIN_STEP_TIME=25 taskset -c 4-7 python3 examples/openpilot/load_pickle.py --run openpilot.pkl --out-of-band --retarget
    - name: openpilot load_pickle big_driving_supercombo
      run: BENCHMARK_LOG=usbgpu_openpilot_big_driving_supercombo_load_pickle GMMU=0 DEV=USB+AMD ASSERT_MIN_LOAD_TIME=25 taskset -c 4-7 python3 examples/openpilot/load_pickle.py openpilot.pkl --out-of-band
    - name: compile driving warp
      run: GMMU=0 DEV=USB+AMD:LLVM taskset -c 4-7 python3 examples/openpilot/compile_warp.py --frame 1928,1208,2048,1216,608,3735552 --warp-to 512x256 --layout yuv420 --frames 2 --output warp.pkl --retargetable
    - name: run warp pickle
      run: BENCHMARK_LOG=usbgpu_openpilot_driving_warp GMMU=0 DEV=USB+AMD taskset -c 4-7 python3 examples/openpilot/load_pickle.py --run warp.pkl --retarget
    - name: Test copy speeds
      run: SIZE=64000000 GMMU=0 DEV=USB+AMD taskset -c 4-7 python3 test/external/external_test_usb_asm24.py

  driverbenchmarks:
    name: PCI Driver Benchmark (DEV=${{ matrix.dev }})
    runs-on: [self-hosted, Linux, tinyboxrandom]
    strategy:
      fail-fast: false
      matrix:
        dev: ['AMD::gfx1201', 'NV']
    timeout-minutes: 5
    defaults:
      run:
        shell: bash -e -o pipefail {0}
    env:
      DEV: ${{ matrix.dev }}
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: Setup
      run: |
        ./extra/hcq/hcq_smi.py ${{ matrix.dev != 'NV' && 'AMD' || 'NV' }} rmmod --expect
        ./extra/hcq/hcq_smi.py ${{ matrix.dev != 'NV' && 'AMD' || 'NV' }} kill_pids --sudoless
        mkdir -p extra/datasets
        ln -s /raid/datasets/imagenet extra/datasets/imagenet
    - name: setup staging db
      if: github.ref == 'refs/heads/update_benchmark_staging'
      run: |
        echo "CACHEDB=/tmp/staging.db" >> $GITHUB_ENV
        rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
    - name: reset process replay
      run: test/external/process_replay/reset.py
    - name: Test driver cold start time
      run: time DEBUG=3 AM_RESET=1 python3 test/test_tiny.py TestTiny.test_plus
    - name: Test driver warm start time
      if: ${{ matrix.dev != 'NV' }}
      run: time DEBUG=3 python3 test/test_tiny.py TestTiny.test_plus
    - name: Test GPU crash recovery
      if: ${{ matrix.dev != 'NV' }}
      run: python3 -m pytest -rA test/external/external_test_gpu_crash.py
    - name: Test tensor cores
      run: |
        if [[ "${{ matrix.dev }}" == "NV" ]]; then
          ALLOW_TF32=1 python3 test/runtime/test_tensor_cores.py
        else
          # Fails on 9070
          # python3 test/test_linearizer.py test/runtime/test_tensor_cores.py
          # DEV=AMD:LLVM python3 test/test_linearizer.py test/runtime/test_tensor_cores.py
          # SHOULD_USE_TC=1 BFLOAT16=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
          SHOULD_USE_TC=1 HALF=1 DEBUG=2 ATOL=2e-2 python3 extra/gemm/simple_matmul.py
        fi
    - name: Test DISK copy time
      run: TESTFILE=/raid/downloads/llama3-8b-sfr/model-00001-of-00004.safetensors python3 test/external/external_benchmark_disk_raw.py
    - name: Test CPU copy time
      run: |
        GRAPH_ONE_KERNEL=1 NSZ=8192 python3 test/speed/external_test_copy_speed.py TestCopySpeed.testCopyDefaulttoCPUJit
        GRAPH_ONE_KERNEL=1 NSZ=8192 python3 test/speed/external_test_copy_speed.py TestCopySpeed.testCopyCPUtoDefaultJit
    # TODO: HEVC decode timing test
    # - name: HEVC Decode Benchmark
    #   if: ${{ matrix.dev == 'NV' }}
    #   run: IGNORE_BEAM_CACHE=1 VALIDATE=1 MAX_FRAMES=100 ASSERT_FPS=1400 JITBEAM=1 PYTHONPATH=. python3 extra/hevc/decode.py
    - name: Run 10 MLPerf ResNet50 training steps (1 gpu)
      if: ${{ matrix.dev == 'NV' }}
      run: BENCHMARK_LOG=resnet_10steps MNISTMOCK=1 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=256 GPUS=1 MODEL=resnet python3 examples/mlperf/model_train.py
    - name: Run 10 MLPerf Bert training steps (1 gpu)
      # TODO: remove BERT_LAYERS once scheduler is fast
      run: BENCHMARK_LOG=bert_10steps CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=66 GPUS=1 BERT_LAYERS=2 MODEL=bert python3 examples/mlperf/model_train.py
    - name: Remote
      if: ${{ matrix.dev != 'NV' }}
      env:
        DEV: PCI+AMD::gfx1201
      run: |
        PYTHONPATH=. python3 extra/remote/serve.py 6482 &
        server_pid=$!
        trap 'kill "$server_pid" 2>/dev/null || true' EXIT
        sleep 1
        DEBUG=2 PYTHONPATH=. REMOTE=127.0.0.1:6482 AM_RESET=1 python3 test/test_tiny.py
        kill "$server_pid"
        wait "$server_pid"
        trap - EXIT
    - name: Run process replay tests
      uses: ./.github/actions/process-replay

  llvmspeed:
    name: LLVM Speed
    runs-on: [self-hosted, Linux, tinyboxrandom]
    timeout-minutes: 10
    if: github.repository_owner == 'tinygrad'
    steps:
    - name: Checkout Code
      uses: actions/checkout@v6
    - name: Speed Test
      run: DEV=CPU:LLVM python3 test/speed/external_test_speed_v_torch.py
    - name: Speed Test (BEAM=2)
      run: IGNORE_BEAM_CACHE=1 BEAM=2 DEV=CPU:LLVM python3 test/speed/external_test_speed_v_torch.py
