name: CI (self-hosted CUDA backend) on: workflow_dispatch: # allows manual triggering push: branches: - master paths: [ '.github/workflows/ci-self-hosted-cuda.yml', 'ci/run.sh', '**/CMakeLists.txt', '**/.cmake', '**/*.h', '**/*.hpp', '**/*.c', '**/*.cpp', '**/*.cu', '**/*.cuh' ] pull_request: types: [opened, synchronize, reopened] paths: [ '.github/workflows/ci-self-hosted-cuda.yml', 'ci/run.sh', '**/CMakeLists.txt', '**/.cmake', 'ggml/src/*', 'ggml/src/ggml-cpu/**', 'ggml/src/ggml-cuda/**' ] concurrency: group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }} cancel-in-progress: true env: # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302) HF_TOKEN: ${{ secrets.HF_TOKEN_CI }} GGML_NLOOP: 3 GGML_N_THREADS: 1 LLAMA_ARG_LOG_COLORS: 1 LLAMA_ARG_LOG_PREFIX: 1 LLAMA_ARG_LOG_TIMESTAMPS: 1 jobs: gpu-cuda: runs-on: "hf-jobs-t4-small:cuda13" steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Install dependencies run: | sudo apt update sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip - name: ccache uses: ggml-org/ccache-action@v1.2.24 with: restore: false save: false - name: ccache-buckets-restore uses: ./.github/actions/ccache-buckets with: key: self-hosted-gpu-cuda folder: llama.cpp hf_bucket: ggml-org/cache - name: Test id: ggml-ci run: | nvidia-smi GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - name: ccache-buckets-save if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} uses: ./.github/actions/ccache-buckets env: HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} with: key: self-hosted-gpu-cuda folder: llama.cpp evict-old-files: 1d hf_bucket: ggml-org/cache save: true gpu-rocm: runs-on: [self-hosted, Linux, AMD] steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Test id: ggml-ci # HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness # issue on integrated RDNA3.5 (gfx1151) where batched inference returns # incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches # restores correctness. Remove once the underlying ROCm/HIP issue is fixed. env: HIP_LAUNCH_BLOCKING: "1" run: | rocminfo GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp # TODO: provision AMD GPU machine # amd-rocm: # runs-on: [self-hosted, Linux, AMD] # steps: # - name: Clone # id: checkout # uses: actions/checkout@v6 # - name: Test # id: ggml-ci # run: | # amd-smi static # GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp