Commit 542e9202d for llama.cpp
commit 542e9202d7d14863562e1e908be0bf32aeec2c2a
Author: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>
Date: Mon Sep 21 12:51:53 2026 +0200
ci : refactor build-self-hosted into backend-specific workflows (#28991)
* refactor build-self-hosted into backends
* update workflow names
* build -> ci
* bump openvino
* trigger on cpu and generic ggml changes
diff --git a/.github/workflows/build-self-hosted.yml b/.github/workflows/build-self-hosted.yml
deleted file mode 100644
index d54f71ac5..000000000
--- a/.github/workflows/build-self-hosted.yml
+++ /dev/null
@@ -1,539 +0,0 @@
-name: CI (self-hosted)
-
-on:
- workflow_dispatch: # allows manual triggering
- push:
- branches:
- - master
- paths: [
- '.github/workflows/build-self-hosted.yml',
- 'ci/run.sh',
- '**/CMakeLists.txt',
- '**/.cmake',
- '**/*.h',
- '**/*.hpp',
- '**/*.c',
- '**/*.cpp',
- '**/*.cu',
- '**/*.cuh',
- '**/*.swift',
- '**/*.m',
- '**/*.metal',
- '**/*.comp',
- '**/*.glsl',
- '**/*.wgsl'
- ]
-
- pull_request:
- types: [opened, synchronize, reopened]
- paths: [
- '.github/workflows/build-self-hosted.yml',
- 'ci/run.sh',
- '**/CMakeLists.txt',
- '**/.cmake',
- '**/*.h',
- '**/*.hpp',
- '**/*.c',
- '**/*.cpp',
- '**/*.cu',
- '**/*.cuh',
- '**/*.swift',
- '**/*.m',
- '**/*.metal',
- '**/*.comp',
- '**/*.glsl',
- '**/*.wgsl'
- ]
-
-concurrency:
- group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
- cancel-in-progress: true
-
-env:
- # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
- HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
- GGML_NLOOP: 3
- GGML_N_THREADS: 1
- LLAMA_ARG_LOG_COLORS: 1
- LLAMA_ARG_LOG_PREFIX: 1
- LLAMA_ARG_LOG_TIMESTAMPS: 1
-
-jobs:
- gpu-cuda:
- runs-on: "hf-jobs-t4-small:cuda13"
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Install dependencies
- run: |
- sudo apt update
- sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip
-
- - name: ccache
- uses: ggml-org/ccache-action@v1.2.24
- with:
- restore: false
- save: false
-
- - name: ccache-buckets-restore
- uses: ./.github/actions/ccache-buckets
- with:
- key: self-hosted-gpu-cuda
- folder: llama.cpp
- hf_bucket: ggml-org/cache
-
- - name: Test
- id: ggml-ci
- run: |
- nvidia-smi
- GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- - name: ccache-buckets-save
- if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
- uses: ./.github/actions/ccache-buckets
- env:
- HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
- with:
- key: self-hosted-gpu-cuda
- folder: llama.cpp
- evict-old-files: 1d
- hf_bucket: ggml-org/cache
- save: true
-
- gpu-rocm:
- runs-on: [self-hosted, Linux, AMD]
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Test
- id: ggml-ci
- # HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness
- # issue on integrated RDNA3.5 (gfx1151) where batched inference returns
- # incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches
- # restores correctness. Remove once the underlying ROCm/HIP issue is fixed.
- env:
- HIP_LAUNCH_BLOCKING: "1"
- run: |
- rocminfo
- GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- gpu-vulkan-nvidia-cm:
- # runs-on: "hf-jobs-t4-small:ubuntu26_04"
- runs-on: [self-hosted, Linux, NVIDIA]
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- # - name: Install dependencies
- # run: |
- # sudo apt update
- # sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip
-
- # - name: ccache
- # uses: ggml-org/ccache-action@v1.2.24
- # with:
- # restore: false
- # save: false
-
- # - name: ccache-buckets-restore
- # uses: ./.github/actions/ccache-buckets
- # with:
- # key: self-hosted-vulkan-nvidia-cm
- # folder: llama.cpp
- # hf_bucket: ggml-org/cache
-
- - name: Test
- id: ggml-ci
- run: |
- vulkaninfo --summary
- GG_BUILD_VULKAN=1 GGML_VK_DISABLE_COOPMAT2=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- # - name: ccache-buckets-save
- # if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
- # uses: ./.github/actions/ccache-buckets
- # env:
- # HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
- # with:
- # key: self-hosted-vulkan-nvidia-cm
- # folder: llama.cpp
- # evict-old-files: 1d
- # hf_bucket: ggml-org/cache
- # save: true
-
- gpu-vulkan-nvidia-cm2:
- # runs-on: "hf-jobs-t4-small:ubuntu26_04"
- runs-on: [self-hosted, Linux, NVIDIA, COOPMAT2]
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- # - name: Install dependencies
- # run: |
- # sudo apt update
- # sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip
-
- # - name: ccache
- # uses: ggml-org/ccache-action@v1.2.24
- # with:
- # restore: false
- # save: false
-
- # - name: ccache-buckets-restore
- # uses: ./.github/actions/ccache-buckets
- # with:
- # key: self-hosted-vulkan-nvidia-cm2
- # folder: llama.cpp
- # hf_bucket: ggml-org/cache
-
- - name: Test
- id: ggml-ci
- run: |
- vulkaninfo --summary
- GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- # - name: ccache-buckets-save
- # if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
- # uses: ./.github/actions/ccache-buckets
- # env:
- # HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
- # with:
- # key: self-hosted-vulkan-nvidia-cm2
- # folder: llama.cpp
- # evict-old-files: 1d
- # hf_bucket: ggml-org/cache
- # save: true
-
- gpu-webgpu-nvidia:
- runs-on: "hf-jobs-t4-small:ubuntu26_04"
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Install dependencies
- run: |
- sudo apt update
- sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan1 mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip
-
- - name: ccache
- uses: ggml-org/ccache-action@v1.2.24
- with:
- restore: false
- save: false
-
- - name: ccache-buckets-restore
- uses: ./.github/actions/ccache-buckets
- with:
- key: self-hosted-webgpu-nvidia
- folder: llama.cpp
- hf_bucket: ggml-org/cache
-
- - name: Dawn Dependency
- id: dawn-depends
- run: |
- DAWN_VERSION="v20260908.214631"
- DAWN_OWNER="google"
- DAWN_REPO="dawn"
- DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-ubuntu-latest-Release"
- echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
- curl -L -o artifact.tar.gz \
- "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
- mkdir dawn
- tar -xvf artifact.tar.gz -C dawn --strip-components=1
-
- - name: Test
- id: ggml-ci
- run: |
- GG_BUILD_WEBGPU=1 \
- GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \
- GG_BUILD_WEBGPU_DAWN_DIR="$GITHUB_WORKSPACE/dawn/lib64/cmake/Dawn" \
- bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- - name: ccache-buckets-save
- if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
- uses: ./.github/actions/ccache-buckets
- env:
- HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
- with:
- key: self-hosted-webgpu-nvidia
- folder: llama.cpp
- evict-old-files: 1d
- hf_bucket: ggml-org/cache
- save: true
-
- # TODO: provision AMX-compatible machine
- #cpu-amx:
- # runs-on: [self-hosted, Linux, CPU, AMX]
-
- # steps:
- # - name: Clone
- # id: checkout
- # uses: actions/checkout@v6
-
- # - name: Test
- # id: ggml-ci
- # run: |
- # bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- # TODO: provision AMD GPU machine
- # amd-vulkan:
- # runs-on: [self-hosted, Linux, AMD]
-
- # steps:
- # - name: Clone
- # id: checkout
- # uses: actions/checkout@v6
-
- # - name: Test
- # id: ggml-ci
- # run: |
- # vulkaninfo --summary
- # GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- # TODO: provision AMD GPU machine
- # amd-rocm:
- # runs-on: [self-hosted, Linux, AMD]
-
- # steps:
- # - name: Clone
- # id: checkout
- # uses: actions/checkout@v6
-
- # - name: Test
- # id: ggml-ci
- # run: |
- # amd-smi static
- # GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- gpu-metal:
- runs-on: [self-hosted, macOS, ARM64]
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Test
- id: ggml-ci
- run: |
- GG_BUILD_METAL=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- gpu-webgpu-apple:
- runs-on: [self-hosted, macOS, ARM64]
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Dawn Dependency
- id: dawn-depends
- run: |
- DAWN_VERSION="v20260908.214631"
- DAWN_OWNER="google"
- DAWN_REPO="dawn"
- DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-macos-latest-Release"
- echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
- curl -L -o artifact.tar.gz \
- "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
- mkdir dawn
- tar -xvf artifact.tar.gz -C dawn --strip-components=1
-
- - name: Test
- id: ggml-ci
- run: |
- GG_BUILD_WEBGPU=1 GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \
- bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- gpu-vulkan-apple:
- runs-on: [self-hosted, macOS, ARM64]
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Test
- id: ggml-ci
- run: |
- vulkaninfo --summary
- GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- gpu-vulkan-intel-linux:
- runs-on: [self-hosted, Linux, Intel]
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
- with:
- persist-credentials: false
-
- - name: Test
- id: ggml-ci
- run: |
- vulkaninfo --summary
- GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- gpu-vulkan-intel-windows:
- runs-on: [self-hosted, Windows, X64, Intel]
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Test
- id: ggml-ci
- shell: C:\msys64\usr\bin\bash.exe --noprofile --norc -eo pipefail "{0}"
- env:
- MSYSTEM: UCRT64
- CHERE_INVOKING: 1
- PATH: C:\msys64\ucrt64\bin;C:\msys64\usr\bin;C:\Windows\System32;${{ env.PATH }}
- run: |
- vulkaninfo --summary
- # Skip python related tests with GG_BUILD_LOW_PERF=1 since Windows MSYS2 UCRT64 currently fails to create
- # a valid python environment for testing
- LLAMA_FATAL_WARNINGS=OFF GG_BUILD_NINJA=1 GG_BUILD_VULKAN=1 GG_BUILD_LOW_PERF=1 ./ci/run.sh ./results/llama.cpp ./mnt/llama.cpp
-
- gpu-openvino-low-perf:
- runs-on: [self-hosted, Linux, Intel, OpenVINO]
-
- env:
- # Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
- OPENVINO_VERSION_MAJOR: "2026.4"
- OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Setup OpenVINO Toolkit
- uses: ./.github/actions/linux-setup-openvino
- with:
- path: ./openvino_toolkit
- version_major: ${{ env.OPENVINO_VERSION_MAJOR }}
- version_full: ${{ env.OPENVINO_VERSION_FULL }}
-
- - name: Install OpenVINO dependencies
- run: |
- cd ./openvino_toolkit
- chmod +x ./install_dependencies/install_openvino_dependencies.sh
- echo "Y" | sudo -E ./install_dependencies/install_openvino_dependencies.sh
-
- - name: Test
- id: ggml-ci
- run: |
- source ./openvino_toolkit/setupvars.sh
- GG_BUILD_OPENVINO=1 GGML_OPENVINO_DEVICE=GPU GG_BUILD_LOW_PERF=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- cpu-x64-high-perf:
- runs-on: [self-hosted, Linux, X64]
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Test
- id: ggml-ci
- run: |
- LLAMA_ARG_THREADS=$(nproc) GG_BUILD_HIGH_PERF=1 GG_BUILD_EXTRA_TESTS_0=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- cpu-arm64-high-perf-graviton4:
- runs-on: ah-ubuntu_24_04-c8g_8x
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Dependencies
- id: depends
- run: |
- set -euxo pipefail
- sudo apt-get update
- sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \
- apt-get install -y \
- build-essential \
- python3-venv \
- gpg \
- wget \
- time \
- git-lfs
-
- git lfs install
-
- # install the latest cmake
- sudo install -d /usr/share/keyrings
- wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \
- | gpg --dearmor \
- | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null
- echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \
- | sudo tee /etc/apt/sources.list.d/kitware.list
- sudo apt-get update
- sudo apt-get install -y cmake
-
- - name: Test
- id: ggml-ci
- run: |
- LLAMA_ARG_THREADS=$(nproc) \
- GG_BUILD_HIGH_PERF=1 \
- GG_BUILD_NO_BF16=1 \
- GG_BUILD_EXTRA_TESTS_0=1 \
- bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
-
- cpu-arm64-graviton4-kleidiai:
- runs-on: ah-ubuntu_24_04-c8g_8x
-
- steps:
- - name: Clone
- id: checkout
- uses: actions/checkout@v6
-
- - name: Dependencies
- id: depends
- run: |
- set -euxo pipefail
- sudo apt-get update
- sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \
- apt-get install -y \
- build-essential \
- python3-venv \
- gpg \
- wget \
- time \
- git-lfs
-
- git lfs install
-
- # install the latest cmake
- sudo install -d /usr/share/keyrings
- wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \
- | gpg --dearmor \
- | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null
- echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \
- | sudo tee /etc/apt/sources.list.d/kitware.list
- sudo apt-get update
- sudo apt-get install -y cmake
-
- - name: Test
- id: ggml-ci
- run: |
- LLAMA_ARG_THREADS=$(nproc) \
- GG_BUILD_KLEIDIAI=1 \
- GG_BUILD_EXTRA_TESTS_0=1 \
- GG_BUILD_HIGH_PERF=1 \
- bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
diff --git a/.github/workflows/ci-self-hosted-cpu.yml b/.github/workflows/ci-self-hosted-cpu.yml
new file mode 100644
index 000000000..18bc9f144
--- /dev/null
+++ b/.github/workflows/ci-self-hosted-cpu.yml
@@ -0,0 +1,112 @@
+name: CI (self-hosted CPU backend)
+
+on:
+ workflow_dispatch: # allows manual triggering
+ push:
+ branches:
+ - master
+ paths: [
+ '.github/workflows/ci-self-hosted-cpu.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ '**/*.h',
+ '**/*.hpp',
+ '**/*.c',
+ '**/*.cpp'
+ ]
+
+ pull_request:
+ types: [opened, synchronize, reopened]
+ paths: [
+ '.github/workflows/ci-self-hosted-cpu.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ 'ggml/src/*',
+ 'ggml/src/ggml-cpu/**'
+ ]
+
+concurrency:
+ group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
+ cancel-in-progress: true
+
+env:
+ # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
+ HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
+ GGML_NLOOP: 3
+ GGML_N_THREADS: 1
+ LLAMA_ARG_LOG_COLORS: 1
+ LLAMA_ARG_LOG_PREFIX: 1
+ LLAMA_ARG_LOG_TIMESTAMPS: 1
+
+jobs:
+ cpu-x64-high-perf:
+ runs-on: [self-hosted, Linux, X64]
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ LLAMA_ARG_THREADS=$(nproc) GG_BUILD_HIGH_PERF=1 GG_BUILD_EXTRA_TESTS_0=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
+
+ cpu-arm64-high-perf-graviton4:
+ runs-on: ah-ubuntu_24_04-c8g_8x
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Dependencies
+ id: depends
+ run: |
+ set -euxo pipefail
+ sudo apt-get update
+ sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \
+ apt-get install -y \
+ build-essential \
+ python3-venv \
+ gpg \
+ wget \
+ time \
+ git-lfs
+
+ git lfs install
+
+ # install the latest cmake
+ sudo install -d /usr/share/keyrings
+ wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \
+ | gpg --dearmor \
+ | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null
+ echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \
+ | sudo tee /etc/apt/sources.list.d/kitware.list
+ sudo apt-get update
+ sudo apt-get install -y cmake
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ LLAMA_ARG_THREADS=$(nproc) \
+ GG_BUILD_HIGH_PERF=1 \
+ GG_BUILD_NO_BF16=1 \
+ GG_BUILD_EXTRA_TESTS_0=1 \
+ bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
+
+ # TODO: provision AMX-compatible machine
+ #cpu-amx:
+ # runs-on: [self-hosted, Linux, CPU, AMX]
+
+ # steps:
+ # - name: Clone
+ # id: checkout
+ # uses: actions/checkout@v6
+
+ # - name: Test
+ # id: ggml-ci
+ # run: |
+ # bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
diff --git a/.github/workflows/ci-self-hosted-cuda.yml b/.github/workflows/ci-self-hosted-cuda.yml
new file mode 100644
index 000000000..5951b52d3
--- /dev/null
+++ b/.github/workflows/ci-self-hosted-cuda.yml
@@ -0,0 +1,124 @@
+name: CI (self-hosted CUDA backend)
+
+on:
+ workflow_dispatch: # allows manual triggering
+ push:
+ branches:
+ - master
+ paths: [
+ '.github/workflows/ci-self-hosted-cuda.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ '**/*.h',
+ '**/*.hpp',
+ '**/*.c',
+ '**/*.cpp',
+ '**/*.cu',
+ '**/*.cuh'
+ ]
+
+ pull_request:
+ types: [opened, synchronize, reopened]
+ paths: [
+ '.github/workflows/ci-self-hosted-cuda.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ 'ggml/src/*',
+ 'ggml/src/ggml-cpu/**',
+ 'ggml/src/ggml-cuda/**'
+ ]
+
+concurrency:
+ group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
+ cancel-in-progress: true
+
+env:
+ # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
+ HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
+ GGML_NLOOP: 3
+ GGML_N_THREADS: 1
+ LLAMA_ARG_LOG_COLORS: 1
+ LLAMA_ARG_LOG_PREFIX: 1
+ LLAMA_ARG_LOG_TIMESTAMPS: 1
+
+jobs:
+ gpu-cuda:
+ runs-on: "hf-jobs-t4-small:cuda13"
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Install dependencies
+ run: |
+ sudo apt update
+ sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip
+
+ - name: ccache
+ uses: ggml-org/ccache-action@v1.2.24
+ with:
+ restore: false
+ save: false
+
+ - name: ccache-buckets-restore
+ uses: ./.github/actions/ccache-buckets
+ with:
+ key: self-hosted-gpu-cuda
+ folder: llama.cpp
+ hf_bucket: ggml-org/cache
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ nvidia-smi
+ GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
+
+ - name: ccache-buckets-save
+ if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
+ uses: ./.github/actions/ccache-buckets
+ env:
+ HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
+ with:
+ key: self-hosted-gpu-cuda
+ folder: llama.cpp
+ evict-old-files: 1d
+ hf_bucket: ggml-org/cache
+ save: true
+
+ gpu-rocm:
+ runs-on: [self-hosted, Linux, AMD]
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Test
+ id: ggml-ci
+ # HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness
+ # issue on integrated RDNA3.5 (gfx1151) where batched inference returns
+ # incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches
+ # restores correctness. Remove once the underlying ROCm/HIP issue is fixed.
+ env:
+ HIP_LAUNCH_BLOCKING: "1"
+ run: |
+ rocminfo
+ GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
+
+ # TODO: provision AMD GPU machine
+ # amd-rocm:
+ # runs-on: [self-hosted, Linux, AMD]
+
+ # steps:
+ # - name: Clone
+ # id: checkout
+ # uses: actions/checkout@v6
+
+ # - name: Test
+ # id: ggml-ci
+ # run: |
+ # amd-smi static
+ # GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
diff --git a/.github/workflows/ci-self-hosted-kleidiai.yml b/.github/workflows/ci-self-hosted-kleidiai.yml
new file mode 100644
index 000000000..c955aa002
--- /dev/null
+++ b/.github/workflows/ci-self-hosted-kleidiai.yml
@@ -0,0 +1,85 @@
+name: CI (self-hosted KleidiAI backend)
+
+on:
+ workflow_dispatch: # allows manual triggering
+ push:
+ branches:
+ - master
+ paths: [
+ '.github/workflows/ci-self-hosted-kleidiai.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ '**/*.h',
+ '**/*.hpp',
+ '**/*.c',
+ '**/*.cpp'
+ ]
+
+ pull_request:
+ types: [opened, synchronize, reopened]
+ paths: [
+ '.github/workflows/ci-self-hosted-kleidiai.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ 'ggml/src/*',
+ 'ggml/src/ggml-cpu/**'
+ ]
+
+concurrency:
+ group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
+ cancel-in-progress: true
+
+env:
+ # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
+ HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
+ GGML_NLOOP: 3
+ GGML_N_THREADS: 1
+ LLAMA_ARG_LOG_COLORS: 1
+ LLAMA_ARG_LOG_PREFIX: 1
+ LLAMA_ARG_LOG_TIMESTAMPS: 1
+
+jobs:
+ cpu-arm64-graviton4-kleidiai:
+ runs-on: ah-ubuntu_24_04-c8g_8x
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Dependencies
+ id: depends
+ run: |
+ set -euxo pipefail
+ sudo apt-get update
+ sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \
+ apt-get install -y \
+ build-essential \
+ python3-venv \
+ gpg \
+ wget \
+ time \
+ git-lfs
+
+ git lfs install
+
+ # install the latest cmake
+ sudo install -d /usr/share/keyrings
+ wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \
+ | gpg --dearmor \
+ | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null
+ echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \
+ | sudo tee /etc/apt/sources.list.d/kitware.list
+ sudo apt-get update
+ sudo apt-get install -y cmake
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ LLAMA_ARG_THREADS=$(nproc) \
+ GG_BUILD_KLEIDIAI=1 \
+ GG_BUILD_EXTRA_TESTS_0=1 \
+ GG_BUILD_HIGH_PERF=1 \
+ bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
diff --git a/.github/workflows/ci-self-hosted-metal.yml b/.github/workflows/ci-self-hosted-metal.yml
new file mode 100644
index 000000000..7af30e294
--- /dev/null
+++ b/.github/workflows/ci-self-hosted-metal.yml
@@ -0,0 +1,59 @@
+name: CI (self-hosted Metal backend)
+
+on:
+ workflow_dispatch: # allows manual triggering
+ push:
+ branches:
+ - master
+ paths: [
+ '.github/workflows/ci-self-hosted-metal.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ '**/*.h',
+ '**/*.hpp',
+ '**/*.c',
+ '**/*.cpp',
+ '**/*.swift',
+ '**/*.m',
+ '**/*.metal'
+ ]
+
+ pull_request:
+ types: [opened, synchronize, reopened]
+ paths: [
+ '.github/workflows/ci-self-hosted-metal.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ 'ggml/src/*',
+ 'ggml/src/ggml-cpu/**',
+ 'ggml/src/ggml-metal/**'
+ ]
+
+concurrency:
+ group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
+ cancel-in-progress: true
+
+env:
+ # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
+ HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
+ GGML_NLOOP: 3
+ GGML_N_THREADS: 1
+ LLAMA_ARG_LOG_COLORS: 1
+ LLAMA_ARG_LOG_PREFIX: 1
+ LLAMA_ARG_LOG_TIMESTAMPS: 1
+
+jobs:
+ gpu-metal:
+ runs-on: [self-hosted, macOS, ARM64]
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ GG_BUILD_METAL=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
diff --git a/.github/workflows/ci-self-hosted-openvino.yml b/.github/workflows/ci-self-hosted-openvino.yml
new file mode 100644
index 000000000..e0947c46e
--- /dev/null
+++ b/.github/workflows/ci-self-hosted-openvino.yml
@@ -0,0 +1,75 @@
+name: CI (self-hosted OpenVINO backend)
+
+on:
+ workflow_dispatch: # allows manual triggering
+ push:
+ branches:
+ - master
+ paths: [
+ '.github/workflows/ci-self-hosted-openvino.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ '**/*.h',
+ '**/*.hpp',
+ '**/*.c',
+ '**/*.cpp'
+ ]
+
+ pull_request:
+ types: [opened, synchronize, reopened]
+ paths: [
+ '.github/workflows/ci-self-hosted-openvino.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ 'ggml/src/*',
+ 'ggml/src/ggml-cpu/**',
+ 'ggml/src/ggml-openvino/**'
+ ]
+
+concurrency:
+ group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
+ cancel-in-progress: true
+
+env:
+ # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
+ HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
+ GGML_NLOOP: 3
+ GGML_N_THREADS: 1
+ LLAMA_ARG_LOG_COLORS: 1
+ LLAMA_ARG_LOG_PREFIX: 1
+ LLAMA_ARG_LOG_TIMESTAMPS: 1
+
+jobs:
+ gpu-openvino-low-perf:
+ runs-on: [self-hosted, Linux, Intel, OpenVINO]
+
+ env:
+ # Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
+ OPENVINO_VERSION_MAJOR: "2026.4"
+ OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Setup OpenVINO Toolkit
+ uses: ./.github/actions/linux-setup-openvino
+ with:
+ path: ./openvino_toolkit
+ version_major: ${{ env.OPENVINO_VERSION_MAJOR }}
+ version_full: ${{ env.OPENVINO_VERSION_FULL }}
+
+ - name: Install OpenVINO dependencies
+ run: |
+ cd ./openvino_toolkit
+ chmod +x ./install_dependencies/install_openvino_dependencies.sh
+ echo "Y" | sudo -E ./install_dependencies/install_openvino_dependencies.sh
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ source ./openvino_toolkit/setupvars.sh
+ GG_BUILD_OPENVINO=1 GGML_OPENVINO_DEVICE=GPU GG_BUILD_LOW_PERF=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
diff --git a/.github/workflows/ci-self-hosted-vulkan.yml b/.github/workflows/ci-self-hosted-vulkan.yml
new file mode 100644
index 000000000..ffed4b09b
--- /dev/null
+++ b/.github/workflows/ci-self-hosted-vulkan.yml
@@ -0,0 +1,201 @@
+name: CI (self-hosted Vulkan backend)
+
+on:
+ workflow_dispatch: # allows manual triggering
+ push:
+ branches:
+ - master
+ paths: [
+ '.github/workflows/ci-self-hosted-vulkan.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ '**/*.h',
+ '**/*.hpp',
+ '**/*.c',
+ '**/*.cpp',
+ '**/*.comp',
+ '**/*.glsl'
+ ]
+
+ pull_request:
+ types: [opened, synchronize, reopened]
+ paths: [
+ '.github/workflows/ci-self-hosted-vulkan.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ 'ggml/src/*',
+ 'ggml/src/ggml-cpu/**',
+ 'ggml/src/ggml-vulkan/**'
+ ]
+
+concurrency:
+ group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
+ cancel-in-progress: true
+
+env:
+ # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
+ HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
+ GGML_NLOOP: 3
+ GGML_N_THREADS: 1
+ LLAMA_ARG_LOG_COLORS: 1
+ LLAMA_ARG_LOG_PREFIX: 1
+ LLAMA_ARG_LOG_TIMESTAMPS: 1
+
+jobs:
+ gpu-vulkan-nvidia-cm:
+ # runs-on: "hf-jobs-t4-small:ubuntu26_04"
+ runs-on: [self-hosted, Linux, NVIDIA]
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ # - name: Install dependencies
+ # run: |
+ # sudo apt update
+ # sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip
+
+ # - name: ccache
+ # uses: ggml-org/ccache-action@v1.2.24
+ # with:
+ # restore: false
+ # save: false
+
+ # - name: ccache-buckets-restore
+ # uses: ./.github/actions/ccache-buckets
+ # with:
+ # key: self-hosted-vulkan-nvidia-cm
+ # folder: llama.cpp
+ # hf_bucket: ggml-org/cache
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ vulkaninfo --summary
+ GG_BUILD_VULKAN=1 GGML_VK_DISABLE_COOPMAT2=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
+
+ # - name: ccache-buckets-save
+ # if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
+ # uses: ./.github/actions/ccache-buckets
+ # env:
+ # HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
+ # with:
+ # key: self-hosted-vulkan-nvidia-cm
+ # folder: llama.cpp
+ # evict-old-files: 1d
+ # hf_bucket: ggml-org/cache
+ # save: true
+
+ gpu-vulkan-nvidia-cm2:
+ # runs-on: "hf-jobs-t4-small:ubuntu26_04"
+ runs-on: [self-hosted, Linux, NVIDIA, COOPMAT2]
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ # - name: Install dependencies
+ # run: |
+ # sudo apt update
+ # sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip
+
+ # - name: ccache
+ # uses: ggml-org/ccache-action@v1.2.24
+ # with:
+ # restore: false
+ # save: false
+
+ # - name: ccache-buckets-restore
+ # uses: ./.github/actions/ccache-buckets
+ # with:
+ # key: self-hosted-vulkan-nvidia-cm2
+ # folder: llama.cpp
+ # hf_bucket: ggml-org/cache
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ vulkaninfo --summary
+ GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
+
+ # - name: ccache-buckets-save
+ # if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
+ # uses: ./.github/actions/ccache-buckets
+ # env:
+ # HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
+ # with:
+ # key: self-hosted-vulkan-nvidia-cm2
+ # folder: llama.cpp
+ # evict-old-files: 1d
+ # hf_bucket: ggml-org/cache
+ # save: true
+
+ gpu-vulkan-apple:
+ runs-on: [self-hosted, macOS, ARM64]
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ vulkaninfo --summary
+ GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
+
+ gpu-vulkan-intel-linux:
+ runs-on: [self-hosted, Linux, Intel]
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+ with:
+ persist-credentials: false
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ vulkaninfo --summary
+ GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
+
+ gpu-vulkan-intel-windows:
+ runs-on: [self-hosted, Windows, X64, Intel]
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Test
+ id: ggml-ci
+ shell: C:\msys64\usr\bin\bash.exe --noprofile --norc -eo pipefail "{0}"
+ env:
+ MSYSTEM: UCRT64
+ CHERE_INVOKING: 1
+ PATH: C:\msys64\ucrt64\bin;C:\msys64\usr\bin;C:\Windows\System32;${{ env.PATH }}
+ run: |
+ vulkaninfo --summary
+ # Skip python related tests with GG_BUILD_LOW_PERF=1 since Windows MSYS2 UCRT64 currently fails to create
+ # a valid python environment for testing
+ LLAMA_FATAL_WARNINGS=OFF GG_BUILD_NINJA=1 GG_BUILD_VULKAN=1 GG_BUILD_LOW_PERF=1 ./ci/run.sh ./results/llama.cpp ./mnt/llama.cpp
+
+ # TODO: provision AMD GPU machine
+ # amd-vulkan:
+ # runs-on: [self-hosted, Linux, AMD]
+
+ # steps:
+ # - name: Clone
+ # id: checkout
+ # uses: actions/checkout@v6
+
+ # - name: Test
+ # id: ggml-ci
+ # run: |
+ # vulkaninfo --summary
+ # GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
diff --git a/.github/workflows/ci-self-hosted-webgpu.yml b/.github/workflows/ci-self-hosted-webgpu.yml
new file mode 100644
index 000000000..a6a36a8eb
--- /dev/null
+++ b/.github/workflows/ci-self-hosted-webgpu.yml
@@ -0,0 +1,130 @@
+name: CI (self-hosted WebGPU backend)
+
+on:
+ workflow_dispatch: # allows manual triggering
+ push:
+ branches:
+ - master
+ paths: [
+ '.github/workflows/ci-self-hosted-webgpu.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ '**/*.h',
+ '**/*.hpp',
+ '**/*.c',
+ '**/*.cpp',
+ '**/*.wgsl'
+ ]
+
+ pull_request:
+ types: [opened, synchronize, reopened]
+ paths: [
+ '.github/workflows/ci-self-hosted-webgpu.yml',
+ 'ci/run.sh',
+ '**/CMakeLists.txt',
+ '**/.cmake',
+ 'ggml/src/*',
+ 'ggml/src/ggml-cpu/**',
+ 'ggml/src/ggml-webgpu/**'
+ ]
+
+concurrency:
+ group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
+ cancel-in-progress: true
+
+env:
+ # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
+ HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
+ GGML_NLOOP: 3
+ GGML_N_THREADS: 1
+ LLAMA_ARG_LOG_COLORS: 1
+ LLAMA_ARG_LOG_PREFIX: 1
+ LLAMA_ARG_LOG_TIMESTAMPS: 1
+
+jobs:
+ gpu-webgpu-nvidia:
+ runs-on: "hf-jobs-t4-small:ubuntu26_04"
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Install dependencies
+ run: |
+ sudo apt update
+ sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan1 mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip
+
+ - name: ccache
+ uses: ggml-org/ccache-action@v1.2.24
+ with:
+ restore: false
+ save: false
+
+ - name: ccache-buckets-restore
+ uses: ./.github/actions/ccache-buckets
+ with:
+ key: self-hosted-webgpu-nvidia
+ folder: llama.cpp
+ hf_bucket: ggml-org/cache
+
+ - name: Dawn Dependency
+ id: dawn-depends
+ run: |
+ DAWN_VERSION="v20260908.214631"
+ DAWN_OWNER="google"
+ DAWN_REPO="dawn"
+ DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-ubuntu-latest-Release"
+ echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
+ curl -L -o artifact.tar.gz \
+ "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
+ mkdir dawn
+ tar -xvf artifact.tar.gz -C dawn --strip-components=1
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ GG_BUILD_WEBGPU=1 \
+ GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \
+ GG_BUILD_WEBGPU_DAWN_DIR="$GITHUB_WORKSPACE/dawn/lib64/cmake/Dawn" \
+ bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
+
+ - name: ccache-buckets-save
+ if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
+ uses: ./.github/actions/ccache-buckets
+ env:
+ HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
+ with:
+ key: self-hosted-webgpu-nvidia
+ folder: llama.cpp
+ evict-old-files: 1d
+ hf_bucket: ggml-org/cache
+ save: true
+
+ gpu-webgpu-apple:
+ runs-on: [self-hosted, macOS, ARM64]
+
+ steps:
+ - name: Clone
+ id: checkout
+ uses: actions/checkout@v6
+
+ - name: Dawn Dependency
+ id: dawn-depends
+ run: |
+ DAWN_VERSION="v20260908.214631"
+ DAWN_OWNER="google"
+ DAWN_REPO="dawn"
+ DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-macos-latest-Release"
+ echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
+ curl -L -o artifact.tar.gz \
+ "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
+ mkdir dawn
+ tar -xvf artifact.tar.gz -C dawn --strip-components=1
+
+ - name: Test
+ id: ggml-ci
+ run: |
+ GG_BUILD_WEBGPU=1 GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \
+ bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp