name: CI (self-hosted) on: workflow_dispatch: # allows manual triggering push: branches: - master paths: [ '.github/workflows/build-self-hosted.yml', 'ci/run.sh', '**/CMakeLists.txt', '**/.cmake', '**/*.h', '**/*.hpp', '**/*.c', '**/*.cpp', '**/*.cu', '**/*.cuh', '**/*.swift', '**/*.m', '**/*.metal', '**/*.comp', '**/*.glsl', '**/*.wgsl' ] pull_request: types: [opened, synchronize, reopened] paths: [ '.github/workflows/build-self-hosted.yml', 'ci/run.sh', '**/CMakeLists.txt', '**/.cmake', '**/*.h', '**/*.hpp', '**/*.c', '**/*.cpp', '**/*.cu', '**/*.cuh', '**/*.swift', '**/*.m', '**/*.metal', '**/*.comp', '**/*.glsl', '**/*.wgsl' ] concurrency: group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }} cancel-in-progress: true env: # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302) HF_TOKEN: ${{ secrets.HF_TOKEN_CI }} GGML_NLOOP: 3 GGML_N_THREADS: 1 LLAMA_ARG_LOG_COLORS: 1 LLAMA_ARG_LOG_PREFIX: 1 LLAMA_ARG_LOG_TIMESTAMPS: 1 jobs: gpu-cuda: runs-on: "hf-jobs-t4-small:cuda13" steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Install dependencies run: | sudo apt update sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip - name: ccache uses: ggml-org/ccache-action@v1.2.24 with: restore: false save: false - name: ccache-buckets-restore uses: ./.github/actions/ccache-buckets with: key: self-hosted-gpu-cuda folder: llama.cpp hf_bucket: ggml-org/cache - name: Test id: ggml-ci run: | nvidia-smi GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - name: ccache-buckets-save if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} uses: ./.github/actions/ccache-buckets env: HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} with: key: self-hosted-gpu-cuda folder: llama.cpp evict-old-files: 1d hf_bucket: ggml-org/cache save: true gpu-rocm: runs-on: [self-hosted, Linux, AMD] steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Test id: ggml-ci # HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness # issue on integrated RDNA3.5 (gfx1151) where batched inference returns # incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches # restores correctness. Remove once the underlying ROCm/HIP issue is fixed. env: HIP_LAUNCH_BLOCKING: "1" run: | rocminfo GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp gpu-vulkan-nvidia-cm: # runs-on: "hf-jobs-t4-small:ubuntu26_04" runs-on: [self-hosted, Linux, NVIDIA] steps: - name: Clone id: checkout uses: actions/checkout@v6 # - name: Install dependencies # run: | # sudo apt update # sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip # - name: ccache # uses: ggml-org/ccache-action@v1.2.24 # with: # restore: false # save: false # - name: ccache-buckets-restore # uses: ./.github/actions/ccache-buckets # with: # key: self-hosted-vulkan-nvidia-cm # folder: llama.cpp # hf_bucket: ggml-org/cache - name: Test id: ggml-ci run: | vulkaninfo --summary GG_BUILD_VULKAN=1 GGML_VK_DISABLE_COOPMAT2=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp # - name: ccache-buckets-save # if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} # uses: ./.github/actions/ccache-buckets # env: # HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} # with: # key: self-hosted-vulkan-nvidia-cm # folder: llama.cpp # evict-old-files: 1d # hf_bucket: ggml-org/cache # save: true gpu-vulkan-nvidia-cm2: # runs-on: "hf-jobs-t4-small:ubuntu26_04" runs-on: [self-hosted, Linux, NVIDIA, COOPMAT2] steps: - name: Clone id: checkout uses: actions/checkout@v6 # - name: Install dependencies # run: | # sudo apt update # sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip # - name: ccache # uses: ggml-org/ccache-action@v1.2.24 # with: # restore: false # save: false # - name: ccache-buckets-restore # uses: ./.github/actions/ccache-buckets # with: # key: self-hosted-vulkan-nvidia-cm2 # folder: llama.cpp # hf_bucket: ggml-org/cache - name: Test id: ggml-ci run: | vulkaninfo --summary GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp # - name: ccache-buckets-save # if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} # uses: ./.github/actions/ccache-buckets # env: # HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} # with: # key: self-hosted-vulkan-nvidia-cm2 # folder: llama.cpp # evict-old-files: 1d # hf_bucket: ggml-org/cache # save: true gpu-webgpu-nvidia: runs-on: "hf-jobs-t4-small:ubuntu26_04" steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Install dependencies run: | sudo apt update sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan1 mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip - name: ccache uses: ggml-org/ccache-action@v1.2.24 with: restore: false save: false - name: ccache-buckets-restore uses: ./.github/actions/ccache-buckets with: key: self-hosted-webgpu-nvidia folder: llama.cpp hf_bucket: ggml-org/cache - name: Dawn Dependency id: dawn-depends run: | DAWN_VERSION="v20260908.214631" DAWN_OWNER="google" DAWN_REPO="dawn" DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-ubuntu-latest-Release" echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" curl -L -o artifact.tar.gz \ "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" mkdir dawn tar -xvf artifact.tar.gz -C dawn --strip-components=1 - name: Test id: ggml-ci run: | GG_BUILD_WEBGPU=1 \ GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \ GG_BUILD_WEBGPU_DAWN_DIR="$GITHUB_WORKSPACE/dawn/lib64/cmake/Dawn" \ bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - name: ccache-buckets-save if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} uses: ./.github/actions/ccache-buckets env: HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} with: key: self-hosted-webgpu-nvidia folder: llama.cpp evict-old-files: 1d hf_bucket: ggml-org/cache save: true # TODO: provision AMX-compatible machine #cpu-amx: # runs-on: [self-hosted, Linux, CPU, AMX] # steps: # - name: Clone # id: checkout # uses: actions/checkout@v6 # - name: Test # id: ggml-ci # run: | # bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp # TODO: provision AMD GPU machine # amd-vulkan: # runs-on: [self-hosted, Linux, AMD] # steps: # - name: Clone # id: checkout # uses: actions/checkout@v6 # - name: Test # id: ggml-ci # run: | # vulkaninfo --summary # GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp # TODO: provision AMD GPU machine # amd-rocm: # runs-on: [self-hosted, Linux, AMD] # steps: # - name: Clone # id: checkout # uses: actions/checkout@v6 # - name: Test # id: ggml-ci # run: | # amd-smi static # GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp gpu-metal: runs-on: [self-hosted, macOS, ARM64] steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Test id: ggml-ci run: | GG_BUILD_METAL=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp gpu-webgpu-apple: runs-on: [self-hosted, macOS, ARM64] steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Dawn Dependency id: dawn-depends run: | DAWN_VERSION="v20260908.214631" DAWN_OWNER="google" DAWN_REPO="dawn" DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-macos-latest-Release" echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" curl -L -o artifact.tar.gz \ "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" mkdir dawn tar -xvf artifact.tar.gz -C dawn --strip-components=1 - name: Test id: ggml-ci run: | GG_BUILD_WEBGPU=1 GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \ bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp gpu-vulkan-apple: runs-on: [self-hosted, macOS, ARM64] steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Test id: ggml-ci run: | vulkaninfo --summary GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp gpu-vulkan-intel-linux: runs-on: [self-hosted, Linux, Intel] steps: - name: Clone id: checkout uses: actions/checkout@v6 with: persist-credentials: false - name: Test id: ggml-ci run: | vulkaninfo --summary GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp gpu-vulkan-intel-windows: runs-on: [self-hosted, Windows, X64, Intel] steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Test id: ggml-ci shell: C:\msys64\usr\bin\bash.exe --noprofile --norc -eo pipefail "{0}" env: MSYSTEM: UCRT64 CHERE_INVOKING: 1 PATH: C:\msys64\ucrt64\bin;C:\msys64\usr\bin;C:\Windows\System32;${{ env.PATH }} run: | vulkaninfo --summary # Skip python related tests with GG_BUILD_LOW_PERF=1 since Windows MSYS2 UCRT64 currently fails to create # a valid python environment for testing LLAMA_FATAL_WARNINGS=OFF GG_BUILD_NINJA=1 GG_BUILD_VULKAN=1 GG_BUILD_LOW_PERF=1 ./ci/run.sh ./results/llama.cpp ./mnt/llama.cpp gpu-openvino-low-perf: runs-on: [self-hosted, Linux, Intel, OpenVINO] env: # Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile OPENVINO_VERSION_MAJOR: "2026.4" OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3" steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Setup OpenVINO Toolkit uses: ./.github/actions/linux-setup-openvino with: path: ./openvino_toolkit version_major: ${{ env.OPENVINO_VERSION_MAJOR }} version_full: ${{ env.OPENVINO_VERSION_FULL }} - name: Install OpenVINO dependencies run: | cd ./openvino_toolkit chmod +x ./install_dependencies/install_openvino_dependencies.sh echo "Y" | sudo -E ./install_dependencies/install_openvino_dependencies.sh - name: Test id: ggml-ci run: | source ./openvino_toolkit/setupvars.sh GG_BUILD_OPENVINO=1 GGML_OPENVINO_DEVICE=GPU GG_BUILD_LOW_PERF=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp cpu-x64-high-perf: runs-on: [self-hosted, Linux, X64] steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Test id: ggml-ci run: | LLAMA_ARG_THREADS=$(nproc) GG_BUILD_HIGH_PERF=1 GG_BUILD_EXTRA_TESTS_0=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp cpu-arm64-high-perf-graviton4: runs-on: ah-ubuntu_24_04-c8g_8x steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Dependencies id: depends run: | set -euxo pipefail sudo apt-get update sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \ apt-get install -y \ build-essential \ python3-venv \ gpg \ wget \ time \ git-lfs git lfs install # install the latest cmake sudo install -d /usr/share/keyrings wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \ | gpg --dearmor \ | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \ | sudo tee /etc/apt/sources.list.d/kitware.list sudo apt-get update sudo apt-get install -y cmake - name: Test id: ggml-ci run: | LLAMA_ARG_THREADS=$(nproc) \ GG_BUILD_HIGH_PERF=1 \ GG_BUILD_NO_BF16=1 \ GG_BUILD_EXTRA_TESTS_0=1 \ bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp cpu-arm64-graviton4-kleidiai: runs-on: ah-ubuntu_24_04-c8g_8x steps: - name: Clone id: checkout uses: actions/checkout@v6 - name: Dependencies id: depends run: | set -euxo pipefail sudo apt-get update sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \ apt-get install -y \ build-essential \ python3-venv \ gpg \ wget \ time \ git-lfs git lfs install # install the latest cmake sudo install -d /usr/share/keyrings wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \ | gpg --dearmor \ | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \ | sudo tee /etc/apt/sources.list.d/kitware.list sudo apt-get update sudo apt-get install -y cmake - name: Test id: ggml-ci run: | LLAMA_ARG_THREADS=$(nproc) \ GG_BUILD_KLEIDIAI=1 \ GG_BUILD_EXTRA_TESTS_0=1 \ GG_BUILD_HIGH_PERF=1 \ bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp