From 0f3070b0c8286fd9ef03193f664b33533ae8c2f0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sigbj=C3=B8rn=20Skj=C3=A6ret?= Date: Wed, 16 Sep 2026 14:59:17 +0200 Subject: [PATCH] refactor build-self-hosted into backends --- .github/workflows/build-self-hosted-cpu.yml | 111 ++++ .github/workflows/build-self-hosted-cuda.yml | 122 ++++ .../workflows/build-self-hosted-kleidiai.yml | 84 +++ .github/workflows/build-self-hosted-metal.yml | 57 ++ .../workflows/build-self-hosted-openvino.yml | 73 +++ .../workflows/build-self-hosted-vulkan.yml | 199 +++++++ .../workflows/build-self-hosted-webgpu.yml | 128 +++++ .github/workflows/build-self-hosted.yml | 539 ------------------ 8 files changed, 774 insertions(+), 539 deletions(-) create mode 100644 .github/workflows/build-self-hosted-cpu.yml create mode 100644 .github/workflows/build-self-hosted-cuda.yml create mode 100644 .github/workflows/build-self-hosted-kleidiai.yml create mode 100644 .github/workflows/build-self-hosted-metal.yml create mode 100644 .github/workflows/build-self-hosted-openvino.yml create mode 100644 .github/workflows/build-self-hosted-vulkan.yml create mode 100644 .github/workflows/build-self-hosted-webgpu.yml delete mode 100644 .github/workflows/build-self-hosted.yml diff --git a/.github/workflows/build-self-hosted-cpu.yml b/.github/workflows/build-self-hosted-cpu.yml new file mode 100644 index 0000000000..5aa0ba404c --- /dev/null +++ b/.github/workflows/build-self-hosted-cpu.yml @@ -0,0 +1,111 @@ +name: CI (self-hosted) + +on: + workflow_dispatch: # allows manual triggering + push: + branches: + - master + paths: [ + '.github/workflows/build-self-hosted-cpu.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + '**/*.h', + '**/*.hpp', + '**/*.c', + '**/*.cpp' + ] + + pull_request: + types: [opened, synchronize, reopened] + paths: [ + '.github/workflows/build-self-hosted-cpu.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + 'ggml/src/ggml-cpu/**' + ] + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }} + cancel-in-progress: true + +env: + # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302) + HF_TOKEN: ${{ secrets.HF_TOKEN_CI }} + GGML_NLOOP: 3 + GGML_N_THREADS: 1 + LLAMA_ARG_LOG_COLORS: 1 + LLAMA_ARG_LOG_PREFIX: 1 + LLAMA_ARG_LOG_TIMESTAMPS: 1 + +jobs: + cpu-x64-high-perf: + runs-on: [self-hosted, Linux, X64] + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Test + id: ggml-ci + run: | + LLAMA_ARG_THREADS=$(nproc) GG_BUILD_HIGH_PERF=1 GG_BUILD_EXTRA_TESTS_0=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp + + cpu-arm64-high-perf-graviton4: + runs-on: ah-ubuntu_24_04-c8g_8x + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Dependencies + id: depends + run: | + set -euxo pipefail + sudo apt-get update + sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \ + apt-get install -y \ + build-essential \ + python3-venv \ + gpg \ + wget \ + time \ + git-lfs + + git lfs install + + # install the latest cmake + sudo install -d /usr/share/keyrings + wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \ + | gpg --dearmor \ + | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null + echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \ + | sudo tee /etc/apt/sources.list.d/kitware.list + sudo apt-get update + sudo apt-get install -y cmake + + - name: Test + id: ggml-ci + run: | + LLAMA_ARG_THREADS=$(nproc) \ + GG_BUILD_HIGH_PERF=1 \ + GG_BUILD_NO_BF16=1 \ + GG_BUILD_EXTRA_TESTS_0=1 \ + bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp + + # TODO: provision AMX-compatible machine + #cpu-amx: + # runs-on: [self-hosted, Linux, CPU, AMX] + + # steps: + # - name: Clone + # id: checkout + # uses: actions/checkout@v6 + + # - name: Test + # id: ggml-ci + # run: | + # bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp diff --git a/.github/workflows/build-self-hosted-cuda.yml b/.github/workflows/build-self-hosted-cuda.yml new file mode 100644 index 0000000000..c6e795c73e --- /dev/null +++ b/.github/workflows/build-self-hosted-cuda.yml @@ -0,0 +1,122 @@ +name: CI (self-hosted) + +on: + workflow_dispatch: # allows manual triggering + push: + branches: + - master + paths: [ + '.github/workflows/build-self-hosted-cuda.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + '**/*.h', + '**/*.hpp', + '**/*.c', + '**/*.cpp', + '**/*.cu', + '**/*.cuh' + ] + + pull_request: + types: [opened, synchronize, reopened] + paths: [ + '.github/workflows/build-self-hosted-cuda.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + 'ggml/src/ggml-cuda/**' + ] + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }} + cancel-in-progress: true + +env: + # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302) + HF_TOKEN: ${{ secrets.HF_TOKEN_CI }} + GGML_NLOOP: 3 + GGML_N_THREADS: 1 + LLAMA_ARG_LOG_COLORS: 1 + LLAMA_ARG_LOG_PREFIX: 1 + LLAMA_ARG_LOG_TIMESTAMPS: 1 + +jobs: + gpu-cuda: + runs-on: "hf-jobs-t4-small:cuda13" + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Install dependencies + run: | + sudo apt update + sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip + + - name: ccache + uses: ggml-org/ccache-action@v1.2.24 + with: + restore: false + save: false + + - name: ccache-buckets-restore + uses: ./.github/actions/ccache-buckets + with: + key: self-hosted-gpu-cuda + folder: llama.cpp + hf_bucket: ggml-org/cache + + - name: Test + id: ggml-ci + run: | + nvidia-smi + GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp + + - name: ccache-buckets-save + if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} + uses: ./.github/actions/ccache-buckets + env: + HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} + with: + key: self-hosted-gpu-cuda + folder: llama.cpp + evict-old-files: 1d + hf_bucket: ggml-org/cache + save: true + + gpu-rocm: + runs-on: [self-hosted, Linux, AMD] + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Test + id: ggml-ci + # HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness + # issue on integrated RDNA3.5 (gfx1151) where batched inference returns + # incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches + # restores correctness. Remove once the underlying ROCm/HIP issue is fixed. + env: + HIP_LAUNCH_BLOCKING: "1" + run: | + rocminfo + GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp + + # TODO: provision AMD GPU machine + # amd-rocm: + # runs-on: [self-hosted, Linux, AMD] + + # steps: + # - name: Clone + # id: checkout + # uses: actions/checkout@v6 + + # - name: Test + # id: ggml-ci + # run: | + # amd-smi static + # GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp diff --git a/.github/workflows/build-self-hosted-kleidiai.yml b/.github/workflows/build-self-hosted-kleidiai.yml new file mode 100644 index 0000000000..98bd64732b --- /dev/null +++ b/.github/workflows/build-self-hosted-kleidiai.yml @@ -0,0 +1,84 @@ +name: CI (self-hosted) + +on: + workflow_dispatch: # allows manual triggering + push: + branches: + - master + paths: [ + '.github/workflows/build-self-hosted-kleidiai.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + '**/*.h', + '**/*.hpp', + '**/*.c', + '**/*.cpp' + ] + + pull_request: + types: [opened, synchronize, reopened] + paths: [ + '.github/workflows/build-self-hosted-kleidiai.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + 'ggml/src/ggml-cpu/kleidiai/**' + ] + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }} + cancel-in-progress: true + +env: + # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302) + HF_TOKEN: ${{ secrets.HF_TOKEN_CI }} + GGML_NLOOP: 3 + GGML_N_THREADS: 1 + LLAMA_ARG_LOG_COLORS: 1 + LLAMA_ARG_LOG_PREFIX: 1 + LLAMA_ARG_LOG_TIMESTAMPS: 1 + +jobs: + cpu-arm64-graviton4-kleidiai: + runs-on: ah-ubuntu_24_04-c8g_8x + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Dependencies + id: depends + run: | + set -euxo pipefail + sudo apt-get update + sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \ + apt-get install -y \ + build-essential \ + python3-venv \ + gpg \ + wget \ + time \ + git-lfs + + git lfs install + + # install the latest cmake + sudo install -d /usr/share/keyrings + wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \ + | gpg --dearmor \ + | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null + echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \ + | sudo tee /etc/apt/sources.list.d/kitware.list + sudo apt-get update + sudo apt-get install -y cmake + + - name: Test + id: ggml-ci + run: | + LLAMA_ARG_THREADS=$(nproc) \ + GG_BUILD_KLEIDIAI=1 \ + GG_BUILD_EXTRA_TESTS_0=1 \ + GG_BUILD_HIGH_PERF=1 \ + bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp diff --git a/.github/workflows/build-self-hosted-metal.yml b/.github/workflows/build-self-hosted-metal.yml new file mode 100644 index 0000000000..f77d65a559 --- /dev/null +++ b/.github/workflows/build-self-hosted-metal.yml @@ -0,0 +1,57 @@ +name: CI (self-hosted) + +on: + workflow_dispatch: # allows manual triggering + push: + branches: + - master + paths: [ + '.github/workflows/build-self-hosted-metal.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + '**/*.h', + '**/*.hpp', + '**/*.c', + '**/*.cpp', + '**/*.swift', + '**/*.m', + '**/*.metal' + ] + + pull_request: + types: [opened, synchronize, reopened] + paths: [ + '.github/workflows/build-self-hosted-metal.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + 'ggml/src/ggml-metal/**' + ] + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }} + cancel-in-progress: true + +env: + # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302) + HF_TOKEN: ${{ secrets.HF_TOKEN_CI }} + GGML_NLOOP: 3 + GGML_N_THREADS: 1 + LLAMA_ARG_LOG_COLORS: 1 + LLAMA_ARG_LOG_PREFIX: 1 + LLAMA_ARG_LOG_TIMESTAMPS: 1 + +jobs: + gpu-metal: + runs-on: [self-hosted, macOS, ARM64] + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Test + id: ggml-ci + run: | + GG_BUILD_METAL=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp diff --git a/.github/workflows/build-self-hosted-openvino.yml b/.github/workflows/build-self-hosted-openvino.yml new file mode 100644 index 0000000000..34c0710d78 --- /dev/null +++ b/.github/workflows/build-self-hosted-openvino.yml @@ -0,0 +1,73 @@ +name: CI (self-hosted) + +on: + workflow_dispatch: # allows manual triggering + push: + branches: + - master + paths: [ + '.github/workflows/build-self-hosted-openvino.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + '**/*.h', + '**/*.hpp', + '**/*.c', + '**/*.cpp' + ] + + pull_request: + types: [opened, synchronize, reopened] + paths: [ + '.github/workflows/build-self-hosted-openvino.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + 'ggml/src/ggml-openvino/**' + ] + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }} + cancel-in-progress: true + +env: + # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302) + HF_TOKEN: ${{ secrets.HF_TOKEN_CI }} + GGML_NLOOP: 3 + GGML_N_THREADS: 1 + LLAMA_ARG_LOG_COLORS: 1 + LLAMA_ARG_LOG_PREFIX: 1 + LLAMA_ARG_LOG_TIMESTAMPS: 1 + +jobs: + gpu-openvino-low-perf: + runs-on: [self-hosted, Linux, Intel, OpenVINO] + + env: + # Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile + OPENVINO_VERSION_MAJOR: "2026.3.1" + OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d" + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Setup OpenVINO Toolkit + uses: ./.github/actions/linux-setup-openvino + with: + path: ./openvino_toolkit + version_major: ${{ env.OPENVINO_VERSION_MAJOR }} + version_full: ${{ env.OPENVINO_VERSION_FULL }} + + - name: Install OpenVINO dependencies + run: | + cd ./openvino_toolkit + chmod +x ./install_dependencies/install_openvino_dependencies.sh + echo "Y" | sudo -E ./install_dependencies/install_openvino_dependencies.sh + + - name: Test + id: ggml-ci + run: | + source ./openvino_toolkit/setupvars.sh + GG_BUILD_OPENVINO=1 GGML_OPENVINO_DEVICE=GPU GG_BUILD_LOW_PERF=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp diff --git a/.github/workflows/build-self-hosted-vulkan.yml b/.github/workflows/build-self-hosted-vulkan.yml new file mode 100644 index 0000000000..2800869de9 --- /dev/null +++ b/.github/workflows/build-self-hosted-vulkan.yml @@ -0,0 +1,199 @@ +name: CI (self-hosted) + +on: + workflow_dispatch: # allows manual triggering + push: + branches: + - master + paths: [ + '.github/workflows/build-self-hosted-vulkan.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + '**/*.h', + '**/*.hpp', + '**/*.c', + '**/*.cpp', + '**/*.comp', + '**/*.glsl' + ] + + pull_request: + types: [opened, synchronize, reopened] + paths: [ + '.github/workflows/build-self-hosted-vulkan.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + 'ggml/src/ggml-vulkan/**' + ] + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }} + cancel-in-progress: true + +env: + # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302) + HF_TOKEN: ${{ secrets.HF_TOKEN_CI }} + GGML_NLOOP: 3 + GGML_N_THREADS: 1 + LLAMA_ARG_LOG_COLORS: 1 + LLAMA_ARG_LOG_PREFIX: 1 + LLAMA_ARG_LOG_TIMESTAMPS: 1 + +jobs: + gpu-vulkan-nvidia-cm: + # runs-on: "hf-jobs-t4-small:ubuntu26_04" + runs-on: [self-hosted, Linux, NVIDIA] + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + # - name: Install dependencies + # run: | + # sudo apt update + # sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip + + # - name: ccache + # uses: ggml-org/ccache-action@v1.2.24 + # with: + # restore: false + # save: false + + # - name: ccache-buckets-restore + # uses: ./.github/actions/ccache-buckets + # with: + # key: self-hosted-vulkan-nvidia-cm + # folder: llama.cpp + # hf_bucket: ggml-org/cache + + - name: Test + id: ggml-ci + run: | + vulkaninfo --summary + GG_BUILD_VULKAN=1 GGML_VK_DISABLE_COOPMAT2=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp + + # - name: ccache-buckets-save + # if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} + # uses: ./.github/actions/ccache-buckets + # env: + # HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} + # with: + # key: self-hosted-vulkan-nvidia-cm + # folder: llama.cpp + # evict-old-files: 1d + # hf_bucket: ggml-org/cache + # save: true + + gpu-vulkan-nvidia-cm2: + # runs-on: "hf-jobs-t4-small:ubuntu26_04" + runs-on: [self-hosted, Linux, NVIDIA, COOPMAT2] + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + # - name: Install dependencies + # run: | + # sudo apt update + # sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip + + # - name: ccache + # uses: ggml-org/ccache-action@v1.2.24 + # with: + # restore: false + # save: false + + # - name: ccache-buckets-restore + # uses: ./.github/actions/ccache-buckets + # with: + # key: self-hosted-vulkan-nvidia-cm2 + # folder: llama.cpp + # hf_bucket: ggml-org/cache + + - name: Test + id: ggml-ci + run: | + vulkaninfo --summary + GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp + + # - name: ccache-buckets-save + # if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} + # uses: ./.github/actions/ccache-buckets + # env: + # HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} + # with: + # key: self-hosted-vulkan-nvidia-cm2 + # folder: llama.cpp + # evict-old-files: 1d + # hf_bucket: ggml-org/cache + # save: true + + gpu-vulkan-apple: + runs-on: [self-hosted, macOS, ARM64] + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Test + id: ggml-ci + run: | + vulkaninfo --summary + GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp + + gpu-vulkan-intel-linux: + runs-on: [self-hosted, Linux, Intel] + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + with: + persist-credentials: false + + - name: Test + id: ggml-ci + run: | + vulkaninfo --summary + GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp + + gpu-vulkan-intel-windows: + runs-on: [self-hosted, Windows, X64, Intel] + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Test + id: ggml-ci + shell: C:\msys64\usr\bin\bash.exe --noprofile --norc -eo pipefail "{0}" + env: + MSYSTEM: UCRT64 + CHERE_INVOKING: 1 + PATH: C:\msys64\ucrt64\bin;C:\msys64\usr\bin;C:\Windows\System32;${{ env.PATH }} + run: | + vulkaninfo --summary + # Skip python related tests with GG_BUILD_LOW_PERF=1 since Windows MSYS2 UCRT64 currently fails to create + # a valid python environment for testing + LLAMA_FATAL_WARNINGS=OFF GG_BUILD_NINJA=1 GG_BUILD_VULKAN=1 GG_BUILD_LOW_PERF=1 ./ci/run.sh ./results/llama.cpp ./mnt/llama.cpp + + # TODO: provision AMD GPU machine + # amd-vulkan: + # runs-on: [self-hosted, Linux, AMD] + + # steps: + # - name: Clone + # id: checkout + # uses: actions/checkout@v6 + + # - name: Test + # id: ggml-ci + # run: | + # vulkaninfo --summary + # GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp diff --git a/.github/workflows/build-self-hosted-webgpu.yml b/.github/workflows/build-self-hosted-webgpu.yml new file mode 100644 index 0000000000..fabc83f0b1 --- /dev/null +++ b/.github/workflows/build-self-hosted-webgpu.yml @@ -0,0 +1,128 @@ +name: CI (self-hosted) + +on: + workflow_dispatch: # allows manual triggering + push: + branches: + - master + paths: [ + '.github/workflows/build-self-hosted-webgpu.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + '**/*.h', + '**/*.hpp', + '**/*.c', + '**/*.cpp', + '**/*.wgsl' + ] + + pull_request: + types: [opened, synchronize, reopened] + paths: [ + '.github/workflows/build-self-hosted-webgpu.yml', + 'ci/run.sh', + '**/CMakeLists.txt', + '**/.cmake', + 'ggml/src/ggml-webgpu/**' + ] + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }} + cancel-in-progress: true + +env: + # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302) + HF_TOKEN: ${{ secrets.HF_TOKEN_CI }} + GGML_NLOOP: 3 + GGML_N_THREADS: 1 + LLAMA_ARG_LOG_COLORS: 1 + LLAMA_ARG_LOG_PREFIX: 1 + LLAMA_ARG_LOG_TIMESTAMPS: 1 + +jobs: + gpu-webgpu-nvidia: + runs-on: "hf-jobs-t4-small:ubuntu26_04" + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Install dependencies + run: | + sudo apt update + sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan1 mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip + + - name: ccache + uses: ggml-org/ccache-action@v1.2.24 + with: + restore: false + save: false + + - name: ccache-buckets-restore + uses: ./.github/actions/ccache-buckets + with: + key: self-hosted-webgpu-nvidia + folder: llama.cpp + hf_bucket: ggml-org/cache + + - name: Dawn Dependency + id: dawn-depends + run: | + DAWN_VERSION="v20260908.214631" + DAWN_OWNER="google" + DAWN_REPO="dawn" + DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-ubuntu-latest-Release" + echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" + curl -L -o artifact.tar.gz \ + "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" + mkdir dawn + tar -xvf artifact.tar.gz -C dawn --strip-components=1 + + - name: Test + id: ggml-ci + run: | + GG_BUILD_WEBGPU=1 \ + GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \ + GG_BUILD_WEBGPU_DAWN_DIR="$GITHUB_WORKSPACE/dawn/lib64/cmake/Dawn" \ + bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp + + - name: ccache-buckets-save + if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} + uses: ./.github/actions/ccache-buckets + env: + HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} + with: + key: self-hosted-webgpu-nvidia + folder: llama.cpp + evict-old-files: 1d + hf_bucket: ggml-org/cache + save: true + + gpu-webgpu-apple: + runs-on: [self-hosted, macOS, ARM64] + + steps: + - name: Clone + id: checkout + uses: actions/checkout@v6 + + - name: Dawn Dependency + id: dawn-depends + run: | + DAWN_VERSION="v20260908.214631" + DAWN_OWNER="google" + DAWN_REPO="dawn" + DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-macos-latest-Release" + echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" + curl -L -o artifact.tar.gz \ + "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" + mkdir dawn + tar -xvf artifact.tar.gz -C dawn --strip-components=1 + + - name: Test + id: ggml-ci + run: | + GG_BUILD_WEBGPU=1 GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \ + bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp diff --git a/.github/workflows/build-self-hosted.yml b/.github/workflows/build-self-hosted.yml deleted file mode 100644 index d54f71ac51..0000000000 --- a/.github/workflows/build-self-hosted.yml +++ /dev/null @@ -1,539 +0,0 @@ -name: CI (self-hosted) - -on: - workflow_dispatch: # allows manual triggering - push: - branches: - - master - paths: [ - '.github/workflows/build-self-hosted.yml', - 'ci/run.sh', - '**/CMakeLists.txt', - '**/.cmake', - '**/*.h', - '**/*.hpp', - '**/*.c', - '**/*.cpp', - '**/*.cu', - '**/*.cuh', - '**/*.swift', - '**/*.m', - '**/*.metal', - '**/*.comp', - '**/*.glsl', - '**/*.wgsl' - ] - - pull_request: - types: [opened, synchronize, reopened] - paths: [ - '.github/workflows/build-self-hosted.yml', - 'ci/run.sh', - '**/CMakeLists.txt', - '**/.cmake', - '**/*.h', - '**/*.hpp', - '**/*.c', - '**/*.cpp', - '**/*.cu', - '**/*.cuh', - '**/*.swift', - '**/*.m', - '**/*.metal', - '**/*.comp', - '**/*.glsl', - '**/*.wgsl' - ] - -concurrency: - group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }} - cancel-in-progress: true - -env: - # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302) - HF_TOKEN: ${{ secrets.HF_TOKEN_CI }} - GGML_NLOOP: 3 - GGML_N_THREADS: 1 - LLAMA_ARG_LOG_COLORS: 1 - LLAMA_ARG_LOG_PREFIX: 1 - LLAMA_ARG_LOG_TIMESTAMPS: 1 - -jobs: - gpu-cuda: - runs-on: "hf-jobs-t4-small:cuda13" - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Install dependencies - run: | - sudo apt update - sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip - - - name: ccache - uses: ggml-org/ccache-action@v1.2.24 - with: - restore: false - save: false - - - name: ccache-buckets-restore - uses: ./.github/actions/ccache-buckets - with: - key: self-hosted-gpu-cuda - folder: llama.cpp - hf_bucket: ggml-org/cache - - - name: Test - id: ggml-ci - run: | - nvidia-smi - GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - - name: ccache-buckets-save - if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} - uses: ./.github/actions/ccache-buckets - env: - HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} - with: - key: self-hosted-gpu-cuda - folder: llama.cpp - evict-old-files: 1d - hf_bucket: ggml-org/cache - save: true - - gpu-rocm: - runs-on: [self-hosted, Linux, AMD] - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Test - id: ggml-ci - # HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness - # issue on integrated RDNA3.5 (gfx1151) where batched inference returns - # incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches - # restores correctness. Remove once the underlying ROCm/HIP issue is fixed. - env: - HIP_LAUNCH_BLOCKING: "1" - run: | - rocminfo - GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - gpu-vulkan-nvidia-cm: - # runs-on: "hf-jobs-t4-small:ubuntu26_04" - runs-on: [self-hosted, Linux, NVIDIA] - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - # - name: Install dependencies - # run: | - # sudo apt update - # sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip - - # - name: ccache - # uses: ggml-org/ccache-action@v1.2.24 - # with: - # restore: false - # save: false - - # - name: ccache-buckets-restore - # uses: ./.github/actions/ccache-buckets - # with: - # key: self-hosted-vulkan-nvidia-cm - # folder: llama.cpp - # hf_bucket: ggml-org/cache - - - name: Test - id: ggml-ci - run: | - vulkaninfo --summary - GG_BUILD_VULKAN=1 GGML_VK_DISABLE_COOPMAT2=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - # - name: ccache-buckets-save - # if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} - # uses: ./.github/actions/ccache-buckets - # env: - # HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} - # with: - # key: self-hosted-vulkan-nvidia-cm - # folder: llama.cpp - # evict-old-files: 1d - # hf_bucket: ggml-org/cache - # save: true - - gpu-vulkan-nvidia-cm2: - # runs-on: "hf-jobs-t4-small:ubuntu26_04" - runs-on: [self-hosted, Linux, NVIDIA, COOPMAT2] - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - # - name: Install dependencies - # run: | - # sudo apt update - # sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip - - # - name: ccache - # uses: ggml-org/ccache-action@v1.2.24 - # with: - # restore: false - # save: false - - # - name: ccache-buckets-restore - # uses: ./.github/actions/ccache-buckets - # with: - # key: self-hosted-vulkan-nvidia-cm2 - # folder: llama.cpp - # hf_bucket: ggml-org/cache - - - name: Test - id: ggml-ci - run: | - vulkaninfo --summary - GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - # - name: ccache-buckets-save - # if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} - # uses: ./.github/actions/ccache-buckets - # env: - # HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} - # with: - # key: self-hosted-vulkan-nvidia-cm2 - # folder: llama.cpp - # evict-old-files: 1d - # hf_bucket: ggml-org/cache - # save: true - - gpu-webgpu-nvidia: - runs-on: "hf-jobs-t4-small:ubuntu26_04" - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Install dependencies - run: | - sudo apt update - sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan1 mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip - - - name: ccache - uses: ggml-org/ccache-action@v1.2.24 - with: - restore: false - save: false - - - name: ccache-buckets-restore - uses: ./.github/actions/ccache-buckets - with: - key: self-hosted-webgpu-nvidia - folder: llama.cpp - hf_bucket: ggml-org/cache - - - name: Dawn Dependency - id: dawn-depends - run: | - DAWN_VERSION="v20260908.214631" - DAWN_OWNER="google" - DAWN_REPO="dawn" - DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-ubuntu-latest-Release" - echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" - curl -L -o artifact.tar.gz \ - "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" - mkdir dawn - tar -xvf artifact.tar.gz -C dawn --strip-components=1 - - - name: Test - id: ggml-ci - run: | - GG_BUILD_WEBGPU=1 \ - GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \ - GG_BUILD_WEBGPU_DAWN_DIR="$GITHUB_WORKSPACE/dawn/lib64/cmake/Dawn" \ - bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - - name: ccache-buckets-save - if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }} - uses: ./.github/actions/ccache-buckets - env: - HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }} - with: - key: self-hosted-webgpu-nvidia - folder: llama.cpp - evict-old-files: 1d - hf_bucket: ggml-org/cache - save: true - - # TODO: provision AMX-compatible machine - #cpu-amx: - # runs-on: [self-hosted, Linux, CPU, AMX] - - # steps: - # - name: Clone - # id: checkout - # uses: actions/checkout@v6 - - # - name: Test - # id: ggml-ci - # run: | - # bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - # TODO: provision AMD GPU machine - # amd-vulkan: - # runs-on: [self-hosted, Linux, AMD] - - # steps: - # - name: Clone - # id: checkout - # uses: actions/checkout@v6 - - # - name: Test - # id: ggml-ci - # run: | - # vulkaninfo --summary - # GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - # TODO: provision AMD GPU machine - # amd-rocm: - # runs-on: [self-hosted, Linux, AMD] - - # steps: - # - name: Clone - # id: checkout - # uses: actions/checkout@v6 - - # - name: Test - # id: ggml-ci - # run: | - # amd-smi static - # GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - gpu-metal: - runs-on: [self-hosted, macOS, ARM64] - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Test - id: ggml-ci - run: | - GG_BUILD_METAL=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - gpu-webgpu-apple: - runs-on: [self-hosted, macOS, ARM64] - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Dawn Dependency - id: dawn-depends - run: | - DAWN_VERSION="v20260908.214631" - DAWN_OWNER="google" - DAWN_REPO="dawn" - DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-macos-latest-Release" - echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" - curl -L -o artifact.tar.gz \ - "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz" - mkdir dawn - tar -xvf artifact.tar.gz -C dawn --strip-components=1 - - - name: Test - id: ggml-ci - run: | - GG_BUILD_WEBGPU=1 GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \ - bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - gpu-vulkan-apple: - runs-on: [self-hosted, macOS, ARM64] - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Test - id: ggml-ci - run: | - vulkaninfo --summary - GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - gpu-vulkan-intel-linux: - runs-on: [self-hosted, Linux, Intel] - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - with: - persist-credentials: false - - - name: Test - id: ggml-ci - run: | - vulkaninfo --summary - GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - gpu-vulkan-intel-windows: - runs-on: [self-hosted, Windows, X64, Intel] - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Test - id: ggml-ci - shell: C:\msys64\usr\bin\bash.exe --noprofile --norc -eo pipefail "{0}" - env: - MSYSTEM: UCRT64 - CHERE_INVOKING: 1 - PATH: C:\msys64\ucrt64\bin;C:\msys64\usr\bin;C:\Windows\System32;${{ env.PATH }} - run: | - vulkaninfo --summary - # Skip python related tests with GG_BUILD_LOW_PERF=1 since Windows MSYS2 UCRT64 currently fails to create - # a valid python environment for testing - LLAMA_FATAL_WARNINGS=OFF GG_BUILD_NINJA=1 GG_BUILD_VULKAN=1 GG_BUILD_LOW_PERF=1 ./ci/run.sh ./results/llama.cpp ./mnt/llama.cpp - - gpu-openvino-low-perf: - runs-on: [self-hosted, Linux, Intel, OpenVINO] - - env: - # Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile - OPENVINO_VERSION_MAJOR: "2026.4" - OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3" - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Setup OpenVINO Toolkit - uses: ./.github/actions/linux-setup-openvino - with: - path: ./openvino_toolkit - version_major: ${{ env.OPENVINO_VERSION_MAJOR }} - version_full: ${{ env.OPENVINO_VERSION_FULL }} - - - name: Install OpenVINO dependencies - run: | - cd ./openvino_toolkit - chmod +x ./install_dependencies/install_openvino_dependencies.sh - echo "Y" | sudo -E ./install_dependencies/install_openvino_dependencies.sh - - - name: Test - id: ggml-ci - run: | - source ./openvino_toolkit/setupvars.sh - GG_BUILD_OPENVINO=1 GGML_OPENVINO_DEVICE=GPU GG_BUILD_LOW_PERF=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - cpu-x64-high-perf: - runs-on: [self-hosted, Linux, X64] - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Test - id: ggml-ci - run: | - LLAMA_ARG_THREADS=$(nproc) GG_BUILD_HIGH_PERF=1 GG_BUILD_EXTRA_TESTS_0=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - cpu-arm64-high-perf-graviton4: - runs-on: ah-ubuntu_24_04-c8g_8x - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Dependencies - id: depends - run: | - set -euxo pipefail - sudo apt-get update - sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \ - apt-get install -y \ - build-essential \ - python3-venv \ - gpg \ - wget \ - time \ - git-lfs - - git lfs install - - # install the latest cmake - sudo install -d /usr/share/keyrings - wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \ - | gpg --dearmor \ - | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null - echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \ - | sudo tee /etc/apt/sources.list.d/kitware.list - sudo apt-get update - sudo apt-get install -y cmake - - - name: Test - id: ggml-ci - run: | - LLAMA_ARG_THREADS=$(nproc) \ - GG_BUILD_HIGH_PERF=1 \ - GG_BUILD_NO_BF16=1 \ - GG_BUILD_EXTRA_TESTS_0=1 \ - bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp - - cpu-arm64-graviton4-kleidiai: - runs-on: ah-ubuntu_24_04-c8g_8x - - steps: - - name: Clone - id: checkout - uses: actions/checkout@v6 - - - name: Dependencies - id: depends - run: | - set -euxo pipefail - sudo apt-get update - sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \ - apt-get install -y \ - build-essential \ - python3-venv \ - gpg \ - wget \ - time \ - git-lfs - - git lfs install - - # install the latest cmake - sudo install -d /usr/share/keyrings - wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \ - | gpg --dearmor \ - | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null - echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \ - | sudo tee /etc/apt/sources.list.d/kitware.list - sudo apt-get update - sudo apt-get install -y cmake - - - name: Test - id: ggml-ci - run: | - LLAMA_ARG_THREADS=$(nproc) \ - GG_BUILD_KLEIDIAI=1 \ - GG_BUILD_EXTRA_TESTS_0=1 \ - GG_BUILD_HIGH_PERF=1 \ - bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp