mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-27 21:46:57 +02:00
* refactor build-self-hosted into backends * update workflow names * build -> ci * bump openvino * trigger on cpu and generic ggml changes
125 lines
3.3 KiB
YAML
125 lines
3.3 KiB
YAML
name: CI (self-hosted CUDA backend)
|
|
|
|
on:
|
|
workflow_dispatch: # allows manual triggering
|
|
push:
|
|
branches:
|
|
- master
|
|
paths: [
|
|
'.github/workflows/ci-self-hosted-cuda.yml',
|
|
'ci/run.sh',
|
|
'**/CMakeLists.txt',
|
|
'**/.cmake',
|
|
'**/*.h',
|
|
'**/*.hpp',
|
|
'**/*.c',
|
|
'**/*.cpp',
|
|
'**/*.cu',
|
|
'**/*.cuh'
|
|
]
|
|
|
|
pull_request:
|
|
types: [opened, synchronize, reopened]
|
|
paths: [
|
|
'.github/workflows/ci-self-hosted-cuda.yml',
|
|
'ci/run.sh',
|
|
'**/CMakeLists.txt',
|
|
'**/.cmake',
|
|
'ggml/src/*',
|
|
'ggml/src/ggml-cpu/**',
|
|
'ggml/src/ggml-cuda/**'
|
|
]
|
|
|
|
concurrency:
|
|
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
|
cancel-in-progress: true
|
|
|
|
env:
|
|
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
|
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
|
GGML_NLOOP: 3
|
|
GGML_N_THREADS: 1
|
|
LLAMA_ARG_LOG_COLORS: 1
|
|
LLAMA_ARG_LOG_PREFIX: 1
|
|
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
|
|
|
jobs:
|
|
gpu-cuda:
|
|
runs-on: "hf-jobs-t4-small:cuda13"
|
|
|
|
steps:
|
|
- name: Clone
|
|
id: checkout
|
|
uses: actions/checkout@v6
|
|
|
|
- name: Install dependencies
|
|
run: |
|
|
sudo apt update
|
|
sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip
|
|
|
|
- name: ccache
|
|
uses: ggml-org/[email protected]
|
|
with:
|
|
restore: false
|
|
save: false
|
|
|
|
- name: ccache-buckets-restore
|
|
uses: ./.github/actions/ccache-buckets
|
|
with:
|
|
key: self-hosted-gpu-cuda
|
|
folder: llama.cpp
|
|
hf_bucket: ggml-org/cache
|
|
|
|
- name: Test
|
|
id: ggml-ci
|
|
run: |
|
|
nvidia-smi
|
|
GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
|
|
|
- name: ccache-buckets-save
|
|
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
|
uses: ./.github/actions/ccache-buckets
|
|
env:
|
|
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
|
with:
|
|
key: self-hosted-gpu-cuda
|
|
folder: llama.cpp
|
|
evict-old-files: 1d
|
|
hf_bucket: ggml-org/cache
|
|
save: true
|
|
|
|
gpu-rocm:
|
|
runs-on: [self-hosted, Linux, AMD]
|
|
|
|
steps:
|
|
- name: Clone
|
|
id: checkout
|
|
uses: actions/checkout@v6
|
|
|
|
- name: Test
|
|
id: ggml-ci
|
|
# HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness
|
|
# issue on integrated RDNA3.5 (gfx1151) where batched inference returns
|
|
# incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches
|
|
# restores correctness. Remove once the underlying ROCm/HIP issue is fixed.
|
|
env:
|
|
HIP_LAUNCH_BLOCKING: "1"
|
|
run: |
|
|
rocminfo
|
|
GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
|
|
|
# TODO: provision AMD GPU machine
|
|
# amd-rocm:
|
|
# runs-on: [self-hosted, Linux, AMD]
|
|
|
|
# steps:
|
|
# - name: Clone
|
|
# id: checkout
|
|
# uses: actions/checkout@v6
|
|
|
|
# - name: Test
|
|
# id: ggml-ci
|
|
# run: |
|
|
# amd-smi static
|
|
# GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|