mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-08-25 22:21:03 +02:00
985b14912b
* ci : apply ccache-clear with older/min/dry-run to all ccache jobs Assisted-by: llama.cpp:DeepSeek-v4-Flash-0731 * ci : install gh in ccache-clear if missing (container jobs) The ccache-clear action relies on the gh CLI, which is not present in container-based jobs. Install it on demand so those jobs can clear caches. Assisted-by: llama.cpp:DeepSeek-v4-Flash-0731 * ci : install gh via apt repo in ccache-clear The install.sh script used previously is no longer served (404). Switch to the official GitHub CLI apt repository, which is still available. Assisted-by: llama.cpp:DeepSeek-v4-Flash-0731 * ci : pass --repo to gh cache commands in ccache-clear In container jobs gh cannot auto-detect the repository from git, so gh cache list/delete fail with 'failed to run git: not a git repository'. Pass the repository explicitly via --repo using GITHUB_REPOSITORY. Assisted-by: llama.cpp:DeepSeek-v4-Flash-0731 * ci : drop -new suffix from vulkan ccache key The -new suffix was only needed to force a fresh cache. With ccache-clear now evicting stale caches, the original key can be used again. The old ccache-vulkan-ubuntu-24.04-arm-new entries still match the ccache-clear key prefix and are cleaned up automatically. Assisted-by: llama.cpp:DeepSeek-v4-Flash-0731 * ci : fix ccache-clear date parsing on macOS (BSD date) macOS ships BSD date, which has no -d option. The older cutoff check was silently disabled there: 'date: illegal option -- d' errors in the log and the loop was only stopped by the min limit, risking deletion of caches not older than the cutoff (e.g. saved by a concurrent job). Parse the ISO-8601 timestamps with GNU date when available and fall back to BSD date otherwise (TZ=UTC, fractional seconds dropped). Assisted-by: llama.cpp:DeepSeek-v4-Flash-0731 * ci : extract ccache-clear logic into scripts/ccache-clear.sh The composite action now consists of a dedicated step that installs the GitHub CLI when missing (e.g. in container jobs) and a thin step that calls the new script. The script follows the make-release-checks.sh conventions (usage/env header, set -euo pipefail, CLI flags) and only checks that gh is available. The action inputs are unchanged, so the workflow steps are untouched. Assisted-by: llama.cpp:DeepSeek-v4-Flash-0731 * ci : remove unused apple ccaches
204 lines
5.4 KiB
YAML
204 lines
5.4 KiB
YAML
name: Server
|
|
|
|
on:
|
|
workflow_dispatch: # allows manual triggering
|
|
inputs:
|
|
sha:
|
|
description: 'Commit SHA1 to build'
|
|
required: false
|
|
type: string
|
|
slow_tests:
|
|
description: 'Run slow tests'
|
|
required: true
|
|
type: boolean
|
|
push:
|
|
branches:
|
|
- master
|
|
paths: [
|
|
'.github/workflows/server.yml',
|
|
'**/CMakeLists.txt',
|
|
'**/Makefile',
|
|
'**/*.h',
|
|
'**/*.hpp',
|
|
'**/*.c',
|
|
'**/*.cpp',
|
|
'**/*.cu',
|
|
'**/*.swift',
|
|
'**/*.m',
|
|
'tools/server/**.*'
|
|
]
|
|
pull_request:
|
|
types: [opened, synchronize, reopened]
|
|
paths: [
|
|
'.github/workflows/server.yml',
|
|
'**/CMakeLists.txt',
|
|
'**/Makefile',
|
|
'**/*.h',
|
|
'**/*.hpp',
|
|
'**/*.c',
|
|
'**/*.cpp',
|
|
'**/*.cu',
|
|
'**/*.swift',
|
|
'**/*.m',
|
|
'tools/server/**.*'
|
|
]
|
|
|
|
env:
|
|
LLAMA_ARG_LOG_COLORS: 1
|
|
LLAMA_ARG_LOG_PREFIX: 1
|
|
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
|
LLAMA_ARG_LOG_VERBOSITY: 10
|
|
|
|
concurrency:
|
|
group: ${{ github.workflow }}-${{ github.ref }}-${{ github.head_ref || github.run_id }}
|
|
cancel-in-progress: true
|
|
|
|
jobs:
|
|
ubuntu:
|
|
runs-on: ubuntu-24.04-arm
|
|
|
|
steps:
|
|
- name: Dependencies
|
|
id: depends
|
|
run: |
|
|
sudo apt-get update
|
|
sudo apt-get -y install \
|
|
build-essential \
|
|
xxd \
|
|
git \
|
|
cmake \
|
|
curl \
|
|
wget \
|
|
language-pack-en \
|
|
libssl-dev
|
|
|
|
- name: Clone
|
|
id: checkout
|
|
uses: actions/checkout@v6
|
|
with:
|
|
fetch-depth: 0
|
|
ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}
|
|
|
|
- name: ccache
|
|
uses: ggml-org/ccache-action@v1.2.21
|
|
with:
|
|
key: server-ubuntu-24.04-arm
|
|
evict-old-files: 1d
|
|
save: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
|
|
|
- name: Build
|
|
id: cmake_build
|
|
run: |
|
|
cmake -B build \
|
|
-DGGML_SCHED_NO_REALLOC=ON
|
|
cmake --build build --config Release -j $(nproc) --target llama-server
|
|
|
|
- name: Python setup
|
|
id: setup_python
|
|
uses: actions/setup-python@v6
|
|
with:
|
|
python-version: '3.11'
|
|
pip-install: -r tools/server/tests/requirements.txt
|
|
|
|
- name: Tests
|
|
id: server_integration_tests
|
|
run: |
|
|
cd tools/server/tests
|
|
./tests.sh
|
|
|
|
- name: Slow tests
|
|
id: server_integration_tests_slow
|
|
if: ${{ github.event.schedule || github.event.inputs.slow_tests == 'true' }}
|
|
run: |
|
|
cd tools/server/tests
|
|
SLOW_TESTS=1 ./tests.sh
|
|
|
|
- name: Tests (Backend sampling)
|
|
id: server_integration_tests_backend_sampling
|
|
run: |
|
|
cd tools/server/tests
|
|
export LLAMA_ARG_BACKEND_SAMPLING=1
|
|
./tests.sh
|
|
|
|
- name: Slow tests (Backend sampling)
|
|
id: server_integration_tests_slow_backend_sampling
|
|
if: ${{ github.event.schedule || github.event.inputs.slow_tests == 'true' }}
|
|
run: |
|
|
cd tools/server/tests
|
|
export LLAMA_ARG_BACKEND_SAMPLING=1
|
|
SLOW_TESTS=1 ./tests.sh
|
|
|
|
- name: ccache-clear
|
|
uses: ./.github/actions/ccache-clear
|
|
env:
|
|
GH_TOKEN: ${{ github.token }}
|
|
with:
|
|
key: server-ubuntu-24.04-arm
|
|
older: 5m
|
|
min: 1
|
|
dry-run: ${{ github.event_name != 'push' || github.ref != 'refs/heads/master' }}
|
|
|
|
windows:
|
|
runs-on: windows-2025
|
|
|
|
steps:
|
|
- name: Clone
|
|
id: checkout
|
|
uses: actions/checkout@v6
|
|
with:
|
|
fetch-depth: 0
|
|
ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}
|
|
|
|
- name: ccache
|
|
uses: ggml-org/ccache-action@v1.2.21
|
|
with:
|
|
key: server-windows-2025-x64
|
|
evict-old-files: 1d
|
|
save: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
|
|
|
- name: Build
|
|
id: cmake_build
|
|
shell: cmd
|
|
run: |
|
|
cmake -B build -G "Ninja Multi-Config" ^
|
|
-DCMAKE_TOOLCHAIN_FILE=cmake/x64-windows-llvm.cmake ^
|
|
-DCMAKE_BUILD_TYPE=Release ^
|
|
-DLLAMA_BUILD_BORINGSSL=ON ^
|
|
-DGGML_SCHED_NO_REALLOC=ON
|
|
set /A NINJA_JOBS=%NUMBER_OF_PROCESSORS%-1
|
|
cmake --build build --config Release -j %NINJA_JOBS% --target llama-server
|
|
|
|
- name: Python setup
|
|
id: setup_python
|
|
uses: actions/setup-python@v6
|
|
with:
|
|
python-version: '3.11'
|
|
pip-install: -r tools/server/tests/requirements.txt
|
|
|
|
- name: Tests
|
|
id: server_integration_tests
|
|
shell: bash
|
|
run: |
|
|
cd tools/server/tests
|
|
export PYTHONIOENCODING=":replace"
|
|
./tests.sh
|
|
|
|
- name: Slow tests
|
|
id: server_integration_tests_slow
|
|
if: ${{ github.event.schedule || github.event.inputs.slow_tests == 'true' }}
|
|
shell: bash
|
|
run: |
|
|
cd tools/server/tests
|
|
export SLOW_TESTS="1"
|
|
./tests.sh
|
|
|
|
- name: ccache-clear
|
|
uses: ./.github/actions/ccache-clear
|
|
env:
|
|
GH_TOKEN: ${{ github.token }}
|
|
with:
|
|
key: server-windows-2025-x64
|
|
older: 5m
|
|
min: 1
|
|
dry-run: ${{ github.event_name != 'push' || github.ref != 'refs/heads/master' }}
|