diff --git a/common/CMakeLists.txt b/common/CMakeLists.txt index 1506bf6479..9a43911d35 100644 --- a/common/CMakeLists.txt +++ b/common/CMakeLists.txt @@ -134,6 +134,8 @@ set_target_properties(${TARGET} PROPERTIES target_include_directories(${TARGET} PUBLIC .) target_link_libraries (${TARGET} PUBLIC vendor::nlohmann vendor::sheredom) target_compile_features (${TARGET} PUBLIC cxx_std_17) +target_precompile_headers (${TARGET} PRIVATE common.h) +target_precompile_headers (${TARGET} PRIVATE chat.h) if (LLAMA_SUBPROCESS) target_compile_definitions(${TARGET} PUBLIC LLAMA_SUBPROCESS) diff --git a/docs/build-profiling.md b/docs/build-profiling.md new file mode 100644 index 0000000000..839e7cca4c --- /dev/null +++ b/docs/build-profiling.md @@ -0,0 +1,122 @@ +## Build profiling +This page is a working document for analyzing the current build and try to +identify ways to improve the build time. + +### Requirements +The profiling script requires clang to be used as the compiler tool chain and +also requires that ClangBuildAnalyzer is installed. + +Mac: +```console +brew install clang-build-analyzer +``` + +Linux: +```console +git clone https://github.com/aras-p/ClangBuildAnalyzer.git +cd ClangBuildAnalyzer +cmake -B build -DCMAKE_BUILD_TYPE=Release +cmake --build build -j$(nproc) +sudo cp build/ClangBuildAnalyzer /usr/local/bin/ +``` + +Windows: install LLVM/clang and Ninja (e.g. via the +[LLVM releases page](https://github.com/llvm/llvm-project/releases) and +`winget install Ninja-build.Ninja`), then build ClangBuildAnalyzer the same +way as on Linux: +```console +git clone https://github.com/aras-p/ClangBuildAnalyzer.git +cd ClangBuildAnalyzer +cmake -B build -G Ninja -DCMAKE_C_COMPILER=clang -DCMAKE_CXX_COMPILER=clang++ -DCMAKE_BUILD_TYPE=Release +cmake --build build --config Release +``` +Then add `ClangBuildAnalyzer\build` to `PATH`. + +### Usage +Mac/Linux: +```console +$ ./scripts/build-profile.sh +``` + +Windows: +```console +> .\scripts\build-profile.ps1 +``` + +Both accept `--full`/`-Full` (include Server, Tools, and Tests) and a jobs +override (`-jN` / `-Jobs N`). + +Note: on Windows, `cmake` defaults to the Visual Studio generator, which +ignores `CMAKE_C_COMPILER`/`CMAKE_CXX_COMPILER` and silently falls back to +MSVC. `build-profile.ps1` passes `-G Ninja` so clang is actually used, this +is required on ARM64. + +### Linux (Ubuntu 24.04) + +Environment: +- Clang: 18.1.3 (Ubuntu clang version 18.1.3 (1ubuntu1)) +- libstdc++: GCC 13.3.0 (Ubuntu 13.3.0-6ubuntu2~24.04.1) +- Target: x86_64-pc-linux-gnu + +```console ++------------------------+-----+------------+------------+------------+ +| Build | TUs | Frontend | Backend | Total | ++------------------------+-----+------------+------------+------------+ +| Minimal, master | 249 | 468.2 s | 270.3 s | 738.5 s | +| Minimal, with PCH | 253 | 177.1 s | 265.8 s | 442.9 s | +| Full, master | 396 | 811.0 s | 692.2 s | 1,503.2 s | +| Full, with PCH | 405 | 380.0 s | 664.7 s | 1,044.7 s | +| Full, with PCH + UB | 264 | 357.7 s | 635.7 s | 993.4 s | ++------------------------+-----+------------+------------+------------+ + +PCH = precompiled header. +Full = includes building Server, Tools, and Tests. +UB = unity build for models +``` +Note that the number of translation units (TUs) increases when using precompiled +headers — each PCH target adds one extra TU for the precompilation step itself. + +### Mac (Apple M3) + +Environment: +- Clang: Apple clang version 17.0.0 (clang-1700.3.19.1) +- libc++: ships with Apple clang 17.0.0 (Xcode toolchain) +- Target: arm64-apple-macosx15.6 + +```console ++------------------------+-----+------------+------------+------------+ +| Build | TUs | Frontend | Backend | Total | ++------------------------+-----+------------+------------+------------+ +| Minimal, master | 256 | 154.5 s | 94.8 s | 249.3 s | +| Minimal, with PCH | 261 | 65.9 s | 90.0 s | 155.9 s | +| Full, master | 407 | 265.7 s | 209.7 s | 475.4 s | +| Full, with PCH | 414 | 154.6 s | 197.5 s | 352.1 s | +| Full, with PCH + UB | 274 | 143.0 s | 192.2 s | 335.2 s | ++------------------------+-----+------------+------------+------------+ + +PCH = precompiled header. +Full = includes building Server, Tools, and Tests. +UB = unity build for models +``` + +### Windows (ARM64) + +Environment: +- Clang: clang version 22.1.8 (LLVM, `C:\Program Files\LLVM`) +- STL: MSVC STL (Visual Studio 2022 Build Tools 14.44.35207) +- Target: aarch64-pc-windows-msvc + +```console ++------------------------+-----+------------+------------+------------+ +| Build | TUs | Frontend | Backend | Total | ++------------------------+-----+------------+------------+------------+ +| Minimal, master | 249 | 159.4 s | 82.2 s | 241.6 s | +| Full, master | 373 | 337.2 s | 167.4 s | 504.6 s | +| Minimal, with PCH + UB | 113 | 62.3 s | 82.4 s | 144.7 s | +| Full, with PCH + UB | 240 | 233.0 s | 185.1 s | 418.1 s | ++------------------------+-----+------------+------------+------------+ + +PCH = precompiled header. +Full = includes building Server, Tools, and Tests. +UB = unity build for models +``` diff --git a/ggml/src/ggml-cpu/CMakeLists.txt b/ggml/src/ggml-cpu/CMakeLists.txt index 1c7338eea4..83088e1471 100644 --- a/ggml/src/ggml-cpu/CMakeLists.txt +++ b/ggml/src/ggml-cpu/CMakeLists.txt @@ -675,6 +675,12 @@ function(ggml_add_cpu_backend_variant_impl tag_name) target_compile_options(${GGML_CPU_NAME} PRIVATE ${ARCH_FLAGS}) target_compile_definitions(${GGML_CPU_NAME} PRIVATE ${ARCH_DEFINITIONS}) + if (CMAKE_C_COMPILER_ID STREQUAL "GNU" AND NOT GGML_SYSTEM_ARCH STREQUAL "x86") + message(STATUS "Skipping PCH for ${GGML_CPU_NAME}: GCC PCH is only enabled for x86 (arch: ${GGML_SYSTEM_ARCH})") + else() + target_precompile_headers(${GGML_CPU_NAME} PRIVATE ggml-impl.h) + endif() + if (EMSCRIPTEN) set_target_properties(${GGML_CPU_NAME} PROPERTIES COMPILE_FLAGS "-msimd128") endif() diff --git a/ggml/src/ggml-cpu/ops.h b/ggml/src/ggml-cpu/ops.h index 4c1642a676..ce2b3e870b 100644 --- a/ggml/src/ggml-cpu/ops.h +++ b/ggml/src/ggml-cpu/ops.h @@ -18,7 +18,15 @@ #endif #endif +// -Winterference-size was introduced in GCC 12 +#if defined(__cplusplus) && defined(__GNUC__) && !defined(__clang__) && __GNUC__ >= 12 +#pragma GCC diagnostic push +#pragma GCC diagnostic ignored "-Winterference-size" +#endif static const size_t CACHE_LINE_SIZE_F32 = CACHE_LINE_SIZE/sizeof(float); +#if defined(__cplusplus) && defined(__GNUC__) && !defined(__clang__) && __GNUC__ >= 12 +#pragma GCC diagnostic pop +#endif // Work buffer size for im2col operations in CONV2D #define GGML_IM2COL_WORK_SIZE (16 * 1024 * 1024) diff --git a/scripts/build-profile.ps1 b/scripts/build-profile.ps1 new file mode 100644 index 0000000000..410ead39d5 --- /dev/null +++ b/scripts/build-profile.ps1 @@ -0,0 +1,136 @@ +# Compile-time profiling using clang -ftime-trace + ClangBuildAnalyzer. +# +# Usage: +# .\scripts\build-profile.ps1 [-Full] [-Jobs N] +# +# -Full : include Server, Tools, and Tests (default: minimal build) +# -Jobs : number of parallel jobs (default: all cores) +# +# Requires ClangBuildAnalyzer: +# https://github.com/aras-p/ClangBuildAnalyzer + +param( + [switch]$Full, + [int]$Jobs = [Environment]::ProcessorCount +) + +$ErrorActionPreference = "Stop" + +$ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path +$RootDir = Split-Path -Parent $ScriptDir + +if ($Full) { + $BuildDir = Join-Path $RootDir "build-profile-full" + $Report = Join-Path $BuildDir "profile-report-full.txt" +} else { + $BuildDir = Join-Path $RootDir "build-profile-baseline" + $Report = Join-Path $BuildDir "profile-report.txt" +} + +$OutputBin = Join-Path $BuildDir "clang_analysis.bin" + +if (-not (Get-Command clang++ -ErrorAction SilentlyContinue)) { + Write-Error "clang++ not found" + exit 1 +} + +if (-not (Get-Command ninja -ErrorAction SilentlyContinue)) { + Write-Error "ninja not found (required so cmake does not fall back to the Visual Studio/MSVC generator)" + exit 1 +} + +if (-not (Get-Command ClangBuildAnalyzer -ErrorAction SilentlyContinue)) { + Write-Error "ClangBuildAnalyzer not found`n https://github.com/aras-p/ClangBuildAnalyzer/releases" + exit 1 +} + +$ClangVer = (clang++ --version | Select-Object -First 1) +Write-Host "compiler : $ClangVer" +Write-Host "build dir: $BuildDir" +Write-Host "output : $OutputBin" +Write-Host "jobs : $Jobs" +Write-Host "" + +if (Get-Command ccache -ErrorAction SilentlyContinue) { + Write-Host "clearing ccache..." + ccache -C -z +} + +$env:CCACHE_DISABLE = "1" + +$TestsFlag = if ($Full) { "ON" } else { "OFF" } +$ToolsFlag = if ($Full) { "ON" } else { "OFF" } +$ServerFlag = if ($Full) { "ON" } else { "OFF" } + +cmake --fresh ` + -S $RootDir ` + -B $BuildDir ` + -G "Ninja" ` + -DCMAKE_BUILD_TYPE=Release ` + -DCMAKE_C_COMPILER=clang ` + -DCMAKE_CXX_COMPILER=clang++ ` + -DCMAKE_C_FLAGS="-ftime-trace" ` + -DCMAKE_CXX_FLAGS="-ftime-trace" ` + -DGGML_CCACHE=OFF ` + -DGGML_OPENMP=ON ` + -DGGML_NATIVE=OFF ` + "-DLLAMA_BUILD_TESTS=$TestsFlag" ` + -DLLAMA_BUILD_EXAMPLES=OFF ` + "-DLLAMA_BUILD_TOOLS=$ToolsFlag" ` + "-DLLAMA_BUILD_SERVER=$ServerFlag" ` + -DLLAMA_BUILD_APP=OFF + +if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } + +$StrayTrace = Join-Path $RootDir "-.json" +if (Test-Path $StrayTrace) { + Remove-Item $StrayTrace -Force +} + +Write-Host "" +Write-Host "Initializing ClangBuildAnalyzer..." +ClangBuildAnalyzer --start $BuildDir +Write-Host "" + +Write-Host "building..." +Write-Host "" + +$StartTime = Get-Date + +cmake --build $BuildDir --clean-first -j $Jobs + +if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } + +$Elapsed = (Get-Date) - $StartTime + +Write-Host "" +Write-Host ("build time: {0}s ({1}m {2}s)" -f [int]$Elapsed.TotalSeconds, [int]$Elapsed.TotalMinutes, $Elapsed.Seconds) +Write-Host "" + +Write-Host "Aggregating profile metrics..." +ClangBuildAnalyzer --stop $BuildDir $OutputBin | Out-Null + +Write-Host "" +Write-Host ("=" * 80) + +$TUs = "?" +if (Test-Path $Report) { + $Match = Select-String -Path $Report -Pattern "Compilation \((\d+)" | Select-Object -First 1 + if ($Match) { $TUs = $Match.Matches[0].Groups[1].Value } +} + +ClangBuildAnalyzer --analyze $OutputBin | Tee-Object -FilePath $Report + +Write-Host "" +Write-Host "translation units: $TUs" +Write-Host "" +Write-Host "largest trace files (top 20 by size):" + +Get-ChildItem -Path $BuildDir -Recurse -Filter "*.json" | + Where-Object { $_.Name -ne "compile_commands.json" } | + Sort-Object Length -Descending | + Select-Object -First 20 | + ForEach-Object { "{0,8:F1} KB {1}" -f ($_.Length / 1024), $_.FullName } + +Write-Host "" +Write-Host "ClangBuildAnalyzer report was generated: $Report" diff --git a/scripts/build-profile.sh b/scripts/build-profile.sh new file mode 100755 index 0000000000..9429949890 --- /dev/null +++ b/scripts/build-profile.sh @@ -0,0 +1,122 @@ +#!/usr/bin/env bash +# Compile-time profiling using clang -ftime-trace + ClangBuildAnalyzer. +# +# Usage: +# ./scripts/build-profile.sh [--full] [-jN] +# +# --full: include Server, Tools, and Tests (default: minimal build) +# -jN : number of parallel jobs (default: all cores) +# +# Requires ClangBuildAnalyzer: +# macOS: brew install clang-build-analyzer +# Linux: https://github.com/aras-p/ClangBuildAnalyzer.git + +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ROOT_DIR="$(cd "${SCRIPT_DIR}/.." && pwd)" + +FULL=0 +JOBS="-j$(nproc 2>/dev/null || sysctl -n hw.ncpu)" + +for arg in "$@"; do + case "${arg}" in + --full) FULL=1 ;; + -j*) JOBS="${arg}" ;; + *) echo "error: unknown argument: ${arg}" >&2; exit 1 ;; + esac +done + +if [ "${FULL}" -eq 1 ]; then + BUILD_DIR="${ROOT_DIR}/build-profile-full" + REPORT="${BUILD_DIR}/profile-report-full.txt" +else + BUILD_DIR="${ROOT_DIR}/build-profile-baseline" + REPORT="${BUILD_DIR}/profile-report.txt" +fi + +OUTPUT_BIN="${BUILD_DIR}/clang_analysis.bin" + +if ! command -v clang++ &>/dev/null; then + echo "error: clang++ not found" >&2 + exit 1 +fi + +if ! command -v ClangBuildAnalyzer &>/dev/null; then + echo "error: ClangBuildAnalyzer not found" >&2 + echo " brew install clangbuildanalyzer (macOS)" >&2 + echo " or: https://github.com/aras-p/ClangBuildAnalyzer/releases" >&2 + exit 1 +fi + +CLANG_VER=$(clang++ --version | head -1) +echo "compiler : ${CLANG_VER}" +echo "build dir: ${BUILD_DIR}" +echo "output : ${OUTPUT_BIN}" +echo "jobs : ${JOBS}" +echo + +if command -v ccache &>/dev/null; then + echo "clearing ccache..." + ccache -C -z +fi + +export CCACHE_DISABLE=1 + +cmake --fresh \ + -S "${ROOT_DIR}" \ + -B "${BUILD_DIR}" \ + -DCMAKE_BUILD_TYPE=Release \ + -DCMAKE_C_COMPILER=clang \ + -DCMAKE_CXX_COMPILER=clang++ \ + -DCMAKE_C_FLAGS="-ftime-trace" \ + -DCMAKE_CXX_FLAGS="-ftime-trace" \ + -DGGML_CCACHE=OFF \ + -DGGML_OPENMP=ON \ + -DGGML_NATIVE=OFF \ + -DLLAMA_BUILD_TESTS=$([ "${FULL}" -eq 1 ] && echo ON || echo OFF) \ + -DLLAMA_BUILD_EXAMPLES=OFF \ + -DLLAMA_BUILD_TOOLS=$([ "${FULL}" -eq 1 ] && echo ON || echo OFF) \ + -DLLAMA_BUILD_SERVER=$([ "${FULL}" -eq 1 ] && echo ON || echo OFF) \ + -DLLAMA_BUILD_APP=OFF + +echo + +echo "Initializing ClangBuildAnalyzer..." +ClangBuildAnalyzer --start "${BUILD_DIR}" +echo + +echo "building..." +echo + +START=$(date +%s) + +cmake --build "${BUILD_DIR}" --clean-first "${JOBS}" + +END=$(date +%s) +ELAPSED=$((END - START)) + +echo +printf "build time: %ds (%dm %ds)\n" "${ELAPSED}" "$((ELAPSED / 60))" "$((ELAPSED % 60))" +echo + +echo "Aggregating profile metrics..." +ClangBuildAnalyzer --stop "${BUILD_DIR}" "${OUTPUT_BIN}" > /dev/null + +echo +echo "================================================================================" +TUS=$(grep -oP "Compilation \(\K[0-9]+" "${REPORT}" 2>/dev/null || echo "?") +ClangBuildAnalyzer --analyze "${OUTPUT_BIN}" | tee "${REPORT}" + +echo +echo "translation units: ${TUS}" +echo +echo "largest trace files (top 20 by size):" +find "${BUILD_DIR}" -name "*.json" ! -name "compile_commands.json" \ + | xargs ls -l 2>/dev/null \ + | awk 'NF>5 {print $5, $NF}' \ + | sort -rn \ + | awk 'NR<=20 {printf "%8.1f KB %s\n", $1/1024, $2}' + +echo +echo "ClangBuildAnalyzer report was generated: ${REPORT}" diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index 221e14f7ff..bc922b6a7b 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -8,40 +8,44 @@ llama_add_compile_flags() file(GLOB LLAMA_MODELS_SOURCES "models/*.cpp") +set(LLAMA_CORE_SOURCES + llama.cpp + llama-adapter.cpp + llama-arch.cpp + llama-batch.cpp + llama-chat.cpp + llama-context.cpp + llama-cparams.cpp + llama-grammar.cpp + llama-graph.cpp + llama-hparams.cpp + llama-impl.cpp + llama-io.cpp + llama-kv-cache.cpp + llama-kv-cache-iswa.cpp + llama-kv-cache-dsa.cpp + llama-kv-cache-dsa-iswa.cpp + llama-kv-cache-msa.cpp + llama-kv-cache-dsv4.cpp + llama-memory.cpp + llama-memory-hybrid.cpp + llama-memory-hybrid-iswa.cpp + llama-memory-hybrid-idx.cpp + llama-memory-recurrent.cpp + llama-mmap.cpp + llama-model-loader.cpp + llama-model-saver.cpp + llama-model.cpp + llama-quant.cpp + llama-sampler.cpp + llama-vocab.cpp + unicode-data.cpp + unicode.cpp +) + add_library(llama ../include/llama.h - llama.cpp - llama-adapter.cpp - llama-arch.cpp - llama-batch.cpp - llama-chat.cpp - llama-context.cpp - llama-cparams.cpp - llama-grammar.cpp - llama-graph.cpp - llama-hparams.cpp - llama-impl.cpp - llama-io.cpp - llama-kv-cache.cpp - llama-kv-cache-iswa.cpp - llama-kv-cache-dsa.cpp - llama-kv-cache-dsa-iswa.cpp - llama-kv-cache-msa.cpp - llama-kv-cache-dsv4.cpp - llama-memory.cpp - llama-memory-hybrid.cpp - llama-memory-hybrid-iswa.cpp - llama-memory-hybrid-idx.cpp - llama-memory-recurrent.cpp - llama-mmap.cpp - llama-model-loader.cpp - llama-model-saver.cpp - llama-model.cpp - llama-quant.cpp - llama-sampler.cpp - llama-vocab.cpp - unicode-data.cpp - unicode.cpp + ${LLAMA_CORE_SOURCES} unicode.h ${LLAMA_MODELS_SOURCES} ) @@ -50,13 +54,20 @@ set_target_properties(llama PROPERTIES VERSION ${LLAMA_VERSION_BASE} SOVERSION ${LLAMA_VERSION_MAJOR} MACHO_CURRENT_VERSION 0 # keep macOS linker from seeing oversized version number + UNITY_BUILD ON + UNITY_BUILD_BATCH_SIZE 16 ) +# exclude non-model sources from unity build +set_source_files_properties(${LLAMA_CORE_SOURCES} ../include/llama.h unicode.h + PROPERTIES SKIP_UNITY_BUILD_INCLUSION ON) + configure_file(llama-version.h.in ${CMAKE_CURRENT_BINARY_DIR}/llama-version.h @ONLY) target_include_directories(llama PRIVATE . ${CMAKE_CURRENT_BINARY_DIR}) target_include_directories(llama PUBLIC ../include) target_compile_features (llama PRIVATE cxx_std_17) # don't bump +target_precompile_headers (llama PRIVATE models/models.h) target_link_libraries(llama PUBLIC ggml) diff --git a/src/models/gemma3n.cpp b/src/models/gemma3n.cpp index ea616db3ba..bb628203aa 100644 --- a/src/models/gemma3n.cpp +++ b/src/models/gemma3n.cpp @@ -82,7 +82,7 @@ std::unique_ptr llama_model_gemma3n::build_arch_graph(const l } // get 2D slice view from a 3D tensor, the idx corresponds to the 3rd dim -static ggml_tensor * ggml_view_2d_slice(ggml_context * ctx0, ggml_tensor * x, int idx) { +static ggml_tensor * gemma3n_view_2d_slice(ggml_context * ctx0, ggml_tensor * x, int idx) { GGML_ASSERT(idx < (int) x->ne[2]); return ggml_view_2d(ctx0, x, x->ne[0], x->ne[1], ggml_row_size(x->type, x->ne[0]), idx * x->ne[0] * x->ne[1] * ggml_element_size(x)); @@ -139,7 +139,7 @@ llama_model_gemma3n::graph::graph(const llama_model & model, const llm_graph_par ggml_tensor * predictions = altup_predict(cur, il); // [n_embd, n_tokens, n_altup] // predicted value will go through self-attention and laurel - ggml_tensor * active_prediction = ggml_view_2d_slice(ctx0, predictions, i_altup_act); // [n_embd, n_tokens] + ggml_tensor * active_prediction = gemma3n_view_2d_slice(ctx0, predictions, i_altup_act); // [n_embd, n_tokens] cur = active_prediction; cb(cur, "active_prediction", il); @@ -236,13 +236,13 @@ llama_model_gemma3n::graph::graph(const llama_model & model, const llm_graph_par ggml_tensor * first_prediction; // [n_embd, n_tokens] { - first_prediction = ggml_view_2d_slice(ctx0, corrected, i_altup_act); // [n_embd, n_tokens] + first_prediction = gemma3n_view_2d_slice(ctx0, corrected, i_altup_act); // [n_embd, n_tokens] first_prediction = ggml_mul(ctx0, first_prediction, model.layers[il].altup_correct_scale); first_prediction = build_lora_mm(model.layers[il].per_layer_inp_gate, first_prediction); first_prediction = ggml_gelu(ctx0, first_prediction); // [n_embd_altup, n_tokens] cb(first_prediction, "first_prediction_gated", il); - ggml_tensor * inp_this_layer = ggml_view_2d_slice(ctx0, inp_per_layer, il); // [n_embd_altup, n_tokens] + ggml_tensor * inp_this_layer = gemma3n_view_2d_slice(ctx0, inp_per_layer, il); // [n_embd_altup, n_tokens] first_prediction = ggml_mul(ctx0, first_prediction, inp_this_layer); // [n_embd_altup, n_tokens] cb(first_prediction, "first_prediction_scaled", il); @@ -253,7 +253,7 @@ llama_model_gemma3n::graph::graph(const llama_model & model, const llm_graph_par } // equivalent to python code: corrected_predictions[1:] += first_prediction { - ggml_tensor * slice_first = ggml_view_2d_slice(ctx0, corrected, 0); + ggml_tensor * slice_first = gemma3n_view_2d_slice(ctx0, corrected, 0); ggml_tensor * slice_rest = ggml_view_3d( ctx0, corrected, n_embd, n_tokens, n_altup - 1, ggml_row_size(corrected->type, n_embd), ggml_row_size(corrected->type, n_embd * n_tokens), n_embd * n_tokens * ggml_element_size(corrected)); @@ -271,7 +271,7 @@ llama_model_gemma3n::graph::graph(const llama_model & model, const llm_graph_par // cur now has multiple altup(s), we want to merge them back to 1 altup { - ggml_tensor * target_magnitude = calc_magnitude(ggml_view_2d_slice(ctx0, cur, i_altup_act)); // [n_embd, n_tokens] + ggml_tensor * target_magnitude = calc_magnitude(gemma3n_view_2d_slice(ctx0, cur, i_altup_act)); // [n_embd, n_tokens] // do a view to skip the first slice (active altup) ggml_tensor * alt_slice = ggml_view_3d(ctx0, cur, n_embd, n_tokens, n_altup - 1, ggml_row_size(cur->type, n_embd), @@ -283,9 +283,9 @@ llama_model_gemma3n::graph::graph(const llama_model & model, const llm_graph_par cb(altup_unembd, "altup_unembd", -1); // equivalent to torch.mean(hidden_states, dim=0) - cur = ggml_view_2d_slice(ctx0, cur, 0); // [n_embd, n_tokens] + cur = gemma3n_view_2d_slice(ctx0, cur, 0); // [n_embd, n_tokens] for (int i = 0; i < n_altup - 1; ++i) { - cur = ggml_add(ctx0, cur, ggml_view_2d_slice(ctx0, altup_unembd, i)); + cur = ggml_add(ctx0, cur, gemma3n_view_2d_slice(ctx0, altup_unembd, i)); } cur = ggml_scale(ctx0, cur, 1.0f / float(n_altup)); // [n_embd, n_tokens] cb(cur, "unembd_merged", -1); @@ -419,7 +419,7 @@ ggml_tensor * llama_model_gemma3n::graph::altup_compute_router_modalities(ggml_t // input cur shape: [n_embd, n_tokens, n_altup] // output shape: [n_embd, n_tokens, n_altup] ggml_tensor * llama_model_gemma3n::graph::altup_predict(ggml_tensor * cur, int il) { - ggml_tensor * activated = ggml_view_2d_slice(ctx0, cur, i_altup_act); // [n_embd, n_tokens] + ggml_tensor * activated = gemma3n_view_2d_slice(ctx0, cur, i_altup_act); // [n_embd, n_tokens] ggml_tensor * modalities = altup_compute_router_modalities(activated, il); // [n_altup, n_tokens] cb(modalities, "modalities", il); @@ -447,7 +447,7 @@ ggml_tensor * llama_model_gemma3n::graph::altup_correct(ggml_tensor * prediction ggml_tensor * modalities = altup_compute_router_modalities(activated, il); // [n_altup, n_tokens] cb(modalities, "modalities", il); - ggml_tensor * active_prediction = ggml_view_2d_slice(ctx0, predictions, i_altup_act); + ggml_tensor * active_prediction = gemma3n_view_2d_slice(ctx0, predictions, i_altup_act); ggml_tensor * innovation = ggml_sub(ctx0, activated, active_prediction); // [n_embd, n_tokens] cb(innovation, "innovation", il); diff --git a/src/models/gemma4.cpp b/src/models/gemma4.cpp index 388126e26a..39e899aa6e 100644 --- a/src/models/gemma4.cpp +++ b/src/models/gemma4.cpp @@ -145,7 +145,7 @@ std::unique_ptr llama_model_gemma4::build_arch_graph(const ll } // get 2D slice view from a 3D tensor, the idx corresponds to the 3rd dim -static ggml_tensor * ggml_view_2d_slice(ggml_context * ctx0, ggml_tensor * x, int idx) { +static ggml_tensor * gemma4_view_2d_slice(ggml_context * ctx0, ggml_tensor * x, int idx) { GGML_ASSERT(idx < (int) x->ne[2]); return ggml_view_2d(ctx0, x, x->ne[0], x->ne[1], ggml_row_size(x->type, x->ne[0]), idx * x->ne[0] * x->ne[1] * ggml_element_size(x)); @@ -372,7 +372,7 @@ llama_model_gemma4::graph::graph(const llama_model & model, const llm_graph_para cur = build_lora_mm(model.layers[il].per_layer_inp_gate, cur); // [n_embd_per_layer, n_tokens] cur = ggml_gelu(ctx0, cur); - ggml_tensor * inp_this_layer = ggml_view_2d_slice(ctx0, inp_per_layer, il); // [n_embd_per_layer, n_tokens] + ggml_tensor * inp_this_layer = gemma4_view_2d_slice(ctx0, inp_per_layer, il); // [n_embd_per_layer, n_tokens] // TODO @ngxson : improve this if (il == n_layer - 1 && inp_out_ids && cparams.embeddings_nextn_masked) { diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 920c58c738..0c4e4d5a9b 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -278,6 +278,8 @@ llama_build_and_test( peg-parser/test-unicode.cpp peg-parser/tests.h ) +target_precompile_headers(test-peg-parser PRIVATE peg-parser/tests.h) + if (NOT ${CMAKE_SYSTEM_PROCESSOR} MATCHES "s390x") set(MODEL_NAME "tinyllamas/stories15M-q4_0.gguf") diff --git a/tools/mtmd/CMakeLists.txt b/tools/mtmd/CMakeLists.txt index 907468e87e..176eb15057 100644 --- a/tools/mtmd/CMakeLists.txt +++ b/tools/mtmd/CMakeLists.txt @@ -84,6 +84,13 @@ target_link_libraries (mtmd PUBLIC ggml llama) target_link_libraries (mtmd PRIVATE Threads::Threads vendor::hash vendor::miniaudio vendor::stb vendor::sheredom) target_include_directories(mtmd PUBLIC .) target_compile_features (mtmd PRIVATE cxx_std_17) +target_precompile_headers (mtmd PRIVATE models/models.h) + +set_source_files_properties( + mtmd-helper.cpp + mtmd-helper-gen.cpp + PROPERTIES SKIP_PRECOMPILE_HEADERS ON +) if (MTMD_VIDEO) target_compile_definitions(mtmd PRIVATE MTMD_VIDEO) diff --git a/tools/server/CMakeLists.txt b/tools/server/CMakeLists.txt index 43c2456333..f02a2ba3b1 100644 --- a/tools/server/CMakeLists.txt +++ b/tools/server/CMakeLists.txt @@ -32,6 +32,7 @@ endif() target_include_directories(${TARGET} PRIVATE ../mtmd) target_include_directories(${TARGET} PRIVATE ${CMAKE_SOURCE_DIR}) target_link_libraries(${TARGET} PUBLIC llama-common mtmd ${CMAKE_THREAD_LIBS_INIT}) +target_precompile_headers(${TARGET} PRIVATE ${CMAKE_SOURCE_DIR}/common/common.h) # llama-server-impl: server logic, reusable by app @@ -49,6 +50,7 @@ set_target_properties(${TARGET} PROPERTIES WINDOWS_EXPORT_ALL_SYMBOLS ON) target_include_directories(${TARGET} PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}) target_include_directories(${TARGET} PRIVATE ../mtmd ${CMAKE_SOURCE_DIR}) target_link_libraries(${TARGET} PUBLIC server-context llama-ui cpp-httplib ${CMAKE_THREAD_LIBS_INIT}) +target_precompile_headers(${TARGET} PRIVATE ${CMAKE_SOURCE_DIR}/common/common.h) add_dependencies(${TARGET} llama-ui-assets)