mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-08-27 15:41:19 +02:00
f280b26983
* metal : per-device tuned (Q, NE) for flash-attn vec (#25750)
* rebase Q-generic FA vec body from 01dc93607 (#23114)
* add 53 f16 (Q,NE) flash-attn vec instantiations (vec 80 -> 133)
* add FA vec (Q,NE) tuning table + dispatch wiring + SMEM cap fallback
* add FA vec (Q,NE) perf sweep
* fill tuning result
* fold family table into a per-family representative SKU
* refactor tuning result format
* extend FA vec tuning to quantized KV caches
* sync fa vec tuner bucketing with runtime, use pointwise tuning regret
* update tuned table
* format and cleanup
* prefix fa_vec tuning procs with ggml_backend_metal_tuning_, drop unused fa_vec_override_active
* add device id -> token lookup for the offline tuning tool
* add ggml-metal-tuning skeleton
* add op-agnostic perf cell + median timing for the tuner
* add FA-vec graph build + tensor init to the tuner
* tools : add FA-vec (Q,NE) sweep, compression and table emit
* cool down and re-measure the dirty window on thermal drift
* test-backend-ops : replace the FA vec tune mode with a bounded (Q,NE) slice
* tools : document the Metal tuner, point the table comment at it
* abort on unknown KV type, single-source fa_vec_legal_ne
* cleanup
* honor -o in the FA vec (Q,NE) slice
* retune FA-vec (Q, NE) under a pointwise no-harm gate
* cont : add fa-vec tunings for M1 Pro, M2 Ultra, M5 Max
---------
Co-authored-by: Georgi Gerganov <ggerganov@gmail.com>
46 lines
1015 B
CMake
46 lines
1015 B
CMake
# dependencies
|
|
|
|
find_package(Threads REQUIRED)
|
|
|
|
# third-party
|
|
|
|
# ...
|
|
|
|
# flags
|
|
|
|
llama_add_compile_flags()
|
|
|
|
# tools
|
|
|
|
if (EMSCRIPTEN)
|
|
else()
|
|
add_subdirectory(batched-bench)
|
|
add_subdirectory(gguf-split)
|
|
add_subdirectory(imatrix)
|
|
add_subdirectory(llama-bench)
|
|
add_subdirectory(completion)
|
|
add_subdirectory(perplexity)
|
|
add_subdirectory(quantize)
|
|
if (LLAMA_BUILD_SERVER)
|
|
add_subdirectory(ui)
|
|
add_subdirectory(cli)
|
|
add_subdirectory(server)
|
|
endif()
|
|
add_subdirectory(tokenize)
|
|
add_subdirectory(tts)
|
|
add_subdirectory(mtmd)
|
|
if (GGML_RPC)
|
|
add_subdirectory(rpc)
|
|
endif()
|
|
if (NOT GGML_BACKEND_DL AND GGML_CPU)
|
|
# these tools use backends directly (no dynamic loading) and depend on CPU backend symbols
|
|
add_subdirectory(cvector-generator)
|
|
add_subdirectory(export-lora)
|
|
endif()
|
|
add_subdirectory(fit-params)
|
|
if (GGML_METAL)
|
|
add_subdirectory(tuning)
|
|
endif()
|
|
add_subdirectory(results)
|
|
endif()
|