mirror of
https://github.com/LostRuins/koboldcpp.git
synced 2026-09-19 09:15:18 +02:00
Merge commit 'ae251b5ff2634108822e0f8bb20ca4cd5c2c5dcc' into concedo_experimental
# Conflicts: # .github/actions/linux-setup-spacemit/action.yml # .github/actions/unarchive-tar/action.yml # .github/workflows/build-android.yml # .github/workflows/build-cmake-pkg.yml # .github/workflows/build-cross.yml # .github/workflows/build-self-hosted.yml # .github/workflows/build.yml # .github/workflows/check-vendor.yml # .github/workflows/code-style.yml # .github/workflows/editorconfig.yml # .github/workflows/pre-tokenizer-hashes.yml # .github/workflows/python-check-requirements.yml # .github/workflows/python-lint.yml # .github/workflows/python-type-check.yml # .github/workflows/server-self-hosted.yml # .github/workflows/ui-build.yml # .github/workflows/ui.yml # .github/workflows/update-ops-docs.yml # ci/run.sh # docs/build-riscv64-spacemit.md # examples/convert_legacy_llama.py # ggml/cmake/ggml-config.cmake.in # ggml/src/CMakeLists.txt # ggml/src/ggml-cpu/CMakeLists.txt # ggml/src/ggml-opencl/ggml-opencl.cpp # scripts/sync_vendor.py # tests/test-chat-auto-parser.cpp # tests/test-chat.cpp # tests/test-gguf.cpp # tools/cli/README.md # tools/perplexity/perplexity.cpp # tools/server/README.md
This commit is contained in:
+2
-1
@@ -877,7 +877,8 @@ extern "C" {
|
||||
// work only with partial states, such as SWA KV cache or recurrent cache (e.g. Mamba)
|
||||
#define LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY 1
|
||||
|
||||
// keeps the tensor data on device buffers (i.e. not accessible in host memory, but faster save/load)
|
||||
// Keeps the tensor data on device buffers (i.e. not accessible in host memory, but faster save/load).
|
||||
// Getting the state for a seq_id with this flag invalidates all prior states gotten for that seq_id with this flag.
|
||||
#define LLAMA_STATE_SEQ_FLAGS_ON_DEVICE 2
|
||||
|
||||
typedef uint32_t llama_state_seq_flags;
|
||||
|
||||
Reference in New Issue
Block a user