unicode : include '~' in collapsed symbol class (#26972)

The collapsed \p{S} class was missing '~', which split " ~" into
separate pre-tokens and prevented the Ġ~ BPE merge used by DeepSeek V4.
This caused re-tokenized prompts to diverge from sampled tokens and
broke KV cache reuse.

Assisted-by: Codex
This commit is contained in:
Thiago Padilha
2026-08-18 10:15:22 -03:00
committed by GitHub
parent 169e4a7ff2
commit afd439df1f
3 changed files with 27 additions and 1 deletions
+1 -1
View File
@@ -1241,7 +1241,7 @@ std::vector<std::string> unicode_regex_split(const std::string & text, const std
{ unicode_cpt_flags::LETTER, "\x41-\x5A\x61-\x7A" }, // A-Za-z
{ unicode_cpt_flags::PUNCTUATION, "\x21-\x23\x25-\x2A\x2C-\x2F\x3A-\x3B\x3F-\x40\\\x5B-\\\x5D\x5F\\\x7B\\\x7D" }, // !-#%-*,-/:-;?-@\[-\]_\{\}
{ unicode_cpt_flags::ACCENT_MARK, "" }, // no sub-128 codepoints
{ unicode_cpt_flags::SYMBOL, "\\\x24\\\x2B\x3C-\x3E\x5E\x60\\\x7C" }, // $+<=>^`|
{ unicode_cpt_flags::SYMBOL, "\\\x24\\\x2B\x3C-\x3E\x5E\x60\\\x7C\\\x7E" }, // $+<=>^`|~
};
// compute collapsed codepoints only if needed by at least one regex
+2
View File
@@ -116,6 +116,8 @@ function(llama_build_and_test source)
set_property(TEST ${TEST_TARGET} PROPERTY LABELS ${LLAMA_TEST_LABEL})
endfunction()
llama_build_and_test(test-unicode.cpp)
# build test-tokenizer-0 target once and add many tests
llama_build(test-tokenizer-0.cpp)
+24
View File
@@ -0,0 +1,24 @@
#include "../src/unicode.h"
#include <cstdio>
#include <string>
#include <vector>
int main() {
const std::vector<std::string> regex_exprs = {
"[~][A-Za-z]+| ?[\\p{S}]+|\\s+",
};
const std::vector<std::string> expected = { " ~", "foo" };
const auto actual = unicode_regex_split(" ~foo", regex_exprs, false);
if (actual != expected) {
fprintf(stderr, "unexpected split:");
for (const auto & piece : actual) {
fprintf(stderr, " [%s]", piece.c_str());
}
fprintf(stderr, "\n");
return 1;
}
return 0;
}