mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-27 21:46:57 +02:00
vocab : add ufakzeka pre-tokenizer (#29033)
* vocab : add ufakzeka pre-tokenizer * vocab : move ufakzeka to the models list and regenerate the hash mapping
This commit is contained in:
@@ -488,6 +488,12 @@ struct llm_tokenizer_bpe : llm_tokenizer {
|
||||
"(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])|[^\\r\\n\\p{L}\\p{N}]?\\p{L}+|\\p{N}{1}| ?[^\\s\\p{L}\\p{N}\\r\\n]+|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
||||
};
|
||||
break;
|
||||
case LLAMA_VOCAB_PRE_TYPE_UFAKZEKA:
|
||||
regex_exprs = {
|
||||
// Qwen2 pattern without the English contraction group, so Turkish apostrophe suffixes stay attached
|
||||
"[^\\r\\n\\p{L}\\p{N}]?\\p{L}+|\\p{N}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
||||
};
|
||||
break;
|
||||
case LLAMA_VOCAB_PRE_TYPE_GROK_2:
|
||||
regex_exprs = {
|
||||
// original regex from tokenizer.json
|
||||
@@ -2376,6 +2382,10 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
||||
tokenizer_pre == "kimi-k2") {
|
||||
pre_type = LLAMA_VOCAB_PRE_TYPE_KIMI_K2;
|
||||
clean_spaces = false;
|
||||
} else if (
|
||||
tokenizer_pre == "ufakzeka") {
|
||||
pre_type = LLAMA_VOCAB_PRE_TYPE_UFAKZEKA;
|
||||
clean_spaces = false;
|
||||
} else if (
|
||||
tokenizer_pre == "grok-2") {
|
||||
pre_type = LLAMA_VOCAB_PRE_TYPE_GROK_2;
|
||||
|
||||
@@ -67,6 +67,7 @@ enum llama_vocab_pre_type {
|
||||
LLAMA_VOCAB_PRE_TYPE_LAGUNA = 56,
|
||||
LLAMA_VOCAB_PRE_TYPE_HY_V4 = 57,
|
||||
LLAMA_VOCAB_PRE_TYPE_SPARK2_5 = 58,
|
||||
LLAMA_VOCAB_PRE_TYPE_UFAKZEKA = 59,
|
||||
};
|
||||
|
||||
struct LLM_KV;
|
||||
|
||||
Reference in New Issue
Block a user