Commit abeada335 for llama.cpp

commit abeada335e2e78bd3fe63febafab7e900ce75810
Author: Toki Nasin <141258697+tokinasin@users.noreply.github.com>
Date:   Wed Oct 7 02:07:54 2026 +0900

    vocab : implement PLaMo-3 tokenizer pre-segmentation (#30045)

    * vocab : implement PLaMo-3 tokenizer pre-segmentation

    The PLaMo-3 tokenizer inserts hard boundaries before running the Unigram
    DP, around <|plamo:...|>-looking text, and around runs of at least 4
    identical characters or 2 spaces. Without them llama.cpp tokenizes code
    indentation and repeated punctuation differently from the reference.

    Reproduce the two re.sub() passes in llm_tokenizer_plamo2 by encoding
    each segment independently.

    * add vocab type "plamo3"

    * Update src/llama-vocab.cpp

    Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>

    * misc change

    ---------

    Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>

diff --git a/conversion/base.py b/conversion/base.py
index e28fad479..3b6fe04c2 100644
--- a/conversion/base.py
+++ b/conversion/base.py
@@ -2511,6 +2511,14 @@ class TextModel(ModelBase):
         with open(tokenizer_config_path, "r", encoding="utf-8") as f:
             tokenizer_config = json.load(f)

+        tokenizer_class = tokenizer_config.get("tokenizer_class")
+        if tokenizer_class == "Plamo2Tokenizer":
+            tokenizer_model = "plamo2"
+        elif tokenizer_class == "Plamo3Tokenizer":
+            tokenizer_model = "plamo3"
+        else:
+            raise ValueError(f"Unsupported PLaMo tokenizer class: {tokenizer_class}")
+
         # Load tokens from JSONL file (actually a list format)
         tokens = []
         scores = []
@@ -2550,7 +2558,7 @@ class TextModel(ModelBase):
                 scores.append(-1000.0)
                 toktypes.append(gguf.TokenType.UNUSED)

-        self.gguf_writer.add_tokenizer_model("plamo2")
+        self.gguf_writer.add_tokenizer_model(tokenizer_model)
         self.gguf_writer.add_tokenizer_pre("default")
         self.gguf_writer.add_token_list(tokens)
         self.gguf_writer.add_token_scores(scores)
diff --git a/include/llama.h b/include/llama.h
index 9bca8ee10..260247e82 100644
--- a/include/llama.h
+++ b/include/llama.h
@@ -78,6 +78,7 @@ extern "C" {
         LLAMA_VOCAB_TYPE_RWKV   = 5, // RWKV tokenizer based on greedy tokenization
         LLAMA_VOCAB_TYPE_PLAMO2 = 6, // PLaMo-2 tokenizer based on Aho-Corasick with dynamic programming
         LLAMA_VOCAB_TYPE_TEST   = 7, // Dummy tokenizer for testing: rolling hash of fixed-size chunks -> tokens, tokens -> hex
+        LLAMA_VOCAB_TYPE_PLAMO3 = 8, // PLaMo-3 tokenizer with pre-segmentation and dynamic programming
     };

     enum llama_rope_type {
diff --git a/src/llama-vocab.cpp b/src/llama-vocab.cpp
index dddb6bab8..de20c757f 100644
--- a/src/llama-vocab.cpp
+++ b/src/llama-vocab.cpp
@@ -1400,7 +1400,7 @@ private:
 };

 struct llm_tokenizer_plamo2 : llm_tokenizer {
-    llm_tokenizer_plamo2(const llama_vocab & vocab) {
+    llm_tokenizer_plamo2(const llama_vocab & vocab, bool pre_segment) : pre_segment_(pre_segment) {
         build(vocab);
     }

@@ -1543,11 +1543,87 @@ struct llm_tokenizer_plamo2 : llm_tokenizer {

     std::vector<llama_token> encode(const std::string & text) const {
         std::vector<uint32_t> unicode_data = unicode_cpts_from_utf8(text);
-        // Skip the first code point if it is a BOM (Byte Order Mark)
-        if (!unicode_data.empty() && unicode_data[0] == 0xFEFF) {
-            unicode_data.erase(unicode_data.begin());
+        // The PLaMo-3 tokenizer keeps a leading U+FEFF in the input.
+        if (!pre_segment_) {
+            if (!unicode_data.empty() && unicode_data[0] == 0xFEFF) {
+                unicode_data.erase(unicode_data.begin());
+            }
+            return encode_cpts(unicode_data);
+        }
+
+        const size_t      n = unicode_data.size();
+        std::vector<bool> cut(n + 1, false);
+
+        // pass 1: <|plamo:...|>
+        {
+            static constexpr uint32_t prefix[]   = { '<', '|', 'p', 'l', 'a', 'm', 'o', ':' };
+            const size_t              prefix_len = std::size(prefix);
+            size_t                    i          = 0;
+            while (i + prefix_len <= n) {
+                if (!std::equal(prefix, prefix + prefix_len, unicode_data.begin() + i)) {
+                    i++;
+                    continue;
+                }
+                // An empty body is valid.
+                size_t j = i + prefix_len;
+                // Treat U+001C..U+001F as whitespace (equivalent to Python \s).
+                while (j < n && j - (i + prefix_len) < 64 && unicode_data[j] != '|' &&
+                       (unicode_data[j] < 0x1C || unicode_data[j] > 0x1F) &&
+                       !unicode_cpt_flags_from_cpt(unicode_data[j]).is_whitespace) {
+                    j++;
+                }
+                if (j + 1 < n && unicode_data[j] == '|' && unicode_data[j + 1] == '>') {
+                    cut[i]     = true;
+                    cut[j + 2] = true;
+                    i          = j + 2;
+                } else {
+                    i++;
+                }
+            }
+        }
+
+        // pass 2: runs of repeated characters / spaces (a run never crosses a boundary from pass 1)
+        {
+            size_t i = 0;
+            while (i < n) {
+                const uint32_t c   = unicode_data[i];
+                size_t         run = 1;
+                while (i + run < n && unicode_data[i + run] == c && !cut[i + run]) {
+                    run++;
+                }
+
+                const bool is_repeated_chars = c != '\n' && run >= 4;
+                const bool is_spaces         = c == ' ' && run >= 2;
+                if (is_repeated_chars || is_spaces) {
+                    cut[i]       = true;
+                    cut[i + run] = true;
+                }
+
+                // a run that does not match cannot match at any later position either (it only gets shorter)
+                i += run;
+            }
+        }
+
+        std::vector<llama_token> output;
+        size_t                   seg_start = 0;
+        for (size_t seg_end = 0; seg_end <= n; ++seg_end) {
+            // U+EE00 is the tokenizer's private-use boundary marker; literal occurrences split segments and are not emitted.
+            const bool is_boundary = seg_end < n && unicode_data[seg_end] == 0xEE00;
+            if (seg_end == n || cut[seg_end] || is_boundary) {
+                if (seg_start < seg_end) {
+                    const std::vector<uint32_t>    segment(unicode_data.begin() + seg_start,
+                                                           unicode_data.begin() + seg_end);
+                    const std::vector<llama_token> tokens = encode_cpts(segment);
+                    output.insert(output.end(), tokens.begin(), tokens.end());
+                }
+                seg_start = seg_end + is_boundary;
+            }
         }

+        return output;
+    }
+
+    std::vector<llama_token> encode_cpts(const std::vector<uint32_t> & unicode_data) const {
         if (unicode_data.empty()) {
             return {};
         }
@@ -1666,6 +1742,8 @@ private:
     // Flattened table representing the Trie structure
     // Each row contains: [piece_length, token_id, score, piece_id]
     std::vector<std::vector<int32_t>> table_;
+
+    const bool pre_segment_;
 };

 struct llm_tokenizer_plamo2_session {
@@ -2132,10 +2210,10 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
             special_sep_id  = LLAMA_TOKEN_NULL;
             special_pad_id  = LLAMA_TOKEN_NULL;
             special_mask_id = LLAMA_TOKEN_NULL;
-        } else if (tokenizer_model == "plamo2") {
-            type = LLAMA_VOCAB_TYPE_PLAMO2;
+        } else if (tokenizer_model == "plamo2" || tokenizer_model == "plamo3") {
+            type = tokenizer_model == "plamo2" ? LLAMA_VOCAB_TYPE_PLAMO2 : LLAMA_VOCAB_TYPE_PLAMO3;

-            // PLaMo-2 default special tokens (these will be overridden by model config)
+            // PLaMo default special tokens (these will be overridden by model config)
             special_bos_id = 1;  // <|plamo:bos|>
             special_eos_id = 2;  // <|plamo:eos|>
             special_unk_id = 0;  // <|plamo:unk|>
@@ -3193,6 +3271,7 @@ std::string llama_vocab::impl::type_name() const{
         case LLAMA_VOCAB_TYPE_UGM:    return "UGM";
         case LLAMA_VOCAB_TYPE_RWKV:   return "RWKV";
         case LLAMA_VOCAB_TYPE_PLAMO2: return "PLaMo2";
+        case LLAMA_VOCAB_TYPE_PLAMO3: return "PLaMo3";
         case LLAMA_VOCAB_TYPE_TEST:   return "TEST";
         default:                      return "unknown";
     }
@@ -3280,7 +3359,10 @@ void llama_vocab::impl::init_tokenizer(enum llama_vocab_type type) {
             tokenizer = std::make_unique<llm_tokenizer_rwkv>(vocab);
             break;
         case LLAMA_VOCAB_TYPE_PLAMO2:
-            tokenizer = std::make_unique<llm_tokenizer_plamo2>(vocab);
+            tokenizer = std::make_unique<llm_tokenizer_plamo2>(vocab, false);
+            break;
+        case LLAMA_VOCAB_TYPE_PLAMO3:
+            tokenizer = std::make_unique<llm_tokenizer_plamo2>(vocab, true);
             break;
         case LLAMA_VOCAB_TYPE_TEST:
             tokenizer = std::make_unique<llm_tokenizer>();
@@ -3642,6 +3724,7 @@ std::vector<llama_token> llama_vocab::impl::tokenize(
                 }
             } break;
         case LLAMA_VOCAB_TYPE_PLAMO2:
+        case LLAMA_VOCAB_TYPE_PLAMO3:
             {
                 if (add_special && add_bos) {
                     GGML_ASSERT(special_bos_id != LLAMA_TOKEN_NULL);
@@ -3813,8 +3896,9 @@ int32_t llama_vocab::impl::token_to_piece(llama_token token, char * buf, int32_t
                 std::string result = format("%x", token);
                 return _try_copy(result.data(), result.size());
             }
-            case LLAMA_VOCAB_TYPE_PLAMO2: {
-                // PLaMo-2 uses similar token handling as BPE/SPM
+            case LLAMA_VOCAB_TYPE_PLAMO2:
+            case LLAMA_VOCAB_TYPE_PLAMO3: {
+                // PLaMo uses similar token handling as BPE/SPM
                 if (vocab.is_byte(token)) {
                     // Handle byte tokens like <0xXX>
                     if (token_text.length() == 6 && token_text.substr(0, 3) == "<0x" && token_text.back() == '>') {
@@ -4077,8 +4161,9 @@ llama_token llama_vocab::byte_to_token(uint8_t ch) const {
         case LLAMA_VOCAB_TYPE_BPE: {
             return pimpl->token_to_id.at(unicode_byte_to_utf8(ch));
         }
-        case LLAMA_VOCAB_TYPE_PLAMO2: {
-            // PLaMo-2 uses byte tokens in format <0xXX>
+        case LLAMA_VOCAB_TYPE_PLAMO2:
+        case LLAMA_VOCAB_TYPE_PLAMO3: {
+            // PLaMo uses byte tokens in format <0xXX>
             char hex_str[8];
             snprintf(hex_str, sizeof(hex_str), "<0x%02X>", ch);
             return pimpl->token_to_id.at(hex_str);