From 2093fa90e8bae9e37fd6e89b81b02f61ec7cd97b Mon Sep 17 00:00:00 2001 From: bong-water-water-bong Date: Sun, 27 Sep 2026 11:25:34 -0300 Subject: [PATCH] convert: ZAYA special tokens are CONTROL, so chat templates tokenize LlamaHfVocab left <|im_start|> (105) and (1) NORMAL although tokenizer.json marks them special. llama.cpp only matches special-token text for CONTROL and USER_DEFINED tokens, so every chat turn's <|im_start|> reached the model as seven text tokens and ZAYA1-8B answered off-template (12 + 30 -> 22). Tokens tokenizer.json lists as special are now CONTROL. Converting ZAYA1-8B now gives the published GGUF's 1283 tensors unchanged and token types with 1 and 105 CONTROL. --- conversion/zaya.py | 22 ++++++++++++++++++++-- 1 file changed, 20 insertions(+), 2 deletions(-) diff --git a/conversion/zaya.py b/conversion/zaya.py index d0af16da1dc5..39b493bef09a 100644 --- a/conversion/zaya.py +++ b/conversion/zaya.py @@ -31,6 +31,24 @@ from .qwenvl import Qwen2VLVisionModel + +def _special_tokens_as_control(tokenizer_dir, toktypes): + """Mark every token tokenizer.json lists as special: true as CONTROL. + + LlamaHfVocab leaves some of them NORMAL (ZAYA1's <|im_start|> and , which are also in the + base vocab), and llama.cpp matches special-token text in prompts only for CONTROL and + USER_DEFINED tokens, so the chat template's <|im_start|> reached the model spelled out as text.""" + import json + from pathlib import Path + path = Path(tokenizer_dir) / "tokenizer.json" + if not path.is_file(): + return toktypes + for added in json.loads(path.read_text(encoding="utf-8")).get("added_tokens", []): + tid = added.get("id") + if added.get("special") and isinstance(tid, int) and 0 <= tid < len(toktypes): + toktypes[tid] = gguf.TokenType.CONTROL + return toktypes + @ModelBase.register("ZayaForCausalLM") class ZayaModel(TextModel): """Zyphra ZAYA1 (transformers naming): every layer is CCA attention + a top-1 MoE.""" @@ -260,7 +278,7 @@ def set_vocab(self): self.gguf_writer.add_tokenizer_model("gemma4") self.gguf_writer.add_token_list(tokens) self.gguf_writer.add_token_scores(scores) - self.gguf_writer.add_token_types(toktypes) + self.gguf_writer.add_token_types(_special_tokens_as_control(self._tokenizer_dir(), toktypes)) special_vocab = gguf.SpecialVocab(self.dir_model, load_merges=True) special_vocab.add_to_gguf(self.gguf_writer) @@ -386,7 +404,7 @@ def set_vocab(self): self.gguf_writer.add_tokenizer_model("gemma4") self.gguf_writer.add_token_list(tokens) self.gguf_writer.add_token_scores(scores) - self.gguf_writer.add_token_types(toktypes) + self.gguf_writer.add_token_types(_special_tokens_as_control(self._tokenizer_dir(), toktypes)) special_vocab = gguf.SpecialVocab(self.dir_model, load_merges=True) special_vocab.chat_template = self._CHAT_TEMPLATE