Created
August 17, 2026 19:03
-
-
Save bartowski1182/82fbf6dfc3fd234d5b50fda4d0bdf218 to your computer and use it in GitHub Desktop.
Rendering script used for shipping the imatrix data, in particular the chat template wrapping
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/usr/bin/env python3 | |
| """Render the v6 imatrix calibration corpus for a target model. | |
| The corpus data is fixed and ships beside this script: | |
| v6_prose.txt prose component, used verbatim | |
| v6_conversations.json canonical conversations (messages + tools) | |
| Only the RENDERING is per-model: the conversations go through the target | |
| model's own chat encoding, so the corpus contains exactly the bytes the | |
| model sees at inference. Three encoders are auto-detected: the jinja chat | |
| template (the normal case), DeepSeek-V4's vendor Python encoder | |
| (<model_dir>/encoding/encoding_dsv4.py), and mistral-common for Mistral | |
| models that ship tekken.json without a chat template. | |
| Rendered conversations are packed into exact 512-token chunks. Models that | |
| render the conversations short fall under the calibrated tool-chunk | |
| fraction band; band-holding then appends conversations from the extension | |
| pool, in its selection order, until the band's lower edge is reached. | |
| The result is fed to llama-imatrix with `-c 512 --parse-special` | |
| """ | |
| import argparse | |
| import copy | |
| import importlib.util | |
| import json | |
| import sys | |
| from pathlib import Path | |
| DATA_DIR = Path(__file__).parent.resolve() | |
| RECIPE = "calibration-v6" | |
| CHUNK_SIZE = 512 # validated; do not change casually | |
| TOOL_FRACTION_BAND = (0.55, 0.68) # validated renders measured 0.58-0.61 | |
| SAMPLE_SEPARATOR = "\n\n========== NEXT CONVERSATION ==========\n\n" | |
| # --- model encoders --------------------------------------------------------- | |
| class TemplateRenderer: | |
| """The normal case: the model ships a jinja chat template.""" | |
| name = "chat_template" | |
| def __init__(self, tokenizer): | |
| self.tokenizer = tokenizer | |
| def render_tokens(self, messages, tools=None) -> list: | |
| encoded = self.tokenizer.apply_chat_template( | |
| messages, tools=tools or None, tokenize=True, | |
| add_generation_prompt=False) | |
| if hasattr(encoded, "keys") and "input_ids" in encoded: # BatchEncoding | |
| encoded = encoded["input_ids"] | |
| if encoded and isinstance(encoded[0], list): | |
| encoded = encoded[0] | |
| return list(encoded) | |
| class DeepSeekV4Renderer: | |
| """DeepSeek-V4 ships no chat template; the prompt format is defined by a | |
| reference Python encoder in <model_dir>/encoding/encoding_dsv4.py. The | |
| encoder wants function.arguments as a JSON string (ours are dicts) and | |
| renders tools on the system message.""" | |
| name = "encoding_dsv4" | |
| def __init__(self, model_dir, tokenizer): | |
| self.tokenizer = tokenizer | |
| path = Path(model_dir) / "encoding" / "encoding_dsv4.py" | |
| spec = importlib.util.spec_from_file_location("encoding_dsv4", path) | |
| self.enc = importlib.util.module_from_spec(spec) | |
| spec.loader.exec_module(self.enc) | |
| def render_tokens(self, messages, tools=None) -> list: | |
| msgs = copy.deepcopy(list(messages)) | |
| for m in msgs: | |
| for tc in m.get("tool_calls") or []: | |
| args = tc.get("function", {}).get("arguments") | |
| if isinstance(args, (dict, list)): | |
| tc["function"]["arguments"] = json.dumps(args, ensure_ascii=False) | |
| if tools: | |
| if msgs and msgs[0].get("role") == "system": | |
| msgs[0] = {**msgs[0], "tools": tools} | |
| else: | |
| msgs.insert(0, {"role": "system", "content": "", "tools": tools}) | |
| text = self.enc.encode_messages(msgs, thinking_mode="thinking") | |
| return list(self.tokenizer(text, add_special_tokens=False)["input_ids"]) | |
| class MistralCommonRenderer: | |
| """Mistral models with tekken.json and no chat template: encode with the | |
| official mistral-common library, decode ids with the HF tokenizer. | |
| Handles 9-char alphanumeric tool-call ids (renamed deterministically) and | |
| assistant-final conversations (continue_final_message=True + EOS).""" | |
| name = "mistral_common" | |
| def __init__(self, model_dir, tokenizer): | |
| from mistral_common.tokens.tokenizers.mistral import MistralTokenizer | |
| self.tokenizer = tokenizer | |
| self.mc = MistralTokenizer.from_file(str(Path(model_dir) / "tekken.json")) | |
| self._eos = self.mc.instruct_tokenizer.tokenizer.eos_id | |
| def render_tokens(self, messages, tools=None) -> list: | |
| from mistral_common.protocol.instruct.request import ChatCompletionRequest | |
| msgs = copy.deepcopy(list(messages)) | |
| id_map = {} | |
| def norm(cid): | |
| if cid not in id_map: | |
| id_map[cid] = f"{len(id_map):09d}" | |
| return id_map[cid] | |
| for m in msgs: | |
| for tc in m.get("tool_calls") or []: | |
| if tc.get("id") is not None: | |
| tc["id"] = norm(tc["id"]) | |
| if m.get("role") == "tool" and m.get("tool_call_id") is not None: | |
| m["tool_call_id"] = norm(m["tool_call_id"]) | |
| last_is_assistant = bool(msgs) and msgs[-1].get("role") == "assistant" | |
| req = ChatCompletionRequest.from_openai( | |
| messages=msgs, tools=tools or None, | |
| continue_final_message=last_is_assistant) | |
| tokens = list(self.mc.encode_chat_completion(req).tokens) | |
| if last_is_assistant: | |
| tokens.append(self._eos) | |
| return tokens | |
| def detect_encoder(model_dir, tokenizer) -> str: | |
| """Name of the encoder this model dir calls for.""" | |
| if (Path(model_dir) / "encoding" / "encoding_dsv4.py").exists(): | |
| return DeepSeekV4Renderer.name | |
| if ((Path(model_dir) / "tekken.json").exists() | |
| and getattr(tokenizer, "chat_template", None) is None): | |
| return MistralCommonRenderer.name | |
| return TemplateRenderer.name | |
| def get_renderer(model_dir, tokenizer): | |
| encoder = detect_encoder(model_dir, tokenizer) | |
| if encoder == DeepSeekV4Renderer.name: | |
| return DeepSeekV4Renderer(model_dir, tokenizer) | |
| if encoder == MistralCommonRenderer.name: | |
| return MistralCommonRenderer(model_dir, tokenizer) | |
| return TemplateRenderer(tokenizer) | |
| # --- per-model quirks ------------------------------------------------------- | |
| # Fixups applied to (messages, tools) before rendering, keyed by a substring | |
| # of the model dir name. Add entries only for template behaviors that would | |
| # otherwise silently corrupt the corpus. | |
| def _quirk_gemma(messages, tools): | |
| """Gemma 4's template renders reasoning ONLY on assistant turns that carry | |
| tool_calls — strip reasoning_content elsewhere (the template would | |
| discard it silently anyway).""" | |
| for msg in messages: | |
| if (msg["role"] == "assistant" and "reasoning_content" in msg | |
| and not msg.get("tool_calls")): | |
| del msg["reasoning_content"] | |
| return messages, tools | |
| MODEL_QUIRKS = { | |
| "gemma-4": _quirk_gemma, | |
| } | |
| def quirks_for(model_name: str) -> list: | |
| lowered = model_name.lower() | |
| return [fn for key, fn in MODEL_QUIRKS.items() if key in lowered] | |
| # --- rendering and packing -------------------------------------------------- | |
| def render_conversation(renderer, tokenizer, conv, quirks) -> list: | |
| """Render one canonical conversation to token ids.""" | |
| messages = copy.deepcopy(conv["messages"]) | |
| tools = copy.deepcopy(conv["tools"]) | |
| for fn in quirks: | |
| messages, tools = fn(messages, tools) | |
| tokens = renderer.render_tokens(messages, tools) | |
| # llama-imatrix prepends its own file-level BOS on add_bos models; a kept | |
| # BOS would be fed as a double BOS in chunk 0 | |
| if (tokenizer.bos_token_id is not None and tokens | |
| and tokens[0] == tokenizer.bos_token_id): | |
| tokens = tokens[1:] | |
| return tokens | |
| def pack_chunks(token_lists, tokenizer, out_file) -> int: | |
| """Pack rendered conversations into exact 512-token chunks written to the | |
| open text handle out_file: sequential fill, split on overflow, pad the | |
| final chunk. Chunks never re-introduce BOS mid-stream — llama-imatrix does | |
| its own per-chunk BOS handling; what matters is that the byte stream | |
| matches inference.""" | |
| n_chunks = 0 | |
| current = [] | |
| for tokens in token_lists: | |
| while tokens: | |
| room = CHUNK_SIZE - len(current) | |
| current.extend(tokens[:room]) | |
| tokens = tokens[room:] | |
| if len(current) == CHUNK_SIZE: | |
| out_file.write(tokenizer.decode(current, skip_special_tokens=False)) | |
| n_chunks += 1 | |
| current = [] | |
| if current: | |
| pad_id = tokenizer.pad_token_id if tokenizer.pad_token_id is not None else 0 | |
| current.extend([pad_id] * (CHUNK_SIZE - len(current))) | |
| out_file.write(tokenizer.decode(current, skip_special_tokens=False)) | |
| n_chunks += 1 | |
| return n_chunks | |
| def apply_band_holding(rendered: dict, ext_ordered: list, prose_chunks: int): | |
| """Hold the tool-chunk fraction at the band's lower edge. | |
| rendered maps kept_index -> tokens for the base set; ext_ordered is a list | |
| of (kept_index, tokens) in the extension pool's selection order. Extension | |
| convs are appended in that order until the fraction reaches the lower band | |
| edge or the pool runs out. Returns (kept, ext_used); kept packs in | |
| kept-index order, so extensions interleave into the stream. | |
| """ | |
| lo = TOOL_FRACTION_BAND[0] | |
| def fraction_of(keep): | |
| # float chunk fraction — the decision rule is deliberately computed | |
| # before the chunks are rounded to whole chunks | |
| tool_chunks_f = sum(len(t) for t in keep.values()) / CHUNK_SIZE | |
| return tool_chunks_f / (tool_chunks_f + prose_chunks) | |
| keep = dict(rendered) | |
| ext_used = 0 | |
| while fraction_of(keep) < lo and ext_used < len(ext_ordered): | |
| idx, tokens = ext_ordered[ext_used] | |
| keep[idx] = tokens | |
| ext_used += 1 | |
| return keep, ext_used | |
| def encoding_anomalies(model_dir, tokenizer, encoder_name) -> list: | |
| """Warnings for on-disk artifacts suggesting the chosen encoder may not be | |
| the model's authoritative chat encoding. | |
| Deliberately conservative: generic trust_remote_code modules | |
| (modeling_*.py, tokenization_*.py, configuration_*.py) are normal and | |
| never flagged. Nothing here changes a rendered byte. | |
| """ | |
| model_dir = Path(model_dir) | |
| warnings = [] | |
| if encoder_name != DeepSeekV4Renderer.name: | |
| enc_dir = model_dir / "encoding" | |
| if enc_dir.is_dir(): | |
| files = sorted(p.name for p in enc_dir.glob("*.py")) | |
| listed = ", ".join(files) if files else "no .py files" | |
| warnings.append( | |
| "model ships an encoding/ directory the renderer does not " | |
| f"recognize ({listed}) — the corpus was rendered through " | |
| f"{encoder_name}; verify the render, or teach " | |
| "render_corpus.py the vendor encoder (see DeepSeekV4Renderer)") | |
| top_level = sorted(p.name for p in model_dir.glob("encoding*.py")) | |
| if top_level: | |
| warnings.append( | |
| "model ships top-level encoding modules the renderer does not " | |
| f"recognize ({', '.join(top_level)}) — the corpus was rendered " | |
| f"through {encoder_name}; verify the render, or teach " | |
| "render_corpus.py the vendor encoder (see DeepSeekV4Renderer)") | |
| if (encoder_name == TemplateRenderer.name | |
| and (model_dir / "tekken.json").exists()): | |
| warnings.append( | |
| "model ships tekken.json but was rendered through its chat " | |
| "template — mistral-common was bypassed; verify which encoding is " | |
| "authoritative") | |
| return warnings | |
| def require_renderable(encoder_name, tokenizer) -> None: | |
| """Exit by name when no chat encoding exists at all, rather than dying | |
| later inside transformers with a generic error.""" | |
| if (encoder_name == TemplateRenderer.name | |
| and getattr(tokenizer, "chat_template", None) is None): | |
| sys.exit( | |
| "model ships no chat template and no encoder the renderer " | |
| "recognizes — the corpus cannot be rendered. Teach " | |
| "render_corpus.py the model's encoding (see DeepSeekV4Renderer) " | |
| "or fall back to the v5 corpus.") | |
| def build_warnings(fraction: float, ext_used: int, ext_pool_size: int) -> list: | |
| lo, hi = TOOL_FRACTION_BAND | |
| warnings = [] | |
| if not lo <= fraction <= hi: | |
| warnings.append( | |
| f"tool chunk fraction {fraction:.3f} outside the validated " | |
| f"{lo}-{hi} band even after exhausting the extension pool — " | |
| "coverage screened clean under the band on the strictest " | |
| "measured MoE (Qwen3-Next-80B, 0/24576 dead experts), but the " | |
| "validated quality margin is not guaranteed at this mix; on a " | |
| "new MoE family, check the imatrix for uncovered experts.") | |
| if ext_used: | |
| warnings.append( | |
| f"band-holding added {ext_used} of {ext_pool_size} extension " | |
| "conversations: the render is coverage-screened but the extension " | |
| "mechanism is not itself det-validated (the base 137-conv " | |
| "selection is)") | |
| return warnings | |
| def main() -> None: | |
| ap = argparse.ArgumentParser( | |
| description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) | |
| ap.add_argument("--model", required=True, | |
| help="Path to the model dir (tokenizer + chat encoding source)") | |
| ap.add_argument("--out-dir", required=True, help="Directory for the outputs") | |
| ap.add_argument("--prefix", required=True, help="Output file name prefix") | |
| args = ap.parse_args() | |
| model_path = Path(args.model).resolve() | |
| if not model_path.is_dir(): | |
| sys.exit(f"Model dir not found: {model_path}") | |
| model_name = model_path.name | |
| out_dir = Path(args.out_dir) | |
| out_dir.mkdir(parents=True, exist_ok=True) | |
| prose = (DATA_DIR / "v6_prose.txt").read_text(encoding="utf-8") | |
| data = json.loads((DATA_DIR / "v6_conversations.json").read_text( | |
| encoding="utf-8")) | |
| from transformers import AutoTokenizer | |
| # trust_remote_code is safe here and nowhere else: this script only runs | |
| # inside the no-network quantization container against a read-only model | |
| # mount, never in the server process. | |
| tokenizer = AutoTokenizer.from_pretrained(str(model_path), | |
| trust_remote_code=True) | |
| renderer = get_renderer(model_path, tokenizer) | |
| print(f"model: {model_name} (encoder: {renderer.name})") | |
| require_renderable(renderer.name, tokenizer) | |
| anomalies = encoding_anomalies(model_path, tokenizer, renderer.name) | |
| quirks = quirks_for(model_name) | |
| rendered = {c["kept_index"]: render_conversation(renderer, tokenizer, c, quirks) | |
| for c in data["conversations"]} | |
| ext_pool = data["extension_conversations"] | |
| ext_ordered = [(c["kept_index"], | |
| render_conversation(renderer, tokenizer, c, quirks)) | |
| for c in ext_pool] | |
| prose_chunks = len( | |
| tokenizer(prose, add_special_tokens=False)["input_ids"]) // CHUNK_SIZE | |
| keep, ext_used = apply_band_holding(rendered, ext_ordered, prose_chunks) | |
| if ext_used: | |
| print(f"band-holding: added {ext_used} extension conversations") | |
| # pack in kept-index order — extension convs interleave into the stream | |
| token_lists = [keep[i] for i in sorted(keep)] | |
| corpus_path = out_dir / f"{args.prefix}-calibration-v6.txt" | |
| with open(corpus_path, "w", encoding="utf-8") as f: | |
| f.write(prose) | |
| tool_chunks = pack_chunks(token_lists, tokenizer, f) | |
| (out_dir / f"{args.prefix}-calibration-v6.samples.txt").write_text( | |
| SAMPLE_SEPARATOR.join( | |
| tokenizer.decode(t, skip_special_tokens=False) | |
| for t in token_lists[:3]), | |
| encoding="utf-8") | |
| total = prose_chunks + tool_chunks | |
| fraction = tool_chunks / total if total else 0.0 | |
| warnings = anomalies + build_warnings(fraction, ext_used, len(ext_ordered)) | |
| manifest = { | |
| "generator": "auto_quant_v2 calibration renderer", | |
| "recipe": RECIPE, | |
| "model": model_name, | |
| "model_path": str(model_path), | |
| "encoder": renderer.name, | |
| "chunk_size": CHUNK_SIZE, | |
| "prose_chunks": prose_chunks, | |
| "tool_chunks": tool_chunks, | |
| "total_chunks": total, | |
| "tool_chunk_fraction": round(fraction, 3), | |
| "n_conversations": len(token_lists), | |
| "extension_convs_used": ext_used, | |
| "conversation_token_lengths": [len(t) for t in token_lists], | |
| "warnings": warnings, | |
| } | |
| manifest_path = out_dir / f"{args.prefix}-calibration-v6.manifest.json" | |
| manifest_path.write_text(json.dumps(manifest, indent=2), encoding="utf-8") | |
| print(f" prose {prose_chunks:4d} chunks") | |
| print(f" tools {tool_chunks:4d} chunks ({len(token_lists)} conversations" | |
| f"{f', {ext_used} from the extension pool' if ext_used else ''})") | |
| print(f" total {total:4d} chunks ({fraction:.0%} tool)") | |
| for w in warnings: | |
| print(f"\nWARNING: {w}") | |
| print(f"\ncorpus: {corpus_path}") | |
| print(f"manifest: {manifest_path}") | |
| if __name__ == "__main__": | |
| main() |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment