Skip to content

Instantly share code, notes, and snippets.

@bartowski1182
Created August 17, 2026 19:03
Show Gist options
  • Select an option

  • Save bartowski1182/82fbf6dfc3fd234d5b50fda4d0bdf218 to your computer and use it in GitHub Desktop.

Select an option

Save bartowski1182/82fbf6dfc3fd234d5b50fda4d0bdf218 to your computer and use it in GitHub Desktop.
Rendering script used for shipping the imatrix data, in particular the chat template wrapping
#!/usr/bin/env python3
"""Render the v6 imatrix calibration corpus for a target model.
The corpus data is fixed and ships beside this script:
v6_prose.txt prose component, used verbatim
v6_conversations.json canonical conversations (messages + tools)
Only the RENDERING is per-model: the conversations go through the target
model's own chat encoding, so the corpus contains exactly the bytes the
model sees at inference. Three encoders are auto-detected: the jinja chat
template (the normal case), DeepSeek-V4's vendor Python encoder
(<model_dir>/encoding/encoding_dsv4.py), and mistral-common for Mistral
models that ship tekken.json without a chat template.
Rendered conversations are packed into exact 512-token chunks. Models that
render the conversations short fall under the calibrated tool-chunk
fraction band; band-holding then appends conversations from the extension
pool, in its selection order, until the band's lower edge is reached.
The result is fed to llama-imatrix with `-c 512 --parse-special`
"""
import argparse
import copy
import importlib.util
import json
import sys
from pathlib import Path
DATA_DIR = Path(__file__).parent.resolve()
RECIPE = "calibration-v6"
CHUNK_SIZE = 512 # validated; do not change casually
TOOL_FRACTION_BAND = (0.55, 0.68) # validated renders measured 0.58-0.61
SAMPLE_SEPARATOR = "\n\n========== NEXT CONVERSATION ==========\n\n"
# --- model encoders ---------------------------------------------------------
class TemplateRenderer:
"""The normal case: the model ships a jinja chat template."""
name = "chat_template"
def __init__(self, tokenizer):
self.tokenizer = tokenizer
def render_tokens(self, messages, tools=None) -> list:
encoded = self.tokenizer.apply_chat_template(
messages, tools=tools or None, tokenize=True,
add_generation_prompt=False)
if hasattr(encoded, "keys") and "input_ids" in encoded: # BatchEncoding
encoded = encoded["input_ids"]
if encoded and isinstance(encoded[0], list):
encoded = encoded[0]
return list(encoded)
class DeepSeekV4Renderer:
"""DeepSeek-V4 ships no chat template; the prompt format is defined by a
reference Python encoder in <model_dir>/encoding/encoding_dsv4.py. The
encoder wants function.arguments as a JSON string (ours are dicts) and
renders tools on the system message."""
name = "encoding_dsv4"
def __init__(self, model_dir, tokenizer):
self.tokenizer = tokenizer
path = Path(model_dir) / "encoding" / "encoding_dsv4.py"
spec = importlib.util.spec_from_file_location("encoding_dsv4", path)
self.enc = importlib.util.module_from_spec(spec)
spec.loader.exec_module(self.enc)
def render_tokens(self, messages, tools=None) -> list:
msgs = copy.deepcopy(list(messages))
for m in msgs:
for tc in m.get("tool_calls") or []:
args = tc.get("function", {}).get("arguments")
if isinstance(args, (dict, list)):
tc["function"]["arguments"] = json.dumps(args, ensure_ascii=False)
if tools:
if msgs and msgs[0].get("role") == "system":
msgs[0] = {**msgs[0], "tools": tools}
else:
msgs.insert(0, {"role": "system", "content": "", "tools": tools})
text = self.enc.encode_messages(msgs, thinking_mode="thinking")
return list(self.tokenizer(text, add_special_tokens=False)["input_ids"])
class MistralCommonRenderer:
"""Mistral models with tekken.json and no chat template: encode with the
official mistral-common library, decode ids with the HF tokenizer.
Handles 9-char alphanumeric tool-call ids (renamed deterministically) and
assistant-final conversations (continue_final_message=True + EOS)."""
name = "mistral_common"
def __init__(self, model_dir, tokenizer):
from mistral_common.tokens.tokenizers.mistral import MistralTokenizer
self.tokenizer = tokenizer
self.mc = MistralTokenizer.from_file(str(Path(model_dir) / "tekken.json"))
self._eos = self.mc.instruct_tokenizer.tokenizer.eos_id
def render_tokens(self, messages, tools=None) -> list:
from mistral_common.protocol.instruct.request import ChatCompletionRequest
msgs = copy.deepcopy(list(messages))
id_map = {}
def norm(cid):
if cid not in id_map:
id_map[cid] = f"{len(id_map):09d}"
return id_map[cid]
for m in msgs:
for tc in m.get("tool_calls") or []:
if tc.get("id") is not None:
tc["id"] = norm(tc["id"])
if m.get("role") == "tool" and m.get("tool_call_id") is not None:
m["tool_call_id"] = norm(m["tool_call_id"])
last_is_assistant = bool(msgs) and msgs[-1].get("role") == "assistant"
req = ChatCompletionRequest.from_openai(
messages=msgs, tools=tools or None,
continue_final_message=last_is_assistant)
tokens = list(self.mc.encode_chat_completion(req).tokens)
if last_is_assistant:
tokens.append(self._eos)
return tokens
def detect_encoder(model_dir, tokenizer) -> str:
"""Name of the encoder this model dir calls for."""
if (Path(model_dir) / "encoding" / "encoding_dsv4.py").exists():
return DeepSeekV4Renderer.name
if ((Path(model_dir) / "tekken.json").exists()
and getattr(tokenizer, "chat_template", None) is None):
return MistralCommonRenderer.name
return TemplateRenderer.name
def get_renderer(model_dir, tokenizer):
encoder = detect_encoder(model_dir, tokenizer)
if encoder == DeepSeekV4Renderer.name:
return DeepSeekV4Renderer(model_dir, tokenizer)
if encoder == MistralCommonRenderer.name:
return MistralCommonRenderer(model_dir, tokenizer)
return TemplateRenderer(tokenizer)
# --- per-model quirks -------------------------------------------------------
# Fixups applied to (messages, tools) before rendering, keyed by a substring
# of the model dir name. Add entries only for template behaviors that would
# otherwise silently corrupt the corpus.
def _quirk_gemma(messages, tools):
"""Gemma 4's template renders reasoning ONLY on assistant turns that carry
tool_calls — strip reasoning_content elsewhere (the template would
discard it silently anyway)."""
for msg in messages:
if (msg["role"] == "assistant" and "reasoning_content" in msg
and not msg.get("tool_calls")):
del msg["reasoning_content"]
return messages, tools
MODEL_QUIRKS = {
"gemma-4": _quirk_gemma,
}
def quirks_for(model_name: str) -> list:
lowered = model_name.lower()
return [fn for key, fn in MODEL_QUIRKS.items() if key in lowered]
# --- rendering and packing --------------------------------------------------
def render_conversation(renderer, tokenizer, conv, quirks) -> list:
"""Render one canonical conversation to token ids."""
messages = copy.deepcopy(conv["messages"])
tools = copy.deepcopy(conv["tools"])
for fn in quirks:
messages, tools = fn(messages, tools)
tokens = renderer.render_tokens(messages, tools)
# llama-imatrix prepends its own file-level BOS on add_bos models; a kept
# BOS would be fed as a double BOS in chunk 0
if (tokenizer.bos_token_id is not None and tokens
and tokens[0] == tokenizer.bos_token_id):
tokens = tokens[1:]
return tokens
def pack_chunks(token_lists, tokenizer, out_file) -> int:
"""Pack rendered conversations into exact 512-token chunks written to the
open text handle out_file: sequential fill, split on overflow, pad the
final chunk. Chunks never re-introduce BOS mid-stream — llama-imatrix does
its own per-chunk BOS handling; what matters is that the byte stream
matches inference."""
n_chunks = 0
current = []
for tokens in token_lists:
while tokens:
room = CHUNK_SIZE - len(current)
current.extend(tokens[:room])
tokens = tokens[room:]
if len(current) == CHUNK_SIZE:
out_file.write(tokenizer.decode(current, skip_special_tokens=False))
n_chunks += 1
current = []
if current:
pad_id = tokenizer.pad_token_id if tokenizer.pad_token_id is not None else 0
current.extend([pad_id] * (CHUNK_SIZE - len(current)))
out_file.write(tokenizer.decode(current, skip_special_tokens=False))
n_chunks += 1
return n_chunks
def apply_band_holding(rendered: dict, ext_ordered: list, prose_chunks: int):
"""Hold the tool-chunk fraction at the band's lower edge.
rendered maps kept_index -> tokens for the base set; ext_ordered is a list
of (kept_index, tokens) in the extension pool's selection order. Extension
convs are appended in that order until the fraction reaches the lower band
edge or the pool runs out. Returns (kept, ext_used); kept packs in
kept-index order, so extensions interleave into the stream.
"""
lo = TOOL_FRACTION_BAND[0]
def fraction_of(keep):
# float chunk fraction — the decision rule is deliberately computed
# before the chunks are rounded to whole chunks
tool_chunks_f = sum(len(t) for t in keep.values()) / CHUNK_SIZE
return tool_chunks_f / (tool_chunks_f + prose_chunks)
keep = dict(rendered)
ext_used = 0
while fraction_of(keep) < lo and ext_used < len(ext_ordered):
idx, tokens = ext_ordered[ext_used]
keep[idx] = tokens
ext_used += 1
return keep, ext_used
def encoding_anomalies(model_dir, tokenizer, encoder_name) -> list:
"""Warnings for on-disk artifacts suggesting the chosen encoder may not be
the model's authoritative chat encoding.
Deliberately conservative: generic trust_remote_code modules
(modeling_*.py, tokenization_*.py, configuration_*.py) are normal and
never flagged. Nothing here changes a rendered byte.
"""
model_dir = Path(model_dir)
warnings = []
if encoder_name != DeepSeekV4Renderer.name:
enc_dir = model_dir / "encoding"
if enc_dir.is_dir():
files = sorted(p.name for p in enc_dir.glob("*.py"))
listed = ", ".join(files) if files else "no .py files"
warnings.append(
"model ships an encoding/ directory the renderer does not "
f"recognize ({listed}) — the corpus was rendered through "
f"{encoder_name}; verify the render, or teach "
"render_corpus.py the vendor encoder (see DeepSeekV4Renderer)")
top_level = sorted(p.name for p in model_dir.glob("encoding*.py"))
if top_level:
warnings.append(
"model ships top-level encoding modules the renderer does not "
f"recognize ({', '.join(top_level)}) — the corpus was rendered "
f"through {encoder_name}; verify the render, or teach "
"render_corpus.py the vendor encoder (see DeepSeekV4Renderer)")
if (encoder_name == TemplateRenderer.name
and (model_dir / "tekken.json").exists()):
warnings.append(
"model ships tekken.json but was rendered through its chat "
"template — mistral-common was bypassed; verify which encoding is "
"authoritative")
return warnings
def require_renderable(encoder_name, tokenizer) -> None:
"""Exit by name when no chat encoding exists at all, rather than dying
later inside transformers with a generic error."""
if (encoder_name == TemplateRenderer.name
and getattr(tokenizer, "chat_template", None) is None):
sys.exit(
"model ships no chat template and no encoder the renderer "
"recognizes — the corpus cannot be rendered. Teach "
"render_corpus.py the model's encoding (see DeepSeekV4Renderer) "
"or fall back to the v5 corpus.")
def build_warnings(fraction: float, ext_used: int, ext_pool_size: int) -> list:
lo, hi = TOOL_FRACTION_BAND
warnings = []
if not lo <= fraction <= hi:
warnings.append(
f"tool chunk fraction {fraction:.3f} outside the validated "
f"{lo}-{hi} band even after exhausting the extension pool — "
"coverage screened clean under the band on the strictest "
"measured MoE (Qwen3-Next-80B, 0/24576 dead experts), but the "
"validated quality margin is not guaranteed at this mix; on a "
"new MoE family, check the imatrix for uncovered experts.")
if ext_used:
warnings.append(
f"band-holding added {ext_used} of {ext_pool_size} extension "
"conversations: the render is coverage-screened but the extension "
"mechanism is not itself det-validated (the base 137-conv "
"selection is)")
return warnings
def main() -> None:
ap = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--model", required=True,
help="Path to the model dir (tokenizer + chat encoding source)")
ap.add_argument("--out-dir", required=True, help="Directory for the outputs")
ap.add_argument("--prefix", required=True, help="Output file name prefix")
args = ap.parse_args()
model_path = Path(args.model).resolve()
if not model_path.is_dir():
sys.exit(f"Model dir not found: {model_path}")
model_name = model_path.name
out_dir = Path(args.out_dir)
out_dir.mkdir(parents=True, exist_ok=True)
prose = (DATA_DIR / "v6_prose.txt").read_text(encoding="utf-8")
data = json.loads((DATA_DIR / "v6_conversations.json").read_text(
encoding="utf-8"))
from transformers import AutoTokenizer
# trust_remote_code is safe here and nowhere else: this script only runs
# inside the no-network quantization container against a read-only model
# mount, never in the server process.
tokenizer = AutoTokenizer.from_pretrained(str(model_path),
trust_remote_code=True)
renderer = get_renderer(model_path, tokenizer)
print(f"model: {model_name} (encoder: {renderer.name})")
require_renderable(renderer.name, tokenizer)
anomalies = encoding_anomalies(model_path, tokenizer, renderer.name)
quirks = quirks_for(model_name)
rendered = {c["kept_index"]: render_conversation(renderer, tokenizer, c, quirks)
for c in data["conversations"]}
ext_pool = data["extension_conversations"]
ext_ordered = [(c["kept_index"],
render_conversation(renderer, tokenizer, c, quirks))
for c in ext_pool]
prose_chunks = len(
tokenizer(prose, add_special_tokens=False)["input_ids"]) // CHUNK_SIZE
keep, ext_used = apply_band_holding(rendered, ext_ordered, prose_chunks)
if ext_used:
print(f"band-holding: added {ext_used} extension conversations")
# pack in kept-index order — extension convs interleave into the stream
token_lists = [keep[i] for i in sorted(keep)]
corpus_path = out_dir / f"{args.prefix}-calibration-v6.txt"
with open(corpus_path, "w", encoding="utf-8") as f:
f.write(prose)
tool_chunks = pack_chunks(token_lists, tokenizer, f)
(out_dir / f"{args.prefix}-calibration-v6.samples.txt").write_text(
SAMPLE_SEPARATOR.join(
tokenizer.decode(t, skip_special_tokens=False)
for t in token_lists[:3]),
encoding="utf-8")
total = prose_chunks + tool_chunks
fraction = tool_chunks / total if total else 0.0
warnings = anomalies + build_warnings(fraction, ext_used, len(ext_ordered))
manifest = {
"generator": "auto_quant_v2 calibration renderer",
"recipe": RECIPE,
"model": model_name,
"model_path": str(model_path),
"encoder": renderer.name,
"chunk_size": CHUNK_SIZE,
"prose_chunks": prose_chunks,
"tool_chunks": tool_chunks,
"total_chunks": total,
"tool_chunk_fraction": round(fraction, 3),
"n_conversations": len(token_lists),
"extension_convs_used": ext_used,
"conversation_token_lengths": [len(t) for t in token_lists],
"warnings": warnings,
}
manifest_path = out_dir / f"{args.prefix}-calibration-v6.manifest.json"
manifest_path.write_text(json.dumps(manifest, indent=2), encoding="utf-8")
print(f" prose {prose_chunks:4d} chunks")
print(f" tools {tool_chunks:4d} chunks ({len(token_lists)} conversations"
f"{f', {ext_used} from the extension pool' if ext_used else ''})")
print(f" total {total:4d} chunks ({fraction:.0%} tool)")
for w in warnings:
print(f"\nWARNING: {w}")
print(f"\ncorpus: {corpus_path}")
print(f"manifest: {manifest_path}")
if __name__ == "__main__":
main()
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment