Skip to content

Instantly share code, notes, and snippets.

@audreyt
Last active September 23, 2026 21:37
Show Gist options
  • Select an option

  • Save audreyt/beb06fea21bebcc5d01bdbb49680eaf0 to your computer and use it in GitHub Desktop.

Select an option

Save audreyt/beb06fea21bebcc5d01bdbb49680eaf0 to your computer and use it in GitHub Desktop.
Unofficial: run Thomson-1.0-Small locally on Splash (Apple silicon) with omp/Paseo, without redistributing weights — hardened recipe + converter + verifier
#!/usr/bin/env python3
"""Pack thomsonreuters/Thomson-1.0-Small (BF16) into a Splash q4-moe package.
Format reference: splash/runtime/model/{WeightStore,QwenTarget,Qwen3_6Moe}.cpp,
runtime/metal/kernels/{common/{q4_mpp_tiles,gdn_primitives,attention_qkv_prepare}.h,
shared/{moe,embedding,normalization,vision}.metal, prefill/{gdn,attention_qkv}.metal}.
N conducted against mlx-community/Qwen3.6-35B-A3B-4bit (the official pack's
source): Splash Q4 sections are bit-identical to MLX 4-bit values repacked
into StorageN=256 tiling, low-nibble-first, affine w = q*scale + bias.
"""
import hashlib
import json
import struct
import sys
from pathlib import Path
import numpy as np
ALIGN = 16384
H = 2048
LAYERS = 40
VOCAB = 248320
GDN_W = 12544
GDN_ACTUAL = 12352
FULL_W = 9216
CONV_DIM = 8192
GDN_VH = 32
ATTN_W = 4096
EXPERTS = 256
EXPERT_INT = 512
# ---------------------------------------------------------------- bf16 ----
def bf16_to_f32(raw: bytes) -> np.ndarray:
u8 = np.frombuffer(raw, dtype=np.uint8).reshape(-1, 2)
bits = (u8[:, 0].astype(np.uint32) + u8[:, 1].astype(np.uint32) * 256) << 16
return bits.view(np.float32).copy()
def f32_to_bf16(a: np.ndarray) -> np.ndarray:
a = np.ascontiguousarray(a, dtype=np.float32)
bits = a.view(np.uint32)
# Round-to-nearest-even into the top 16 bits; preserve NaN/Inf.
exp = (bits >> 23) & 0xFF
bias = np.where(exp == 0xFF, np.uint32(0),
((bits >> 16) & np.uint32(1)) + np.uint32(0x7FFF))
return ((bits + bias) >> 16).astype(np.uint16)
# --------------------------------------------------------- safetensors ----
class ShardSet:
"""Reads BF16 tensors (as float32) from local safetensors shards."""
def __init__(self, directory: Path, index_name: str):
index = json.loads((directory / index_name).read_text())
self.weight_map = index["weight_map"]
self.files = {}
self.entries = {}
for shard in sorted(set(self.weight_map.values())):
path = directory / shard
with open(path, "rb") as f:
(n,) = struct.unpack("<Q", f.read(8))
header = json.loads(f.read(n).decode())
base = 8 + n
self.files[shard] = path
for name, meta in header.items():
if name == "__metadata__":
continue
o0, o1 = meta["data_offsets"]
self.entries[name] = (path, base + o0, base + o1,
meta["dtype"], tuple(meta["shape"]))
def load(self, name: str) -> np.ndarray:
path, o0, o1, dtype, shape = self.entries[name]
assert dtype == "BF16", (name, dtype)
with open(path, "rb") as f:
f.seek(o0)
raw = f.read(o1 - o0)
assert len(raw) == o1 - o0, name
return bf16_to_f32(raw).reshape(shape)
# ------------------------------------------------------------ quantize ----
def _group_stats(mat: np.ndarray):
O, I = mat.shape
assert I % 64 == 0, mat.shape
g = mat.reshape(O, I // 64, 64)
mn = g.min(axis=2)
mx = g.max(axis=2)
return mn, mx
def quant_affine(mat: np.ndarray, levels: int):
"""Per-64 affine quant. Returns (q uint8, scales f32, biases f32)."""
mat = np.ascontiguousarray(mat, dtype=np.float32)
mn, mx = _group_stats(mat)
span = (mx - mn) / (levels - 1)
span = np.where(span == 0, np.ones_like(span), span)
q = np.clip(np.rint((mat.reshape(mn.shape[0], mn.shape[1], 64) - mn[:, :, None])
/ span[:, :, None]), 0, levels - 1).astype(np.uint8)
return q.reshape(mat.shape), span, mn
def _tile_params(param: np.ndarray):
"""(O, G) bf16 params -> tiled [tile][group][row] bytes. O % 256 == 0."""
O, G = param.shape
assert O % 256 == 0, param.shape
return f32_to_bf16(param).reshape(O // 256, 256, G).transpose(0, 2, 1) \
.tobytes()
def pack_q4_fused(mat: np.ndarray) -> bytes:
"""Fused Q4 section: tiled nibbles ++ tiled bf16 scales ++ biases."""
O, I = mat.shape
assert O % 256 == 0 and I % 64 == 0, mat.shape
q, s, b = quant_affine(mat, 16)
T, G = O // 256, I // 64
# (T, 256 rows, G, 64) -> (T, G, 256, 32 bytes), low nibble first.
t = q.reshape(T, 256, G, 64).transpose(0, 2, 1, 3).reshape(T, G, 256, 32, 2)
packed = (t[..., 0] | (t[..., 1] << 4)).astype(np.uint8).tobytes()
return packed + _tile_params(s) + _tile_params(b)
def pack_q4_expert_slabs(mat: np.ndarray) -> bytes:
"""Per-expert Q4 slabs: [nibbles ++ scales ++ biases] per expert.
The runtime reads MoE projections with readExpertQ4Projection as
``experts`` contiguous slabs of q4PackedBytes(Oe, I) (see
splash/runtime/model/WeightStore.cpp). A flat fused pack of the
reshaped (experts*Oe, I) matrix has the same total size but groups
ALL nibbles before ALL scales, so slab reads land on wrong bytes
(nibble bytes decoded as bf16 scales -> inf -> NaN logits, and the
sampler emits its 0xffffffff no-candidate sentinel). Pack each
expert's rows separately instead; per-64-group affine quant touches
only (row, 64-col) groups, so values are identical to the flat pack.
"""
E, Oe, I = mat.shape
assert Oe % 256 == 0 and I % 64 == 0, mat.shape
return b"".join(
pack_q4_fused(np.ascontiguousarray(mat[e])) for e in range(E))
def pack_q8_fused(mat: np.ndarray) -> bytes:
O, I = mat.shape
assert O % 256 == 0 and I % 64 == 0, mat.shape
q, s, b = quant_affine(mat, 256)
T, G = O // 256, I // 64
packed = q.reshape(T, 256, G, 64).transpose(0, 2, 1, 3).tobytes()
return packed + _tile_params(s) + _tile_params(b)
def pack_q4_embed(mat: np.ndarray):
"""Row-major components variant for embedding.bin. Returns 3 blobs."""
O, I = mat.shape
assert I % 64 == 0, mat.shape
q, s, b = quant_affine(mat, 16)
w = (q.reshape(O, I // 2, 2)[..., 0] |
(q.reshape(O, I // 2, 2)[..., 1] << 4)).astype(np.uint8).tobytes()
return w, f32_to_bf16(s).tobytes(), f32_to_bf16(b).tobytes()
# -------------------------------------------------------- section I/O ----
def align16k(x: int) -> int:
return (x + ALIGN - 1) & ~(ALIGN - 1)
class PackFile:
def __init__(self, path: Path, magic: bytes, layer: int, typ: int):
assert len(magic) == 8
self.f = open(path, "wb")
self.f.write(magic + struct.pack("<II", layer, typ))
self.off = 16
def section(self, blob: bytes):
assert len(blob) > 0
pad = align16k(self.off) - self.off
if pad:
self.f.write(b"\x00" * pad)
self.off += pad
self.f.write(blob)
self.off += len(blob)
def close(self) -> int:
pad = align16k(self.off) - self.off
if pad:
self.f.write(b"\x00" * pad)
self.off += pad
self.f.close()
return self.off
# ---------------------------------------------------- layer assembly ----
P = "model.language_model.layers.{}.{}"
MAGIC_LAYER = b"MDFM0001"
def norm1p(w: np.ndarray) -> bytes:
"""Zero-centered RMSNorm weight -> packed (1 + w) bf16."""
return f32_to_bf16(w + 1.0).tobytes()
def pack_gdn_input(src: ShardSet, i: int) -> bytes:
qkv = src.load(P.format(i, "linear_attn.in_proj_qkv.weight"))
assert qkv.shape == (8192, H), qkv.shape
z = src.load(P.format(i, "linear_attn.in_proj_z.weight"))
assert z.shape == (ATTN_W, H), z.shape
b = src.load(P.format(i, "linear_attn.in_proj_b.weight"))
assert b.shape == (GDN_VH, H), b.shape
a = src.load(P.format(i, "linear_attn.in_proj_a.weight"))
assert a.shape == (GDN_VH, H), a.shape
full = np.zeros((GDN_W, H), dtype=np.float32)
full[0:8192] = qkv
full[8192:8192 + ATTN_W] = z
full[8192 + ATTN_W:8192 + ATTN_W + GDN_VH] = b
full[8192 + ATTN_W + GDN_VH:GDN_ACTUAL] = a
return pack_q4_fused(full)
def pack_attn_input(src: ShardSet, i: int) -> bytes:
q = src.load(P.format(i, "self_attn.q_proj.weight"))
assert q.shape == (8192, H), q.shape # per-head [q(256); gate(256)]
k = src.load(P.format(i, "self_attn.k_proj.weight"))
assert k.shape == (512, H), k.shape
v = src.load(P.format(i, "self_attn.v_proj.weight"))
assert v.shape == (512, H), v.shape
return pack_q4_fused(np.concatenate([q, k, v], axis=0))
def pack_ffn_tail(src: ShardSet, i: int, pf: PackFile):
gate = src.load(P.format(i, "mlp.gate.weight"))
assert gate.shape == (EXPERTS, H), gate.shape
pf.section(pack_q8_fused(gate))
gu = src.load(P.format(i, "mlp.experts.gate_up_proj"))
assert gu.shape == (EXPERTS, 2 * EXPERT_INT, H), gu.shape
pf.section(pack_q4_expert_slabs(
np.ascontiguousarray(gu[:, 0:EXPERT_INT, :])))
pf.section(pack_q4_expert_slabs(
np.ascontiguousarray(gu[:, EXPERT_INT:2 * EXPERT_INT, :])))
down = src.load(P.format(i, "mlp.experts.down_proj"))
assert down.shape == (EXPERTS, H, EXPERT_INT), down.shape
pf.section(pack_q4_expert_slabs(np.ascontiguousarray(down)))
sg = src.load(P.format(i, "mlp.shared_expert.gate_proj.weight"))
assert sg.shape == (EXPERT_INT, H), sg.shape
pf.section(pack_q4_fused(sg))
su = src.load(P.format(i, "mlp.shared_expert.up_proj.weight"))
assert su.shape == (EXPERT_INT, H), su.shape
pf.section(pack_q4_fused(su))
sd = src.load(P.format(i, "mlp.shared_expert.down_proj.weight"))
assert sd.shape == (H, EXPERT_INT), sd.shape
pf.section(pack_q4_fused(sd))
sgate = src.load(P.format(i, "mlp.shared_expert_gate.weight"))
assert sgate.shape == (1, H), sgate.shape
pf.section(pack_q8_fused(np.tile(sgate, (256, 1))))
def pack_layer(src: ShardSet, out_dir: Path, i: int, full: bool) -> int:
pre = P.format(i, "")
pf = PackFile(out_dir / f"layer-{i}.bin", MAGIC_LAYER, i, 1 if full else 0)
pf.section(norm1p(src.load(pre + "input_layernorm.weight")))
if full:
pf.section(pack_attn_input(src, i))
pf.section(norm1p(src.load(pre + "self_attn.q_norm.weight")))
pf.section(norm1p(src.load(pre + "self_attn.k_norm.weight")))
out = src.load(pre + "self_attn.o_proj.weight")
assert out.shape == (H, ATTN_W), out.shape
pf.section(pack_q4_fused(out))
else:
pf.section(pack_gdn_input(src, i))
conv = src.load(pre + "linear_attn.conv1d.weight")
assert conv.shape == (CONV_DIM, 1, 4), conv.shape
pf.section(f32_to_bf16(conv.reshape(CONV_DIM, 4)).tobytes())
alog = src.load(pre + "linear_attn.A_log")
assert alog.shape == (GDN_VH,), alog.shape
pf.section((-np.exp(alog)).astype(np.float32).tobytes())
dtb = src.load(pre + "linear_attn.dt_bias")
assert dtb.shape == (GDN_VH,), dtb.shape
pf.section(f32_to_bf16(dtb).tobytes())
# Gated RMSNorm uses ones-init plain weights: store as-is.
pf.section(f32_to_bf16(src.load(pre + "linear_attn.norm.weight"))
.tobytes())
out = src.load(pre + "linear_attn.out_proj.weight")
assert out.shape == (H, ATTN_W), out.shape
pf.section(pack_q4_fused(out))
pf.section(norm1p(src.load(pre + "post_attention_layernorm.weight")))
pack_ffn_tail(src, i, pf)
return pf.close()
# --------------------------------------------------- vision/head/emb ----
VH, VPD, VPOS, VDEPTH, VINT, VPAD, VMERGED, VOUT = \
1152, 1536, 2304, 27, 4304, 4352, 4608, 2048
VP = "model.visual.{}"
def _affine(pf: PackFile, w: np.ndarray, b: np.ndarray):
pf.section(f32_to_bf16(w).tobytes())
pf.section(f32_to_bf16(b).tobytes())
def pack_vision(src: ShardSet, path: Path) -> int:
pf = PackFile(path, b"MDFV0001", VDEPTH, 0)
pw = src.load(VP.format("patch_embed.proj.weight"))
assert pw.shape == (VH, 3, 2, 16, 16), pw.shape
# Runtime patch rows are (channel, temporal, row, col): each 16x16 frame
# appears twice (single image duplicated over temporal taps), matching
# the Conv3d [2, 16, 16] kernel applied to duplicated frames. Row-major
# flatten of (out, channel, temporal, h, w) is exactly that order.
_affine(pf, pw.reshape(VH, VPD),
src.load(VP.format("patch_embed.proj.bias")))
pos = src.load(VP.format("pos_embed.weight"))
assert pos.shape == (VPOS, VH), pos.shape
pf.section(f32_to_bf16(pos).tobytes())
for blk in range(VDEPTH):
b = VP.format(f"blocks.{blk}.")
n1w = src.load(b + "norm1.weight")
pf.section(f32_to_bf16(n1w).tobytes())
pf.section(f32_to_bf16(src.load(b + "norm1.bias")).tobytes())
_affine(pf, src.load(b + "attn.qkv.weight"),
src.load(b + "attn.qkv.bias"))
_affine(pf, src.load(b + "attn.proj.weight"),
src.load(b + "attn.proj.bias"))
pf.section(f32_to_bf16(src.load(b + "norm2.weight")).tobytes())
pf.section(f32_to_bf16(src.load(b + "norm2.bias")).tobytes())
f1w = src.load(b + "mlp.linear_fc1.weight")
assert f1w.shape == (VINT, VH), f1w.shape
f1 = np.zeros((VPAD, VH), dtype=np.float32)
f1[:VINT] = f1w
f1b = np.zeros((VPAD,), dtype=np.float32)
f1b[:VINT] = src.load(b + "mlp.linear_fc1.bias")
_affine(pf, f1, f1b)
f2w = src.load(b + "mlp.linear_fc2.weight")
assert f2w.shape == (VH, VINT), f2w.shape
f2 = np.zeros((VH, VPAD), dtype=np.float32)
f2[:, :VINT] = f2w
_affine(pf, f2, src.load(b + "mlp.linear_fc2.bias"))
pf.section(f32_to_bf16(src.load(VP.format("merger.norm.weight"))).tobytes())
pf.section(f32_to_bf16(src.load(VP.format("merger.norm.bias"))).tobytes())
_affine(pf, src.load(VP.format("merger.linear_fc1.weight")),
src.load(VP.format("merger.linear_fc1.bias")))
_affine(pf, src.load(VP.format("merger.linear_fc2.weight")),
src.load(VP.format("merger.linear_fc2.bias")))
return pf.close()
def pack_head_emb(src: ShardSet, out_dir: Path):
head = PackFile(out_dir / "head.bin", b"MDFM0002", LAYERS, 2)
head.section(norm1p(src.load("model.language_model.norm.weight")))
lm = src.load("lm_head.weight")
assert lm.shape == (VOCAB, H), lm.shape
head.section(pack_q4_fused(lm))
n_head = head.close()
emb = PackFile(out_dir / "embedding.bin", b"MDFE0001", VOCAB, H)
et = src.load("model.language_model.embed_tokens.weight")
assert et.shape == (VOCAB, H), et.shape
for blob in pack_q4_embed(et):
emb.section(blob)
n_emb = emb.close()
return n_head, n_emb
# ------------------------------------------------------------- manifest ----
def sha256_file(path: Path) -> str:
h = hashlib.sha256()
with open(path, "rb") as f:
while chunk := f.read(8 * 1024 * 1024):
h.update(chunk)
return h.hexdigest()
def main():
root = Path(__file__).resolve().parent
src = ShardSet(root / "src-thomson", "model.safetensors.index.json")
out = root / "out" / "Thomson-1.0-Small-Splash"
(out / "target").mkdir(parents=True, exist_ok=True)
(out / "draft").mkdir(parents=True, exist_ok=True)
(out / "vision").mkdir(parents=True, exist_ok=True)
(out / "tokenizer").mkdir(parents=True, exist_ok=True)
base_manifest = json.loads((root / "moe-manifest.json").read_text())
expect = {r["path"]: r["size"] for r in base_manifest["artifacts"]}
layer_types = base_manifest["target"]["layer_types"]
assert len(layer_types) == LAYERS
for i in range(LAYERS):
full = layer_types[i] == "attention"
dest = out / "target" / f"layer-{i}.bin"
key = f"target/layer-{i}.bin"
if dest.exists() and dest.stat().st_size == expect[key]:
print(f"[SKIP] {key} already packed", flush=True)
continue
n = pack_layer(src, out / "target", i, full)
key = f"target/layer-{i}.bin"
status = "OK " if n == expect[key] else "SIZE-MISMATCH"
print(f"[{status}] {key} {n} (expected {expect[key]})", flush=True)
assert n == expect[key], key
n_head, n_emb = pack_head_emb(src, out / "target")
for key, n in (("target/head.bin", n_head),
("target/embedding.bin", n_emb)):
print(f"[{'OK ' if n == expect[key] else 'SIZE-MISMATCH'}] {key} {n}",
flush=True)
assert n == expect[key], key
n_vis = pack_vision(src, out / "vision" / "model.bin")
print(f"vision/model.bin {n_vis}", flush=True)
import shutil
draft_map = [(f"draft-layer-{i}.bin", f"draft/layer-{i}.bin")
for i in range(6)] + [("draft-model.bin", "draft/model.bin")]
for local, dest in draft_map:
shutil.copyfile(root / "base-draft" / local, out / dest)
tok_map = [("config.json", "tokenizer/config.json"),
("chat_template.jinja", "tokenizer/chat_template.jinja"),
("tokenizer.json", "tokenizer/tokenizer.json"),
("tokenizer_config.json", "tokenizer/tokenizer_config.json"),
("vocab.json", "tokenizer/vocab.json")]
for local, dest in tok_map:
shutil.copyfile(root / "src-thomson" / local, out / dest)
records = []
for sub in ("draft", "target", "vision", "tokenizer"):
for path in sorted((out / sub).rglob("*")):
if path.is_file():
rel = path.relative_to(out).as_posix()
records.append({"path": rel, "size": path.stat().st_size,
"sha256": sha256_file(path)})
manifest = {
"format": base_manifest["format"],
"schema_version": 4,
"model": "Thomson-1.0-Small",
"execution_geometry": base_manifest["execution_geometry"],
"target": base_manifest["target"],
"draft": base_manifest["draft"],
"converter": {
"tool": "splash-thomson-pack/convert.py",
"method": "full BF16->packed conversion (group-64 affine Q4/Q8, "
"StorageN=256, low-nibble-first)",
"moe_experts": "per-expert [nibbles++scales++biases] slabs "
"(readExpertQ4Projection); shared expert and "
"router stay flat fused",
"target_text_norms": "zero-centered RMSNorm stored as (1+w); "
"gated mixer norm stored as-is",
"target_decay": "fp32 -exp(A_log)",
"vision_patch": "row-major Conv3d [2,16,16] weights; runtime "
"patch rows are (channel, temporal, row, col) over "
"the duplicated frame",
"shared_scalar_gate": "single BF16 row tiled 256x then Q8",
},
"upstream": {
"target": {"repo_id": "thomsonreuters/Thomson-1.0-Small",
"revision": "6f58dd819061240192285228e8c0ffcffd0fc665"},
"draft": base_manifest["upstream"]["draft"],
"tokenizer": {"repo_id": "thomsonreuters/Thomson-1.0-Small",
"revision": "6f58dd819061240192285228e8c0ffcffd0fc665"},
"vision": {"repo_id": "thomsonreuters/Thomson-1.0-Small",
"revision": "6f58dd819061240192285228e8c0ffcffd0fc665"},
},
"artifacts": records,
}
manifest["artifact_set_sha256"] = hashlib.sha256(
"\n".join(f"{r['path']} {r['size']} {r['sha256']}"
for r in records).encode()).hexdigest()
(out / "manifest.json").write_text(json.dumps(manifest, indent=2) + "\n")
print(f"wrote {out / 'manifest.json'} with {len(records)} artifacts")
if __name__ == "__main__":
sys.exit(main())

Recipe: local Thomson-1.0-Small on Splash, driven by omp and Paseo

This recipe runs thomsonreuters/Thomson-1.0-Small locally on Splash (Apple silicon). It then connects it to omp as a coding agent, to Paseo as the agent app, and optionally to paseo-spacedock as a decision layer. Only code and instructions are shared. Every operator downloads the weights from upstream and converts them on their own machine.

Layer What you get Required?
1. Core (steps 0–7) Thomson served by Splash on 127.0.0.1:8001 yes
2. omp (step 8) a coding agent that uses the local Thomson recommended
3. Paseo (step 9) desktop/mobile app driving that omp agent optional
4. paseo-spacedock (step 10) gate panel whose judge is the local Thomson optional

If you are an agent following this recipe

  • Run the steps in order. Every step ends with a Check. If a check fails, stop and report it. Don't improvise around it.
  • STOP FOR HUMAN marks a decision only the human can make. Ask, wait for an explicit yes, then continue. Never accept a license on the human's behalf.
  • Stop after the layer the human asked for. Layers 3 and 4 are opt-in.
  • Never upload, commit, push, sync, or copy anything under src-thomson/, out/, or the Splash snapshot directory (step 6). Never run hf upload on them.
  • Use the exact pinned versions and commits below. Don't replace a pin with "latest".

Security notes (read once)

  • Model name: use an owner you control. Splash only serves owner/repo names. If the local package ever fails its check at startup (moved out/, truncated file), the installer falls back to downloading owner/repo from Hugging Face. So never invent an owner such as local: that is a real HF user, and whoever controls it could publish a model under your name and have your agent run it. Use your own HF username as the owner (a free account reserves it), and never create that repo on Hugging Face.
  • Serve with HF_HUB_OFFLINE=1. With it set, a failed local check stops with offline mode is enabled instead of downloading anything. Steps 6 and 7 and the launchd plist set it.
  • Everything executed is pinned: Hugging Face revisions, the Splash source commit that verify.py imports, Python package versions, and the paseo-spacedock commit. The brew taps (incoai/tap, can1357/tap, spacedock-dev/tap) follow their publishers' releases. Trust those publishers or build from source.
  • This gist can change. If you were given a revision-pinned link, check out that revision (step 0), and compare the script hashes below against the ones published with the link.
  • convert.py and verify.py read safetensors with their own parser. They don't use pickle or torch, so tampered weights can't run code during conversion.
  • Splash rejects requests whose Host isn't local and cross-origin browser requests, so web pages can't drive 127.0.0.1:8001. Local processes can. Set SPLASH_API_KEY if that matters on your machine.

What gets shared, what stays local

Item Where it comes from Share it?
convert.py, verify.py, this RECIPE.md this gist yes
Upstream repo IDs, pinned revisions, expected hashes this recipe yes
src-thomson/: Thomson BF16 shards and tokenizer, 65 GB operator downloads from HF no
out/Thomson-1.0-Small-Splash/: packed Splash model, ~20 GB operator runs convert.py no
Splash snapshot and model link (step 6) operator creates them no
base-draft/, moe-manifest.json operator downloads from HF no need to (Apache-2.0, every operator fetches them anyway)

License (read before step 2)

Thomson-1.0-Small is released under PolyForm Strict 1.0.0. That license:

  • permits use only for permitted purposes: noncommercial purposes, personal research/experiment/testing, and use by noncommercial organizations (charities, education, public research, public safety/health, environmental, government);
  • forbids distributing the software; and
  • forbids "making changes or new works based on the software."

This recipe covers the first two restrictions: nothing is distributed. It does not resolve the third. Repacking BF16 weights into 4-bit Splash format on your own machine can plausibly be read as "making changes". For institutional use, check with the licensor or counsel. Splash, the Qwen3.6 draft package (incoai/Qwen3.6-35B-A3B-Splash), Paseo, and Spacedock are Apache-2.0. omp is MIT. paseo-spacedock (v0.1.0) is CC0 1.0 (public domain).

Pinned versions

Role Source Pin
Target, tokenizer, vision thomsonreuters/Thomson-1.0-Small 6f58dd819061240192285228e8c0ffcffd0fc665
Draft and geometry template incoai/Qwen3.6-35B-A3B-Splash 0f4714b2db37b5f3c42a10de07281e74f88e4adc
Splash brew install incoai/tap/splash 1.0.2 or newer (first release with --port and /v1/systemone)
Splash source (read-only, for verify.py) incoai/splash (tag 1.0.2) e8fffde2c3a1d1c4120028d9e5399bb917b8b917
Python packages PyPI numpy==2.5.3, huggingface_hub==1.28.0
paseo-spacedock audreyt/paseo-spacedock release v0.1.0 7aea6066f8434c811a3b5d06970dd9c18e7b89df

convert.py writes the Thomson revision into the package manifest's upstream block, so download exactly that revision.

Script hashes for this revision (SHA-256):

convert.py  883f49dfed4e8d6c701d83660aa666c9d7585c9bc0c2770b13e238d614413fe4
verify.py   84736a2c4e700b26db45cd98ac6e8bfbdb4e1730d0ff91b20416aa3dc31e27d5

Tested with: brew Splash 1.0.2 (steps 6–7: installer check, offline serve, chat, /v1/systemone); the omp tool-call probe (step 8) on omp 18.2.11; Paseo 0.9.1; Spacedock 0.27.3.


Layer 1: Core

0. Preflight

sw_vers -productVersion                                # need 26.4 or newer
sysctl -n machdep.cpu.brand_string                     # need Apple M3 or newer
echo "$(( $(sysctl -n hw.memsize) / 1073741824 )) GB"  # need 36+ (48+ recommended)
df -h "$HOME"                                          # need ~90 GB free
python3 --version                                      # need 3.12–3.14

The ~90 GB covers 65 GB of BF16 source, 20 GB packed, and 0.5 GB of draft. You can delete the BF16 source after step 5 if you won't rebuild.

STOP FOR HUMAN. Ask for the human's Hugging Face username (or an org they control). It becomes the model's owner name. Never use local or any name they don't own.

Set up the tools and folders:

export HF_OWNER=your-hf-username      # replace: your own HF username or org
export MODEL_ID="$HF_OWNER/Thomson-1.0-Small-Splash"
export W=~/w                          # any parent dir
export PACK=$W/splash-thomson-pack
export MODELS_DIR=~/Models            # where the big downloads go
mkdir -p "$W" "$MODELS_DIR"

python3 -m venv ~/.venvs/thomson-pack
~/.venvs/thomson-pack/bin/pip install numpy==2.5.3 huggingface_hub==1.28.0
export PATH=~/.venvs/thomson-pack/bin:$PATH   # provides python3 (with numpy) and hf

cd "$W"
git clone https://gist.github.com/beb06fea21bebcc5d01bdbb49680eaf0.git splash-thomson-pack
# If you were given a revision-pinned link: git -C splash-thomson-pack checkout --detach <revision>
git clone --filter=blob:none https://github.com/incoai/splash.git splash
git -C splash checkout --detach e8fffde2c3a1d1c4120028d9e5399bb917b8b917
printf 'src-thomson\nbase-draft\nout\n__pycache__\n' > "$PACK/.gitignore"

verify.py imports ../splash/install/models.py, so the Splash source clone must sit next to the pack folder. It doesn't need to be built. The .gitignore matters because the gist clone is a git repo: it keeps weights out of any accidental commit or push.

Check:

  • every preflight value meets its minimum;
  • curl -s -o /dev/null -w '%{http_code}\n' "https://huggingface.co/api/models/$MODEL_ID" does not print 200 (the repo must not exist; 401 is the normal anonymous answer for a missing repo);
  • python3 -c "import numpy" and hf version both succeed;
  • git -C "$W/splash" rev-parse HEAD prints e8fffde2c3a1d1c4120028d9e5399bb917b8b917;
  • shasum -a 256 "$PACK/convert.py" "$PACK/verify.py" matches the script hashes above.

1. Install Splash

brew install incoai/tap/splash

Check: splash serve --help | grep -- --port prints the --port option. If it doesn't, run brew upgrade splash. Versions before 1.0.1 lack --port, and versions before 1.0.2 lack /v1/systemone.

2. Download Thomson BF16 (pinned)

STOP FOR HUMAN. Confirm that the human has read the PolyForm Strict license above, that their use is a permitted purpose, and that they accept the "making changes" question for step 4. Continue only on an explicit yes.

hf download thomsonreuters/Thomson-1.0-Small \
  --revision 6f58dd819061240192285228e8c0ffcffd0fc665 \
  --local-dir "$MODELS_DIR/Thomson-1.0-Small"

ln -sfn "$MODELS_DIR/Thomson-1.0-Small" "$PACK/src-thomson"

Check: ls "$PACK"/src-thomson/model-*.safetensors | wc -l prints 16, and model.safetensors.index.json, config.json, chat_template.jinja, tokenizer.json, tokenizer_config.json, and vocab.json are all present.

3. Download the draft and geometry template (pinned, ~0.5 GB)

Thomson has the same geometry as Qwen3.6-35B-A3B (40 layers, hidden 2048, vocab 248320, 256 experts, top-8, MoE width 512). The pack reuses the official Qwen3.6 Splash package's DFlash2 draft unchanged and takes its manifest.json as the size template. Download only those files, not the 20 GB Qwen target:

Q=$MODELS_DIR/Qwen3.6-35B-A3B-Splash-draft
hf download incoai/Qwen3.6-35B-A3B-Splash \
  --revision 0f4714b2db37b5f3c42a10de07281e74f88e4adc \
  --local-dir "$Q" \
  manifest.json draft/model.bin \
  draft/layer-0.bin draft/layer-1.bin draft/layer-2.bin \
  draft/layer-3.bin draft/layer-4.bin draft/layer-5.bin

cp "$Q/manifest.json" "$PACK/moe-manifest.json"
mkdir -p "$PACK/base-draft"
for i in 0 1 2 3 4 5; do
  ln -sfn "$Q/draft/layer-$i.bin" "$PACK/base-draft/draft-layer-$i.bin"
done
ln -sfn "$Q/draft/model.bin" "$PACK/base-draft/draft-model.bin"

Check: shasum -a 256 "$PACK/moe-manifest.json" prints 22aa0f68a76fa8b84245b81eb1606f25aaa002259663f84f361fd7c66f244418.

4. Convert

cd "$PACK"
python3 convert.py

This writes out/Thomson-1.0-Small-Splash/{target,draft,vision,tokenizer,manifest.json} (55 artifacts, ~20 GB). Each target layer prints [OK] … (expected …), and a size mismatch stops the run. Reruns skip layers that are already packed at the right size. It takes a few minutes on recent Apple silicon.

What the converter does:

  • Group-64 affine Q4 for projections and Q8 for the router and shared-expert gate, in Splash's StorageN=256 tiling, low nibble first.
  • MoE experts are packed as per-expert slabs ([nibbles ++ scales ++ biases] per expert). A flat fused pack has the same total size, so every size check passes, but the runtime then decodes nibble bytes as bf16 scales, the logits go all-NaN, and bootstrap fails with the sampler's 0xffffffff sentinel. Keep the slab layout.
  • Zero-centered text RMSNorms are stored as (1 + w). The gated mixer norm is stored as-is. GDN decay is stored as fp32 -exp(A_log).
  • Tokenizer and vision weights come from Thomson. The draft comes from the Qwen3.6 package byte for byte.

Check: the run ends with wrote …/manifest.json with 55 artifacts.

5. Verify

cd "$PACK"
python3 verify.py

It checks the manifest schema, full SHA-256 of every artifact, that the draft is identical to the template, 16 KiB alignment, Q4 dequant error against the BF16 source (layers 0, 3, 39, including expert slab 0), norm means near 1.0, and the tokenizer geometry.

Check: the last line is ALL VERIFY CHECKS PASSED.

Optional reproducibility check: a bit-identical build has

artifact_set_sha256 = 6f01760271cd6d0eda797e8c1711ede50d6e0b478e915024ed0a00537b35ca53
python3 -c "import json;print(json.load(open('out/Thomson-1.0-Small-Splash/manifest.json'))['artifact_set_sha256'])"

A mismatch alongside a passing verify.py is not necessarily an error, but mention it when you report problems.

6. Install into Splash as a local Hub snapshot

splash serve accepts only owner/repo IDs. Its installer treats a model as installed only if <models>/<owner>/<repo> resolves to a directory shaped like a Hub snapshot, models--<owner>--<repo>/snapshots/<40-hex>/. Otherwise it tries to download from Hugging Face (see Security notes). So build a snapshot out of per-file symlinks into out/, under your own owner name:

PKG=$PACK/out/Thomson-1.0-Small-Splash
SNAP=${HF_HUB_CACHE:-$HOME/.cache/huggingface/hub}/models--${HF_OWNER}--Thomson-1.0-Small-Splash/snapshots/0000000000000000000000000000000000000000

mkdir -p "$SNAP"
(cd "$PKG" && find . -type f) | while read -r f; do
  mkdir -p "$SNAP/$(dirname "$f")"
  ln -sfn "$PKG/${f#./}" "$SNAP/${f#./}"
done

SPLASH_MODELS="$HOME/Library/Application Support/Splash/models"   # brew install
mkdir -p "$SPLASH_MODELS/$HF_OWNER"
ln -sfn "$SNAP" "$SPLASH_MODELS/$HF_OWNER/Thomson-1.0-Small-Splash"

If you run Splash from a source checkout instead of brew, use SPLASH_MODELS=<checkout>/install/models. The installer adds a pin under refs/splash/ next to snapshots/, which is expected. The snapshot holds symlinks, not copies, so don't move or delete out/ while the model is installed.

Check:

SB=$(brew --prefix splash)/libexec
HF_HUB_OFFLINE=1 "$SB/python/bin/python3" "$SB/install/models.py" \
  --model "$MODEL_ID" verify --full

prints Splash model <your-owner>/Thomson-1.0-Small-Splash preflight passed (full).

7. Serve

HF_HUB_OFFLINE=1 splash serve --model "$MODEL_ID" --port 8001

Always serve this model with HF_HUB_OFFLINE=1 (see Security notes). Port 8001 leaves Splash's default 8000 free for another model. Any free port works, as long as you use the same one in steps 8 and 10. The first start maps ~20 GB into unified memory. Wait for Ready.

Check (from another terminal with MODEL_ID set):

curl -s http://127.0.0.1:8001/v1/models      # lists your MODEL_ID

curl -s http://127.0.0.1:8001/v1/chat/completions \
  -H 'Content-Type: application/json' -d @- <<EOF
{"model":"$MODEL_ID","max_tokens":2048,
 "messages":[{"role":"user","content":"Summarize the IRAC method in two sentences."}]}
EOF

curl -s http://127.0.0.1:8001/v1/systemone \
  -H 'Content-Type: application/json' -d @- <<EOF
{"model":"$MODEL_ID",
 "state":{"message":"I was charged twice. Please fix this today."},
 "questions":{"department":{"type":"choice",
   "instructions":"Which team should handle this?",
   "criteria":{"billing":null,"technical":null,"sales":null}}}}
EOF

The chat reply is coherent. The systemone reply has answers.department.choice = billing.

Thomson is a reasoning model. The trace comes back in reasoning_content. With a small max_tokens, the reply can end with finish_reason: "length" and content: null, so give it a generous budget, or send "reasoning_effort": "none". Context capacity is up to 262,144 tokens, limited by available memory.

The draft was trained for Qwen3.6, not Thomson. Speculative decoding still has the target verify every token, so output quality is Thomson's. Only speed depends on how often the draft's proposals get accepted.

Optional: keep it running with launchd

~/Library/LaunchAgents/local.splash-thomson.plist (replace YOUR_HF_OWNER and YOU):

<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
  <key>Label</key><string>local.splash-thomson</string>
  <key>ProgramArguments</key>
  <array>
    <string>/opt/homebrew/bin/splash</string>
    <string>serve</string>
    <string>--model</string><string>YOUR_HF_OWNER/Thomson-1.0-Small-Splash</string>
    <string>--port</string><string>8001</string>
  </array>
  <key>EnvironmentVariables</key>
  <dict>
    <key>HF_HUB_OFFLINE</key><string>1</string>
  </dict>
  <key>RunAtLoad</key><true/>
  <key>KeepAlive</key><true/>
  <key>ThrottleInterval</key><integer>30</integer>
  <key>StandardOutPath</key><string>/Users/YOU/Library/Logs/splash-thomson.log</string>
  <key>StandardErrorPath</key><string>/Users/YOU/Library/Logs/splash-thomson.log</string>
</dict>
</plist>
plutil -lint ~/Library/LaunchAgents/local.splash-thomson.plist
launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/local.splash-thomson.plist

Layer 2: omp

8. Use the local Thomson from omp

Install omp (brew install can1357/tap/omp, or the official installer at omp.sh) and set up at least one provider the way omp's first run asks.

omp's built-in vllm provider only talks to port 8000, so add a dedicated provider. Put this under the top-level providers: key in ~/.omp/agent/models.yml (create the file with a providers: line if it doesn't exist), replacing YOUR_HF_OWNER:

providers:
  splash-thomson:
    baseUrl: http://127.0.0.1:8001/v1
    auth: none
    api: openai-completions
    models:
      - id: YOUR_HF_OWNER/Thomson-1.0-Small-Splash
        name: Thomson 1.0 Small (Splash)
        reasoning: true
        supportsTools: true
        input: [text, image]
        contextWindow: 262144
        maxTokens: 32768
        cost:
          input: 0
          output: 0
          cacheRead: 0
          cacheWrite: 0

Set contextWindow to what your Splash reports: curl -s http://127.0.0.1:8001/status → maximum_context_tokens.

Check: omp drives a real tool call through Thomson:

mkdir -p /tmp/thomson-probe && cd /tmp/thomson-probe && echo ZEBRA-4417 > probe.txt
omp -p --model "splash-thomson/$MODEL_ID" \
  "Use the read tool on probe.txt and report its exact contents."

The reply contains ZEBRA-4417.


Layer 3: Paseo (optional)

9. Drive the omp agent from Paseo

Install Paseo: the desktop app from paseo.sh/download, or npm install -g @getpaseo/cli for a headless daemon. Paseo supports omp natively and runs the omp installed in step 8 with its own models.yml. If omp is turned off, enable it in Paseo's provider settings, or set "agents": {"providers": {"omp": {"enabled": true}}} in ~/.paseo/config.json and restart the daemon.

In Paseo, start a new OMP agent and pick the model splash-thomson/<your MODEL_ID>. You can also save that choice as an agent profile.

If paseo isn't on PATH with the desktop app, the bundled CLI is at /Applications/Paseo.app/Contents/Resources/bin/paseo.

Check: the Paseo agent answers the same probe prompt from step 8 with ZEBRA-4417.


Layer 4: paseo-spacedock (optional)

10. Gate decisions judged by the local Thomson

paseo-spacedock adds Spacedock's decision layer to Paseo: a panel of stage gates with Approve / Revise / Hold, and a Judge button that asks a TypeSafe System One model for a typed recommendation. The plugin doesn't depend on Splash: its judge works with any TypeSafe-compatible /v1/systemone endpoint (the hosted Jev by default). This step only points it at the local Thomson, which Splash serves from 1.0.2.

STOP FOR HUMAN. A Paseo plugin runs with the daemon's permissions. Confirm the human wants to install it.

  1. Paseo 0.8.0 or newer. Set "pluginsEnabled": true in ~/.paseo/config.json and restart the daemon.

  2. Install Spacedock and the plugin, pinned to the v0.1.0 release commit:

    brew tap spacedock-dev/tap && brew install spacedock
    paseo plugin install https://github.com/audreyt/paseo-spacedock.git \
      --ref 7aea6066f8434c811a3b5d06970dd9c18e7b89df

    The commit pin can't be moved the way a tag can. It's the commit that release v0.1.0 points to.

  3. In the plugin's settings screen, set:

    • TypeSafe base URL: http://127.0.0.1:8001
    • Judge model: your MODEL_ID
    • TypeSafe API key: leave empty (Splash authentication is off by default), or your SPLASH_API_KEY if you set one

    Use the settings screen rather than TYPESAFE_BASE_URL / TYPESAFE_DEFAULT_MODEL environment variables. A daemon started by the desktop app doesn't read your shell rc files.

  4. Open a workspace, then run /spacedock. If the repo has no commissioned workflow, the panel offers to bootstrap one.

Check: on a pending gate, Judge shows a verdict with confidence, evidence, and risk, and the model shown is your MODEL_ID.

The judge is advisory. The plugin never records a decision by itself, and Splash's own docs say these scores are local model scores, not calibrated confidence. Before trusting the plugin's delegate outcomes, measure how Thomson's judgments line up with your own calls on real gates.


Troubleshooting

Symptom Cause
error: could not download … offline mode is enabled The local package failed its check and HF_HUB_OFFLINE=1 correctly blocked a Hub download. Redo step 6 (is out/ still there? does SPLASH_MODELS match how you installed Splash?). Don't unset HF_HUB_OFFLINE to "fix" it.
refusing to replace non-symlink model path <models>/<owner>/Thomson-1.0-Small-Splash is a real directory. Move it aside and make it a symlink.
unrecognized arguments: --port Splash is older than 1.0.1. Run brew upgrade splash.
/v1/systemone returns 404 Splash is older than 1.0.2. Run brew upgrade splash.
All-NaN logits / bootstrap fails with 0xffffffff Experts packed flat instead of per-expert slabs. Use this convert.py unchanged and rerun verify.py.
draft drift in verify.py Wrong Qwen3.6 package revision in step 3.
SIZE-MISMATCH in convert.py Wrong Thomson revision, or moe-manifest.json isn't the pinned Qwen3.6 manifest.
ModuleNotFoundError: No module named 'install' in verify.py The Splash source clone isn't a sibling of the pack folder (step 0).
Startup prints a memory budget and exits Not enough unified memory. Stop other models or pass --max-context.
omp reply empty or finish_reason: length The reasoning budget ran out. Raise maxTokens, or lower the thinking level.
#!/usr/bin/env python3
"""Validate a packed Thomson Splash package (project collateral, rerunnable)."""
import json
import struct
import sys
from pathlib import Path
import numpy as np
ROOT = Path(__file__).resolve().parent
sys.path.insert(0, str(ROOT.parent / "splash"))
sys.path.insert(0, str(ROOT))
from install.models import ( # noqa: E402
validate_package_manifest, verify_artifacts)
from convert import ( # noqa: E402
ShardSet, bf16_to_f32, ALIGN, H, LAYERS, VOCAB)
OUT = ROOT / "out" / "Thomson-1.0-Small-Splash"
def read_section(path: Path, offset: int, n: int) -> bytes:
with open(path, "rb") as f:
f.seek(offset)
return f.read(n)
def dequant_q4_section(blob: bytes, O: int, I: int):
T, G = O // 256, I // 64
W = O * I // 2
wb = np.frombuffer(blob[:W], dtype=np.uint8).reshape(T, G, 256, 32)
s = bf16_to_f32(blob[W:W + O * I // 32]).reshape(T, G, 256)
b = bf16_to_f32(blob[W + O * I // 32:]).reshape(T, G, 256)
out = np.empty((O, I), np.float32)
for t in range(T):
for g in range(G):
cols = np.arange(64)
nib = ((wb[t, g, :, cols // 2].astype(np.int32).T
>> (cols % 2) * 4) & 15).astype(np.float32)
out[t * 256:(t + 1) * 256, g * 64:(g + 1) * 64] = \
nib * s[t, g][:, None] + b[t, g][:, None]
return out
def main():
manifest = json.loads((OUT / "manifest.json").read_text())
validate_package_manifest(OUT / "manifest.json")
print("manifest schema: OK")
verify_artifacts(OUT, manifest, full=True)
print(f"artifacts full SHA256: OK ({len(manifest['artifacts'])} files)")
base = json.loads((ROOT / "moe-manifest.json").read_text())
base_draft = {r["path"]: r["sha256"] for r in base["artifacts"]
if r["path"].startswith("draft/")}
mine = {r["path"]: r["sha256"] for r in manifest["artifacts"]}
assert all(mine[k] == v for k, v in base_draft.items()), "draft drift"
print("draft bytes identical to base package: OK")
for name in sorted((OUT / "target").glob("*.bin")):
with open(name, "rb") as f:
magic, layer, typ = f.read(8), struct.unpack("<I", f.read(4))[0], \
struct.unpack("<I", f.read(4))[0]
size = name.stat().st_size
assert size % ALIGN == 0, name
print("headers + 16KiB alignment: OK")
# Value spot checks against BF16 source (layers 0, 3, 39).
src = ShardSet(ROOT / "src-thomson", "model.safetensors.index.json")
pre = "model.language_model.layers.{}.{}"
def off_after(*sizes):
off = 16
for s in sizes:
off = (off + ALIGN - 1) & ~(ALIGN - 1)
off += s
return (off + ALIGN - 1) & ~(ALIGN - 1)
off_gdn = off_after(H * 2, 12544 * H * 9 // 16, 8192 * 4 * 2, 32 * 4,
32 * 2, 128 * 2)
off_attn = off_after(H * 2, 9216 * H * 9 // 16, 256 * 2, 256 * 2)
assert off_gdn == 14598144, off_gdn # matches official pack layout
worst = 0.0
for i, full in ((0, False), (3, True), (39, True)):
raw = (OUT / "target" / f"layer-{i}.bin").read_bytes()
if full:
tag = "self_attn"
o = src.load(pre.format(i, f"{tag}.o_proj.weight"))
sec = raw[off_attn:off_attn + 2048 * 4096 * 9 // 16]
else:
tag = "linear_attn"
o = src.load(pre.format(i, f"{tag}.out_proj.weight"))
sec = raw[off_gdn:off_gdn + 2048 * 4096 * 9 // 16]
dq = dequant_q4_section(sec, 2048, 4096)
err = float(np.abs(dq - o).max())
worst = max(worst, err)
print(f"layer {i} {tag}.out_proj: dequant maxerr={err:.5f}")
assert worst < 0.05, worst
# MoE experts are per-expert slabs ([nibbles ++ scales ++ biases] per
# expert, read by readExpertQ4Projection), NOT one flat fused section:
# flat packing has the same total size but puts all nibbles before all
# scales, so slab reads decode nibble bytes as bf16 scales (inf), the
# logits go all-NaN, and bootstrap fails with the sampler's 0xffffffff
# sentinel. Dequantize slab 0 in place and compare to source expert 0.
def q4bytes(O, I):
return O * I // 2 + 2 * (O * I // 32)
def q8bytes(O, I):
return O * I + 4 * (O * I // 64)
for i, full in ((0, False), (3, True), (39, True)):
raw = (OUT / "target" / f"layer-{i}.bin").read_bytes()
if full:
sizes = [H * 2, q4bytes(9216, H), 256 * 2, 256 * 2,
q4bytes(H, 4096), H * 2, q8bytes(256, H)]
else:
sizes = [H * 2, q4bytes(12544, H), 8192 * 4 * 2, 32 * 4,
32 * 2, 128 * 2, q4bytes(H, 4096), H * 2,
q8bytes(256, H)]
gate_off = off_after(*sizes)
slab = q4bytes(512, H)
dq = dequant_q4_section(raw[gate_off:gate_off + slab], 512, H)
gu = src.load(pre.format(i, "mlp.experts.gate_up_proj"))
err = float(np.abs(dq - gu[0, 0:512, :]).max())
print(f"layer {i} experts-gate slab0: dequant maxerr={err:.5f}")
assert err < 0.05, (i, err)
dslab = q4bytes(H, 512)
down_off = off_after(*sizes, 256 * slab, 256 * slab)
dq = dequant_q4_section(raw[down_off:down_off + dslab], H, 512)
down = src.load(pre.format(i, "mlp.experts.down_proj"))
err = float(np.abs(dq - down[0]).max())
print(f"layer {i} experts-down slab0: dequant maxerr={err:.5f}")
assert err < 0.05, (i, err)
# Norms stored as (1+w): mean must sit near 1, not near 0.
nrm = bf16_to_f32(read_section(OUT / "target" / "layer-0.bin", 16384, 4096))
print(f"input-norm mean={nrm.mean():.4f} (expect ~1.0)")
assert 0.9 < nrm.mean() < 1.1
tok = json.loads((OUT / "tokenizer" / "config.json").read_text())
t = tok["text_config"]
assert (t["model_type"], t["hidden_size"], t["vocab_size"],
t["max_position_embeddings"]) == \
("qwen3_5_moe_text", 2048, 248320, 262144), t
print("tokenizer config: OK")
print("ALL VERIFY CHECKS PASSED")
if __name__ == "__main__":
sys.exit(main())
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment