Janus-35B-HERETIC / scripts /verify_arch.py
FoolDev's picture
Claude Opus 5 (1M context)
release 0.9.9: second audit tier — scripts that lied, and card claims that were false
e023532
Raw History Blame Contribute Delete
8.06 kB
#!/usr/bin/env python3
"""
Janus-35B — verify the README "Architecture" claims and the MoE
forward-pass structure against the actual GGUF metadata.
Reads the qwen35moe-stamped bundle (or any GGUF that declares that
`general.architecture` value), prints each claim alongside the metadata key
it derives from, and exits non-zero if any value mismatches the expected
claim. Useful as a manual audit after the bundle is re-stamped or after
upstream re-conversion.
Usage:
python3 scripts/verify_arch.py # default bundle
python3 scripts/verify_arch.py Janus-35B-A3B.Q4_K_M.gguf
python3 scripts/verify_arch.py /path/to/some-other.gguf
Exit code 0 = all claims verify, 1 = at least one mismatch, 2 = no readable
GGUF (missing, or an un-smudged git-LFS pointer) / usage error.
Requires: pip install gguf
Note: this does NOT verify the ~34.7B-total / ~3B-active parameter counts
directly (no such KV in the GGUF) — they follow from the layer count, hidden
size and the expert count/width below and llama.cpp's `qwen35moe` type branch,
not from a single metadata field.
`block_count` is checked against 40, which is this model's natural depth (10
Gated-Attention + 30 Gated-DeltaNet layers), NOT a post-strip value: the
bundled quant carries no MTP/NextN tensors, so scripts/strip_mtp.py is a
defensive no-op for it. llmfan46's separate *-Native-MTP-Preserved-GGUF variant
— which this repo does not ship — reports 41 here, with a NextN block at index
40 that stock llama.cpp / Ollama cannot load.
"""
from __future__ import annotations
import sys
from pathlib import Path
from gguf import GGUFReader
EXPECTED = {
"block_count": (40, "40 transformer layers"),
"context_length": (262144, "262 144 native context"),
"embedding_length": (2048, "Hidden size 2048"),
"expert_count": (256, "MoE: 256 experts"),
"expert_used_count": (8, "MoE: 8 experts active per token"),
"expert_feed_forward_length": (512, "MoE: per-expert FFN 512"),
"expert_shared_feed_forward_length": (512, "MoE: shared-expert FFN 512"),
"attention.head_count": (16, "Gated Attention: 16 Q-heads"),
"attention.head_count_kv": (2, "Gated Attention: 2 KV-heads (GQA)"),
"attention.key_length": (256, "Gated Attention: head_dim 256 (key)"),
"attention.value_length": (256, "Gated Attention: head_dim 256 (value)"),
"rope.dimension_count": (64, "Partial RoPE: 64 of 256 dims (factor 0.25)"),
"full_attention_interval": (4, "Hybrid stack: every 4th layer is Gated Attention (10 cycles)"),
"ssm.conv_kernel": (4, "Gated DeltaNet: conv kernel 4"),
"ssm.state_size": (128, "Gated DeltaNet: head_dim 128"),
"ssm.time_step_rank": (32, "Gated DeltaNet: 32 V-heads"),
"ssm.group_count": (16, "Gated DeltaNet: 16 QK-heads"),
"ssm.inner_size": (4096, "Gated DeltaNet: inner size 4096 (32 V-heads x 128)"),
}
EXPECTED_VOCAB = 248320
EXPECTED_ARCHS = {"qwen35moe"}
def read_scalar(reader: GGUFReader, key: str):
f = reader.fields.get(key)
if f is None:
return None
arr = f.parts[f.data[0]]
val = arr.tolist() if hasattr(arr, "tolist") else arr
if isinstance(val, list) and len(val) == 1:
return val[0]
return val
def read_arch(reader: GGUFReader) -> str:
f = reader.fields["general.architecture"]
return bytes(f.parts[f.data[0]]).decode()
def main() -> int:
if len(sys.argv) > 2:
print(f"usage: {sys.argv[0]} [path/to/Janus-35B-A3B.Q4_K_M.gguf]", file=sys.stderr)
return 2
root = Path(__file__).resolve().parent.parent
default_paths = [
root / "Janus-35B-A3B.Q4_K_M.gguf",
]
if len(sys.argv) == 2:
path = Path(sys.argv[1])
else:
path = next((p for p in default_paths if p.exists() and p.stat().st_size > 1024), None)
if path is None:
# A fresh clone made without git-lfs leaves the bundle on disk as a
# ~136-byte pointer. The size filter above correctly refuses it, but
# "not found" is the wrong diagnosis for a file that is right there -
# load_bundle.sh and live_check.sh both detect this case properly.
pointer = next((p for p in default_paths if p.exists()), None)
if pointer is not None:
head = pointer.open("rb").read(64)
if head.startswith(b"version https://git-lfs.github.com"):
print(f"[!] {pointer} is an un-smudged git-LFS pointer, not the weights.",
file=sys.stderr)
print(" Fetch it with 'git lfs pull', or run ./scripts/load_bundle.sh",
file=sys.stderr)
print(" (it downloads into .cache/ and prints the path), then pass that path.",
file=sys.stderr)
return 2
print("[!] no Janus-35B GGUF found in repo root; pass a path explicitly", file=sys.stderr)
return 2
print(f"[*] reading: {path}")
reader = GGUFReader(str(path), "r")
arch = read_arch(reader)
if arch not in EXPECTED_ARCHS:
print(f"[!] unexpected general.architecture: {arch!r} (expected one of {EXPECTED_ARCHS})", file=sys.stderr)
return 1
print(f"[*] general.architecture: {arch}")
print()
mismatches = 0
fmt = " {marker} {claim:55s} {key:45s} = {actual}"
for suffix, (expected, claim) in EXPECTED.items():
key = f"{arch}.{suffix}"
actual = read_scalar(reader, key)
ok = actual == expected
marker = "[ ok ]" if ok else "[FAIL]"
print(fmt.format(marker=marker, claim=claim, key=key, actual=actual))
if not ok:
mismatches += 1
# Vocab count comes from the tokenizer tokens array length, not a scalar KV.
f = reader.fields.get("tokenizer.ggml.tokens")
vocab_actual = len(f.data) if f is not None else None
ok = vocab_actual == EXPECTED_VOCAB
marker = "[ ok ]" if ok else "[FAIL]"
print(fmt.format(marker=marker, claim=f"Vocab {EXPECTED_VOCAB}", key="tokenizer.ggml.tokens (length)", actual=vocab_actual))
if not ok:
mismatches += 1
# ---- embedded chat template vs the repo's chat_template.jinja -------------
# Since 0.7.0 this repo ships NO chat_template.jinja: the bundled GGUF carries the
# Qwen 3.6 base's own embedded template, untouched, so there are no two copies to
# drift and the block below skips. It is kept for the case where a maintainer
# reintroduces a repo-side template (as 0.2.0-0.6.5 had): back then the two did
# drift twice, and a stale embedded copy silently reinstates the tool-calling
# failure for every loader that reads the GGUF instead of an external file.
repo_template = Path(__file__).resolve().parent.parent / "chat_template.jinja"
if repo_template.exists():
f = reader.fields.get("tokenizer.chat_template")
embedded = None
if f is not None and f.parts:
embedded = str(bytes(f.parts[-1]), encoding="utf-8")
expected = repo_template.read_text(encoding="utf-8")
ok = embedded == expected
marker = "[ ok ]" if ok else "[FAIL]"
if embedded is None:
actual = "(absent)"
elif ok:
actual = f"{len(embedded)} chars, identical"
else:
actual = f"{len(embedded)} chars, DIFFERS from the {len(expected)}-char repo file"
print(fmt.format(marker=marker, claim="Embedded chat template == chat_template.jinja",
key="tokenizer.chat_template", actual=actual))
if not ok:
mismatches += 1
print(" re-stamp with: python3 scripts/strip_mtp.py IN.gguf OUT.gguf "
"--chat-template chat_template.jinja")
print()
if mismatches:
print(f"[!] {mismatches} mismatch(es) — README Architecture claims disagree with GGUF metadata.")
return 1
print("[+] all Architecture claims verify against GGUF metadata.")
return 0
if __name__ == "__main__":
sys.exit(main())