{ "model": "checkpoints/sludge-small", "preset": "small", "parameters": 12011719, "platform": "macOS-27.0-arm64-arm-64bit-Mach-O", "processor": "arm64", "torch_threads": 1, "mean_nodes_per_screen": 13.175, "formats": { "pytorch": { "ms_mean": 10.71451664007327, "ms_p50": 10.695083000427985, "ms_p95": 11.57509119984752, "size_mb": 48.046876, "available": true }, "onnx": { "ms_mean": 33.21225328021683, "ms_p50": 33.24016600072355, "ms_p95": 33.34932440011471, "size_mb": 48.260867, "available": true, "max_abs_diff_vs_pytorch": 1.239776611328125e-05, "max_prob_diff_vs_pytorch": 6.6356733441352844e-09, "node_decision_agreement_vs_pytorch": 1.0, "screen_decision_agreement_vs_pytorch": 1.0, "max_screen_prob_diff_vs_pytorch": 2.2402343025085258e-07, "decision_agreement_vs_pytorch": 1.0, "path": "sludge-small.onnx", "screen_decision_agreement_over_corpus": 1.0, "screens_with_any_flip": 0, "corpus_screens": 40 }, "onnx_int8": { "ms_mean": 13.350311599970155, "ms_p50": 13.336750000235043, "ms_p95": 13.508375000674278, "size_mb": 12.41416, "available": true, "max_abs_diff_vs_pytorch": 10.417057991027832, "max_prob_diff_vs_pytorch": 0.9890071749687195, "node_decision_agreement_vs_pytorch": 0.9997106481481481, "screen_decision_agreement_vs_pytorch": 0.9629629629629629, "max_screen_prob_diff_vs_pytorch": 0.6227820437142244, "decision_agreement_vs_pytorch": 0.9629629629629629, "path": "sludge-small.int8.onnx", "note": "int8 weight quantisation shifts logits; the meaningful check is decision agreement at the detector's threshold, reported above.", "screen_decision_agreement_over_corpus": 0.9898148148148148, "screens_with_any_flip": 11, "corpus_screens": 40 }, "coreml": { "ms_mean": 2.3157748799712863, "ms_p50": 2.360292000048503, "ms_p95": 2.4256751998109394, "size_mb": 24.183565, "available": true, "path": "sludge-small.mlpackage", "max_abs_diff_vs_pytorch": 1.2060108184814453, "max_prob_diff_vs_pytorch": 0.0005344669334590435, "node_decision_agreement_vs_pytorch": 1.0, "screen_decision_agreement_vs_pytorch": 1.0, "max_screen_prob_diff_vs_pytorch": 0.01530037641045523, "decision_agreement_vs_pytorch": 1.0, "precision": "float16" }, "gguf": { "size_mb": 24.030848, "available": true, "path": "sludge-small.f16.gguf", "dtype": "f16", "max_abs_diff_vs_pytorch": 0.00128173828125, "max_prob_diff_vs_pytorch": 2.1227169781923294e-06, "node_decision_agreement_vs_pytorch": 1.0, "screen_decision_agreement_vs_pytorch": 1.0, "max_screen_prob_diff_vs_pytorch": 2.3298150106043636e-05, "decision_agreement_vs_pytorch": 1.0, "runnable_by_llama_cpp": false, "note": "Valid GGUF v3 container and round-trips through sludge.gguf. llama.cpp cannot execute it because this is not one of its LLM graph architectures; the graph lives in sludge.model." } }, "note": "Per-screen figures are the model forward pass only, on this machine. The end-to-end number in the CLI's `sludge bench` additionally includes featurisation and decoding.", "graph_shape": "static: the node axis is fixed at max_nodes and the caller pads and masks. A dynamic node axis bakes the traced node count into the attention reshapes and only runs for that one screen size.", "end_to_end": { "ms_mean": 4.9734986666483865, "ms_p50": 4.9054374999286665, "ms_p95": 5.962366649646356, "ms_p99": 6.116143729759642, "n": 120, "rss_delta_mb": 0.147456, "rss_mb": 1305.837568, "params": 12011719 } }