Buckets:

soixantesept's picture
download
raw
26.6 kB
#!/usr/bin/env python3
"""Emit /root/hutter67/swarm67/B6-bundle/REPORT.md and bundle67.json from measured artifacts."""
from __future__ import annotations
import json
import pathlib
B6 = pathlib.Path("/root/hutter67/swarm67/B6-bundle")
OUT = B6 / "out"
WORK = B6 / "work"
TABLE = json.loads((OUT / "variant_table.json").read_text())
CODECS = json.loads((OUT / "codec_compare.json").read_text()) if (OUT / "codec_compare.json").exists() else {}
PROOF = (B6 / "test/decode_proof.tsv").read_text().strip().splitlines()
BASELINE_BYTES = 766308
BASELINE_SHA = "debed42c9ec425e5461dd886148814e55c71c52ea43b2675d80b563030074dce"
WINNER = "dyn_Os"
BUILD_CMD = """\
# zlib objects (9 C files)
gcc -c -DNDEBUG -I<src>/zlib -Os -fno-fast-math -ffp-contract=off -fexceptions -include unistd.h <zlib>.c -o <obj>.o
# paq8px link (single g++ invocation, sequential TUs)
g++ -DNDEBUG -I<src>/zlib -fno-rtti -std=gnu++17 -ffp-contract=off -fno-fast-math \\
-Os -flto=2 -ffunction-sections -fdata-sections -Wl,--gc-sections \\
-fno-unwind-tables -fno-asynchronous-unwind-tables -Wl,-s \\
<zlib-objects> src/file/*.cpp src/filter/*.cpp src/model/*.cpp src/text/*.cpp src/lstm/*.cpp src/*.cpp \\
-o paq8px
strip --strip-all paq8px"""
PACK_CMD = """\
python3 work/make_bundle.py --binary out/build/dyn_Os --out out/bundle_dyn_Os --label dyn_Os
# == inside make_bundle.py:
# b.xz = lzma.compress(binary, FORMAT_XZ, CHECK_NONE,
# filters=[FILTER_X86, FILTER_LZMA2(dict_size=1<<19, lc=3, lp=0, pb=0,
# mode=MODE_NORMAL, mf=MF_BT2, nice_len=96, depth=0)])
# v.xz, d = byte-identical copies of the P5 paq8px-blob members
# zip = STORED b.xz, STORED v.xz, DEFLATED(9) d with mode 0755 and DOS-epoch timestamps"""
def fmt_row(label: str, doc: dict) -> str:
bl = doc.get("binary_level_proof", {}).get("status", "not run")
bd = doc.get("bundle_level_proof", {}).get("status", "not run")
proof_s = f"binary: {bl} / bundle: {bd}"
return (f"| `{label}` | {doc['raw_binary_bytes']:,} | {doc['b_xz_bytes']:,} | **{doc['bundle_bytes']:,}** | "
f"`{doc['bundle_sha256'][:16]}…` | {proof_s} | {doc['description']} |")
def proven_binaries() -> dict:
"""sha256 -> list of archives a binary-level decode proved, from decode_proof.tsv."""
out = {}
for line in PROOF[1:]:
f = line.split("\t")
if len(f) < 5 or f[0] not in ("orig", "dyn_O3", "dyn_Os", "static_Os", "dyn_Os_hard"):
continue
out.setdefault(f[0], []).append((f[1], f[4]))
return out
def main() -> int:
prov = proven_binaries()
NAME2BIN = {"orig": WORK / "paq8px.orig", "dyn_Os": OUT / "build/dyn_Os",
"dyn_O3": OUT / "build/dyn_O3", "static_Os": OUT / "build/static_Os",
"dyn_Os_hard": OUT / "build/dyn_Os_hard"}
import hashlib as _h
proven_shas = {_h.sha256(p.read_bytes()).hexdigest(): (n, prov[n]) for n, p in NAME2BIN.items() if p.exists()}
for label, doc in TABLE.items():
sha = doc.get("binary_sha256")
bl = proven_shas.get(sha)
if bl:
doc["binary_level_proof"] = {"status": "PASS", "same_binary_as": bl[0],
"archives": [a for a, m in bl[1]]}
else:
doc["binary_level_proof"] = {"status": "not run", "reason": "larger than the winner; not a shipping candidate"}
doc["bundle_level_proof"] = {"status": "PASS"} if doc.get("decode_proof") else {
"status": "not run", "reason": "larger than the winner; not a shipping candidate"}
rows = [fmt_row(k, v) for k, v in sorted(TABLE.items(), key=lambda kv: kv[1]["bundle_bytes"])]
win = TABLE[WINNER]
saved = BASELINE_BYTES - win["bundle_bytes"]
codec_rows = "\n".join(
f"| {c['name']} | {c['bytes']:,} | {c['note']} |" for c in CODECS.get("rows", []))
body = PROOF[1:] if PROOF and PROOF[0].startswith("binary\tarchive") else PROOF
flag = (
"**By director decision (after V1's timing correction) B6 did not run a full-archive decode.**\n\n"
"Decoding the 60,634,501 B blob at `-9` costs roughly as long as the encode (~7 h at ~4.3 GB RSS; "
"P1's own receipt for the 1 MB blob arm at `-9` is 7:36 min / 4,042,460 kB, which scales linearly), so a "
"B6 run would have duplicated the verifier's certification decode hour for hour. The waiter B6 had armed "
"under `locked_run.sh 14400` (full67 + a poc slot, MemAvailable >= 2,500,000 kB, archive-quiescence check) "
"was cancelled before it ever acquired `full67`; the lock was never taken by B6. "
"**The definitive full-archive proof is therefore V1's single certification decode, running exactly the "
"shipped 204,244 B artifact** (`7a389922…`, components in §5) against "
"`runs/full_blob_m9/blob_full.paq8px216`, expecting blob 60,634,501 B / `9c869f14…` and then "
"`enwik8` 100,000,000 B / `2b49720e…`. What this report contributes to that decode: the artifact receipt, "
"the measured performance envelope (level-1 decode peak RSS 563,896 kB; the `-Os` payload runs ~10-20 % "
"slower than the `-O3` binary in §6a's single-run timings, so size the certificate window at ~7-9 h and "
"~4.3 GB), and the fact that both stages were already exercised end-to-end on real project archives below."
)
proof_tbl = "\n".join("| " + " | ".join(l.split("\t")) + " |" for l in body)
e2e = win.get("e2e", {})
e2e_rows = "\n".join(
f"| {c['name']} | {c['rc']} | {c['out_bytes']} | `{str(c['out_sha256'])[:24]}…` | "
f"{'YES' if c['match'] else 'NO'} | {c['peak_rss_kb']} | {c['wall']} | {c['what']} |"
for c in e2e.get("cases", []))
md = f"""# B6 — minimizing the shipped paq8px decoder bundle
**Scope.** Decoder-side only: no compression runs, no changes to the archive, nothing touched under
`V1-verify-queue/`, `publish67/` or the flagship run directory, no publication or HF API calls.
All work under `/root/hutter67/swarm67/B6-bundle/`.
## Headline
| | bytes | sha256 | notes |
|---|---:|---|---|
| P5/published paq8px blob-chain bundle | {BASELINE_BYTES:,} | `{BASELINE_SHA}` | b.xz 760,856 + v.xz 4,948 + d 236 + 268 framing |
| **B6 winner (`dyn_Os`)** | **{win['bundle_bytes']:,}** | `{win['bundle_sha256']}` | same launcher/inverse, new payload |
| saving | **{saved:,}** | | one-line: **the winner ships {saved:,} bytes smaller than the 766,308 B baseline.** |
## SHIP THIS
| | |
|---|---|
| artefact | `/root/hutter67/swarm67/B6-bundle/out/bundle_dyn_Os/decompressor.zip` |
| bytes / sha256 | **204,244** / `7a3899220efe888bd0d0e3affed875094afaea382ef1eda67eb1268a15b0071a` |
| contents | `b.xz` 198,792 (XZ of the 476,864 B `dyn_Os` ELF, sha256 `94cc29db…`) + `v.xz` 4,948 (unchanged P5 member) + `d` 409 raw / 236 deflated (unchanged P5 launcher) + 268 B ZIP framing |
| fallback if the verifier's `-9` decode hits anything unexpected | `/root/hutter67/swarm67/B6-bundle/out/bundle_dyn_O3/decompressor.zip` — 443,156 B, sha256 `411bd72a8deb791e…`, P1's exact `-O3 -flto` recipe with only `-static` removed, also small-chain proven (3/3) |
| evidence table for every variant | §4, with per-binary proofs in §6a/§6b and machine-readable form in `bundle67.json` |
**{saved:,} B** off the counted total, from a rebuild + repack of the payload only. On the assets basis
the entry is `A_blob + 204,244` (was `A_blob + 766,308`); with the flagship blob archive projecting to
~16.1 MB the entry lands at ≈16.29 MB, versus 17,969,555 B for the published v4 baseline. The launcher `d`
and the blob inverse `v.xz` are byte-identical to the published P5 package, so the decode chain a
verifier runs is unchanged; only the paq8px executable it unpacks is smaller.
## 1. The decoder chain that has to be shipped
Flagship production chain (`P1-paq8px-cal/fullrun_blob.sh`, `-9`):
```
archive.paq8px216 --paq8px -d--> blob (60,634,501 B) --blob inverse--> enwik8 (100,000,000 B)
```
P5 accounted that whole chain as a 766,308 B ZIP holding exactly three members
(`paq8px-blob/package.json`): `b.xz` 760,856 (XZ of the 1,895,080 B static paq8px binary),
`v.xz` 4,948 (the dedicated cmix-store blob inverse), `d` 409 raw / 236 deflated (Python launcher),
plus 268 B of ZIP framing. **B6 changes only `b.xz`.**
Byte-accounting validation: `work/make_bundle.py --verify-baseline` rebuilds the P5 ZIP from its
members and reproduces **766,308 B / sha256 `{BASELINE_SHA}` byte-for-byte**. Stated plainly, because
it is the whole basis of the claim: **because the packer reproduces P5's ZIP byte-for-byte, the
−562,064 B is an apples-to-apples accounting change (same three members, same launcher, same inverse,
same 268 B of framing) and not a change of accounting basis.** The only member whose bytes differ is
the payload `b.xz`.
## 2. Lever 1 — rebuild the binary smaller (this is where the bytes were)
The shipped binary is a `-O3 -flto -march=native -static` build: 1,895,080 B raw. Static linking of
libc/libstdc++ plus `-O3` inlining is most of it. Rebuilding the **pinned source**
(`paq8px-29237fb44cb1995690e3eb72c6c3b1e4aede5791`, the same commit P1/P5 used) with size flags and
dynamic linking, **changing no source line**, gives:
| build | raw B | XZ payload B | bundle B |
|---|---:|---:|---:|
| P1 shipped (`-O3 -flto -march=native -static`), P5's own tuned XZ | 1,895,080 | 760,856 | 766,308 (the baseline) |
| `-O3 -flto=2` dynamic (identical codegen flags, only the link mode changes) | 1,012,192 | 437,704 | 443,156 |
| `-O2 -flto=2` dynamic | 647,648 | 264,480 | 269,932 |
| clang++-17 `-Oz` dynamic | 523,208 | 211,248 | 216,700 |
| **`-Os -flto=2` dynamic + gc-sections + strip** | **476,864** | **198,792** | **204,244** |
| `-Os -flto=2` dynamic + hard flags (`-fno-plt -fno-stack-protector …`) | 473,552 | 201,004 | 206,456 (smaller raw, *worse* packed) |
| `-Os` dynamic, no LTO | 509,632 | 204,672 | 210,124 |
| `-Os` static (self-contained fallback) | 1,277,800 | 480,876 | 486,328 |
Exact commands are in §5. `-flto=1` and `-flto=2` produced a byte-identical binary here
(sha256 `94cc29db…`), so that knob is a no-op. Every payload column below uses the same swept-optimal
XZ parameters as the winner, so variants are comparable at bundle level.
Cost of the size build: in the single-run measurements of §6a the `-Os` decode is **12.25 s** on the
32 KiB binary slice where the original binary takes **10.42 s** (≈ +18 %, within session noise on a
box that was simultaneously running a 4.5 GB encode; `dyn_O3` measured 9.38 s). This affects *decode
time only* — it is not a scoring term — but it is the honest trade: the smallest bundle decodes
somewhat slower than the `-O3` binary, and the fallback `dyn_O3` variant does not have that penalty.
**`-march=native` was deliberately NOT used** even though P1's build had it: it would bake AVX-512
instructions from this EPYC into the shipped decoder and could crash on a verifier's CPU. The
`-Os` build targets baseline x86-64. This is a portability gain, not just a size choice.
## 3. Lever 2 — pack the payload with the strongest *shippable* codec
All measured on the winner binary (`dyn_Os`, 476,864 B). "Shippable" means the decoder is either
already present on the host (Python's `lzma`) or small enough to pay for itself:
| codec | bytes | note |
|---|---:|---|
{codec_rows}
XZ + x86/BCJ wins: it needs **zero extra shipped bytes** (the launcher already uses Python's `lzma`),
while zpaq's 11,595 B head start is eaten by the 16,476 B libzpaq extractor it would require
(187,197 + 16,476 = 203,673 > 198,792), and UPX's self-extracting stub lands at 202,768.
Swept per-parameter: `dict 1–64 MiB`, `lc 0–4`, `lp 0/1`, `pb 0–4`, `nice 128/192/273`,
`mf ∈ {{bt2,bt3,bt4}}`, BCJ `start` offsets 0…0x7fffffff — best is `dict=512 KiB, lc=3, lp=0, pb=0, mf=bt2, nice=96`,
whole-file BCJ. A `.text`/`.rodata` split with BCJ on code only saves a further 452 B but needs a
reassembly step in the launcher (≈ +60 B of launcher, +90 B of ZIP framing), so it was not shipped.
## 4. Every bundle variant built (size, sha256, decode round-trip)
Bundle = `b.xz` + `v.xz` + `d` + 268 B ZIP framing, all built with the same deterministic packer.
"binary" = the candidate executable decoded a real project archive byte-exactly (§6a);
"bundle" = the *packed bundle* ran the complete chain (`launcher → payload → paq8px -d → inverse`) with
the expected sha256 (§6b). Non-candidates were not decoded; `dyn_Os_flto1` is byte-identical to
`dyn_Os` (sha256 `94cc29db…`) so it inherits that proof by identity.
| label | raw binary B | payload (`b.xz`) B | bundle B | bundle sha256 | decode round-trip | what it is |
|---|---:|---:|---:|---|---|---|
{chr(10).join(rows)}
The last row re-packs the *baseline* binary with the winner's XZ parameters (761,640 B); P5's own
tuned parameters do 784 B better on that large binary (760,856 B → the 766,308 B published baseline),
which is why the baseline comparison uses 766,308 and not 767,092. On the ~477 KB winner the swept
parameters above are the better ones.
## 5. Winner and exact commands
Winner: **`dyn_Os`**, bundle **{win['bundle_bytes']:,} B**, sha256
`{win['bundle_sha256']}`, payload `b.xz` {win['b_xz_bytes']:,} B sha256
`{win['b_xz_sha256']}`.
```sh
{BUILD_CMD}
{PACK_CMD}
```
Rebuild the same bundle from scratch (all in `work/`):
```sh
./build_variant.sh dyn_Os dyn_Os # → out/build/dyn_Os (476,864 B)
python3 make_bundle.py --binary ../out/build/dyn_Os --out ../out/bundle_dyn_Os --label dyn_Os
bash e2e_bundle.sh dyn_Os ../out/bundle_dyn_Os # end-to-end proof, writes test/e2e_dyn_Os.json
```
## 6. Proof — what was actually decoded
### 6a. Binary-level proof (real project archives, `test/decode_proof.tsv`)
| binary | archive | out bytes | out sha256 | match expected | rss kB | wall |
|---|---|---:|---|---|---:|---|
{proof_tbl}
`slice32k.paq8px216` is a **fresh `-1` archive of the first 32,768 bytes of the project's real
`store67/blob67.bin`**, encoded with the original binary (block type `default` = generic/binary path,
12,962 B archive). `small.paq8px216` is P1's calibration fixture (text path). Every candidate binary
reproduces the original binary's output exactly — including the two binaries that were *not* expected
to win (`dyn_Os_hard`, `static_Os`).
### 6b. Full-chain proof through the packed bundle (`test/e2e_{WINNER}.json`)
Fresh directory, `decompressor.zip` unpacked exactly as a verifier would, then the shipped launcher
was run on real archives:
| case | exit | out bytes | out sha256 | match | rss kB | wall | what |
|---|---:|---:|---|---:|---:|---:|---|
{e2e_rows}
Case 1 is the P5-certified fixture (`blob-paq.paq`, a genuine `paq8px -0L` archive of the real
35-byte blob `blob35.bin` whose chain output P5 recorded as sha256 `4ab9cb28…`) — **the shipped
bundle produces that same 17-byte output**. Case 2 is a *fresh* `-1` archive of the same blob, so the
chain is also proven through the CM model path at level 1. Case 3 decodes the 32 KiB real-blob-slice
archive with the payload extracted from `b.xz` (extracted payload sha256
`{e2e.get('extracted_payload_sha256', '—')}` = the built `dyn_Os` binary, i.e. the XZ round-trip is
byte-exact), exercising the generic/binary model path.
### 6c. Inverse stage on a real 2 MiB blob, and independent `unzip` re-verification
The bundle's `v` member (extracted from `v.xz`, sha256 `60015c60…`) was run on P5's real
1,666,504 B blob fixture
(`R1-store-recovery/b2case/blob2m.bin`): exit 0, 2,097,152 B out, sha256
`9dd214e64458c0c36f75a6100aed1a90a7b76ff9dfbb85dc032dea17a1963f50` — identical to P5's certified
`unblob2m.out`, in 0.13 s at 6,780 kB RSS. So both stages of the shipped chain are exercised on real
project data: the payload stage in §6a/§6b, the inverse stage here.
The shipped `out/bundle_dyn_Os/decompressor.zip` (204,244 B, sha256
`7a3899220efe888bd0d0e3affed875094afaea382ef1eda67eb1268a15b0071a`) was unpacked with the system
`unzip` in a fresh directory (`unzip -t`: "No errors detected"), and the launcher was run on both
archives: two runs, exit 0 each, 17-byte outputs, both sha256
`4ab9cb28f180436abbd3a76866c13eb3dc0797888908947f298eec7fb546bcd6` (expected).
### 6d. Flagship `-9` acceptance test
{{FLAGSHIP}}
### 6e. What is *not* yet proven
{{OUTSTANDING}}
```sh
cd <unpacked bundle> && ./d /path/blob_full.paq8px216 /path/restored_blob.bin # the inverse stage
# paq8px stage alone (no inverse), to compare against the blob hash directly:
python3 -c "import lzma,pathlib;p=pathlib.Path('.');(p/'b').write_bytes(lzma.open(p/'b.xz').read())"
./b -d /path/blob_full.paq8px216 /path/restored_blob.bin
sha256sum /path/restored_blob.bin # must equal 9c869f141bc91a9b164e2cd4faecfea5bb648bd996c8c0f5e865b4e4010be425
```
Residual risk is bounded by construction: same source commit, same `-ffp-contract=off -fno-fast-math`
semantics, no `-march=native`, and `Shared::mem = 65536 << level` means the level changes only model
*table sizes*, not which models run — the level-1 proof therefore exercises the same model code as
`-9`. The remaining uncertainty is codegen at larger tables, which the single `-9` decode settles.
## 7. Fallback ladder and lower-risk alternatives
If the verifier's `-9` decode of the winner hits anything unexpected, use the next rung — each rung is
packaged, hashed and small-chain proven, so no rework is needed:
1. **`dyn_O3`** — `out/bundle_dyn_O3/decompressor.zip`, **443,156 B**, sha256
`411bd72a8deb791e304f7d2ee56b86d35c6990166d90515b6d8f3de4be83b570` (proven 3/3 at bundle level).
2. **`static_Os`** — `out/bundle_static_Os/decompressor.zip`, 486,328 B (proven 3/3; no host DSOs).
3. **P5's original** — `P5-bundles/paq8px-blob/decompressor.zip`, 766,308 B, sha256
`debed42c…` (the published baseline, unchanged on disk).
Rungs 1–3 cost 238,912 / 282,084 / 562,064 B more than the winner respectively.
## 7b. Lower-risk alternatives (if a dynamic payload is unacceptable)
* **`dyn_O3`** (bundle {TABLE['dyn_O3']['bundle_bytes']:,} B): P1's exact `-O3 -flto=2` flags, only
`-static` removed. Same optimization level as the proven binary, so the codegen risk is the
smallest of any candidate; costs {TABLE['dyn_O3']['bundle_bytes'] - win['bundle_bytes']:,} B more.
* **`static_Os`** (bundle {TABLE['static_Os']['bundle_bytes']:,} B): fully static, needs no host
libstdc++/libc at all. Costs {TABLE['static_Os']['bundle_bytes'] - win['bundle_bytes']:,} B more
than the winner but is self-contained.
* The winner and `dyn_O3`/`static_Os` are dynamic/static respectively: `dyn_Os` requires
`ld-linux-x86-64.so.2`, `libc.so.6`, `libstdc++.so.6` (+ transitive `libm`/`libgcc_s`) on the
decode host, exactly like the published 16,306 B baseline ELF and like cmix/paq8l/paq9a in P5's
table. `static_Os` removes that requirement entirely.
## 8. Boundaries
* No source line was modified; no transform/model was removed; nothing about what the decoder outputs
changed. Only build flags, link mode, and payload packing.
* Not attempted (and reported as such): pruning encoder-only code or the LSTM (would change the
decoder's capability set and needs its own behavioural proof), and `-march=native` (portability).
* Measured-but-rejected levers, with numbers, are in §2/§3 so nothing is silently dropped.
* Light-lane discipline held: every build ≤ 285,260 kB peak RSS, every decode test ≤ 564,120 kB;
no heavy encode, no lock taken, no pattern-kill, nothing written outside `swarm67/B6-bundle/`.
* The full-archive `-9` decode was planned, armed under `locked_run.sh`, and then **cancelled before it
ever acquired `full67`** once V1's timing correction (~7 h, not ~23 min) showed it would duplicate
the verifier's certification decode. That decode, on exactly this 204,244 B artifact, is the
definitive full-archive proof; B6's evidence is the small-chain/real-archive work in §6.
"""
outstanding = ("The `-9` level itself has not been exercised by B6 — every B6 measurement is at `-0L`/`-1` "
"(≤ 564 MB RSS) because a `-9` decode needs ~4.3 GB and hours, outside the light lane and "
"duplicative of V1's certification decode. The residual risk is therefore exactly this: codegen "
"behaviour at `-9`'s larger model tables. It is bounded by construction — identical source "
"commit, `-ffp-contract=off -fno-fast-math` preserved, no `-march=native`, and "
"`Shared::mem = 65536 << level` means the level changes only table *sizes*, not which models "
"run, so the level-1 proof exercises the same model code — but it is retired only by the "
"verifier's single `-9` certification decode of this artifact, and by nothing B6 did. If that "
"decode fails, the fallback is `dyn_O3` (443,156 B, §7, also proven on the small chain, "
"packaged and waiting at `out/bundle_dyn_O3/decompressor.zip`), whose codegen is P1's exact "
"`-O3 -flto` recipe with only `-static` removed.")
md = md.replace("{OUTSTANDING}", outstanding)
md = md.replace("{FLAGSHIP}", flag)
(B6 / "REPORT.md").write_text(md)
doc = {
"task": "minimize shipped paq8px decoder bundle without changing what it decodes",
"baseline": {"bundle_bytes": BASELINE_BYTES, "bundle_sha256": BASELINE_SHA,
"source": "P5-bundles/paq8px-blob/decompressor.zip",
"members": {"b.xz": 760856, "v.xz": 4948, "d_raw": 409, "d_deflated": 236, "framing": 268},
"reproduced_byte_for_byte_by": "work/make_bundle.py --verify-baseline"},
"winner": {"label": WINNER, **{k: win[k] for k in
("bundle_bytes", "bundle_sha256", "b_xz_bytes", "b_xz_sha256", "binary_bytes",
"binary_sha256", "raw_binary_bytes", "bundle_path")}},
"saving_vs_baseline_bytes": saved,
"decode_chain": {
"steps": ["unpack decompressor.zip", "launcher d: b.xz -> paq8px, v.xz -> inverse",
"paq8px -d archive blob", "inverse blob -> original"],
"unchanged_vs_p5": ["v.xz (4,948 B, sha 55e5f625…)", "launcher d (409 B, sha 9a9132bb…)"],
"changed": ["b.xz payload (760,856 -> 198,792 B)"],
},
"build": {
"source_commit": "29237fb44cb1995690e3eb72c6c3b1e4aede5791",
"source_tree": "P1-paq8px-cal/src/paq8px-29237fb…(copied to work/paq8px-src, unmodified)",
"cxx_flags": ("-DNDEBUG -fno-rtti -std=gnu++17 -ffp-contract=off -fno-fast-math -Os -flto=2 "
"-ffunction-sections -fdata-sections -Wl,--gc-sections -fno-unwind-tables "
"-fno-asynchronous-unwind-tables -Wl,-s"),
"link": "dynamic (no -static), then strip --strip-all",
"march_native": False,
"march_native_reason": "AVX-512 portability risk on the verifier's host",
},
"packing": {
"payload_codec": "xz FORMAT_XZ CHECK_NONE, filters [FILTER_X86, FILTER_LZMA2]",
"lzma2": {"dict_size": 524288, "lc": 3, "lp": 0, "pb": 0, "mode": "NORMAL",
"mf": "BT2", "nice_len": 96, "depth": 0},
"zip": "b.xz STORED, v.xz STORED, d DEFLATED(9) mode 0755, DOS epoch timestamps",
},
"variants": {k: {kk: v[kk] for kk in ("description", "raw_binary_bytes", "binary_sha256",
"b_xz_bytes", "b_xz_sha256", "bundle_bytes", "bundle_sha256",
"bundle_path", "binary_level_proof", "bundle_level_proof",
"decode_proof") if kk in v}
for k, v in TABLE.items()},
"codec_comparison_on_winner_binary": CODECS,
"proofs": {
"binary_level": [dict(zip(["binary", "archive", "out_bytes", "out_sha256", "match", "rss_kb", "wall"],
l.split("\t"))) for l in PROOF[1:]],
"end_to_end": e2e,
"flagship_minus9": {
"status": "NOT RUN BY B6 - cancelled by director after V1's timing correction",
"reason": ("a full -9 decode of the 60.6 MB blob costs ~7 h at ~4.3 GB RSS (P1 receipt: "
"1 MB blob arm at -9 = 7:36 min / 4,042,460 kB, linear in blob size), so it would "
"duplicate the verifier's certification decode; the waiter B6 had armed under "
"locked_run.sh 14400 was cancelled before it ever acquired full67 and the lock was "
"never taken by B6"),
"owner_of_the_full_archive_proof": "V1-verify-queue, single certification decode of exactly this artifact",
"artifact_to_certify": {"path": win["bundle_path"], "bytes": win["bundle_bytes"],
"sha256": win["bundle_sha256"]},
"archive": "/root/hutter67/swarm67/P1-paq8px-cal/runs/full_blob_m9/blob_full.paq8px216",
"expected_blob_bytes": 60634501,
"expected_blob_sha256": "9c869f141bc91a9b164e2cd4faecfea5bb648bd996c8c0f5e865b4e4010be425",
"expected_enwik8_bytes": 100000000,
"expected_enwik8_sha256": "2b49720ec4d78c3c9fabaee6e4179a5e997302b3a70029f30f2d582218c024a8",
"performance_envelope": {"level1_decode_peak_rss_kb": 563896,
"original_binary_minus9_1mb_arm_peak_rss_kb": 4042460,
"estimated_full_decode": "~7-9 h wall, ~4.3 GB RSS"},
},
},
"artifacts": {
"bundle_zip": win["bundle_path"],
"binary": str(OUT / "build/dyn_Os"),
"packer": "work/make_bundle.py",
"build_script": "work/build_variant.sh",
"e2e_proof": "work/e2e_bundle.sh -> test/e2e_dyn_Os.json",
"decode_proof": "work/decode_proof.sh -> test/decode_proof.tsv",
"independent_verify": "test/final_verify.json (system unzip, fresh directory)",
},
}
(B6 / "bundle67.json").write_text(json.dumps(doc, indent=1))
print(f"REPORT.md and bundle67.json written. winner={WINNER} bundle={win['bundle_bytes']} saved={saved}")
return 0
if __name__ == "__main__":
raise SystemExit(main())

Xet Storage Details

Size:
26.6 kB
·
Xet hash:
9ccc1b464ad375a40b632c1c248cfcea74abf01f0f2529588319a588333b774a

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.