Download examples/coupling.py from zeechimp/gnomon: direct link, hf CLI and curl.
- Browser
- Download file 2.66 kB
-
https://huggingface.co/zeechimp/gnomon/resolve/main/examples/coupling.py
- Command line
-
hf download hf://zeechimp/gnomon/examples/coupling.py
-
curl -L -o coupling.py https://huggingface.co/zeechimp/gnomon/resolve/main/examples/coupling.py
2.66 kB
| """First coupling: does attention survive on IB-reconstructed K/V? | |
| Runs the same random-weight GNOMON twice β once with raw attention, once | |
| where every layer attends over the previous layer's reconstructed K/V. | |
| Reports the logit divergence and both latencies. | |
| """ | |
| import os | |
| import sys | |
| import time | |
| sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..")) | |
| import jax | |
| import jax.numpy as jnp | |
| from gnomon import init_gnomon, init_mode_state, make_forward_jit | |
| def kl(p_logits, q_logits): | |
| p = jax.nn.softmax(p_logits, axis=-1) | |
| q = jax.nn.log_softmax(q_logits, axis=-1) | |
| return jnp.mean(jnp.sum(p * (jnp.log(p + 1e-9) - q), axis=-1)) | |
| def main(): | |
| params = init_gnomon( | |
| jax.random.PRNGKey(0), | |
| vocab_size=128, d_model=64, n_layers=3, n_heads=4, | |
| ib_k=16, seq_len=24, d_signal=16, | |
| ) | |
| ids = jax.random.randint(jax.random.PRNGKey(1), (1, 24), 0, 128) | |
| mode_sig = jax.random.normal(jax.random.PRNGKey(2), (1, 24, 16)) | |
| mode = init_mode_state(d_mod=4) | |
| raw_fn = make_forward_jit(n_heads=4, use_reconstructed_kv=False) | |
| rec_fn = make_forward_jit(n_heads=4, use_reconstructed_kv=True) | |
| # Warmup β forces both traces | |
| _r = raw_fn(params, ids, mode, mode_sig, jax.random.PRNGKey(3)) | |
| _c = rec_fn(params, ids, mode, mode_sig, jax.random.PRNGKey(3)) | |
| jax.block_until_ready(_r["logits"]) | |
| jax.block_until_ready(_c["logits"]) | |
| t0 = time.perf_counter() | |
| out_raw = raw_fn(params, ids, mode, mode_sig, jax.random.PRNGKey(3)) | |
| jax.block_until_ready(out_raw["logits"]) | |
| t_raw = time.perf_counter() - t0 | |
| t0 = time.perf_counter() | |
| out_rec = rec_fn(params, ids, mode, mode_sig, jax.random.PRNGKey(3)) | |
| jax.block_until_ready(out_rec["logits"]) | |
| t_rec = time.perf_counter() - t0 | |
| divergence = float(kl(out_raw["logits"], out_rec["logits"])) | |
| print("Coupling test: raw attention vs IB-reconstructed attention") | |
| print(f" layers: 3 (IB state dim = 16)") | |
| print(f" raw forward: {t_raw * 1000:.2f} ms") | |
| print(f" coupled forward: {t_rec * 1000:.2f} ms") | |
| print(f" logit KL divergence: {divergence:.4f}") | |
| print() | |
| if divergence < 0.5: | |
| print(" β Attention survives on reconstructed K/V.") | |
| else: | |
| print(" β Reconstruction too lossy (random weights).") | |
| print(" Adding KL loss to the training objective closes this gap.") | |
| print() | |
| print("Note: this is with random IB weights. Training drives the") | |
| print("divergence down β that's what the coupling loss term is for.") | |
| if __name__ == "__main__": | |
| main() |