betterwithage commited on
Commit
082de2a
·
verified ·
1 Parent(s): 47c7eb2

Add torch-universal variant for kernels 0.16 get_kernel (build.toml universal=true; no model conversion)

Browse files
build/torch-universal/metadata.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "szl-lambda-gate",
3
+ "version": 1,
4
+ "license": "Apache-2.0",
5
+ "universal": true,
6
+ "python-depends": [],
7
+ "id": "_szl_lambda_gate_universal_cto1",
8
+ "backend": {
9
+ "type": "cpu"
10
+ },
11
+ "lambda": "Conjecture 1 (advisory, never a theorem)",
12
+ "locked_8": [
13
+ "F1",
14
+ "F4",
15
+ "F7",
16
+ "F11",
17
+ "F12",
18
+ "F18",
19
+ "F19",
20
+ "F22"
21
+ ],
22
+ "github_recipe": "build.toml [torch] universal=true; kernels 0.16 loads repo_type=kernel",
23
+ "digest": {
24
+ "algorithm": "sha256",
25
+ "files": {
26
+ "szl_lambda_gate/layers.py": "cde2de167d89002ad0cc63a4de205848f14eae0c649c19e2a8a96665a63806f2",
27
+ "szl_lambda_gate/_lambda.py": "7c8a04eeaa1d7851b99b498c6b85d3556da6cab1b0adc274e42ff2a3814353ee",
28
+ "szl_lambda_gate/_ops.py": "6adf1dc42e7a9e6da914e304da9720438e59c234104e11f3e5ccaa45ecee23b6",
29
+ "szl_lambda_gate/__init__.py": "3d65508cdf9b9dc2d525cdd2063e432752d42894cd96f84efa5f27866dfaf126",
30
+ "szl_lambda_gate/governed_norm/layers.py": "d79c83355cb2332b9d0e9d98584b86434e486532e27d30652df10158b30fc37b",
31
+ "szl_lambda_gate/governed_norm/_norm.py": "cf72ace281ec86c504426f75e23eebdb661a1e3a12e4dfcc6c8b9fa634b03942",
32
+ "szl_lambda_gate/governed_norm/_receipt.py": "db1244bd184affdca929bd38ac728b2696f82ad565aaf82d14ca3845d4d748a9",
33
+ "szl_lambda_gate/governed_norm/__init__.py": "50fe4fb09a165d0ae9a781f0ef3515388c752a14cde95f73687854f59a7dfadc"
34
+ }
35
+ }
36
+ }
build/torch-universal/szl_lambda_gate/__init__.py ADDED
@@ -0,0 +1,198 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # © 2026 SZL Holdings · Stephen P. Lutar · ORCID 0009-0001-0110-4173
3
+ """szl_lambda_gate — the Lambda-Spine aggregator (Λ) as a universal kernel.
4
+
5
+ A pure-PyTorch (universal) kernel from SZL Holdings for the Hugging Face
6
+ Kernel Hub. It ports the canonical Λ aggregator into a differentiable,
7
+ torch.compile-friendly torch op:
8
+
9
+ Λ(x) = ∏ xᵢ^{wᵢ}, Σwᵢ = 1, wᵢ > 0, xᵢ ∈ [0,1] (weighted geometric mean)
10
+
11
+ plus an ADVISORY governance gate (Λ vs threshold), the four carried axioms as
12
+ real runtime self-checks, and pure nn.Module layers.
13
+
14
+ Load from the Hub:
15
+
16
+ import torch
17
+ from kernels import get_kernel
18
+
19
+ lg = get_kernel("SZLHOLDINGS/szl-lambda-gate")
20
+ axes = torch.tensor([0.9, 0.8, 0.95]) # axis scores in [0,1]
21
+ score = lg.lambda_aggregate(axes) # Λ(x) ∈ [0,1]
22
+ res = lg.lambda_gate(axes, threshold=0.5) # ADVISORY pass/fail
23
+ print(res.score, res.passed, res.advisory)
24
+
25
+ WHAT Λ IS / IS NOT (HONESTY — SZL Holdings doctrine v11):
26
+ Λ is the weighted-geometric-mean aggregator — a non-compensatory, ADVISORY
27
+ way to roll axis scores in [0,1] into one number (any zeroed axis zeroes the
28
+ aggregate). It is NOT "proven trust" and NOT a closed theorem: Λ-uniqueness
29
+ remains Conjecture 1 (OPEN — an unresolved CAUCHY_ND step plus a missing
30
+ symmetry axiom). Label it honestly everywhere; a gate "pass" is advisory.
31
+
32
+ PROVENANCE: backed by the Lean 4 formalization szl-holdings/lutar-lean
33
+ (749 declarations / 14 axioms / 163 tracked sorries),
34
+ DOI 10.5281/zenodo.20434308 (lutar-lean). Λ uniqueness = Conjecture 1 (open).
35
+ """
36
+ from typing import Optional
37
+
38
+ import torch
39
+
40
+ from . import layers # noqa: F401 (must be importable for Hub layer mapping)
41
+ # CONSOLIDATION (Wave D): the governed-norm universal kernel is folded in here
42
+ # as a subpackage so szl-lambda-gate is the ONE canonical kernels package. The
43
+ # source repo szl-holdings/szl-governed-norm is DEPRECATED and points here;
44
+ # nothing was deleted (additive, reversible copy). Λ stays Conjecture 1.
45
+ from . import governed_norm # noqa: F401 (folded-in governed normalization kernels)
46
+ from ._lambda import YUYAY_AXES, YUYAY_FLOORS, LambdaGateResult
47
+ from ._lambda import find_axiom_violation as _find_axiom_violation
48
+ from ._lambda import is_bounded_by_max as _is_bounded_by_max
49
+ from ._lambda import is_egyptian_exact as _is_egyptian_exact
50
+ from ._lambda import is_homogeneous as _is_homogeneous
51
+ from ._lambda import is_monotone as _is_monotone
52
+ from ._lambda import lambda_aggregate as _lambda_aggregate
53
+ from ._lambda import lambda_gate as _lambda_gate
54
+ from ._lambda import lambda_gate_batch as _lambda_gate_batch
55
+ from ._lambda import selfcheck as _selfcheck
56
+ from ._lambda import yuyay_weights as _yuyay_weights
57
+
58
+ __all__ = [
59
+ "lambda_aggregate",
60
+ "lambda_gate",
61
+ "lambda_gate_batch",
62
+ "LambdaGateResult",
63
+ "is_monotone",
64
+ "is_egyptian_exact",
65
+ "is_bounded_by_max",
66
+ "is_homogeneous",
67
+ "find_axiom_violation",
68
+ "selfcheck",
69
+ "yuyay_weights",
70
+ "YUYAY_AXES",
71
+ "YUYAY_FLOORS",
72
+ "layers",
73
+ "DOCTRINE_FOOTER",
74
+ "PROVENANCE",
75
+ "__version__",
76
+ # ---- folded-in governed-norm kernels (Wave D consolidation) ----
77
+ "governed_norm",
78
+ "rms_norm",
79
+ "layer_norm",
80
+ "fused_add_rms_norm",
81
+ ]
82
+
83
+ # ---- folded-in governed-norm surface (Wave D consolidation) ---------------- #
84
+ # Convenience top-level re-exports of the governed normalization kernels that
85
+ # were absorbed from szl-governed-norm. The full surface (ReceiptChain,
86
+ # emit_receipt, receipt_* helpers, selfcheck, layers) lives under
87
+ # ``szl_lambda_gate.governed_norm``. These are a DIFFERENT kernel family from Λ
88
+ # (normalization, not the Λ aggregator); Λ itself remains Conjecture 1
89
+ # (advisory, uniqueness OPEN) and is never described as proven trust.
90
+ rms_norm = governed_norm.rms_norm
91
+ layer_norm = governed_norm.layer_norm
92
+ fused_add_rms_norm = governed_norm.fused_add_rms_norm
93
+
94
+ __version__ = "0.2.0"
95
+ DOCTRINE_FOOTER = (
96
+ "SZL Holdings · Λ = Conjecture 1 (ADVISORY, weighted geometric mean) · "
97
+ "uniqueness OPEN · NOT proven trust · honesty over checklist"
98
+ )
99
+ PROVENANCE = {
100
+ "lean_repo": "szl-holdings/lutar-lean",
101
+ "lean_declarations": 749,
102
+ "lean_axioms": 14,
103
+ "lean_tracked_sorries": 163,
104
+ "doi_lutar_lean": "10.5281/zenodo.20434308",
105
+ "lambda_status": "Conjecture 1 (open) — uniqueness unproven; advisory only",
106
+ }
107
+
108
+
109
+ def lambda_aggregate(
110
+ axes: torch.Tensor,
111
+ weights: Optional[torch.Tensor] = None,
112
+ ) -> torch.Tensor:
113
+ """Λ(x) = ∏ xᵢ^{wᵢ}, the weighted geometric mean over the last dim of axes.
114
+
115
+ See ``szl_lambda_gate._lambda.lambda_aggregate``. Axis scores in [0,1],
116
+ uniform weights when ``weights`` is None. Differentiable, batched, and
117
+ torch.compile-friendly. ADVISORY — NOT proven trust.
118
+ """
119
+ return _lambda_aggregate(axes, weights=weights)
120
+
121
+
122
+ def lambda_gate(
123
+ axes: torch.Tensor,
124
+ weights: Optional[torch.Tensor] = None,
125
+ threshold: float = 0.5,
126
+ ) -> LambdaGateResult:
127
+ """ADVISORY Λ governance gate: returns LambdaGateResult(score, passed,
128
+ threshold, advisory). ``passed`` = Λ(axes) >= threshold. ``threshold`` must
129
+ lie within Λ's range [0,1] (a value outside it is a misconfiguration — a
130
+ negative threshold would advisory-pass a fully-failing Λ=0 candidate — and
131
+ is rejected). A pass is an advisory, non-compensatory signal — NOT proven
132
+ trust (Λ = Conjecture 1).
133
+ """
134
+ return _lambda_gate(axes, weights=weights, threshold=threshold)
135
+
136
+
137
+ def lambda_gate_batch(
138
+ candidates: torch.Tensor,
139
+ weights: Optional[torch.Tensor] = None,
140
+ threshold: float = 0.5,
141
+ ) -> LambdaGateResult:
142
+ """ADVISORY batch gate over many candidate action-vectors (shape (..., N, k)).
143
+
144
+ The realistic per-inference-step call: score all N candidates at once and
145
+ return the advisory pass mask. Returns LambdaGateResult(score, passed,
146
+ threshold, advisory) with score/passed of shape (..., N). ``threshold``
147
+ must lie within Λ's range [0,1] (same domain guard as ``lambda_gate``).
148
+ NOT proven trust.
149
+ """
150
+ return _lambda_gate_batch(candidates, weights=weights, threshold=threshold)
151
+
152
+
153
+ def yuyay_weights(dtype: torch.dtype = torch.float64, device=None) -> torch.Tensor:
154
+ """Canonical 13-axis Yuyay Λ weight vector (uniform 1/13), ADVISORY only.
155
+
156
+ Use as ``weights`` over the 13 ``YUYAY_AXES``. The yuyay_v3 gate is a
157
+ conjunctive AND with per-axis floors (``YUYAY_FLOORS``); this Λ roll-up is
158
+ the weighted geometric mean and is ADVISORY — NOT proven trust.
159
+ """
160
+ return _yuyay_weights(dtype=dtype, device=device)
161
+
162
+
163
+ def find_axiom_violation(k=5, trials=200, weights=None, seed=0, tol=1e-6):
164
+ """Random-search for any A1–A4 violation; returns (axiom, axes, weights) or
165
+ None. An honest falsification attempt — finding nothing is evidence, not a
166
+ proof (Λ-uniqueness is Conjecture 1, open).
167
+ """
168
+ return _find_axiom_violation(k=k, trials=trials, weights=weights, seed=seed, tol=tol)
169
+
170
+
171
+ def selfcheck(k=5, trials=64, seed=0) -> dict:
172
+ """Expose the A1–A4 empirical self-checks + version as a single verdict dict.
173
+
174
+ Callable as get_kernel(...).selfcheck(). EMPIRICAL checks on sampled inputs,
175
+ NOT a proof of Λ-uniqueness (Conjecture 1, open). Advisory only.
176
+ """
177
+ return _selfcheck(k=k, trials=trials, seed=seed)
178
+
179
+
180
+ # ---- axiom runtime self-checks (real, verifiable; NOT a uniqueness proof) -- #
181
+ def is_monotone(axes, weights=None, delta=0.05, tol=1e-7) -> bool:
182
+ """A1 IsMonotone self-check: Λ is non-decreasing in each axis (on this data)."""
183
+ return _is_monotone(axes, weights=weights, delta=delta, tol=tol)
184
+
185
+
186
+ def is_egyptian_exact(c, k=3, weights=None, tol=1e-5) -> bool:
187
+ """A3 IsEgyptianExact self-check: Λ(c, …, c) = c."""
188
+ return _is_egyptian_exact(c, k=k, weights=weights, tol=tol)
189
+
190
+
191
+ def is_bounded_by_max(axes, weights=None, tol=1e-6) -> bool:
192
+ """A4 IsBounded self-check: Λ(x) ≤ maxᵢ xᵢ."""
193
+ return _is_bounded_by_max(axes, weights=weights, tol=tol)
194
+
195
+
196
+ def is_homogeneous(axes, t, weights=None, tol=1e-5) -> bool:
197
+ """A2 IsHomogeneous(degree 1) self-check: Λ(t·x) = t·Λ(x)."""
198
+ return _is_homogeneous(axes, t, weights=weights, tol=tol)
build/torch-universal/szl_lambda_gate/_lambda.py ADDED
@@ -0,0 +1,504 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # © 2026 SZL Holdings · Stephen P. Lutar · ORCID 0009-0001-0110-4173
3
+ """Pure-PyTorch Lambda-Spine aggregator (Λ) for the szl-lambda-gate kernel.
4
+
5
+ Λ(x) = ∏ xᵢ^{wᵢ}, Σwᵢ = 1, wᵢ > 0, xᵢ ∈ [0,1] (weighted geometric mean)
6
+
7
+ This is a TORCH port of the canonical pure-Python reference
8
+ (packages/puriq-os/puriq_os/lambda_aggregator.py — saved alongside this kernel
9
+ as lambda_aggregator_source.py). It is a correctness reference, computed via
10
+ logs in float32 for stability, differentiable (autograd works), and
11
+ torch.compile-friendly. Depends ONLY on torch + the Python standard library
12
+ (a Kernel Hub requirement for universal kernels).
13
+
14
+ WHAT Λ IS / IS NOT (HONESTY — SZL Holdings doctrine v11):
15
+ Λ is the *weighted-geometric-mean aggregator*: a non-compensatory way to
16
+ combine axis scores in [0,1] into one number. It is ADVISORY governance
17
+ signal — a conservative roll-up where any single zeroed axis drives the
18
+ aggregate to 0. It is NOT "proven trust" and NOT a closed theorem. Its
19
+ *uniqueness* (that the weighted geometric mean is the only aggregator
20
+ satisfying the carried axioms) remains Conjecture 1 — OPEN (an unresolved
21
+ CAUCHY_ND step plus a missing symmetry axiom in the Lean development). Do
22
+ not describe Λ as proven trust anywhere.
23
+
24
+ PRIOR ART (honest attribution): the weighted geometric mean as a *less-
25
+ compensatory* composite-indicator aggregator is established practice — the
26
+ UN HDI (arithmetic→geometric switch, 2010), the OECD Handbook on
27
+ Constructing Composite Indicators (2008), and the UNECE well-being
28
+ guidelines all use it "to limit the compensation effect". The veto / cut-off
29
+ idea (a single failing criterion blocks a pass regardless of the others) is
30
+ the ELECTRE veto threshold / "satisficing" minimum-threshold screen. The
31
+ 13-axis conjunctive form exposed by :func:`yuyay_weights` is SZL's own
32
+ yuyay_v3 "Heart" gate. None of this makes Λ "proven trust"; the gate is
33
+ ADVISORY (a11oy: "the advisory Λ trust score is a research conjecture, not a
34
+ pass/fail oracle").
35
+
36
+ PROVENANCE: backed by the Lean 4 formalization szl-holdings/lutar-lean
37
+ (749 declarations / 14 axioms / 163 tracked sorries),
38
+ DOI 10.5281/zenodo.20434308 (lutar-lean).
39
+ Λ uniqueness = Conjecture 1 (open).
40
+
41
+ Axioms carried (Lutar/Axioms.lean), available below as runtime self-checks:
42
+ A1 IsMonotone — Λ is non-decreasing in each axis
43
+ A2 IsHomogeneous — Λ(t·x) = t·Λ(x) (degree 1)
44
+ A3 IsEgyptianExact — Λ(c,…,c) = c (the uniform-diagonal fixpoint)
45
+ A4 IsBounded(by max) — Λ(x) ≤ maxᵢ xᵢ
46
+ """
47
+ from typing import Optional
48
+
49
+ import torch
50
+
51
+ # Compute reductions/log-sum in float32 for stability when inputs are low
52
+ # precision; keep float64 inputs in float64 (downcasting would break gradcheck
53
+ # and silently lose precision).
54
+ _SUPPORTED_DTYPES = (torch.float16, torch.bfloat16, torch.float32, torch.float64)
55
+
56
+
57
+ def _compute_dtype(in_dtype: torch.dtype) -> torch.dtype:
58
+ return torch.float32 if in_dtype in (torch.float16, torch.bfloat16) else in_dtype
59
+
60
+
61
+ def _check_axes(axes: torch.Tensor) -> None:
62
+ """Cheap, allocation-free metadata guards on the axis-score tensor.
63
+
64
+ Inspects only type / dtype / rank / last-dim, so it constant-folds under
65
+ torch.compile and adds no tensor work on the happy path.
66
+ """
67
+ if not isinstance(axes, torch.Tensor):
68
+ raise TypeError(f"axes must be a torch.Tensor, got {type(axes).__name__}")
69
+ if axes.dtype not in _SUPPORTED_DTYPES:
70
+ raise TypeError(
71
+ f"axes has unsupported dtype {axes.dtype}; "
72
+ f"expected one of {tuple(str(d) for d in _SUPPORTED_DTYPES)}"
73
+ )
74
+ if axes.dim() < 1:
75
+ raise ValueError(
76
+ "axes must have at least 1 dimension (the k axis scores live on "
77
+ f"the last dim); got a {axes.dim()}-d tensor"
78
+ )
79
+ if axes.shape[-1] < 1:
80
+ raise ValueError("axes last dimension (k = number of axes) must be >= 1")
81
+
82
+
83
+ def _resolve_weights(
84
+ axes: torch.Tensor,
85
+ weights: Optional[torch.Tensor],
86
+ cdt: torch.dtype,
87
+ ) -> torch.Tensor:
88
+ """Return a normalized (Σw = 1) weight vector of shape (k,) in compute dtype.
89
+
90
+ ``weights=None`` -> uniform 1/k (the Egyptian-exact diagonal). Otherwise the
91
+ weights must be 1-D of length k, strictly positive, with a positive sum;
92
+ they are normalized so Σwᵢ = 1.
93
+ """
94
+ k = axes.shape[-1]
95
+ if weights is None:
96
+ return torch.full((k,), 1.0 / k, dtype=cdt, device=axes.device)
97
+ if not isinstance(weights, torch.Tensor):
98
+ raise TypeError(f"weights must be a torch.Tensor or None, got {type(weights).__name__}")
99
+ if weights.device != axes.device:
100
+ raise ValueError(
101
+ f"weights is on device {weights.device} but axes is on {axes.device}; "
102
+ "move them to the same device"
103
+ )
104
+ if weights.dim() != 1 or weights.shape[0] != k:
105
+ raise ValueError(
106
+ f"weights must be 1-D with shape ({k},) to match the last dim of axes; "
107
+ f"got shape {tuple(weights.shape)}"
108
+ )
109
+ wf = weights.to(cdt)
110
+ # Reject non-finite weights up front: a NaN/Inf weight is meaningless for a
111
+ # governance roll-up and would silently poison the normalization.
112
+ if not bool(torch.all(torch.isfinite(wf))):
113
+ raise ValueError("weights must all be finite (no NaN/Inf)")
114
+ # Positivity / sum guards mirror the pure-Python reference (wᵢ>0, Σw>0).
115
+ if bool(torch.any(wf <= 0.0)):
116
+ raise ValueError("weights must be strictly positive (wᵢ > 0)")
117
+ sw = wf.sum()
118
+ if not bool(sw > 0.0):
119
+ raise ValueError("weights must sum to a positive value")
120
+ return wf / sw
121
+
122
+
123
+ def lambda_aggregate(
124
+ axes: torch.Tensor,
125
+ weights: Optional[torch.Tensor] = None,
126
+ ) -> torch.Tensor:
127
+ """Weighted geometric mean Λ(x) = ∏ xᵢ^{wᵢ} over the last dim of ``axes``.
128
+
129
+ Λ is the (ADVISORY) Lambda-Spine aggregator. Axis scores are expected in
130
+ [0,1] and are clamped into [0,1]; uniform weights (1/k) are used when
131
+ ``weights`` is None — the Egyptian-exact diagonal. Computed via logs in
132
+ float32 (or float64 for float64 inputs) for numerical stability:
133
+
134
+ Λ(x) = exp( Σᵢ wᵢ · log(clamp(xᵢ, 0, 1)) )
135
+
136
+ Non-compensatory zero-routing (A4-consistent): any axis that is zero, OR
137
+ that is NON-FINITE (NaN / ±Inf), is treated as a FAILING axis and drives
138
+ the whole aggregate to exactly 0. This is the conservative governance
139
+ choice — a garbage/invalid axis must never silently pass as a "perfect"
140
+ (clamped-to-1) axis, and the output (and its gradient) stay finite and in
141
+ [0,1] for every input. Zeros/non-finite axes are routed explicitly so
142
+ log(0) = -inf and log(NaN) = NaN never produce a NaN value or gradient.
143
+
144
+ Args:
145
+ axes: tensor of shape (..., k) of axis scores in [0,1]. Batched:
146
+ the reduction is over the last dim, leading dims are batch.
147
+ weights: optional 1-D tensor of shape (k,); None -> uniform. Normalized
148
+ internally so Σwᵢ = 1.
149
+
150
+ Returns:
151
+ tensor of shape (...) — Λ(x) ∈ [0,1] per batch row. Differentiable
152
+ w.r.t. ``axes`` (and ``weights``).
153
+
154
+ HONESTY: this is a non-compensatory governance roll-up, NOT proven trust.
155
+ Λ-uniqueness is Conjecture 1 (open).
156
+ """
157
+ _check_axes(axes)
158
+ in_dtype = axes.dtype
159
+ cdt = _compute_dtype(in_dtype)
160
+ xf = axes.to(cdt)
161
+ w = _resolve_weights(axes, weights, cdt) # (k,), Σw=1
162
+
163
+ # A "bad" axis is one that fails non-compensatorily: a non-positive score
164
+ # OR a non-finite value (NaN / ±Inf). clamp(+inf)=1 would otherwise count a
165
+ # garbage axis as perfect, and clamp(NaN)=NaN would poison the product — we
166
+ # treat BOTH as failing (zeroing) axes. Detect non-finite on the RAW input.
167
+ finite_mask = torch.isfinite(xf)
168
+ xc = xf.clamp(0.0, 1.0)
169
+ bad_mask = (~finite_mask) | (xc <= 0.0)
170
+ any_bad = torch.any(bad_mask, dim=-1) # (...)
171
+
172
+ # Replace bad axes with 1.0 before the log purely to keep log finite and the
173
+ # gradient well-defined; the bad-axis contribution is reinstated via any_bad.
174
+ safe = torch.where(bad_mask, torch.ones_like(xc), xc)
175
+ logx = torch.log(safe) # (..., k)
176
+ acc = (logx * w).sum(dim=-1) # (...) weighted log-sum
177
+ val = torch.exp(acc) # (...) Λ before zero-routing
178
+
179
+ out = torch.where(any_bad, torch.zeros_like(val), val)
180
+ out = out.clamp(0.0, 1.0)
181
+ return out.to(in_dtype)
182
+
183
+
184
+ def lambda_gate(
185
+ axes: torch.Tensor,
186
+ weights: Optional[torch.Tensor] = None,
187
+ threshold: float = 0.5,
188
+ ):
189
+ """ADVISORY governance gate over Λ(x): score plus a pass/fail vs threshold.
190
+
191
+ Computes Λ(x) (see :func:`lambda_aggregate`) and compares it to
192
+ ``threshold``: pass := Λ(x) >= threshold.
193
+
194
+ ``threshold`` must be a finite float within Λ's range ``[0, 1]`` (Λ is the
195
+ weighted geometric mean over [0,1]). This bound is enforced: a threshold
196
+ below 0 or above 1 is meaningless for the advisory gate and is rejected —
197
+ see the non-compensatory rationale below. The domain edges are valid:
198
+ ``0.0`` admits every candidate (a permissive "no-gate" boundary) and
199
+ ``1.0`` admits only a Λ == 1 candidate.
200
+
201
+ Returns a :class:`LambdaGateResult` namedtuple with fields:
202
+ score — Λ(x) tensor of shape (...), in [0,1]
203
+ passed — boolean tensor of shape (...), Λ(x) >= threshold
204
+ threshold — the float threshold used
205
+ advisory — always True; a STANDING reminder that this is a
206
+ non-compensatory governance signal, NOT proven trust.
207
+
208
+ Non-compensatory threshold hardening: because a failing/garbage candidate
209
+ (a zero, NaN, or ±Inf axis) is routed to Λ = 0, a NEGATIVE threshold would
210
+ advisory-"pass" exactly those fully-failing candidates (0 >= t for t < 0) —
211
+ the opposite of a conservative admission gate. A threshold above 1 can
212
+ never pass. Both are misconfigurations, so the [0,1] domain is enforced up
213
+ front rather than silently producing a wrong pass mask.
214
+
215
+ HONESTY: a "pass" is an ADVISORY signal only. Λ is the weighted-geometric-
216
+ mean aggregator; its uniqueness is Conjecture 1 (open). Do not treat a
217
+ pass as proven trust or a closed theorem.
218
+ """
219
+ t = float(threshold)
220
+ if t != t or t == float("inf") or t == float("-inf"):
221
+ raise ValueError(f"threshold must be a finite float, got {threshold!r}")
222
+ if t < 0.0 or t > 1.0:
223
+ raise ValueError(
224
+ "threshold must be within Λ's range [0, 1] (Λ is the weighted "
225
+ f"geometric mean over [0,1]); got {t!r}. A threshold below 0 would "
226
+ "advisory-pass a fully-failing (Λ=0) candidate and one above 1 can "
227
+ "never pass — both signal a misconfigured gate."
228
+ )
229
+ score = lambda_aggregate(axes, weights)
230
+ passed = score >= t
231
+ return LambdaGateResult(score=score, passed=passed, threshold=t, advisory=True)
232
+
233
+
234
+ def lambda_gate_batch(
235
+ candidates: torch.Tensor,
236
+ weights: Optional[torch.Tensor] = None,
237
+ threshold: float = 0.5,
238
+ ):
239
+ """ADVISORY batch gate: score MANY candidate action-vectors in one call.
240
+
241
+ This is the realistic way a model/agent uses the gate — one call per
242
+ inference step that scores every proposed action-vector at once and returns
243
+ the advisory pass mask (which candidates clear the threshold).
244
+
245
+ ``candidates`` is a tensor of shape (..., N, k): the last dim ``k`` holds
246
+ the per-axis scores of a single candidate, and the second-to-last dim ``N``
247
+ enumerates the candidates (any leading dims are extra batch). Equivalent to
248
+ calling :func:`lambda_gate` on the whole tensor — the reduction is over the
249
+ last dim — but named to make the agent-loop intent explicit. ``threshold``
250
+ inherits the same [0,1] domain guard as :func:`lambda_gate` (a threshold
251
+ outside Λ's range is a misconfiguration and is rejected).
252
+
253
+ Returns a :class:`LambdaGateResult` with:
254
+ score — Λ tensor of shape (..., N), one score per candidate
255
+ passed — boolean mask of shape (..., N): score >= threshold
256
+ threshold — the float threshold used
257
+ advisory — always True (NOT proven trust)
258
+
259
+ HONESTY: the pass mask is an ADVISORY, non-compensatory signal. A "pass"
260
+ is not proven trust; Λ-uniqueness is Conjecture 1 (open).
261
+ """
262
+ _check_axes(candidates)
263
+ if candidates.dim() < 2:
264
+ raise ValueError(
265
+ "candidates must be at least 2-D, shape (..., N, k): the last dim is "
266
+ f"the k axis scores and the one before it enumerates the N candidates; "
267
+ f"got a {candidates.dim()}-d tensor"
268
+ )
269
+ # Reuse the single-call gate — its reduction over the last dim already gives
270
+ # one score per candidate, so the (..., N) layout falls out for free.
271
+ return lambda_gate(candidates, weights=weights, threshold=threshold)
272
+
273
+
274
+ # ---- A1..A4 axiom RUNTIME self-checks (real, verifiable) ------------------- #
275
+ # These are honest empirical checks callers can run on concrete inputs. They
276
+ # verify the carried axioms hold for THIS implementation on the given data —
277
+ # they are NOT a proof of Λ-uniqueness (that is Conjecture 1, open).
278
+
279
+ def is_egyptian_exact(
280
+ c: float,
281
+ k: int = 3,
282
+ weights: Optional[torch.Tensor] = None,
283
+ tol: float = 1e-5,
284
+ ) -> bool:
285
+ """A3 IsEgyptianExact: Λ(c, …, c) = c for a constant axis vector of length k.
286
+
287
+ Builds the uniform vector (c repeated k times) and checks Λ equals c within
288
+ ``tol``. ``c`` is clamped into [0,1] to match the aggregator's domain.
289
+ """
290
+ if k < 1:
291
+ raise ValueError("k must be >= 1")
292
+ cc = min(max(float(c), 0.0), 1.0)
293
+ axes = torch.full((k,), cc, dtype=torch.float64)
294
+ val = lambda_aggregate(axes, weights)
295
+ return bool(torch.abs(val - cc) <= tol)
296
+
297
+
298
+ def is_bounded_by_max(
299
+ axes: torch.Tensor,
300
+ weights: Optional[torch.Tensor] = None,
301
+ tol: float = 1e-6,
302
+ ) -> bool:
303
+ """A4 IsBounded: Λ(x) ≤ maxᵢ xᵢ (over the last dim), within ``tol``.
304
+
305
+ Returns True iff the bound holds for every batch row. Non-finite axis
306
+ values are clamped/zero-routed the same way the aggregator treats them, so
307
+ the bound is checked on the conservative (finite) domain.
308
+ """
309
+ _check_axes(axes)
310
+ val = lambda_aggregate(axes, weights) # (...)
311
+ xf = axes.to(_compute_dtype(axes.dtype))
312
+ # Mirror the aggregator: non-finite axes are failing (treated as 0) for the
313
+ # purposes of the max bound, so the check matches the routed semantics.
314
+ xf = torch.where(torch.isfinite(xf), xf, torch.zeros_like(xf))
315
+ mx = xf.clamp(0.0, 1.0).amax(dim=-1) # (...)
316
+ return bool(torch.all(val.to(mx.dtype) <= mx + tol))
317
+
318
+
319
+ def is_homogeneous(
320
+ axes: torch.Tensor,
321
+ t: float,
322
+ weights: Optional[torch.Tensor] = None,
323
+ tol: float = 1e-5,
324
+ ) -> bool:
325
+ """A2 IsHomogeneous (degree 1): Λ(t·x) = t·Λ(x) for scalar t in [0,1].
326
+
327
+ Verified on the clamped domain: both ``axes`` and ``t*axes`` must remain in
328
+ [0,1] for the identity to be meaningful, so ``axes`` is clamped to [0,1] and
329
+ ``t`` to [0,1] before the comparison.
330
+ """
331
+ _check_axes(axes)
332
+ tt = min(max(float(t), 0.0), 1.0)
333
+ x = axes.to(torch.float64).clamp(0.0, 1.0)
334
+ lhs = lambda_aggregate(x * tt, weights)
335
+ rhs = tt * lambda_aggregate(x, weights)
336
+ return bool(torch.all(torch.abs(lhs - rhs) <= tol))
337
+
338
+
339
+ def is_monotone(
340
+ axes: torch.Tensor,
341
+ weights: Optional[torch.Tensor] = None,
342
+ delta: float = 0.05,
343
+ tol: float = 1e-7,
344
+ ) -> bool:
345
+ """A1 IsMonotone: Λ is non-decreasing in each axis.
346
+
347
+ For each axis j, nudges that axis UP by ``delta`` (clamped to stay ≤ 1) on
348
+ every batch row and checks Λ does not decrease (within ``tol``). Rows that
349
+ cannot move (already at 1) are skipped for that axis. A real check on the
350
+ given data — not a symbolic proof.
351
+ """
352
+ _check_axes(axes)
353
+ x = axes.to(torch.float64).clamp(0.0, 1.0)
354
+ base = lambda_aggregate(x, weights)
355
+ k = x.shape[-1]
356
+ ok = True
357
+ for j in range(k):
358
+ bumped = x.clone()
359
+ bumped[..., j] = (bumped[..., j] + float(delta)).clamp(0.0, 1.0)
360
+ bumped_val = lambda_aggregate(bumped, weights)
361
+ # Λ must not go DOWN when an axis goes UP.
362
+ ok = ok and bool(torch.all(bumped_val - base >= -tol))
363
+ return ok
364
+
365
+
366
+ # ---- Adversarial axiom search (honest: a falsification attempt) ------------ #
367
+ def find_axiom_violation(
368
+ k: int = 5,
369
+ trials: int = 200,
370
+ weights: Optional[torch.Tensor] = None,
371
+ seed: Optional[int] = 0,
372
+ tol: float = 1e-6,
373
+ ):
374
+ """Random-search for ANY A1–A4 violation on random axis/weight draws.
375
+
376
+ Returns the first ``(axiom, axes, weights)`` triple that violates a carried
377
+ axiom within ``tol``, or ``None`` if none is found in ``trials`` draws. This
378
+ is an honest FALSIFICATION attempt on this implementation — finding nothing
379
+ is empirical evidence, NOT a proof (Λ-uniqueness is Conjecture 1, open).
380
+ """
381
+ gen = torch.Generator()
382
+ if seed is not None:
383
+ gen.manual_seed(int(seed))
384
+ for _ in range(int(trials)):
385
+ x = torch.rand(k, generator=gen, dtype=torch.float64)
386
+ w = weights
387
+ if w is None:
388
+ w = torch.rand(k, generator=gen, dtype=torch.float64) + 1e-3
389
+ # A3 on a constant draw
390
+ c = float(torch.rand(1, generator=gen).item())
391
+ if not is_egyptian_exact(c, k=k, weights=w, tol=max(tol, 1e-5)):
392
+ return ("A3_IsEgyptianExact", torch.full((k,), c, dtype=torch.float64), w)
393
+ # A4 bounded-by-max
394
+ if not is_bounded_by_max(x, w, tol=max(tol, 1e-6)):
395
+ return ("A4_IsBounded", x, w)
396
+ # A2 homogeneous at a random t
397
+ t = float(torch.rand(1, generator=gen).item())
398
+ if not is_homogeneous(x, t, weights=w, tol=max(tol, 1e-5)):
399
+ return ("A2_IsHomogeneous", x, w)
400
+ # A1 monotone (leave headroom so an up-bump stays in range)
401
+ if not is_monotone(x * 0.9, w, tol=max(tol, 1e-7)):
402
+ return ("A1_IsMonotone", x * 0.9, w)
403
+ return None
404
+
405
+
406
+ # ---- Canonical 13-axis Yuyay preset (ADVISORY ONLY) ------------------------ #
407
+ # SZL's own yuyay_v3 "Heart" gate is a 13-axis CONJUNCTIVE-AND screen (each axis
408
+ # independently clears its floor — no compensation). We expose its published
409
+ # axis NAMES and per-axis FLOORS as advisory metadata, and a uniform Λ weight
410
+ # vector over the 13 axes. This is ADVISORY: Λ here is still the weighted
411
+ # geometric mean, and a "pass" is a research-conjecture signal, NOT proven
412
+ # trust. Source: yuyay_v3 spec (Lutar, 2026).
413
+ YUYAY_AXES = (
414
+ "moralGrounding",
415
+ "measurabilityHonesty",
416
+ "empiricalGrounding",
417
+ "logicalConsistency",
418
+ "sourceTransparency",
419
+ "reproducibility",
420
+ "licenseHygiene",
421
+ "scopeDiscipline",
422
+ "claimCalibration",
423
+ "evalAwareness",
424
+ "deceptionKeywords",
425
+ "conflictingDirectives",
426
+ "reversalDirective",
427
+ )
428
+ # Published per-axis advisory floors for the CONJUNCTIVE screen: two "sacred"
429
+ # axes at 0.95, seven "structural" at 0.90, four "introspection" at 0.90.
430
+ YUYAY_FLOORS = (
431
+ 0.95, 0.95, # sacred
432
+ 0.90, 0.90, 0.90, 0.90, 0.90, 0.90, 0.90, # structural (7)
433
+ 0.90, 0.90, 0.90, 0.90, # introspection (4)
434
+ )
435
+
436
+
437
+ def yuyay_weights(
438
+ dtype: torch.dtype = torch.float64,
439
+ device: Optional[torch.device] = None,
440
+ ) -> torch.Tensor:
441
+ """Canonical 13-axis Yuyay Λ weight vector (uniform 1/13), ADVISORY only.
442
+
443
+ Returns a length-13 weight tensor for use as the ``weights`` argument to
444
+ :func:`lambda_aggregate` / :func:`lambda_gate` over the 13 :data:`YUYAY_AXES`.
445
+ Uniform by default (the Egyptian-exact diagonal). The published yuyay_v3
446
+ gate is a conjunctive AND with per-axis floors (:data:`YUYAY_FLOORS`); the
447
+ Λ roll-up here is the weighted geometric mean and is ADVISORY — NOT proven
448
+ trust (Λ-uniqueness is Conjecture 1, open).
449
+ """
450
+ k = len(YUYAY_AXES)
451
+ return torch.full((k,), 1.0 / k, dtype=dtype, device=device)
452
+
453
+
454
+ # ---- Kernel self-check surface --------------------------------------------- #
455
+ def selfcheck(
456
+ k: int = 5,
457
+ trials: int = 64,
458
+ seed: Optional[int] = 0,
459
+ ) -> dict:
460
+ """Run the A1–A4 empirical self-checks and report a verdict + version.
461
+
462
+ Returns a dict:
463
+ version — kernel version string
464
+ axioms — {A1..A4: bool} empirical pass on sampled inputs
465
+ all_axioms_hold — bool, every sampled axiom check passed
466
+ adversarial — {trials, violation} from a random falsification search
467
+ (violation is None when no violation was found)
468
+ advisory — always True
469
+ lambda_status — Conjecture 1 (open) honesty string
470
+
471
+ HONESTY: these are EMPIRICAL checks on sampled inputs, NOT a proof of
472
+ Λ-uniqueness (Conjecture 1, open). A clean run is evidence, not proof.
473
+ """
474
+ x = torch.rand(k, dtype=torch.float64) * 0.9 # headroom for the A1 up-bump
475
+ w = torch.rand(k, dtype=torch.float64) + 1e-3
476
+ axioms = {
477
+ "A1_IsMonotone": is_monotone(x, w),
478
+ "A2_IsHomogeneous": is_homogeneous(x, float(torch.rand(1).item()), weights=w),
479
+ "A3_IsEgyptianExact": is_egyptian_exact(float(torch.rand(1).item()), k=k, weights=w),
480
+ "A4_IsBounded": is_bounded_by_max(x, w),
481
+ }
482
+ violation = find_axiom_violation(k=k, trials=trials, seed=seed)
483
+ return {
484
+ "version": __version__,
485
+ "axioms": axioms,
486
+ "all_axioms_hold": all(axioms.values()) and violation is None,
487
+ "adversarial": {"trials": int(trials), "violation": violation},
488
+ "advisory": True,
489
+ "lambda_status": "Conjecture 1 (open) — uniqueness unproven; advisory only",
490
+ }
491
+
492
+
493
+ # Kept in sync with the package __version__ (single source of truth lives in
494
+ # __init__; duplicated here so _lambda is importable/selfcheck-able standalone).
495
+ __version__ = "0.2.0"
496
+
497
+
498
+ # Namedtuple result type for the gate. Defined after functions so docstrings
499
+ # above can reference it; imported by __init__ and layers.
500
+ from collections import namedtuple # noqa: E402
501
+
502
+ LambdaGateResult = namedtuple(
503
+ "LambdaGateResult", ["score", "passed", "threshold", "advisory"]
504
+ )
build/torch-universal/szl_lambda_gate/_ops.py ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # Auto-style ops namespace shim for the universal kernel. Unique suffix lets
3
+ # multiple versions load in the same process (Kernel Hub requirement).
4
+ import torch
5
+
6
+ ops = torch.ops._szl_lambda_gate_20260623081355
7
+
8
+
9
+ def add_op_namespace_prefix(op_name: str) -> str:
10
+ return f"_szl_lambda_gate_20260623081355::{op_name}"
build/torch-universal/szl_lambda_gate/governed_norm/__init__.py ADDED
@@ -0,0 +1,279 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # © 2026 SZL Holdings · Stephen P. Lutar · ORCID 0009-0001-0110-4173
3
+ """szl_lambda_gate.governed_norm — governed normalization kernels (folded in).
4
+
5
+ CONSOLIDATION (Wave D): this subpackage is the ``szl-governed-norm`` universal
6
+ kernel folded into the canonical ``szl-lambda-gate`` kernels package so the two
7
+ duplicate micro-repos become ONE canonical home. The source repo
8
+ ``szl-holdings/szl-governed-norm`` is DEPRECATED (see its DEPRECATED.md) and
9
+ points here; nothing was deleted — this is the additive, reversible copy.
10
+
11
+ It provides correctness-verified RMSNorm and LayerNorm that run on CPU and CUDA
12
+ and are torch.compile-friendly, plus an optional *governed* path that emits
13
+ content-addressed, SHA3-256 hash-chained receipts of each call — provenance at
14
+ the kernel layer, in the spirit of the a11oy governed-AI platform
15
+ (https://a-11-oy.com).
16
+
17
+ Usage (as a subpackage of the canonical kernel)::
18
+
19
+ import torch
20
+ from szl_lambda_gate import governed_norm as gn
21
+
22
+ print(gn.selfcheck()) # one-shot correctness + receipt check
23
+ x = torch.randn(4, 1024, dtype=torch.float16)
24
+ y = gn.rms_norm(x, eps=1e-6) # plain path
25
+ y2 = gn.rms_norm(x, eps=1e-6, governed=True) # records to the default chain
26
+ chain = gn.ReceiptChain()
27
+ y3 = gn.rms_norm(x, eps=1e-6, chain=chain) # records into YOUR chain only
28
+ print(chain.verify()) # (ok, depth, first_break_seq)
29
+
30
+ Honesty: this is a universal (pure-Python) kernel — a correctness reference,
31
+ not a hand-tuned CUDA speed record. No fabricated benchmarks. Its
32
+ differentiator is verifiable governance, not raw FLOPs. Λ = Conjecture 1
33
+ (advisory, uniqueness OPEN) — never described as proven trust anywhere.
34
+
35
+ Note on torch.compile: every op is torch.compile(fullgraph=True)-compatible.
36
+ Receipt emission is an eager-only side effect (it hashes materialized tensor
37
+ bytes), so when a *governed* call is captured into a compiled graph the
38
+ numerics are unchanged but NO receipt is recorded — govern at the eager audit
39
+ boundary. This is documented honestly and covered by tests.
40
+ """
41
+ from typing import Any, Dict, List, Optional, Tuple
42
+
43
+ import torch
44
+
45
+ from . import layers # noqa: F401 (must be importable for Hub layer mapping)
46
+ from ._norm import fused_add_rms_norm as _fused_add_rms_norm
47
+ from ._norm import layer_norm as _layer_norm
48
+ from ._norm import rms_norm as _rms_norm
49
+ from ._receipt import _GENESIS as _GENESIS_HEAD
50
+ from ._receipt import ReceiptChain, default_chain, emit_receipt
51
+
52
+ __all__ = [
53
+ "rms_norm",
54
+ "layer_norm",
55
+ "fused_add_rms_norm",
56
+ "layers",
57
+ "ReceiptChain",
58
+ "emit_receipt",
59
+ "receipt_head",
60
+ "receipt_count",
61
+ "receipt_tail",
62
+ "receipt_verify",
63
+ "selfcheck",
64
+ "DOCTRINE_FOOTER",
65
+ "__version__",
66
+ ]
67
+
68
+ __version__ = "0.2.0"
69
+ DOCTRINE_FOOTER = (
70
+ "SZL Holdings · governed normalization · provenance at the kernel layer · "
71
+ "Lambda = Conjecture 1 (advisory) · honesty over checklist"
72
+ )
73
+
74
+
75
+ def _is_tracing() -> bool:
76
+ """True while torch.compile / Dynamo is tracing this code.
77
+
78
+ Receipt emission reads materialized tensor bytes (hashing on CPU), which is
79
+ an inherently eager, side-effecting host operation that cannot live inside
80
+ a traced FX graph — so under torch.compile we skip the emit. This keeps
81
+ EVERY op torch.compile(fullgraph=True)-compatible while remaining honest:
82
+ when a governed call is captured into a compiled graph, NO receipt is
83
+ recorded (the numerics are unchanged and identical to the eager path).
84
+ Governance is intended for the eager audit boundary; record receipts there.
85
+ """
86
+ is_compiling = getattr(torch.compiler, "is_compiling", None)
87
+ return bool(is_compiling()) if is_compiling is not None else False
88
+
89
+
90
+ def _emit(
91
+ chain: Optional[ReceiptChain],
92
+ op: str,
93
+ x: torch.Tensor,
94
+ out: torch.Tensor,
95
+ eps: float,
96
+ sign_key: Optional[Any] = None,
97
+ organ: str = "szl-governed-norm",
98
+ ) -> None:
99
+ """Append a receipt to ``chain`` (or the process default chain if None).
100
+
101
+ No-op while torch.compile is tracing (see ``_is_tracing``). When
102
+ ``sign_key`` (a PEM ECDSA-P256 private key) is supplied and szl-receipt is
103
+ installed, the receipt carries an additive DSSE ``signature`` envelope;
104
+ keyless is UNSIGNED-honest.
105
+ """
106
+ if _is_tracing():
107
+ return
108
+ target = chain if chain is not None else default_chain()
109
+ target.emit(op, x, out, eps, sign_key=sign_key, organ=organ)
110
+
111
+
112
+ def rms_norm(
113
+ x: torch.Tensor,
114
+ weight: Optional[torch.Tensor] = None,
115
+ eps: float = 1e-6,
116
+ governed: bool = False,
117
+ chain: Optional[ReceiptChain] = None,
118
+ sign_key: Optional[Any] = None,
119
+ organ: str = "szl-governed-norm",
120
+ ) -> torch.Tensor:
121
+ """RMSNorm over the last dim.
122
+
123
+ If ``governed=True``, append an audit receipt. By default the receipt goes
124
+ to the process-wide default chain (convenient). Pass your own ``chain`` (a
125
+ ``ReceiptChain`` instance) to record into a caller-owned chain instead —
126
+ this avoids global-state contention when many threads/requests govern
127
+ independently. Passing ``chain`` implies governance even if
128
+ ``governed=False`` is left at its default. Pass ``sign_key`` (PEM
129
+ ECDSA-P256) to additively sign the receipt via szl-receipt.
130
+ """
131
+ out = _rms_norm(x, weight=weight, eps=eps)
132
+ if governed or chain is not None:
133
+ _emit(chain, "rms_norm", x, out, eps, sign_key=sign_key, organ=organ)
134
+ return out
135
+
136
+
137
+ def layer_norm(
138
+ x: torch.Tensor,
139
+ weight: Optional[torch.Tensor] = None,
140
+ bias: Optional[torch.Tensor] = None,
141
+ eps: float = 1e-5,
142
+ governed: bool = False,
143
+ chain: Optional[ReceiptChain] = None,
144
+ sign_key: Optional[Any] = None,
145
+ organ: str = "szl-governed-norm",
146
+ ) -> torch.Tensor:
147
+ """LayerNorm over the last dim.
148
+
149
+ If ``governed=True`` (or a ``chain`` is supplied), append an audit receipt
150
+ to ``chain`` when given, otherwise to the process default chain. See
151
+ ``rms_norm`` for the per-call ``chain`` rationale and ``sign_key``.
152
+ """
153
+ out = _layer_norm(x, weight=weight, bias=bias, eps=eps)
154
+ if governed or chain is not None:
155
+ _emit(chain, "layer_norm", x, out, eps, sign_key=sign_key, organ=organ)
156
+ return out
157
+
158
+
159
+ def fused_add_rms_norm(
160
+ x: torch.Tensor,
161
+ residual: torch.Tensor,
162
+ weight: Optional[torch.Tensor] = None,
163
+ eps: float = 1e-6,
164
+ governed: bool = False,
165
+ chain: Optional[ReceiptChain] = None,
166
+ sign_key: Optional[Any] = None,
167
+ organ: str = "szl-governed-norm",
168
+ ) -> Tuple[torch.Tensor, torch.Tensor]:
169
+ """Residual-add + RMSNorm (transformer block pattern).
170
+
171
+ Returns ``(y, new_residual)`` where ``new_residual = x + residual`` and
172
+ ``y = rms_norm(new_residual, weight, eps)``. If ``governed=True`` (or a
173
+ ``chain`` is supplied), append an audit receipt over the normalized output
174
+ to ``chain`` when given, otherwise to the process default chain. Pass
175
+ ``sign_key`` to additively sign the receipt via szl-receipt.
176
+ """
177
+ out, new_residual = _fused_add_rms_norm(x, residual, weight=weight, eps=eps)
178
+ if governed or chain is not None:
179
+ _emit(chain, "fused_add_rms_norm", x, out, eps, sign_key=sign_key, organ=organ)
180
+ return out, new_residual
181
+
182
+
183
+ # ---- governance receipt surface (operates on the default in-process chain) --
184
+ def receipt_head() -> str:
185
+ """SHA3-256 head of the governed-call receipt chain ('0'*64 if empty)."""
186
+ return default_chain().head()
187
+
188
+
189
+ def receipt_count() -> int:
190
+ """Number of governed calls recorded."""
191
+ return default_chain().count()
192
+
193
+
194
+ def receipt_tail(n: int = 10) -> List[Dict[str, Any]]:
195
+ """Last n receipts."""
196
+ return default_chain().tail(n)
197
+
198
+
199
+ def receipt_verify() -> Dict[str, Any]:
200
+ """Re-walk the receipt chain. Returns {ok, depth, first_break_seq}."""
201
+ ok, depth, brk = default_chain().verify()
202
+ return {"ok": ok, "depth": depth, "first_break_seq": brk, "head": default_chain().head()}
203
+
204
+
205
+ # ---- one-shot self-verification --------------------------------------------
206
+ def selfcheck() -> Dict[str, Any]:
207
+ """Verify correctness + governance in a single call; never raises.
208
+
209
+ Runs a tiny, self-contained, CPU-only smoke test against PyTorch references
210
+ so downstream code (and SZL's own a11oy / hatun-mcp) can confirm the loaded
211
+ kernel is the real, working article before trusting it.
212
+
213
+ Checks (all on a *private, throwaway* ReceiptChain so the process default
214
+ chain is never touched):
215
+ * ``rms_norm`` matches a Llama-style float32 reference,
216
+ * ``layer_norm`` matches ``torch.nn.functional.layer_norm``,
217
+ * ``fused_add_rms_norm`` matches the unfused add-then-norm path,
218
+ * a governed call emits exactly one receipt and the chain verifies.
219
+
220
+ Returns a JSON-able dict:
221
+ ``{ok, version, checks: {name: bool}, receipt_ok, receipt_head, error}``
222
+ ``ok`` is True iff every check passed. On unexpected failure ``ok`` is
223
+ False and ``error`` carries the message — this function is designed to be
224
+ safe to call in a health probe and will not raise.
225
+ """
226
+ checks: Dict[str, bool] = {}
227
+ receipt_ok = False
228
+ receipt_head = _GENESIS_HEAD
229
+ error = None
230
+ try:
231
+ torch.manual_seed(0)
232
+ x = torch.randn(4, 64, dtype=torch.float32)
233
+ w = torch.randn(64, dtype=torch.float32)
234
+ b = torch.randn(64, dtype=torch.float32)
235
+ res = torch.randn(4, 64, dtype=torch.float32)
236
+ eps_r, eps_l = 1e-6, 1e-5
237
+
238
+ # rms_norm vs Llama-style fp32 reference
239
+ xf = x.to(torch.float32)
240
+ ref_rms = (xf * torch.rsqrt(xf.pow(2).mean(-1, keepdim=True) + eps_r)) * w
241
+ checks["rms_norm"] = bool(
242
+ torch.allclose(rms_norm(x, weight=w, eps=eps_r), ref_rms, rtol=1e-5, atol=1e-5)
243
+ )
244
+
245
+ # layer_norm vs torch reference
246
+ ref_ln = torch.nn.functional.layer_norm(x, (64,), weight=w, bias=b, eps=eps_l)
247
+ checks["layer_norm"] = bool(
248
+ torch.allclose(layer_norm(x, weight=w, bias=b, eps=eps_l), ref_ln,
249
+ rtol=1e-5, atol=1e-5)
250
+ )
251
+
252
+ # fused_add_rms_norm vs unfused path
253
+ y_f, new_res = fused_add_rms_norm(x, res, weight=w, eps=eps_r)
254
+ h = x.to(torch.float32) + res.to(torch.float32)
255
+ ref_y = rms_norm(h.to(x.dtype), weight=w, eps=eps_r)
256
+ checks["fused_add_rms_norm"] = bool(
257
+ torch.allclose(y_f, ref_y, rtol=1e-5, atol=1e-5)
258
+ and torch.allclose(new_res, h.to(x.dtype), rtol=1e-6, atol=1e-6)
259
+ )
260
+
261
+ # governance on a private chain: one emit, chain verifies
262
+ probe_chain = ReceiptChain()
263
+ rms_norm(x, weight=w, eps=eps_r, chain=probe_chain)
264
+ ok, depth, brk = probe_chain.verify()
265
+ receipt_ok = bool(ok and depth == 1 and brk == -1)
266
+ receipt_head = probe_chain.head()
267
+ checks["governance"] = receipt_ok
268
+ except Exception as exc: # never raise from a health probe
269
+ error = f"{type(exc).__name__}: {exc}"
270
+
271
+ ok = bool(checks) and all(checks.values()) and error is None
272
+ return {
273
+ "ok": ok,
274
+ "version": __version__,
275
+ "checks": checks,
276
+ "receipt_ok": receipt_ok,
277
+ "receipt_head": receipt_head,
278
+ "error": error,
279
+ }
build/torch-universal/szl_lambda_gate/governed_norm/_norm.py ADDED
@@ -0,0 +1,246 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # © 2026 SZL Holdings · Stephen P. Lutar · ORCID 0009-0001-0110-4173
3
+ """Pure-PyTorch normalization primitives for the SZL governed-norm kernel.
4
+
5
+ These are correctness-verified reference implementations (RMSNorm, LayerNorm,
6
+ and the residual-fused RMSNorm pattern used by transformer blocks) written in
7
+ pure PyTorch. They run on CPU and CUDA, are torch.compile-friendly, and depend
8
+ ONLY on torch + the Python standard library (a Kernel Hub requirement for
9
+ universal kernels).
10
+
11
+ HONESTY: this is a *universal* (pure-Python) kernel. It does NOT ship a
12
+ hand-tuned CUDA/Triton binary, so it is a correctness reference, not a
13
+ speed record. We make no fabricated benchmark claims. Where it adds value
14
+ is the optional *governed* path (see _receipt.py): every normalization call
15
+ can emit a content-addressed, hash-chained receipt of its inputs/outputs so
16
+ the operation is auditable — SZL Holdings' provenance doctrine applied at
17
+ the kernel layer.
18
+
19
+ Numerical convention (all ops): reductions and the normalization math are
20
+ computed in float32 for stability, then the result is cast back to the input
21
+ dtype. This is the standard Llama-style convention and is what makes
22
+ float16 / bfloat16 inputs numerically well-behaved.
23
+
24
+ Validation convention: guards below are cheap, branch-only checks on metadata
25
+ (dtype / ndim / shape / device) — they allocate nothing on the happy path and
26
+ constant-fold away under torch.compile, so they do not perturb traced graphs.
27
+ They exist to turn silent broadcasting / device-mismatch bugs into clear,
28
+ early errors. A zero-size normalized last dimension is rejected (normalizing
29
+ over zero elements is undefined); a single-element last dimension is allowed
30
+ (RMSNorm yields sign(x); LayerNorm yields 0, matching F.layer_norm).
31
+
32
+ Non-finite convention (NaN / Inf inputs): these ops do NOT sanitize their
33
+ input. A NaN or Inf in the input propagates through the reduction and appears
34
+ in the output, exactly as it would in torch.nn.functional.layer_norm / a
35
+ hand-written kernel. We deliberately do NOT silently replace non-finite values
36
+ (that would hide upstream numerical bugs); detecting/handling them is the
37
+ caller's responsibility. This propagation behavior is covered by regression
38
+ tests so it cannot change unnoticed.
39
+ """
40
+ from typing import Optional
41
+
42
+ import torch
43
+
44
+ # Floating dtypes this kernel supports. Integer / complex inputs are rejected
45
+ # early with a clear message rather than silently producing garbage.
46
+ _SUPPORTED_DTYPES = (torch.float16, torch.bfloat16, torch.float32, torch.float64)
47
+
48
+
49
+ def _compute_dtype(in_dtype: torch.dtype) -> torch.dtype:
50
+ """Reduction/normalization compute dtype.
51
+
52
+ Low-precision inputs (fp16/bf16) are upcast to float32 for stability — the
53
+ standard Llama-style convention. float64 inputs are NOT downcast: doing so
54
+ would silently lose precision (and break gradcheck), so we keep float64.
55
+ """
56
+ return torch.float32 if in_dtype in (torch.float16, torch.bfloat16) else in_dtype
57
+
58
+
59
+ def _check_input(x: torch.Tensor, name: str = "x") -> None:
60
+ """Cheap, allocation-free guards on the primary input tensor.
61
+
62
+ Only inspects metadata (type / dtype / ndim), so it is constant-folded by
63
+ torch.compile and adds no runtime tensor work on the happy path.
64
+ """
65
+ if not isinstance(x, torch.Tensor):
66
+ raise TypeError(f"{name} must be a torch.Tensor, got {type(x).__name__}")
67
+ if x.dtype not in _SUPPORTED_DTYPES:
68
+ raise TypeError(
69
+ f"{name} has unsupported dtype {x.dtype}; "
70
+ f"expected one of {tuple(str(d) for d in _SUPPORTED_DTYPES)}"
71
+ )
72
+ if x.dim() < 1:
73
+ raise ValueError(
74
+ f"{name} must have at least 1 dimension (the normalized dim); "
75
+ f"got a {x.dim()}-d tensor"
76
+ )
77
+ # A zero-size normalized (last) dimension is mathematically undefined:
78
+ # mean/RMS over zero elements is NaN, so normalization has no meaning.
79
+ # Reject it early with a clear message instead of silently returning an
80
+ # empty/NaN tensor (the classic shape-bug-masquerading-as-success case).
81
+ if x.shape[-1] == 0:
82
+ raise ValueError(
83
+ f"{name} has a zero-size normalized last dimension {tuple(x.shape)}; "
84
+ f"normalization over zero elements is undefined"
85
+ )
86
+
87
+
88
+ def _check_affine(
89
+ x: torch.Tensor,
90
+ param: Optional[torch.Tensor],
91
+ name: str,
92
+ ) -> None:
93
+ """Validate an optional affine parameter (weight/bias/residual peer).
94
+
95
+ Enforces that the parameter is 1-D and matches the normalized (last)
96
+ dimension, and lives on the same device as ``x``. This catches the
97
+ classic silent-broadcast bug where a mis-shaped weight would broadcast
98
+ instead of erroring. Metadata-only: no allocations.
99
+ """
100
+ if param is None:
101
+ return
102
+ if not isinstance(param, torch.Tensor):
103
+ raise TypeError(f"{name} must be a torch.Tensor or None, got {type(param).__name__}")
104
+ if param.device != x.device:
105
+ raise ValueError(
106
+ f"{name} is on device {param.device} but x is on {x.device}; "
107
+ f"move them to the same device"
108
+ )
109
+ last = x.shape[-1]
110
+ if param.dim() != 1 or param.shape[0] != last:
111
+ raise ValueError(
112
+ f"{name} must be 1-D with shape ({last},) to match the normalized "
113
+ f"last dimension of x; got shape {tuple(param.shape)}"
114
+ )
115
+
116
+
117
+ def _check_eps(eps: float) -> None:
118
+ """eps must be a positive, finite scalar (rsqrt(var+eps) must be safe)."""
119
+ e = float(eps)
120
+ if not (e > 0.0) or e != e or e == float("inf"):
121
+ raise ValueError(f"eps must be a positive finite float, got {eps!r}")
122
+
123
+
124
+ def rms_norm(
125
+ x: torch.Tensor,
126
+ weight: Optional[torch.Tensor] = None,
127
+ eps: float = 1e-6,
128
+ ) -> torch.Tensor:
129
+ """Root-mean-square layer normalization over the last dimension.
130
+
131
+ y = x / sqrt(mean(x^2, dim=-1) + eps) * weight
132
+
133
+ Computed in float32 for numerical stability, then cast back to the input
134
+ dtype (the standard, correctness-preserving convention used by Llama-style
135
+ RMSNorm). `weight` is optional; when omitted, no affine scale is applied.
136
+
137
+ Raises clear TypeError/ValueError on bad dtype, rank, eps, or a weight
138
+ whose shape/device does not match x's normalized dimension.
139
+ """
140
+ _check_input(x)
141
+ _check_eps(eps)
142
+ _check_affine(x, weight, "weight")
143
+
144
+ in_dtype = x.dtype
145
+ xf = x.to(_compute_dtype(in_dtype))
146
+ variance = xf.pow(2).mean(dim=-1, keepdim=True)
147
+ xf = xf * torch.rsqrt(variance + eps)
148
+ out = xf.to(in_dtype)
149
+ if weight is not None:
150
+ out = out * weight
151
+ return out
152
+
153
+
154
+ def layer_norm(
155
+ x: torch.Tensor,
156
+ weight: Optional[torch.Tensor] = None,
157
+ bias: Optional[torch.Tensor] = None,
158
+ eps: float = 1e-5,
159
+ ) -> torch.Tensor:
160
+ """Standard layer normalization over the last dimension.
161
+
162
+ Mean/variance computed in float32 for stability, then cast back. Matches
163
+ torch.nn.functional.layer_norm semantics for the normalized-shape = last
164
+ dim case; verified against it in the test suite.
165
+
166
+ Raises clear TypeError/ValueError on bad dtype, rank, eps, or a
167
+ weight/bias whose shape/device does not match x's normalized dimension.
168
+ """
169
+ _check_input(x)
170
+ _check_eps(eps)
171
+ _check_affine(x, weight, "weight")
172
+ _check_affine(x, bias, "bias")
173
+
174
+ in_dtype = x.dtype
175
+ xf = x.to(_compute_dtype(in_dtype))
176
+ mean = xf.mean(dim=-1, keepdim=True)
177
+ # Biased (population) variance = mean of squared deviations. We compute it
178
+ # directly rather than via Tensor.var(unbiased=False): torch's .var emits a
179
+ # "degrees of freedom <= 0" UserWarning when the normalized dim has a single
180
+ # element, even though unbiased=False is well-defined there (variance 0).
181
+ # Computing it ourselves matches F.layer_norm exactly and stays silent and
182
+ # torch.compile(fullgraph=True)-clean for the single-element edge case.
183
+ centered = xf - mean
184
+ var = centered.pow(2).mean(dim=-1, keepdim=True)
185
+ xf = centered * torch.rsqrt(var + eps)
186
+ out = xf.to(in_dtype)
187
+ if weight is not None:
188
+ out = out * weight
189
+ if bias is not None:
190
+ out = out + bias
191
+ return out
192
+
193
+
194
+ def fused_add_rms_norm(
195
+ x: torch.Tensor,
196
+ residual: torch.Tensor,
197
+ weight: Optional[torch.Tensor] = None,
198
+ eps: float = 1e-6,
199
+ ):
200
+ """Residual-add followed by RMSNorm — the canonical transformer block pattern.
201
+
202
+ h = x + residual # updated residual stream
203
+ y = rms_norm(h, weight, eps)
204
+ return y, h
205
+
206
+ This mirrors the `fused_add_rms_norm` used in real LLM inference stacks
207
+ (e.g. the pre-norm transformer block: the normalized output `y` feeds the
208
+ sublayer, while the un-normalized sum `h` is carried forward as the next
209
+ residual). We return BOTH so callers can thread the residual stream, which
210
+ is exactly why the fused form exists.
211
+
212
+ HONESTY: "fused" here means *logically* fused (one Python op, one float32
213
+ cast path, the add done in float32 alongside the norm) — it is a correct,
214
+ allocation-conscious pure-PyTorch reference, not a hand-written fused CUDA
215
+ kernel. No speed claims are made.
216
+
217
+ The add is performed in float32 so that, for float16/bfloat16 inputs, the
218
+ residual accumulation does not lose precision before normalization — this
219
+ matches high-quality reference implementations.
220
+ """
221
+ _check_input(x, "x")
222
+ _check_input(residual, "residual")
223
+ _check_eps(eps)
224
+ if residual.shape != x.shape:
225
+ raise ValueError(
226
+ f"residual shape {tuple(residual.shape)} must equal x shape "
227
+ f"{tuple(x.shape)} for the residual add"
228
+ )
229
+ if residual.device != x.device:
230
+ raise ValueError(
231
+ f"residual is on device {residual.device} but x is on {x.device}; "
232
+ f"move them to the same device"
233
+ )
234
+ _check_affine(x, weight, "weight")
235
+
236
+ in_dtype = x.dtype
237
+ cdt = _compute_dtype(in_dtype)
238
+ # Add in compute dtype, keep both the normalized output and the residual.
239
+ hf = x.to(cdt) + residual.to(cdt)
240
+ new_residual = hf.to(in_dtype)
241
+ variance = hf.pow(2).mean(dim=-1, keepdim=True)
242
+ yf = hf * torch.rsqrt(variance + eps)
243
+ out = yf.to(in_dtype)
244
+ if weight is not None:
245
+ out = out * weight
246
+ return out, new_residual
build/torch-universal/szl_lambda_gate/governed_norm/_receipt.py ADDED
@@ -0,0 +1,253 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # © 2026 SZL Holdings · Stephen P. Lutar · ORCID 0009-0001-0110-4173
3
+ """Content-addressed governance receipts for normalization calls.
4
+
5
+ SZL Holdings' provenance doctrine applied at the kernel layer: when a
6
+ normalization runs in *governed* mode, it emits a small, deterministic
7
+ receipt describing the call — input shape/dtype, eps, and a SHA3-256 digest
8
+ of the (quantized) output tensor — and hash-chains it to the previous
9
+ receipt. This makes a sequence of kernel calls independently auditable
10
+ without trusting the caller.
11
+
12
+ HONESTY:
13
+ - The digest is a real SHA3-256 over the output bytes (rounded to a fixed
14
+ decimal precision so it is reproducible across runs/devices). It is an
15
+ integrity fingerprint, NOT a cryptographic signature — we never claim
16
+ it proves authorship. DSSE signing is a separate, out-of-band concern.
17
+ - Receipts are kept in an in-process, append-only chain. Nothing is written
18
+ to disk or the network from inside the kernel.
19
+ - Stdlib + torch (+ numpy for the output digest) only — Kernel Hub
20
+ universal-kernel requirement.
21
+ - The canonical szl-receipt v0.2.0 evidence binding (``emit_receipt``) is
22
+ ADDITIVE and IMPORT-GUARDED: with szl-receipt absent it returns ``None`` and
23
+ the kernel runs unchanged. It binds subject / input-digest / output-digest /
24
+ policy-id / energy; energy is the literal string "UNAVAILABLE" because this
25
+ kernel measures NO joules — a value is never fabricated. Like the SHA3-256
26
+ chain, it is an EVIDENCE trail, NOT a proof of correctness.
27
+ """
28
+ import hashlib
29
+ import json
30
+ import threading
31
+ import time
32
+ from typing import Any, Dict, List, Optional, Union
33
+
34
+ import torch
35
+
36
+ _GENESIS = "0" * 64
37
+
38
+ # Logical signing-authority label stamped onto signature envelopes.
39
+ _ORGAN = "szl-governed-norm"
40
+
41
+ # Governing policy id bound into every canonical szl-receipt evidence binding.
42
+ _POLICY_ID = "szl-governed-norm/provenance@v1"
43
+
44
+ # This universal kernel measures NO joules. The honesty doctrine forbids
45
+ # fabricating an energy value, so the canonical binding records the literal
46
+ # string "UNAVAILABLE" rather than a placeholder number.
47
+ _ENERGY_UNAVAILABLE = "UNAVAILABLE"
48
+
49
+
50
+ def _maybe_sign(
51
+ body: Dict[str, Any],
52
+ sign_key: Optional[Union[str, bytes]],
53
+ organ: str,
54
+ ) -> Optional[Dict[str, Any]]:
55
+ """ADDITIVE szl-receipt signature layer over the receipt *body*.
56
+
57
+ Returns a DSSE envelope (from ``szl_receipt.sign_receipt``) covering the
58
+ exact canonical body, or ``None`` when szl-receipt is not installed (the
59
+ kernel then behaves exactly as before). Doctrine: with no *sign_key* the
60
+ envelope is UNSIGNED-honest (``signed=False``); a signature is NEVER
61
+ fabricated. This is distinct from and additive to the SHA3-256 chain
62
+ integrity hash (``digest``) — szl-receipt's envelope carries its own
63
+ SHA-256 ``digest``/``algo`` so the two integrity hashes are explicit.
64
+ """
65
+ try:
66
+ from szl_receipt import Receipt, sign_receipt
67
+ except Exception: # noqa: BLE001 - signing is optional; absence is honest
68
+ return None
69
+ env = sign_receipt(Receipt(kind="governed-norm", body=body),
70
+ sign_key, organ=organ)
71
+ return env
72
+
73
+
74
+ def _tensor_digest(t: torch.Tensor, decimals: int = 6) -> str:
75
+ """Deterministic SHA3-256 over a tensor's rounded float32 contents.
76
+
77
+ Rounding to a fixed number of decimals makes the digest stable across
78
+ devices/dtypes for the same logical values (tiny FP noise won't change
79
+ it). This is an integrity fingerprint, not a signature.
80
+ """
81
+ flat = t.detach().to(torch.float32).reshape(-1)
82
+ # Round to `decimals` places, integerize, hash the raw bytes. CPU move is
83
+ # required to read bytes; kept O(n) and allocation-light.
84
+ scaled = torch.round(flat * (10 ** decimals)).to(torch.int64).cpu().numpy().tobytes()
85
+ h = hashlib.sha3_256()
86
+ h.update(scaled)
87
+ return h.hexdigest()
88
+
89
+
90
+ def _input_digest(x: torch.Tensor, eps: float) -> str:
91
+ """SHA3-256 over the canonical JSON of a call's input spec.
92
+
93
+ Binds {input shape, dtype, eps} — the *shape* of the call, not the input
94
+ bytes — so the receipt is a compact fingerprint of what produced the
95
+ output. Deterministic and stdlib-only (json + hashlib).
96
+ """
97
+ spec = {
98
+ "in_shape": list(x.shape),
99
+ "in_dtype": str(x.dtype).replace("torch.", ""),
100
+ "eps": float(eps),
101
+ }
102
+ raw = json.dumps(spec, sort_keys=True, separators=(",", ":")).encode("utf-8")
103
+ return hashlib.sha3_256(raw).hexdigest()
104
+
105
+
106
+ def emit_receipt(
107
+ op: str,
108
+ x: torch.Tensor,
109
+ out: torch.Tensor,
110
+ eps: float,
111
+ subject: Optional[str] = None,
112
+ policy_id: str = _POLICY_ID,
113
+ sign_key: Optional[Union[str, bytes]] = None,
114
+ organ: str = _ORGAN,
115
+ ) -> Optional[Dict[str, Any]]:
116
+ """Canonical szl-receipt v0.2.0 evidence binding for a governed-norm call.
117
+
118
+ ADDITIVE and IMPORT-GUARDED: returns ``None`` when szl-receipt is not
119
+ installed, so this universal Kernel-Hub kernel still imports and runs on
120
+ stdlib + torch + numpy alone. When szl-receipt is present it binds an
121
+ EVIDENCE trail and wraps it in a DSSE envelope via ``sign_receipt``:
122
+
123
+ subject organ / norm-call id (who/what emitted this)
124
+ input_digest SHA3-256 over canonical {input shape, dtype, eps}
125
+ output_digest the EXISTING SHA3-256 rounded-tensor digest of ``out``
126
+ policy_id the governing policy id
127
+ energy the literal string "UNAVAILABLE"
128
+
129
+ Doctrine (non-negotiable):
130
+ * A receipt is an integrity/EVIDENCE trail, NOT a proof of correctness.
131
+ * ``energy == "UNAVAILABLE"`` — this kernel measures NO joules; a joule is
132
+ NEVER fabricated.
133
+ * Keyless => UNSIGNED-honest (``signature["signed"] is False``); a
134
+ signature is NEVER fabricated. A real ``sign_key`` yields a real DSSE
135
+ signature over the exact canonical binding.
136
+
137
+ Returns the binding dict (subject/input_digest/output_digest/policy_id/
138
+ energy) with the DSSE envelope under ``signature``, or ``None`` when
139
+ szl-receipt is absent.
140
+ """
141
+ try:
142
+ from szl_receipt import Receipt, sign_receipt
143
+ except Exception: # noqa: BLE001 - canonical binding is optional; absence is honest
144
+ return None
145
+ body = {
146
+ "subject": subject if subject is not None else f"{organ}/{op}",
147
+ "input_digest": _input_digest(x, eps),
148
+ "output_digest": _tensor_digest(out),
149
+ "policy_id": policy_id,
150
+ "energy": _ENERGY_UNAVAILABLE,
151
+ }
152
+ env = sign_receipt(Receipt(kind="governed-norm", body=body), sign_key, organ=organ)
153
+ return dict(body, signature=env)
154
+
155
+
156
+ class ReceiptChain:
157
+ """Append-only, SHA3-256 hash-chained log of normalization receipts.
158
+
159
+ Each receipt: {seq, op, in_shape, in_dtype, eps, out_digest, prev, digest, ts}
160
+ digest = SHA3-256 over the canonical JSON body (excluding digest/ts).
161
+ verify() re-walks the chain and returns (ok, depth, first_break_seq).
162
+ """
163
+
164
+ def __init__(self) -> None:
165
+ self._lock = threading.RLock()
166
+ self._records: List[Dict[str, Any]] = []
167
+
168
+ @staticmethod
169
+ def _digest_body(body: Dict[str, Any]) -> str:
170
+ raw = json.dumps(body, sort_keys=True, separators=(",", ":")).encode("utf-8")
171
+ return hashlib.sha3_256(raw).hexdigest()
172
+
173
+ def emit(
174
+ self,
175
+ op: str,
176
+ x: torch.Tensor,
177
+ out: torch.Tensor,
178
+ eps: float,
179
+ sign_key: Optional[Union[str, bytes]] = None,
180
+ organ: str = _ORGAN,
181
+ policy_id: str = _POLICY_ID,
182
+ ) -> Dict[str, Any]:
183
+ with self._lock:
184
+ prev = self._records[-1]["digest"] if self._records else _GENESIS
185
+ seq = len(self._records)
186
+ body = {
187
+ "seq": seq,
188
+ "op": op,
189
+ "in_shape": list(x.shape),
190
+ "in_dtype": str(x.dtype).replace("torch.", ""),
191
+ "eps": float(eps),
192
+ "out_digest": _tensor_digest(out),
193
+ "prev": prev,
194
+ }
195
+ digest = self._digest_body(body)
196
+ rec = dict(body, digest=digest, ts=time.time())
197
+ sig = _maybe_sign(body, sign_key, organ)
198
+ if sig is not None:
199
+ rec["signature"] = sig
200
+ # ADDITIVE canonical szl-receipt v0.2.0 evidence binding. Import-
201
+ # guarded: None when szl-receipt is absent, so the universal kernel
202
+ # keeps working on stdlib + torch + numpy only. Binds subject /
203
+ # input-digest / output-digest / policy-id / energy; energy is
204
+ # "UNAVAILABLE" (no joules measured here). It does NOT enter the
205
+ # SHA3-256 chain body, so verify() is unaffected.
206
+ binding = emit_receipt(
207
+ op, x, out, eps,
208
+ subject=f"{organ}/{op}#{seq}",
209
+ policy_id=policy_id,
210
+ sign_key=sign_key,
211
+ organ=organ,
212
+ )
213
+ if binding is not None:
214
+ rec["receipt"] = binding
215
+ self._records.append(rec)
216
+ return rec
217
+
218
+ def head(self) -> str:
219
+ with self._lock:
220
+ return self._records[-1]["digest"] if self._records else _GENESIS
221
+
222
+ def count(self) -> int:
223
+ with self._lock:
224
+ return len(self._records)
225
+
226
+ def tail(self, n: int = 10) -> List[Dict[str, Any]]:
227
+ with self._lock:
228
+ return list(self._records[-n:])
229
+
230
+ def verify(self):
231
+ """Re-walk the chain. Returns (ok: bool, depth: int, first_break: int)."""
232
+ with self._lock:
233
+ prev = _GENESIS
234
+ for i, rec in enumerate(self._records):
235
+ body = {k: rec[k] for k in
236
+ ("seq", "op", "in_shape", "in_dtype", "eps", "out_digest", "prev")}
237
+ if rec["prev"] != prev or rec["digest"] != self._digest_body(body):
238
+ return (False, len(self._records), i)
239
+ prev = rec["digest"]
240
+ return (True, len(self._records), -1)
241
+
242
+
243
+ # Module-level default chain (opt-in: only written when governed=True is used).
244
+ _DEFAULT_CHAIN: Optional[ReceiptChain] = None
245
+ _chain_lock = threading.Lock()
246
+
247
+
248
+ def default_chain() -> ReceiptChain:
249
+ global _DEFAULT_CHAIN
250
+ with _chain_lock:
251
+ if _DEFAULT_CHAIN is None:
252
+ _DEFAULT_CHAIN = ReceiptChain()
253
+ return _DEFAULT_CHAIN
build/torch-universal/szl_lambda_gate/governed_norm/layers.py ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # © 2026 SZL Holdings · Stephen P. Lutar · ORCID 0009-0001-0110-4173
3
+ """Hub-compliant kernel layers.
4
+
5
+ Per the Kernel Hub `kernel-requirements`, layers exposed for extension must
6
+ be PURE torch.nn.Module subclasses:
7
+ - no custom __init__,
8
+ - no class variables,
9
+ - only a `forward` method,
10
+ - forward signature compatible with the module it extends.
11
+
12
+ These layers therefore read their parameters (weight/bias/eps) off the
13
+ module instance they are bound to (set by the host model), and only define
14
+ `forward`. They are drop-in replacements for an existing RMSNorm/LayerNorm
15
+ module via the `kernels` layer-mapping mechanism.
16
+ """
17
+ import torch
18
+ from torch import nn
19
+
20
+ from ._norm import fused_add_rms_norm, layer_norm, rms_norm
21
+
22
+
23
+ class RMSNorm(nn.Module):
24
+ """Pure RMSNorm layer. Expects the host module to provide `self.weight`
25
+ (optional) and `self.variance_epsilon` or `self.eps`."""
26
+
27
+ def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
28
+ weight = getattr(self, "weight", None)
29
+ eps = getattr(self, "variance_epsilon", None)
30
+ if eps is None:
31
+ eps = getattr(self, "eps", 1e-6)
32
+ return rms_norm(hidden_states, weight=weight, eps=float(eps))
33
+
34
+
35
+ class LayerNorm(nn.Module):
36
+ """Pure LayerNorm layer. Expects the host module to provide `self.weight`
37
+ (optional), `self.bias` (optional), and `self.eps`."""
38
+
39
+ def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
40
+ weight = getattr(self, "weight", None)
41
+ bias = getattr(self, "bias", None)
42
+ eps = getattr(self, "eps", 1e-5)
43
+ return layer_norm(hidden_states, weight=weight, bias=bias, eps=float(eps))
44
+
45
+
46
+ class FusedAddRMSNorm(nn.Module):
47
+ """Pure residual-add + RMSNorm layer for pre-norm transformer blocks.
48
+
49
+ Expects the host module to provide `self.weight` (optional) and
50
+ `self.variance_epsilon` or `self.eps`. Returns `(normalized, new_residual)`
51
+ where `new_residual = hidden_states + residual` is carried forward as the
52
+ next block's residual stream.
53
+ """
54
+
55
+ def forward(self, hidden_states: torch.Tensor, residual: torch.Tensor):
56
+ weight = getattr(self, "weight", None)
57
+ eps = getattr(self, "variance_epsilon", None)
58
+ if eps is None:
59
+ eps = getattr(self, "eps", 1e-6)
60
+ return fused_add_rms_norm(hidden_states, residual, weight=weight, eps=float(eps))
build/torch-universal/szl_lambda_gate/layers.py ADDED
@@ -0,0 +1,51 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # © 2026 SZL Holdings · Stephen P. Lutar · ORCID 0009-0001-0110-4173
3
+ """Hub-compliant kernel layer for the szl-lambda-gate kernel.
4
+
5
+ Per the Kernel Hub `kernel-requirements`, layers exposed for extension must be
6
+ PURE torch.nn.Module subclasses:
7
+ - no custom __init__,
8
+ - no class variables,
9
+ - only a `forward` method.
10
+
11
+ The layer therefore reads its parameters (weights / threshold) off the module
12
+ instance it is bound to (set by the host model) and only defines `forward`.
13
+
14
+ HONESTY: `LambdaGate` emits an ADVISORY governance signal (the weighted
15
+ geometric mean Λ plus a pass/fail vs threshold). Λ is NOT proven trust; its
16
+ uniqueness is Conjecture 1 (open).
17
+ """
18
+ import torch
19
+ from torch import nn
20
+
21
+ from ._lambda import lambda_aggregate, lambda_gate
22
+
23
+
24
+ class LambdaGate(nn.Module):
25
+ """Pure Λ-gate layer.
26
+
27
+ Reads optional ``self.weights`` (1-D, length k) and ``self.threshold``
28
+ (float, default 0.5) off the bound module instance.
29
+
30
+ forward(axes) -> LambdaGateResult(score, passed, threshold, advisory) where
31
+ ``score`` = Λ(axes) over the last dim and ``passed`` = score >= threshold.
32
+ Differentiable in ``score`` w.r.t. ``axes``.
33
+ """
34
+
35
+ def forward(self, axes: torch.Tensor):
36
+ weights = getattr(self, "weights", None)
37
+ threshold = getattr(self, "threshold", 0.5)
38
+ return lambda_gate(axes, weights=weights, threshold=float(threshold))
39
+
40
+
41
+ class LambdaAggregate(nn.Module):
42
+ """Pure Λ-aggregator layer: forward(axes) -> Λ(axes) tensor in [0,1].
43
+
44
+ Reads optional ``self.weights`` (1-D, length k) off the bound module
45
+ instance; uniform weights when absent. Returns just the score (no gate),
46
+ fully differentiable w.r.t. ``axes``.
47
+ """
48
+
49
+ def forward(self, axes: torch.Tensor) -> torch.Tensor:
50
+ weights = getattr(self, "weights", None)
51
+ return lambda_aggregate(axes, weights=weights)