PairwisePM / tests /test_engine.py
brettleehari's picture
PairwisePM v0.1.0 — mechanical pairwise judge for GenAI product decisions (Spine and Leaf build)
3eb6857 verified
Raw History Blame Contribute Delete
46.5 kB
"""PairwisePM engine test suite (pytest, engine-only — no Gradio required).
Implements the Compliance leaf's executable specifications
(tests/test_engine_spec.md) against the AS-BUILT API in ``pairwisepm``.
Renames from the spec's provisional §0 shapes are deliberate and declared:
- ``compare(cfg, raw_a, raw_b, *, ...) -> ComparisonResult`` (spec: Verdict);
probabilities at ``result.m0.p`` / ``result.m1.p`` (spec: p_m0/p_m1),
verdict strings "A"/"B"/"too_close" (spec: "a"/"b"/"too_close").
- ``fit_m1(delta_rows, picks, cfg, lam=...) -> M1Fit`` with ``.active`` as the
activation gate (spec: m1_active) and ``.free_intercept`` always fitted as
the diagnostic (spec: free_intercept=True flag).
- ``DecisionLog`` methods ``kendall_zeta``/``brier`` (spec: zeta/brier free
functions); JSONL via ``to_jsonl``/``from_jsonl``.
- Engineering's declared choices (per the exchange round): infeasible parity
paths are KEPT in the list marked ``feasible=False`` (T-PAR-03 option 1);
fragility epsilon is ``FRAGILITY_Z = 0.75`` z-units pending Decisions D6
(T-FRAG constants below are derived for 0.75, not the spec's provisional
0.5); order-bias significance is a Wald z from the penalized Hessian with
|z| > 2 as the gate.
Every test that pins a numeric constant derived from the D1/D2 default prior
weights says so — regenerate on resolution.
"""
from __future__ import annotations
import inspect
import json
import math
import re
import sys
import numpy as np
import pytest
import pairwisepm
from pairwisepm import (
FRAGILITY_Z,
M1_ACTIVATION_N,
DecisionLog,
Factor,
ModeConfig,
blended_stats,
compare,
fit_m1,
load_default_configs,
load_mode_config,
make_record,
verdict_from,
)
from pairwisepm.strings import (
PINNED_CEILING,
PINNED_FRAGILITY_TMPL,
STRINGS,
)
# ---------------------------------------------------------------------------
# Fixtures
# ---------------------------------------------------------------------------
W1N_KEYS = ["impact_primary_metric", "capability_feasibility", "reach",
"unit_economics", "data_flywheel", "risk_surface", "effort"]
W1N = [0.25, 0.20, 0.15, 0.15, 0.10, 0.05, -0.10] # D1 defaults (spec §2)
W1N_TAGS = ["lever", "lever", "fact", "lever", "fact", "lever", "lever"]
# D_FIX z-deltas in schema order; L = W1N · d = 0.26 (hand-computed).
D_FIX = [1.0, 0.5, -0.2, 0.0, 0.3, -1.0, 0.4]
L_FIX = 0.26
P_FIX = 0.5646362918030292
def std1_config(stds=None, ranges=None, weights=None) -> ModeConfig:
"""A 1→N-shaped config with seed mean 0 / std 1 (raw values ARE z-values)
unless overridden — isolates the scorer from the z pipeline (spec §0)."""
stds = stds or [1.0] * 7
ranges = ranges or [(-100.0, 100.0)] * 7
weights = weights or W1N
factors = [
Factor(key=k, name=k.replace("_", " "), scale="test",
prior_weight=abs(w), sign=1 if w >= 0 else -1, tag=t,
rationale="test fixture", seed_mean=0.0, seed_std=s,
transform="identity", plausible_range=r)
for k, w, t, s, r in zip(W1N_KEYS, weights, W1N_TAGS, stds, ranges)
]
return ModeConfig(mode="one_to_n", label="1toN-test", banner=None,
factors=factors)
STD1 = std1_config()
def ideas_from_deltas(d):
"""raw_a = d, raw_b = 0 so z_A − z_B = d under STD1."""
return (dict(zip(W1N_KEYS, [float(x) for x in d])),
dict(zip(W1N_KEYS, [0.0] * 7)))
@pytest.fixture(scope="module")
def shipped():
return load_default_configs()
# ---------------------------------------------------------------------------
# T-SYM — swap A/B ⇒ p exactly 1 − p
# ---------------------------------------------------------------------------
def _assert_swap(cfg, raw_a, raw_b, seed=7):
va = compare(cfg, raw_a, raw_b, seed=seed)
vb = compare(cfg, raw_b, raw_a, seed=seed)
assert vb.m0.logit == -va.m0.logit # bitwise (zero intercept)
assert abs(va.m0.p + vb.m0.p - 1.0) <= 1e-12
assert abs(vb.m0.interval[0] - (1 - va.m0.interval[1])) <= 1e-12
assert abs(vb.m0.interval[1] - (1 - va.m0.interval[0])) <= 1e-12
if va.m0.verdict == "A":
assert vb.m0.verdict == "B"
elif va.m0.verdict == "B":
assert vb.m0.verdict == "A"
else:
assert vb.m0.verdict == "too_close"
# equal-weight ablation preserves symmetry (T-SYM-02)
assert vb.m0_equal.logit == -va.m0_equal.logit
assert abs(va.m0_equal.p + vb.m0_equal.p - 1.0) <= 1e-12
def test_sym_fixture():
_assert_swap(STD1, *ideas_from_deltas(D_FIX))
def test_sym_property_both_modes(shipped):
rng = np.random.default_rng(20260830)
for _ in range(25):
d = rng.standard_normal(7)
_assert_swap(STD1, *ideas_from_deltas(d))
z2o = shipped["zero_to_one"]
for _ in range(25):
raw_a, raw_b = {}, {}
for f in z2o.factors:
lo, hi = f.plausible_range
raw_a[f.key] = float(rng.uniform(lo, hi))
raw_b[f.key] = float(rng.uniform(lo, hi))
_assert_swap(z2o, raw_a, raw_b)
def test_sym_m1_scoring():
"""Swap symmetry holds for M1 too (zero intercept in the scoring fit)."""
rng = np.random.default_rng(3)
rows = [dict(zip(W1N_KEYS, rng.standard_normal(7))) for _ in range(12)]
picks = [1 if sum(w * r[k] for w, k in zip(W1N, W1N_KEYS)) > 0 else 0
for r in rows]
fit = fit_m1(rows, picks, STD1)
raw_a, raw_b = ideas_from_deltas(D_FIX)
va = compare(STD1, raw_a, raw_b, m1_fit=fit)
vb = compare(STD1, raw_b, raw_a, m1_fit=fit)
assert vb.m1.logit == -va.m1.logit
assert abs(va.m1.p + vb.m1.p - 1.0) <= 1e-12
# ---------------------------------------------------------------------------
# T-EQ — identical ideas ⇒ exactly 0.5 and too_close
# ---------------------------------------------------------------------------
def test_equal_ideas(shipped):
for cfg in (STD1, shipped["one_to_n"], shipped["zero_to_one"]):
raw = {f.key: (f.plausible_range[0] + f.plausible_range[1]) / 2
for f in cfg.factors}
v = compare(cfg, dict(raw), dict(raw))
assert v.m0.logit == 0.0
assert v.m0.p == 0.5
assert v.m0.verdict == "too_close"
assert all(r.contribution == 0.0 for r in v.leverage)
# ---------------------------------------------------------------------------
# T-M0 / T-CFG — hand-computed case; weights are config, not code
# ---------------------------------------------------------------------------
def test_m0_hand_value():
# Pins the D1 default weights deliberately; regenerate if D1 changes.
v = compare(STD1, *ideas_from_deltas(D_FIX))
assert abs(v.m0.logit - L_FIX) <= 1e-12
assert abs(v.m0.p - P_FIX) <= 1e-9
def test_weights_are_config():
w = list(W1N)
w[0] = 0.40 # impact .25 -> .40
cfg = std1_config(weights=w)
v = compare(cfg, *ideas_from_deltas(D_FIX))
assert abs(v.m0.logit - 0.41) <= 1e-12
def test_config_shapes(shipped):
n2n, z2o = shipped["one_to_n"], shipped["zero_to_one"]
assert len(n2n.factors) == 7 and len(z2o.factors) == 6
for cfg in (n2n, z2o):
for f in cfg.factors:
assert isinstance(f.prior_weight, float) and f.prior_weight > 0
assert f.sign in (-1, 1)
assert f.tag in ("lever", "fact")
assert f.rationale.strip()
assert f.scale.strip()
assert f.seed_std > 0
# Ceiling: a Flag with NO weight field at all (unweighted by construction)
ceiling = [g for g in z2o.flags if g.key == "ceiling"]
assert len(ceiling) == 1
assert not hasattr(ceiling[0], "prior_weight")
assert not hasattr(ceiling[0], "tag")
# |weights| sum to 1.0 in both modes (spec §2 defaults)
for cfg in (n2n, z2o):
assert abs(sum(f.prior_weight for f in cfg.factors) - 1.0) <= 1e-9
def test_zero_to_one_banner_mandatory(shipped, tmp_path):
assert shipped["zero_to_one"].banner == STRINGS["str.banner.01"]
assert shipped["one_to_n"].banner is None
bad = tmp_path / "bad.yaml"
bad.write_text(
"mode: zero_to_one\nlabel: x\nbanner: null\nfactors:\n"
" - {key: k, name: n, scale: s, prior_weight: 1.0, sign: 1,\n"
" tag: fact, rationale: r, seed_mean: 0, seed_std: 1}\n",
encoding="utf-8")
with pytest.raises(ValueError):
load_mode_config(bad)
# ---------------------------------------------------------------------------
# T-MAP — MAP regularization never diverges on degenerate logs
# ---------------------------------------------------------------------------
def _converged(fit):
w = np.array(list(fit.weights.values()))
assert np.all(np.isfinite(w)) and np.max(np.abs(w)) <= 10.0
def test_map_empty_log():
fit = fit_m1([], [], STD1)
for k, w in zip(W1N_KEYS, W1N):
assert fit.weights[k] == w # bitwise: prior returned exactly
assert fit.active is False
def test_map_one_decision():
rows = [dict(zip(W1N_KEYS, D_FIX))]
_converged(fit_m1(rows, [1], STD1))
def test_map_complete_separation():
rng = np.random.default_rng(1)
rows, picks = [], []
for _ in range(12):
d = rng.standard_normal(7)
if sum(w * x for w, x in zip(W1N, d)) < 0:
d = -d
rows.append(dict(zip(W1N_KEYS, d)))
picks.append(1) # pick "A" every time
fit = fit_m1(rows, picks, STD1)
_converged(fit)
assert np.isfinite(fit.free_intercept)
def test_map_rank_deficient_stays_at_prior():
"""A factor with zero observed variation stays AT its prior weight —
shrinkage centered on the prior, not zero (the classic mistake §3 forbids)."""
rng = np.random.default_rng(4)
rows = []
for _ in range(12):
d = rng.standard_normal(7)
d[2] = 0.0 # reach never varies
rows.append(dict(zip(W1N_KEYS, d)))
picks = [1 if sum(w * r[k] for w, k in zip(W1N, W1N_KEYS)) > 0 else 0
for r in rows]
fit = fit_m1(rows, picks, STD1)
_converged(fit)
assert abs(fit.weights["reach"] - W1N[2]) <= 1e-9
# ---------------------------------------------------------------------------
# T-VERDICT — interval straddling 0.5 ⇒ too_close (boundary breaks toward it)
# ---------------------------------------------------------------------------
def test_verdict_rule_units():
assert verdict_from(0.55, (0.48, 0.61)) == "too_close"
assert verdict_from(0.56, (0.52, 0.61)) == "A"
assert verdict_from(0.44, (0.39, 0.48)) == "B"
assert verdict_from(0.53, (0.50, 0.57)) == "too_close" # touching counts
def test_verdict_end_to_end():
near = [0.05, -0.03, 0.02, 0.0, -0.02, 0.04, 0.01]
v = compare(STD1, *ideas_from_deltas(near), seed=42, n_boot=1000)
lo, hi = v.m0.interval
assert lo <= 0.5 <= hi and v.m0.verdict == "too_close"
# Clear winner: the canonical seed-comparison deltas (contributions spread
# across factors). NOTE for compliance: the spec's 3×D_FIX fixture is NOT
# a clear winner under bootstrap-over-factors — D_FIX's dominant single
# contribution (+0.75) against a large negative (−0.30) makes the
# resampled interval straddle 50% even at L = 0.78. Flagged in the
# exchange round; T-VERDICT-02's clear-winner fixture needs regenerating.
clear = [0.5, 0.4, -0.4 / 3.0, 0.7, 0.8, -0.5, 0.0]
v2 = compare(STD1, *ideas_from_deltas(clear))
assert v2.m0.verdict == "A" and v2.m0.interval[0] > 0.5
# even on a too-close, lever paths are computed (S4 "cheapest evidence")
assert v.parity_paths and all(p.tag == "lever" for p in v.parity_paths)
def test_interval_always_present(shipped):
for d in (D_FIX, [0.0] * 7):
v = compare(STD1, *ideas_from_deltas(d))
for sc in (v.m0, v.m0_equal):
lo, hi = sc.interval
assert 0.0 <= lo <= sc.p <= hi <= 1.0
# ---------------------------------------------------------------------------
# T-CEIL — ceiling flag never changes the score
# ---------------------------------------------------------------------------
def test_ceiling_never_scores(shipped):
z2o = shipped["zero_to_one"]
raw_a = {"problem_severity_frequency": 3.0,
"capability_trajectory_exposure": 1.0, "evidence_of_pull": 2.0,
"genai_necessity": 3.0, "data_distribution_advantage": 2.0,
"cost_to_test": 3.0}
raw_b = {"problem_severity_frequency": 2.0,
"capability_trajectory_exposure": -1.0, "evidence_of_pull": 1.0,
"genai_necessity": 3.0, "data_distribution_advantage": 1.0,
"cost_to_test": 2.0}
runs = [compare(z2o, raw_a, raw_b, seed=11, ceiling_a=ca, ceiling_b=cb)
for ca, cb in ((False, False), (True, False), (True, True))]
base = runs[0]
for v in runs[1:]:
assert v.m0.p == base.m0.p and v.m0.logit == base.m0.logit
assert v.m0.interval == base.m0.interval
assert [r.contribution for r in v.leverage] == \
[r.contribution for r in base.leverage]
assert [(p.key, p.raw_target, p.feasible) for p in v.parity_paths] == \
[(p.key, p.raw_target, p.feasible) for p in base.parity_paths]
assert base.ceiling_notes == []
assert PINNED_CEILING in runs[1].ceiling_notes[0]
assert len(runs[2].ceiling_notes) == 2
# ---------------------------------------------------------------------------
# T-EVAL — evaluative inputs are excluded from the score BY CONSTRUCTION
# ---------------------------------------------------------------------------
def test_evaluative_excluded_by_type():
# The engine's compare() has no evaluative parameter at all: exclusion is
# by type, not convention. They round-trip through the log record instead.
assert "evaluative" not in inspect.signature(compare).parameters
v = compare(STD1, *ideas_from_deltas(D_FIX))
rec = make_record(
timestamp="2026-08-30T12:00:00Z", mode="one_to_n",
name_a="A", name_b="B",
raw_a=ideas_from_deltas(D_FIX)[0], raw_b=ideas_from_deltas(D_FIX)[1],
result=v, pick="A",
evaluative_a={"strategic_fit": "core bet"},
evaluative_b={"strategic_fit": "adjacent"},
)
assert rec["evaluative"]["a"]["strategic_fit"] == "core bet"
# ---------------------------------------------------------------------------
# T-PAR — path to parity: levers only, targets always reported, range flagged
# ---------------------------------------------------------------------------
def test_parity_arithmetic():
# Impact std 1.2: z-shift = 0.26/0.25 = 1.04; raw = 1.04 × 1.2 = 1.248.
cfg = std1_config(stds=[1.2, 1.0, 500.0, 1.0, 1.0, 1.0, 2.0],
ranges=[(-2, 2), (0, 4), (0, 1e7), (0, 4), (0, 4),
(0, 4), (0.5, 52)])
raw_a = dict(zip(W1N_KEYS, [1.2 * 1.0, 2.5, 900.0, 2.0, 2.3, 1.0, 6.8]))
raw_b = dict(zip(W1N_KEYS, [0.0, 2.0, 1000.0, 2.0, 2.0, 2.0, 6.0]))
v = compare(cfg, raw_a, raw_b)
assert abs(v.m0.logit - L_FIX) <= 1e-9 # deltas reproduce D_FIX
path = next(p for p in v.parity_paths if p.key == "impact_primary_metric")
assert path.loser == "B"
assert abs(path.z_shift - 1.04) <= 1e-9
assert abs(path.raw_shift - 1.248) <= 1e-9
assert abs(path.raw_target - (0.0 + 1.248)) <= 1e-9
assert path.feasible
# 60/40 for the loser: (0.26 + ln 1.5) / 0.25
assert abs(path.z_shift_6040 - 2.6618604324326583) <= 1e-9
def test_parity_levers_only():
v = compare(STD1, *ideas_from_deltas(D_FIX))
named = {p.key for p in v.parity_paths}
assert "reach" not in named and "data_flywheel" not in named # facts
for p in v.parity_paths:
assert p.tag == "lever"
def test_parity_infeasible_marked_not_hidden():
# risk (lever, w=.05): z-shift 0.26/0.05 = 5.2 ⇒ target 2.0+5.2 = 7.2 > 4.
cfg = std1_config(ranges=[(-2, 2), (0, 4), (0, 1e7), (0, 4), (0, 4),
(0, 4), (0.5, 52)])
raw_a = dict(zip(W1N_KEYS, [1.0, 2.5, 0.0, 2.0, 0.3, 1.0, 6.4]))
raw_b = dict(zip(W1N_KEYS, [0.0, 2.0, 0.2, 2.0, 0.0, 2.0, 6.0]))
v = compare(cfg, raw_a, raw_b)
risk = next(p for p in v.parity_paths if p.key == "risk_surface")
assert risk.feasible is False # marked infeasible AND kept in the list
assert abs(risk.raw_target - 7.2) <= 1e-9 # target reported, not clamped
def test_loses_on_fundamentals(shipped):
z2o = shipped["zero_to_one"]
# Gap entirely on facts; both lever paths infeasible (pull at scale top,
# cost_to_test at its floor for both ideas).
raw_a = {"problem_severity_frequency": 3.5,
"capability_trajectory_exposure": 1.0, "evidence_of_pull": 4.0,
"genai_necessity": 3.0, "data_distribution_advantage": 2.0,
"cost_to_test": 0.25}
raw_b = {"problem_severity_frequency": 2.0,
"capability_trajectory_exposure": 0.0, "evidence_of_pull": 4.0,
"genai_necessity": 2.0, "data_distribution_advantage": 1.5,
"cost_to_test": 0.25}
v = compare(z2o, raw_a, raw_b)
assert v.m0.verdict == "A"
assert not any(p.feasible for p in v.parity_paths)
assert v.loses_on_fundamentals is True
# ---------------------------------------------------------------------------
# T-FRAG — fragility flag (FRAGILITY_Z = 0.75 pending Decisions D6)
# ---------------------------------------------------------------------------
def test_fragility_fragile():
# d = [0.4, 0...] ⇒ L = 0.10; impact needs 0.4σ ≤ 0.75σ ⇒ fragile.
v = compare(STD1, *ideas_from_deltas([0.4, 0, 0, 0, 0, 0, 0]))
assert v.fragility.fragile is True
assert v.fragility.factor_key == "impact_primary_metric"
assert PINNED_FRAGILITY_TMPL.format(
factor=v.fragility.factor_name) in v.fragility.message
def test_fragility_robust():
# D_FIX: L = 0.26; smallest flip needs 0.26/0.25 = 1.04σ > 0.75σ.
v = compare(STD1, *ideas_from_deltas(D_FIX))
assert v.fragility.fragile is False
# ---------------------------------------------------------------------------
# T-ZETA — Kendall's ζ on known cases (most recent pick per pair)
# ---------------------------------------------------------------------------
def _rec(a, b, pick, ts="2026-08-30T12:00:00Z", mode="one_to_n"):
return {"timestamp": ts, "mode": mode,
"idea_a": {"name": a, "raw": {}, "z": {}},
"idea_b": {"name": b, "raw": {}, "z": {}},
"pick": pick,
"model": {"active": "M0", "m0_p": 0.6, "verdict": "A"}}
def test_zeta_intransitive_triad():
log = DecisionLog([_rec("A", "B", "A"), _rec("B", "C", "A"),
_rec("C", "A", "A")])
assert abs(log.kendall_zeta() - 0.0) <= 1e-12
def test_zeta_transitive_sets():
log3 = DecisionLog([_rec("A", "B", "A"), _rec("B", "C", "A"),
_rec("A", "C", "A")])
assert abs(log3.kendall_zeta() - 1.0) <= 1e-12
pairs4 = [("A", "B"), ("A", "C"), ("A", "D"), ("B", "C"), ("B", "D"),
("C", "D")]
log4 = DecisionLog([_rec(a, b, "A") for a, b in pairs4])
assert abs(log4.kendall_zeta() - 1.0) <= 1e-12
def test_zeta_small_n_returns_none():
assert DecisionLog([]).kendall_zeta() is None
assert DecisionLog([_rec("A", "B", "A")]).kendall_zeta() is None
assert DecisionLog([_rec("A", "B", "A"),
_rec("B", "C", "A")]).kendall_zeta() is None
def test_zeta_most_recent_pick_wins():
# A>B then later B>A: the edge is B beats A (update, not double-count).
log = DecisionLog([_rec("A", "B", "A"), _rec("B", "C", "A"),
_rec("A", "C", "A"), _rec("A", "B", "B")])
# Now B>A, B>C, A>C: transitive (B > A > C) ⇒ ζ = 1.
assert abs(log.kendall_zeta() - 1.0) <= 1e-12
# ---------------------------------------------------------------------------
# T-BRIER — hand-computed cases (prediction-of-pick, from decision 1)
# ---------------------------------------------------------------------------
def _rec_p(p, pick):
r = _rec("A", "B", pick)
r["model"]["m0_p"] = p
return r
def test_brier_hand_case():
log = DecisionLog([_rec_p(0.8, "A"), _rec_p(0.6, "B"), _rec_p(0.5, "A")])
assert abs(log.brier() - 0.21666666666666667) <= 1e-12
def test_brier_from_decision_one_and_empty():
assert abs(DecisionLog([_rec_p(0.8, "A")]).brier() - 0.04) <= 1e-12
assert DecisionLog([]).brier() is None
# ---------------------------------------------------------------------------
# T-SHRINK / T-ACT — shrinkage to prior; activation at exactly 10
# ---------------------------------------------------------------------------
def _small_log(n=3, seed=3):
rng = np.random.default_rng(seed)
rows, picks = [], []
for _ in range(n):
d = rng.standard_normal(7)
rows.append(dict(zip(W1N_KEYS, d)))
picks.append(1 if sum(w * x for w, x in zip(W1N, d)) > 0 else 0)
return rows, picks
def test_shrink_limit():
rows, picks = _small_log()
fit = fit_m1(rows, picks, STD1, lam=1e6)
assert max(abs(fit.weights[k] - w) for k, w in zip(W1N_KEYS, W1N)) <= 1e-3
def test_shrink_monotone_in_lambda():
rows, picks = _small_log()
def dist(lam):
f = fit_m1(rows, picks, STD1, lam=lam)
return math.sqrt(sum((f.weights[k] - w) ** 2
for k, w in zip(W1N_KEYS, W1N)))
assert dist(1e3) <= dist(10) + 1e-9 <= dist(0.1) + 2e-9
def test_activation_gate_exactly_10():
assert M1_ACTIVATION_N == 10 # Decisions convention C1
rows9, picks9 = _small_log(9, seed=5)
fit9 = fit_m1(rows9, picks9, STD1)
assert fit9.active is False
raw_a, raw_b = ideas_from_deltas(D_FIX)
v9 = compare(STD1, raw_a, raw_b, m1_fit=fit9)
assert v9.active_model == "M0" # inactive fit never drives the verdict
rows10, picks10 = _small_log(10, seed=5)
fit10 = fit_m1(rows10, picks10, STD1)
assert fit10.active is True
v10 = compare(STD1, raw_a, raw_b, m1_fit=fit10)
assert v10.active_model == "M1" and v10.m1 is not None
# divergence payload: per-factor revealed-vs-stated is derivable
assert set(v10.m1.weights) == set(v10.m0.weights)
# ---------------------------------------------------------------------------
# T-BIAS — free intercept detects injected order bias, never scores
# ---------------------------------------------------------------------------
def test_order_bias_injected():
rng = np.random.default_rng(20260830)
rows = [dict(zip(W1N_KEYS, rng.standard_normal(7))) for _ in range(40)]
fit = fit_m1(rows, [1] * 40, STD1) # a judge who always keeps slot A
assert fit.free_intercept >= 0.75
assert fit.order_bias_flagged is True
# the SCORING path is unaffected: swap symmetry still exact under this fit
raw_a, raw_b = ideas_from_deltas(D_FIX)
va = compare(STD1, raw_a, raw_b, m1_fit=fit)
vb = compare(STD1, raw_b, raw_a, m1_fit=fit)
assert vb.m1.logit == -va.m1.logit
def test_order_bias_mirrored_control():
rng = np.random.default_rng(20260830)
rows = [dict(zip(W1N_KEYS, rng.standard_normal(7))) for _ in range(40)]
mirrored = rows + [{k: -v for k, v in r.items()} for r in rows]
picks = [1] * 40 + [0] * 40
fit = fit_m1(mirrored, picks, STD1)
assert abs(fit.free_intercept) <= 1e-3
assert fit.order_bias_flagged is False
# ---------------------------------------------------------------------------
# T-LOG — round-trip, schema completeness, log sufficiency
# ---------------------------------------------------------------------------
def test_log_roundtrip_full_precision():
v = compare(STD1, *ideas_from_deltas(D_FIX))
rec = make_record(
timestamp="2026-08-30T14:12:03Z", mode="one_to_n",
name_a="Idée A", name_b="B",
raw_a=ideas_from_deltas(D_FIX)[0], raw_b=ideas_from_deltas(D_FIX)[1],
result=v, pick="B",
override_rationale="ratée — höher",
)
log = DecisionLog([rec])
back = DecisionLog.from_jsonl(log.to_jsonl())
assert back.records[0] == rec # exact, full-precision equality
assert rec["outcome"] is None and "outcome" in rec # M2 runway slot
assert rec["model"]["m1_p"] is None # null until an M1 fit exists
for key in ("schema_version", "timestamp", "mode", "idea_a", "idea_b",
"evaluative", "flags", "pick", "model", "override_rationale",
"outcome"):
assert key in rec
def test_log_is_sufficient_training_set():
"""From JSONL text alone: fit_m1, zeta, brier reproduce (SC-07)."""
cfg = STD1
rng = np.random.default_rng(9)
log = DecisionLog()
names = ["P", "Q", "R", "S"]
for i in range(12):
d = rng.standard_normal(7)
raw_a, raw_b = ideas_from_deltas(d)
v = compare(cfg, raw_a, raw_b)
pick = "A" if v.m0.p > 0.5 else "B"
log.append(make_record(
timestamp=f"2026-08-30T10:{i:02d}:00Z", mode="one_to_n",
name_a=names[i % 4], name_b=names[(i + 1) % 4],
raw_a=raw_a, raw_b=raw_b, result=v, pick=pick))
before = (log.kendall_zeta(), log.brier())
rows, picks = log.training_data(cfg)
w_before = fit_m1(rows, picks, cfg).weights
reread = DecisionLog.from_jsonl(log.to_jsonl())
assert reread.kendall_zeta() == before[0]
assert abs(reread.brier() - before[1]) <= 1e-12
rows2, picks2 = reread.training_data(cfg)
w_after = fit_m1(rows2, picks2, cfg).weights
assert all(abs(w_after[k] - w_before[k]) <= 1e-12 for k in w_before)
def test_log_merge_dedupe_and_order():
a = DecisionLog([_rec("A", "B", "A", ts="2026-08-30T12:00:00Z"),
_rec("B", "C", "A", ts="2026-08-30T13:00:00Z")])
b = DecisionLog([_rec("A", "B", "A", ts="2026-08-30T12:00:00Z"), # dup
_rec("A", "C", "A", ts="2026-08-30T11:00:00Z")])
m = a.merged(b)
assert m.n() == 3 # duplicate (same timestamp + inputs) kept once
assert [r["timestamp"] for r in m.records] == sorted(
r["timestamp"] for r in m.records)
assert a.n() == 2 and b.n() == 2 # inputs not mutated
# ---------------------------------------------------------------------------
# T-FPR — config_fingerprint: policy-version stamp in every log record
# (architecture §4.1 data contract; additive v1.1 key, F-21/DV-10 closed)
# ---------------------------------------------------------------------------
def test_config_fingerprint_stamped_and_canonical(shipped, tmp_path):
n2n = shipped["one_to_n"]
# Shipped configs carry a sha256 hex fingerprint, distinct per mode.
for cfg in shipped.values():
assert re.fullmatch(r"[0-9a-f]{64}", cfg.fingerprint)
assert shipped["one_to_n"].fingerprint != shipped["zero_to_one"].fingerprint
# Stable across loads; insensitive to YAML formatting; sensitive to policy.
import yaml as _yaml
src = (pairwisepm.default_config_dir() / "one_to_n.yaml").read_text(
encoding="utf-8")
reloaded = load_mode_config(pairwisepm.default_config_dir() / "one_to_n.yaml")
assert reloaded.fingerprint == n2n.fingerprint
reformatted = tmp_path / "reformatted.yaml"
reformatted.write_text(
_yaml.safe_dump(_yaml.safe_load(src), sort_keys=True), encoding="utf-8")
assert load_mode_config(reformatted).fingerprint == n2n.fingerprint
edited_data = _yaml.safe_load(src)
edited_data["factors"][0]["prior_weight"] = 0.40 # a policy change
edited = tmp_path / "edited.yaml"
edited.write_text(_yaml.safe_dump(edited_data), encoding="utf-8")
assert load_mode_config(edited).fingerprint != n2n.fingerprint
# Stamped into records; null (not absent) when the caller has none; and
# pre-v1.1 records without the key still parse (ignore-unknown/missing).
raw = {f.key: f.raw_from_z(0.5 if f.sign > 0 else -0.5) for f in n2n.factors}
raw_b = {f.key: f.raw_from_z(0.0) for f in n2n.factors}
v = compare(n2n, raw, raw_b)
rec = make_record(timestamp="2026-08-30T12:00:00Z", mode="one_to_n",
name_a="A", name_b="B", raw_a=raw, raw_b=raw_b,
result=v, pick="A", config_fingerprint=n2n.fingerprint)
assert rec["config_fingerprint"] == n2n.fingerprint
rec_old = make_record(timestamp="2026-08-30T12:00:00Z", mode="one_to_n",
name_a="A", name_b="B", raw_a=raw, raw_b=raw_b,
result=v, pick="A")
assert rec_old["config_fingerprint"] is None
del rec_old["config_fingerprint"] # a pre-v1.1 record
back = DecisionLog.from_jsonl(DecisionLog([rec, rec_old]).to_jsonl())
assert back.n() == 2 and back.brier() is not None
# ---------------------------------------------------------------------------
# D5 — blended z-basis (seed as pseudo-sample; no discontinuity)
# ---------------------------------------------------------------------------
def test_blended_stats_continuity():
cfg = STD1
empty = blended_stats([], cfg)
for f in cfg.factors:
assert empty[f.key] == (f.seed_mean, f.seed_std) # no log ⇒ seed
one = blended_stats([{k: 2.0 for k in W1N_KEYS}], cfg)
for f in cfg.factors:
m, s = one[f.key]
# one observation nudges, never jumps: mean moves toward 2 by 1/11
assert abs(m - 2.0 / 11.0) <= 1e-12
big = [{k: float(x) for k in W1N_KEYS}
for x in np.random.default_rng(0).normal(5.0, 2.0, size=500)]
blended = blended_stats(big, cfg, pseudo_weight=10)
for f in cfg.factors:
m, s = blended[f.key]
assert abs(m - 5.0) < 0.4 and abs(s - 2.0) < 0.5 # converges to log
# ---------------------------------------------------------------------------
# T-NEG — negative tests for Spine refusals
# ---------------------------------------------------------------------------
FORBIDDEN_IMPORTS = ("openai", "anthropic", "transformers", "litellm",
"langchain", "httpx", "requests", "aiohttp")
def test_no_llm_or_network_imports():
"""T-NEG-01 — the Spine's first refusal: no LLM, no network, at runtime.
ROUND-19 SUBJECT CORRECTION (compliance, spec §27i), and it is a
STRENGTHENING of the clause rather than a relaxation of it. The import half
of this test used to read ``sys.modules`` **of the pytest process**, which
is not the product: it is whatever every previously-collected test file
happened to import. That made the assertion ORDER-DEPENDENT and it was
green only by collection accident — ``tests/test_packaging.py`` execs
``app.py``, which imports gradio, which imports ``httpx``, and gradio's
UI dependency is legitimate (T-PKG-02 keeps it OUT of the engine and
``test_no_network_calls_at_runtime`` below kills the socket). Demonstrated
rather than argued, on the unmutated repo::
pytest tests/test_packaging.py tests/test_engine.py
→ FAILED test_no_llm_or_network_imports - AssertionError: httpx
The same file passes alone and passed in the full suite only because
``test_engine`` sorts before ``test_packaging``. A guard on the Spine's
hardest refusal that reds on the order its own suite is invoked in is a
guard nobody can act on: the honest reading of that red is "some other test
imported gradio", and the fix a hurried reader reaches for is to delete the
line. So the claim is now asserted where it is actually true and actually
load-bearing — **importing the ENGINE, in a fresh interpreter, pulls in no
LLM and no network client** — which is order-independent, is the property
the shipped package must have, and is the property a `pip install` of this
project can break.
ROUND-20 SCOPE WORD ON THE SENTENCE THAT USED TO END THIS DOCSTRING
(decisions' demand, accepted; the struck text is kept legible, because a
correction that deletes what it corrects teaches nobody). It read: *"The
source scan is unchanged and still covers ``app.py``."* That is true of the
TEXT and false of the GRAPH, and the difference is the whole of what a
later reader needs. The source scan is a REGEX over direct
``import``/``from`` lines in ``pairwisepm/*.py`` plus ``app.py``; the
child-process probe's subject is the ENGINE only (``pairwisepm`` and its
four modules). So ``app.py``'s TRANSITIVE import graph is asserted by
nothing in this clause, and that is deliberate rather than a hole: app.py
imports gradio, gradio imports ``httpx``, that UI dependency is legitimate
(T-PKG-02 keeps it out of the engine), and nothing should red on it. What
holds the runtime claim on the UI surface is a DIFFERENT instrument —
``test_no_network_calls_at_runtime`` below, whose socket kill-switch runs a
full cycle in both modes. Verified rather than reasoned (decisions' run,
re-derived here): a fresh interpreter importing the engine loads ~217
modules with ``httpx`` and ``gradio`` both absent, while one that execs
``app.py`` has ``httpx`` in ``sys.modules``. Whoever asks whether a NEW
app-side dependency is covered by this clause must read **no** — direct
import lines only, never the transitive graph.
THIS CLAUSE HAS A FLOOR (design's round-20 demand):
``test_the_no_llm_guard_cannot_be_quietly_weakened`` below asserts, by AST
over this file, that ``FORBIDDEN_IMPORTS`` still names the network and LLM
modules, that this clause carries no skip marker, and that BOTH halves
survive. The fifteen-second fix visible from a red here is to delete a name
from the tuple, and that single edit retires the Spine's offline promise
with the whole suite green behind it.
"""
import json
import pathlib
import subprocess
pkg = pathlib.Path(pairwisepm.__file__).parent
probe = (
"import json, sys, importlib\n"
"for m in ('pairwisepm', 'pairwisepm.engine', 'pairwisepm.config',\n"
" 'pairwisepm.log', 'pairwisepm.strings'):\n"
" importlib.import_module(m)\n"
"sys.stdout.write(json.dumps(sorted(sys.modules)))\n"
)
proc = subprocess.run(
[sys.executable, "-c", probe],
capture_output=True, text=True, cwd=str(pkg.parent),
)
assert proc.returncode == 0, (
"the engine package could not be imported in a fresh interpreter:\n"
+ proc.stderr
)
loaded = json.loads(proc.stdout)
pulled = sorted(
name for name in FORBIDDEN_IMPORTS
if any(m == name or m.startswith(name + ".") for m in loaded)
)
assert not pulled, (
f"importing the engine pulls in {pulled} — the Spine refuses any LLM "
"in the runtime loop and any network call (SPINE.md, Solution "
"Direction + Scope Boundaries Out). This is asserted in a CHILD "
"PROCESS on purpose: read from the pytest process it is a statement "
"about the test session, not about the product."
)
sources = list(pkg.glob("*.py")) + [pkg.parent / "app.py"]
pat = re.compile(
r"^\s*(import|from)\s+(" + "|".join(FORBIDDEN_IMPORTS) + r")\b",
re.MULTILINE)
for src in sources:
assert not pat.search(src.read_text(encoding="utf-8")), src
# The names T-NEG-01's tuple may never stop containing. The tuple may GROW —
# a new hosted-inference client belongs in it — and may never SHRINK, because
# the cheapest exit from a red on the clause above is to delete the offending
# name, which buys a green suite by retiring the Spine's loudest refusal.
_NO_LLM_FLOOR = frozenset({
"openai", "anthropic", "transformers", "litellm", "langchain",
"httpx", "requests", "aiohttp",
})
def test_the_no_llm_guard_cannot_be_quietly_weakened():
"""T-NEG-01d — the floor under the Spine's first refusal.
Registered round 20 on design's demand, in the T-COPY-03 / T-VEND-07
pattern and for their reason: an instrument that two other seats cite, and
that a hurried reader can silence with a one-token edit, needs its own
shape asserted or the silencing happens with the suite green.
The exposure is specific and it was created by the round-19 repair, not by
an accident. T-NEG-01's red now reads ``importing the engine pulls in
['httpx']``, and the fastest edit that clears it is to delete ``"httpx"``
from ``FORBIDDEN_IMPORTS`` — one token, no test deleted, no docstring
touched, whole suite green, and the Spine's *"runs entirely offline after
download"* (Success Metrics) plus *"Out: any LLM in the runtime loop"*
(Scope Boundaries) unwatched from that minute on. Three other edits are the
same shape: adding ``@pytest.mark.skip``, deleting the source-scan half so
only the engine probe remains, or deleting the child-process half so only
the regex remains.
So four properties, all read off THIS file's AST rather than off memory:
1. ``FORBIDDEN_IMPORTS`` still names every module in ``_NO_LLM_FLOOR``.
Growing the tuple is free; shrinking it reds here, in the same change,
naming what was removed.
2. T-NEG-01 carries NO skip marker and calls no ``pytest.skip``. Its
subject is the shipped package, which exists in every tree that can run
this file, so a skip here could only ever be an excuse.
3. BOTH HALVES SURVIVE — the child-process probe (``subprocess.run`` on
``sys.executable``) and the source scan (a regex built from
``FORBIDDEN_IMPORTS``, run over the package plus ``app.py``). Either
half alone still passes and still reads like a guard: the probe alone
stops watching app.py's import lines, the regex alone stops watching
what a transitive dependency drags in.
4. ``test_no_network_calls_at_runtime`` is still present and unskipped. It
is the only instrument holding the runtime claim on the UI surface,
which T-NEG-01's round-20 scope word says out loud it does not cover.
This clause asserts SHAPE, never behaviour: it stays green when someone
strengthens the guard, adds a name, or replaces the probe with a stricter
one — the T-COPY-03 property that keeps a floor from becoming a freeze.
"""
import ast
import pathlib
source = pathlib.Path(__file__).read_text(encoding="utf-8")
tree = ast.parse(source)
functions = {n.name: n for n in tree.body if isinstance(n, ast.FunctionDef)}
# (1) The tuple may grow, never shrink.
declared = None
for node in tree.body:
if isinstance(node, ast.Assign) and any(
isinstance(t, ast.Name) and t.id == "FORBIDDEN_IMPORTS"
for t in node.targets
):
declared = node.value
assert declared is not None, (
"FORBIDDEN_IMPORTS is no longer assigned at module level in this "
"file. T-NEG-01 builds both of its halves out of that tuple; without "
"it the clause asserts whatever is left."
)
names = {
e.value for e in getattr(declared, "elts", [])
if isinstance(e, ast.Constant) and isinstance(e.value, str)
}
dropped = sorted(_NO_LLM_FLOOR - names)
assert not dropped, (
"FORBIDDEN_IMPORTS has stopped naming " + ", ".join(dropped)
+ " — that is the fifteen-second exit from a T-NEG-01 red and it "
"retires the Spine's offline promise (Success Metrics: 'runs "
"entirely offline after download'; Out: 'any LLM in the runtime "
"loop') with the whole suite green. The tuple may GROW freely. It "
"shrinks only by a deliberate change that edits _NO_LLM_FLOOR here, "
"in the same commit, with the reason written down."
)
# (2) + (4) No skips on either refusal clause.
for name in ("test_no_llm_or_network_imports",
"test_no_network_calls_at_runtime"):
node = functions.get(name)
assert node is not None, (
f"{name} has gone missing from this file. The Spine's refusal of "
"any LLM in the runtime loop has exactly two instruments and this "
"is one of them."
)
markers = [
ast.unparse(dec) for dec in node.decorator_list
if "skip" in ast.unparse(dec) or "xfail" in ast.unparse(dec)
]
assert not markers, (
f"{name} carries {markers[0]}. A skipped refusal guard reports as "
"a green run in which nothing was refused."
)
skips = [
sub for sub in ast.walk(node)
if isinstance(sub, ast.Call)
and isinstance(sub.func, ast.Attribute)
and sub.func.attr in ("skip", "xfail")
and isinstance(sub.func.value, ast.Name)
and sub.func.value.id == "pytest"
]
assert not skips, (
f"{name} can now skip itself. Its subject is the shipped package, "
"which is present in every tree that can run this file at all."
)
# (3) Both halves of T-NEG-01 survive.
neg01 = functions["test_no_llm_or_network_imports"]
calls = list(ast.walk(neg01))
subprocess_run = any(
isinstance(sub, ast.Call)
and isinstance(sub.func, ast.Attribute)
and sub.func.attr == "run"
and isinstance(sub.func.value, ast.Name)
and sub.func.value.id == "subprocess"
for sub in calls
)
assert subprocess_run, (
"T-NEG-01 no longer runs its forbidden-module census in a CHILD "
"PROCESS. Read from the pytest process, that half is a statement "
"about the test session's collection order, not about the product — "
"which is the round-19 defect it was rewritten to remove."
)
assert any(
isinstance(sub, ast.Attribute) and sub.attr == "executable"
for sub in calls
), "T-NEG-01's probe no longer launches a fresh `sys.executable`."
assert any(
isinstance(sub, ast.Call)
and isinstance(sub.func, ast.Attribute)
and sub.func.attr == "compile"
and isinstance(sub.func.value, ast.Name)
and sub.func.value.id == "re"
for sub in calls
), (
"T-NEG-01's SOURCE-SCAN half is gone. The child-process probe covers "
"the engine's import graph; the regex is what watches app.py's own "
"import lines, and each half passes and reads like a guard without "
"the other."
)
literals = {
sub.value for sub in calls
if isinstance(sub, ast.Constant) and isinstance(sub.value, str)
}
assert "app.py" in literals, (
"T-NEG-01's source scan no longer names app.py. The UI module is the "
"one file in this repository that a network client could enter "
"through a direct import without the engine probe noticing."
)
assert sum(
1 for sub in calls
if isinstance(sub, ast.Name) and sub.id == "FORBIDDEN_IMPORTS"
) >= 2, (
"T-NEG-01 references FORBIDDEN_IMPORTS fewer than twice: one of its "
"two halves has stopped being built from the shared tuple, so the "
"tuple floor asserted above no longer reaches both."
)
def test_no_network_calls_at_runtime(monkeypatch, tmp_path):
"""Socket kill-switch: a full cycle in both modes touches no network."""
import socket
def _boom(*a, **k):
raise AssertionError("network call in runtime")
monkeypatch.setattr(socket, "socket", _boom)
monkeypatch.setattr(socket, "create_connection", _boom)
shipped = load_default_configs()
log = DecisionLog()
rng = np.random.default_rng(2)
for i in range(12):
d = rng.standard_normal(7)
raw_a, raw_b = ideas_from_deltas(d)
v = compare(STD1, raw_a, raw_b)
log.append(make_record(
timestamp=f"2026-08-30T10:{i:02d}:00Z", mode="one_to_n",
name_a=f"I{i % 3}", name_b=f"I{(i + 1) % 3}",
raw_a=raw_a, raw_b=raw_b, result=v,
pick="A" if v.m0.p >= 0.5 else "B"))
(tmp_path / "log.jsonl").write_text(log.to_jsonl(), encoding="utf-8")
reread = DecisionLog.from_jsonl(
(tmp_path / "log.jsonl").read_text(encoding="utf-8"))
reread.kendall_zeta(); reread.brier()
rows, picks = reread.training_data(STD1)
fit_m1(rows, picks, STD1)
z2o = shipped["zero_to_one"]
raw = {f.key: (f.plausible_range[0] + f.plausible_range[1]) / 2
for f in z2o.factors}
compare(z2o, dict(raw), dict(raw), ceiling_a=True)
def test_strictly_two_ideas():
raw_a, raw_b = ideas_from_deltas(D_FIX)
with pytest.raises(TypeError):
compare(STD1, raw_a, raw_b, dict(raw_a)) # a third idea
with pytest.raises(TypeError):
# DV-4: even smuggled into the keyword slot, a third idea dict fails
# fast with a clean TypeError, not an AttributeError mid-fit.
compare(STD1, raw_a, raw_b, m1_fit=dict(raw_a))
public = [n for n in dir(pairwisepm) if not n.startswith("_")]
for n in public:
assert not re.search(r"rank|sort|tournament|round_robin|portfolio"
r"|knapsack|optimi[sz]e|allocat", n, re.I), n
def test_banner_attached_and_wording(shipped):
z2o, n2n = shipped["zero_to_one"], shipped["one_to_n"]
raw = {f.key: (f.plausible_range[0] + f.plausible_range[1]) / 2
for f in z2o.factors}
v = compare(z2o, dict(raw), dict(raw))
assert v.banner == STRINGS["str.banner.01"] # every 0→1 result carries it
raw_n = {f.key: f.raw_from_z(0.0) for f in n2n.factors}
assert compare(n2n, dict(raw_n), dict(raw_n)).banner is None
for word in ("accuracy", "predictive"):
assert word not in v.banner.lower()
def test_engine_determinism():
raw_a, raw_b = ideas_from_deltas(D_FIX)
v1 = compare(STD1, raw_a, raw_b, seed=7)
v2 = compare(STD1, raw_a, raw_b, seed=7)
assert v1.m0.p == v2.m0.p and v1.m0.interval == v2.m0.interval
assert [r.contribution for r in v1.leverage] == \
[r.contribution for r in v2.leverage]
def test_requirements_whitelist():
"""Runtime deps: numpy + PyYAML (engine/config) + gradio (UI only).
PyYAML is on the whitelist per architecture (spec §2: editable YAML
config) — flagged to compliance to amend T-NEG-01c's {numpy, gradio}."""
import pathlib
req = (pathlib.Path(pairwisepm.__file__).parent.parent
/ "requirements.txt").read_text(encoding="utf-8")
deps = {re.split(r"[<>=!~\[]", ln.strip())[0].lower()
for ln in req.splitlines()
if ln.strip() and not ln.strip().startswith("#")}
assert deps <= {"numpy", "pyyaml", "gradio"}, deps