File size: 3,023 Bytes
75263bf
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
"""Capability stack: linked ingest + stable compose + substrate decoder."""

from palimseste.lsh import LSHConfig
from palimseste.memory import Memory
from palimseste.phi import Phi, KernelConfig
from palimseste.linked_ingest import LinkedEncoder
from palimseste.compose import analogize, compose_chain, ComposeConfig
from palimseste.substrate_decoder import SubstrateDecoder


def _stack(D=1024):
    mem = Memory(D=D, lsh_config=LSHConfig.tune(D=D, target_radius=0.20, recall=0.95))
    enc = LinkedEncoder(mem)
    phi = Phi(KernelConfig(radius=max(80, D // 8), topk=12, min_weight=1e-6))
    dec = SubstrateDecoder(enc, phi)
    return enc, phi, dec


CORPUS = {
    "capitals": (
        "Paris is the capital of France. London is the capital of England. "
        "Rome is the capital of Italy."
    ),
    "cortex": (
        "Palimpseste is an append-only hypervector cortex. "
        "Learning is a single write. "
        "Phi reconstructs values from a Hamming neighborhood."
    ),
}


def test_ingest_links_not_raw_dump():
    enc, phi, dec = _stack()
    rep = enc.ingest_document("capitals", CORPUS["capitals"], topics=["capital", "city"])
    assert rep.n_sentences == 3
    assert rep.n_traces > 10
    tags = [tr.tag or "" for tr in enc.mem.traces]
    assert any(t.startswith("NEXT:") for t in tags)
    assert any(t.startswith("PREV:") for t in tags)
    assert any(t.startswith("TOPIC:") for t in tags)
    assert any(t.startswith("IDX:") for t in tags)


def test_encoder_deterministic():
    enc, _, _ = _stack()
    a = enc.encode_text("Paris is the capital of France")
    b = enc.encode_text("Paris is the capital of France")
    assert a == b


def test_analogize_identity():
    enc, _, _ = _stack()
    a = enc.encode_text("paris")
    c = enc.encode_text("london")
    assert analogize(a, a, c) == c


def test_decoder_recovers_ingested_tokens():
    enc, phi, dec = _stack()
    enc.ingest_document("capitals", CORPUS["capitals"], topics=["capital"])
    bag = dec.decode_bag(enc.encode_text("Paris is the capital of France"), n=8, min_sim=-1.0)
    joined = " ".join(bag.tokens)
    assert "paris" in joined or "france" in joined or "capital" in joined


def test_compose_chain_does_not_rot_on_exact_items():
    enc, phi, dec = _stack()
    enc.ingest_document("cortex", CORPUS["cortex"], topics=["palimpseste", "phi"])
    start = enc.encode_text("Palimpseste is an append-only hypervector cortex")
    result = compose_chain(
        enc.mem,
        phi,
        start,
        steps=[("bundle", enc.encode_text("learning write"), None)],
        config=ComposeConfig(floor=-0.2, max_hops=2),
    )
    assert result.hv is not None
    assert result.hops


def test_answer_uses_substrate_not_llm():
    enc, phi, dec = _stack()
    enc.ingest_document("cortex", CORPUS["cortex"], topics=["palimpseste"])
    out = dec.answer("what is palimpseste")
    assert isinstance(out.text, str)
    # decoder is a codebook peel + NEXT walk — no external model
    assert out.tokens is not None