nexa / tests /test_embedding.py
DBax127's picture
Claude Opus 5
retrieval: the embedder's cut, measured instead of deduced
b18750b
Raw History Blame Contribute Delete
2.86 kB
"""What the embedder silently drops, and the first test that needs a live model.
pytest.ini has declared an `ollama` marker since the suite was written and
nothing had ever used it: every other test either fakes the server or needs no
model at all, which is the right default and leaves one kind of claim untestable.
ARCHITECTURE 3.31 makes exactly that kind. `nexa check` hands retrieval a
question built from a file, nomic-embed-text truncates whatever exceeds its
context, and Ollama returns a vector without mentioning it. The size of that cut
decides whether the compact retrieval query is worth adopting, and it was a
deduction until somebody appended a marker and watched the vector fail to move.
Skipped when there is no server, so the suite still runs on a laptop with
nothing installed and in CI, where OLLAMA_HOST points at a closed port.
"""
import os
import pytest
import corpus
import subject
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
LONG_SUBJECT = os.path.join(ROOT, "subjects", "odoo_module", "models", "fsm_visit.py")
pytestmark = pytest.mark.ollama
@pytest.fixture(scope="module")
def live():
try:
corpus.embed(["probe"])
except Exception as exc: # noqa: BLE001
pytest.skip("no live Ollama: {0}".format(str(exc).splitlines()[0]))
def test_the_embedder_drops_the_end_of_a_long_query(live):
"""The measurement behind 3.31, pinned so the number cannot rot.
If a future embedding model has a larger context this fails, and the right
response is to re-measure and rewrite the section rather than to relax the
assertion.
"""
prompt = subject.plan(LONG_SUBJECT)["question"]
assert len(prompt) > 7000, "the subject has to be long enough to overflow"
marker = " ZZZ_UNIQUE_TAIL_MARKER_about_sudo_and_record_rules"
whole = corpus.embed([prompt])[0]
extended = corpus.embed([prompt + marker])[0]
assert whole == extended, "text past the cut changed the vector -- re-measure 3.31"
# And the near half is not dropped: a marker early on does move it, so this
# is a limit rather than an embedder that ignores its input.
head = prompt[:2000]
assert corpus.embed([head])[0] != corpus.embed([head + marker])[0]
def test_the_compact_query_fits_where_the_prompt_does_not(live):
"""The compact query exists because of the cut above. Whether it retrieves
better is `nexa eval --compare-retrieval`; that it is small enough to be read
whole is this."""
plan = subject.plan(LONG_SUBJECT)
query = plan["retrieval_query"]
assert len(query) < len(plan["question"])
marker = " ZZZ_UNIQUE_TAIL_MARKER_about_sudo_and_record_rules"
assert corpus.embed([query])[0] != corpus.embed([query + marker])[0], \
"the compact query is over the cut too, which defeats its purpose"