File size: 5,507 Bytes
494a4bf
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
fcdba1d
494a4bf
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
fcdba1d
494a4bf
fcdba1d
 
 
494a4bf
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
"""
make_testset β€” generate a corpus-grounded test set from the indexed calendar chunks.

A test set is only meaningful if its gold ids point at chunks that are actually in the
index. So instead of hand-writing questions, we sample real indexed chunks and ask the
local LLM to write, for each one, a specific question that chunk answers plus a concise
reference answer grounded in it. The chunk's own id becomes the gold id β€” which is what
makes the deterministic retrieval metrics (hit@k etc.) work.

Landing pages and tiny fragments make useless questions, so sampling skips chunks with
page_type == "landing" or very short text.

This produces *candidates*. The intended workflow is to then CURATE by hand β€” delete vague
or wrong questions, fix wording, and (for near-duplicate chunks, e.g. the same requirement
in the 2025/26 vs 2026/27 edition or across cohort pages) add sibling ids to gold_ids β€”
and commit the frozen result. Start small and grow.

Run (needs a built index; uses the local LM Studio backend β€” see config.CHAT_MODEL):
    python -m eval.make_testset --num 15 --out eval/testset.json
"""

from __future__ import annotations

import argparse
import json
import random

from openai import OpenAI

from src import config, retrieve
from eval.judge import _extract_json, JudgeParseError  # reuse robust JSON parsing

_MIN_CHUNK_CHARS = 300  # too little text to ground a specific question + answer

_SYSTEM = (
    "You write evaluation data for a UBC academic-calendar question-answering system. "
    "Respond with ONLY valid JSON β€” no prose, no markdown, no code fences."
)


def _prompt(chunk: dict) -> str:
    return (
        "Given the CALENDAR EXCERPT below, write ONE specific question that a UBC student "
        "might ask and that THIS excerpt answers well, plus a concise reference answer "
        "grounded only in the excerpt. The question should be specific (name the course, "
        "program, or policy it is about β€” and the student cohort or calendar year when the "
        "excerpt is cohort- or edition-specific) and answerable from the excerpt.\n\n"
        f"EXCERPT TITLE: {chunk['title']}\n"
        f"SOURCE URL: {chunk['url']}\n"
        f"EXCERPT TEXT: {chunk['text']}\n\n"
        'Return JSON: {"question": "<question>", "reference_answer": "<answer>"}'
    )


def _eligible(chunk: dict) -> bool:
    return chunk.get("page_type") != "landing" and len(chunk.get("text", "")) >= _MIN_CHUNK_CHARS


def generate(num: int, seed: int) -> list[dict]:
    """Sample `num` eligible indexed chunks and turn each into a test item."""
    _, chunks = retrieve._load_index()  # the indexed corpus (id-aligned metadata)
    pool = [c for c in chunks if _eligible(c)]
    print(f"Sampling from {len(pool)} eligible chunks (of {len(chunks)} indexed).")
    if num > len(pool):
        num = len(pool)
    sampled = random.Random(seed).sample(pool, num)

    client = OpenAI(base_url=config.OPENAI_BASE_URL, api_key=config.OPENAI_API_KEY)
    items: list[dict] = []
    for chunk in sampled:
        response = client.chat.completions.create(
            model=config.CHAT_MODEL,
            messages=[
                {"role": "system", "content": _SYSTEM},
                {"role": "user", "content": _prompt(chunk)},
            ],
            temperature=0.3,
            max_tokens=400,
        )
        try:
            parsed = _extract_json(response.choices[0].message.content or "")
            question = parsed["question"].strip()
            reference_answer = parsed["reference_answer"].strip()
        except (JudgeParseError, KeyError, AttributeError, TypeError):
            print(f"  skipped chunk {chunk['id']} ({chunk['title']!r}): bad JSON")
            continue

        items.append(
            {
                "id": f"q{len(items) + 1:03d}",
                "question": question,
                "reference_answer": reference_answer,
                "gold_ids": [chunk["id"]],
                "gold_keys": [chunk["chunk_key"]],
                "source_chunk_id": chunk["id"],
            }
        )
        print(f"  [{len(items)}] id={chunk['id']:<4} {chunk['title']}")
    return items


def main() -> None:
    parser = argparse.ArgumentParser(description="Generate a corpus-grounded eval test set.")
    parser.add_argument("--num", type=int, default=15, help="how many questions to generate")
    parser.add_argument("--seed", type=int, default=42, help="sampling seed (reproducible)")
    parser.add_argument("--out", default=str(config.TESTSET_PATH), help="output JSON path")
    args = parser.parse_args()

    print(f"Generating {args.num} questions from the indexed corpus "
          f"(model={config.CHAT_MODEL}) ...")
    items = generate(args.num, args.seed)

    payload = {
        "description": (
            "Auto-generated, corpus-grounded test set (eval/make_testset.py). Each gold key "
            "points at an indexed calendar chunk. REVIEW AND CURATE before trusting the "
            "numbers: delete vague/wrong items, fix wording, add sibling keys to gold_keys "
            "for near-duplicate chunks (edition/cohort twins). gold_keys are what the eval "
            "resolves against the index; gold_ids are a positional cache it overwrites."
        ),
        "items": items,
    }
    with open(args.out, "w", encoding="utf-8") as f:
        json.dump(payload, f, ensure_ascii=False, indent=2)
    print(f"\nWrote {len(items)} questions to {args.out}")


if __name__ == "__main__":
    main()