File size: 5,851 Bytes
112812c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
"""Deterministic Oracle engine over a JSON attribute database.

- load_items(category)            -> list of {name, category, attributes}
- filter_candidates(items, facts) -> items consistent with facts (yes/no/unknown)
- choose_attribute(items, asked)  -> attribute that best splits the set (info gain)

All pure Python — filtering is always factually correct (the JSON is the truth).
"""

from __future__ import annotations

import json
import os
import shutil

HERE = os.path.dirname(os.path.abspath(__file__))

# Read-only seed shipped with the app.
SEED_DIR = os.path.join(HERE, "data")
# Where the live DB lives. Point ORACLE_DATA_DIR at a mounted HF Storage Bucket
# (e.g. /data) so items taught during play persist across Space restarts.
DATA_DIR = os.environ.get("ORACLE_DATA_DIR") or SEED_DIR

# category -> json filename
_FILES = {"animal": "animals.json", "fruit": "fruits.json", "vegetable": "vegetables.json"}


def _seed_if_missing(path: str, fname: str) -> None:
    """On first run against an empty bucket, copy the bundled seed DB over."""
    if os.path.exists(path):
        return
    seed = os.path.join(SEED_DIR, fname)
    if os.path.exists(seed) and os.path.abspath(path) != os.path.abspath(seed):
        os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
        shutil.copyfile(seed, path)
        print(f"[engine] seeded {fname} -> {path}", flush=True)

# Fallback question phrasing per attribute (used if the LLM question-maker is
# unavailable or returns SKIP). Mirrors the attribute keys in the JSON.
ATTR_QUESTIONS = {
    # animal
    "mammal": "Is it a mammal?",
    "bird": "Is it a bird?",
    "water": "Does it live mainly in water?",
    "carnivore": "Does it mainly eat meat?",
    "big": "Is it bigger than a human?",
    "domestic": "Is it a pet or farm animal?",
    "can_fly": "Can it fly?",
    "stripes": "Does it have stripes?",
    "horns": "Does it have horns or antlers?",
    # fruit
    "red": "Is it red?",
    "sweet": "Is it sweet?",
    "has_pit": "Does it have a hard pit or stone inside?",
    "tree": "Does it grow on a tree?",
    "tropical": "Is it a tropical fruit?",
    "peel": "Do you usually peel it before eating?",
    # vegetable
    "root": "Is it a root vegetable?",
    "leafy": "Is it a leafy vegetable?",
    "green": "Is it green?",
    "underground": "Does it grow underground?",
    "raw": "Is it usually eaten raw?",
    "long": "Is it long in shape?",
    "spicy": "Is it spicy or pungent?",
    "starchy": "Is it starchy?",
    # animal (added)
    "climbs": "Can it climb trees well?",
    "hops": "Does it hop or jump to move around?",
    "wool": "Does it have wool?",
    "hump": "Does it have a hump on its back?",
    "black_white": "Is it black and white?",
    "long_tail": "Does it have a long tail?",
    "pack": "Does it live in a group or pack?",
    # fruit (added)
    "small": "Is it small or bite-sized?",
    "seeds_outside": "Does it have seeds on the outside?",
    "hard_shell": "Does it have a hard shell?",
    "spiky": "Does it have a spiky skin?",
    # vegetable (added)
    "round": "Is it round in shape?",
    "white": "Is it white in colour?",
    "cooked": "Is it usually cooked before eating?",
    "pod": "Does it grow in a pod?",
    "thin": "Is it thin and slender?",
    "knobbly": "Is it knobbly or bumpy?",
}


# cache keyed by file path -> (mtime, items). Reloads automatically when the
# JSON changes, so hand-edits are picked up on the next game with NO restart.
_CACHE: dict = {}


def load_items(category: str) -> tuple:
    """Load the JSON DB for a category. Cached, but auto-reloads if the file
    has changed on disk (compares modification time)."""
    fname = _FILES.get(category)
    if not fname:
        return ()
    path = os.path.join(DATA_DIR, fname)
    _seed_if_missing(path, fname)
    try:
        mtime = os.path.getmtime(path)
    except OSError:
        return ()
    cached = _CACHE.get(path)
    if cached and cached[0] == mtime:
        return cached[1]
    with open(path, "r", encoding="utf-8") as f:
        items = tuple(json.load(f))
    _CACHE[path] = (mtime, items)
    return items


def _cache_clear() -> None:
    """Drop the in-memory cache (used after writing the DB programmatically)."""
    _CACHE.clear()


load_items.cache_clear = _cache_clear  # keep the old API used by discovery/tests


def filter_candidates(items: list, facts: list) -> list:
    """Keep items consistent with the facts.

    facts: [{"attribute": str, "answer": "yes"|"no"|"unknown"}].
    Unknown answers (and missing attribute values) never eliminate an item.
    """
    out = list(items)
    for fact in facts:
        attr = fact.get("attribute")
        ans = str(fact.get("answer", "")).strip().lower()
        if not attr or ans not in ("yes", "no"):
            continue  # unknown / not-sure -> no filtering
        expected = ans == "yes"
        out = [it for it in out
               if it["attributes"].get(attr) is None or it["attributes"].get(attr) == expected]
    return out


def choose_attribute(category: str, items: list, asked: list) -> str | None:
    """Pick the unused attribute whose yes/no split is closest to 50/50."""
    asked = set(asked or [])
    best, best_score = None, -1
    # consider attributes that actually appear in the data, in a stable order
    seen = []
    for it in items:
        for k in it["attributes"]:
            if k not in seen:
                seen.append(k)
    for attr in seen:
        if attr in asked:
            continue
        yes = sum(1 for it in items if it["attributes"].get(attr) is True)
        no = sum(1 for it in items if it["attributes"].get(attr) is False)
        if yes == 0 or no == 0:
            continue
        score = min(yes, no)
        if score > best_score:
            best, best_score = attr, score
    return best