File size: 7,067 Bytes
112812c
 
 
036e689
 
 
 
 
 
112812c
 
 
 
 
 
 
036e689
112812c
 
 
 
036e689
 
112812c
f0d3c87
112812c
f0d3c87
 
 
 
112812c
f0d3c87
 
 
 
 
 
 
 
112812c
 
036e689
 
f0d3c87
 
036e689
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f0d3c87
036e689
f0d3c87
 
 
 
036e689
 
112812c
 
036e689
 
 
 
 
 
 
112812c
036e689
 
 
 
 
112812c
 
 
036e689
112812c
f0d3c87
112812c
 
 
 
 
 
036e689
 
 
112812c
036e689
112812c
036e689
 
 
f0d3c87
036e689
 
f0d3c87
036e689
 
 
 
 
f0d3c87
036e689
 
 
 
 
f0d3c87
036e689
 
 
 
f0d3c87
 
 
 
 
 
 
 
 
036e689
 
 
 
 
 
 
 
 
 
112812c
 
 
036e689
 
 
112812c
 
 
 
 
 
 
036e689
112812c
 
036e689
112812c
 
 
 
 
036e689
112812c
036e689
112812c
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
"""LLM question-maker: attribute name -> one natural yes/no question.

The model ONLY phrases questions — it never decides elimination (the engine does).

Performance note: there are only ~42 possible questions (one per attribute per
category), so we generate them with the model ONCE (at boot, via prewarm_questions)
and cache them to disk (persisted in the bucket). During gameplay make_question
is a pure dict lookup — instant, even on a weak CPU — yet the questions are still
genuinely model-written. Falls back to built-in phrasing until the cache fills.
"""

from __future__ import annotations

import json
import os

import engine
from engine import ATTR_QUESTIONS
from discovery import ATTR_MEANING

MODEL = os.environ.get("ORACLE_LLAMA_MODEL", "Llama-3.2-3B-Instruct")  # Built with Llama 🦙
USE_LLM = os.environ.get("ORACLE_QUESTION_LLM", "1") == "1"        # model writes the questions
USE_REVEAL_LLM = os.environ.get("ORACLE_REVEAL_LLM", "0") == "1"   # off: instant templated reveal

SYSTEM = ("You write simple, clear, kid-friendly yes/no questions for a guessing game.")

PROMPT = """Turn the fact below into ONE simple yes/no question for a kids' guessing game.
Keep it short, clear, and natural — plain everyday words, nothing weird or confusing.
Mention "this {category}" or "the {category}". Do NOT use the word "{attribute}".
Output only the question, nothing else.

The {category} either {meaning} — or not. Ask about exactly that.

Examples of the style:
- Does this animal have a long tail?
- Do you usually peel the fruit before eating it?
- Does this animal eat meat?

Now write the question:"""


# --- question cache (persisted next to the DB, e.g. in the bucket) ----------

CACHE_VERSION = "v3"  # bump when the prompt/style changes to auto-regenerate

_qcache: dict | None = None
_qcache_path: str | None = None


def _cache_file() -> str:
    return os.path.join(engine.DATA_DIR, "questions_cache.json")


def _cache() -> dict:
    global _qcache, _qcache_path
    path = _cache_file()
    if _qcache is not None and _qcache_path == path:
        return _qcache
    try:
        with open(path, "r", encoding="utf-8") as f:
            data = json.load(f)
    except Exception:  # noqa: BLE001 — no cache yet
        data = {}
    if data.get("__version__") != CACHE_VERSION:   # stale style -> start fresh
        data = {"__version__": CACHE_VERSION}
    _qcache = data
    _qcache_path = path
    return _qcache


def _save_cache() -> None:
    try:
        with open(_cache_file(), "w", encoding="utf-8") as f:
            json.dump(_qcache or {}, f, ensure_ascii=False, indent=0)
    except Exception as exc:  # noqa: BLE001
        print(f"[question_maker] cache save failed: {exc}")


# --- generation (only runs at boot/prewarm, never during a turn) ------------

def _llm_question(category: str, attribute: str) -> str | None:
    import llm  # runs the model through llama.cpp
    meaning = ATTR_MEANING.get(attribute, f"relates to '{attribute}'")
    messages = [
        {"role": "system", "content": SYSTEM},
        {"role": "user", "content": PROMPT.format(
            category=category, attribute=attribute, meaning=meaning)},
    ]
    text = llm.chat(messages, temperature=0.3, max_tokens=60).strip()
    if "</think>" in text:
        text = text.split("</think>")[-1].strip()
    text = text.splitlines()[-1].strip().strip('"') if text else ""
    return text or None


def prewarm_questions() -> None:
    """Generate every category/attribute question once and cache it. Safe to run
    in a background thread at startup; persists progress so restarts are instant."""
    if not USE_LLM:
        return
    try:
        from discovery import CATEGORY_ATTRS
    except Exception:  # noqa: BLE001
        return
    import time
    cache = _cache()
    made = 0
    t_start = time.time()
    for cat, attrs in CATEGORY_ATTRS.items():
        for attr in attrs:
            key = f"{cat}:{attr}"
            if cache.get(key):
                continue
            t0 = time.time()
            try:
                q = _llm_question(cat, attr)
            except Exception as exc:  # noqa: BLE001
                print(f"[question_maker] prewarm {key} failed: {exc}")
                q = None
            dt = time.time() - t0
            if q and q.upper() != "SKIP" and "?" in q:
                cache[key] = q
                _save_cache()
                made += 1
                print(f"[question_maker] {key} ({dt:.1f}s): {q}", flush=True)
            else:
                print(f"[question_maker] {key} ({dt:.1f}s): no question -> fallback", flush=True)
    total = sum(1 for k in cache if ":" in k)
    elapsed = time.time() - t_start
    if made:
        print(f"[question_maker] generated {made} questions in {elapsed:.1f}s "
              f"({elapsed / made:.1f}s each)", flush=True)
    print(f"[question_maker] question cache ready: {total} questions (+{made} new), style {CACHE_VERSION}",
          flush=True)


# --- used during gameplay: instant lookup, never blocks ---------------------

def make_question(category: str, attribute: str, asked: list | None = None) -> str:
    """Return a natural yes/no question — a cached model-written one if available,
    otherwise the built-in phrasing. Never calls the model (so it's instant)."""
    fallback = ATTR_QUESTIONS.get(attribute, f"Is your {category} related to '{attribute}'?")
    return _cache().get(f"{category}:{attribute}", fallback)


def make_reveal(category: str, yes_attrs: list) -> str:
    """A short, theatrical line said just before guessing, built from the traits
    the player confirmed. Templated (instant) by default; set ORACLE_REVEAL_LLM=1
    to have the model write it. Never names the item (keeps the suspense)."""
    short = {"big": "is large", "carnivore": "eats meat",
             "domestic": "is a pet or farm creature", "can_fly": "can fly"}
    traits = [short.get(a, ATTR_MEANING[a]) for a in (yes_attrs or []) if a in ATTR_MEANING]
    if traits:
        fallback = "I sense something that " + ", ".join(traits[:3]) + "…"
    else:
        fallback = "The mists are clearing… I see it now…"
    if not USE_REVEAL_LLM:
        return fallback
    try:
        import llm
        desc = ", ".join(traits) if traits else "a mysterious thing"
        messages = [
            {"role": "system", "content": "You are a theatrical crystal-ball fortune teller."},
            {"role": "user", "content":
                f"In ONE short sentence (max 18 words), tease that you are about to "
                f"reveal a {category} that {desc}. Be mystical and playful. Do NOT name it."},
        ]
        line = llm.chat(messages, temperature=0.8, max_tokens=60, timeout=8).strip()
        if "</think>" in line:
            line = line.split("</think>")[-1].strip()
        line = line.splitlines()[-1].strip().strip('"')
        return line or fallback
    except Exception as exc:  # noqa: BLE001
        print(f"[question_maker] reveal LLM failed, using fallback: {exc}")
        return fallback