DecisionLab / tests /test_demos.py
Michael Stattelman
Version updates
d8c255d
Raw History Blame Contribute Delete
3.61 kB
"""Integrity of the demo set in app/demos.py: every demo is well-formed and scorable."""
import unittest
from app.demos import DEMOS, GROUPS
from app.validation import validate_questions
class DemoSetTest(unittest.TestCase):
def test_demo_ids_are_unique(self):
ids = [d["id"] for d in DEMOS]
dupes = sorted({i for i in ids if ids.count(i) > 1})
self.assertEqual(dupes, [])
def test_every_demo_group_is_listed_in_groups(self):
known = {g["id"] for g in GROUPS}
unknown = sorted({d["group"] for d in DEMOS} - known)
self.assertEqual(unknown, [])
def test_every_group_has_at_least_one_demo(self):
used = {d["group"] for d in DEMOS}
empty = [g["id"] for g in GROUPS if g["id"] not in used]
self.assertEqual(empty, [])
def test_only_agent_decision_groups_remain(self):
self.assertEqual({g["family"] for g in GROUPS}, {"Agent decisions"})
self.assertEqual([g["id"] for g in GROUPS], ["triage", "routing", "loop", "guardrails", "agent_security"])
self.assertEqual(len(DEMOS), 47)
def test_demos_are_ordered_by_group(self):
order = [g["id"] for g in GROUPS]
positions = [order.index(d["group"]) for d in DEMOS]
self.assertEqual(positions, sorted(positions))
def test_every_demo_passes_request_validation(self):
for d in DEMOS:
with self.subTest(demo=d["id"]):
validate_questions(d["questions"], max_options=40)
def test_every_scored_demo_has_a_reference_for_every_question(self):
for d in DEMOS:
if d["group"] == "agent_security":
continue
with self.subTest(demo=d["id"]):
self.assertEqual(sorted(d["reference"]), sorted(d["questions"]))
def test_nano_set_is_states_and_questions_only(self):
"""Operator ruling 2026-09-30: only the states and questions of decisionlab_nano_demos.json are added."""
nano = [d for d in DEMOS if d["group"] == "agent_security"]
self.assertEqual(len(nano), 23)
label = next(g["label"] for g in GROUPS if g["id"] == "agent_security")
self.assertEqual(label, "Agent security") # operator, 2026-10-01: no "(nano set)"
self.assertEqual(sum(len(d["questions"]) for d in nano), 34)
for d in nano:
with self.subTest(demo=d["id"]):
self.assertEqual(set(d), {"id", "group", "title", "blurb", "state", "questions"})
self.assertTrue(d["id"].startswith("nano-"))
def test_every_reference_is_a_valid_answer(self):
for d in DEMOS:
for name, q in d["questions"].items():
if name not in d.get("reference", {}):
continue
ref = d["reference"][name]
with self.subTest(demo=d["id"], question=name):
qtype = q.get("type", "choice")
if qtype == "noul":
self.assertIn(ref, ("true", "false"))
elif qtype == "score":
self.assertIn(ref, [str(i) for i in range(len(q["criteria"]))])
else:
self.assertIn(ref, q["criteria"])
def test_stakes_name_existing_questions_and_only_high(self):
for d in DEMOS:
stakes = d.get("stakes", {})
with self.subTest(demo=d["id"]):
self.assertLessEqual(set(stakes), set(d["questions"]))
self.assertLessEqual(set(stakes.values()), {"high"})
if __name__ == "__main__":
unittest.main()