Spaces:
Running on Zero
Running on Zero
Download tests/test_demos.py from RealFalconsAI/DecisionLab: direct link, hf CLI and curl.
- Browser
- Download file 3.61 kB
-
https://huggingface.co/spaces/RealFalconsAI/DecisionLab/resolve/main/tests/test_demos.py
- Command line
-
hf download hf://spaces/RealFalconsAI/DecisionLab/tests/test_demos.py
-
curl -L -o test_demos.py https://huggingface.co/spaces/RealFalconsAI/DecisionLab/resolve/main/tests/test_demos.py
3.61 kB
| """Integrity of the demo set in app/demos.py: every demo is well-formed and scorable.""" | |
| import unittest | |
| from app.demos import DEMOS, GROUPS | |
| from app.validation import validate_questions | |
| class DemoSetTest(unittest.TestCase): | |
| def test_demo_ids_are_unique(self): | |
| ids = [d["id"] for d in DEMOS] | |
| dupes = sorted({i for i in ids if ids.count(i) > 1}) | |
| self.assertEqual(dupes, []) | |
| def test_every_demo_group_is_listed_in_groups(self): | |
| known = {g["id"] for g in GROUPS} | |
| unknown = sorted({d["group"] for d in DEMOS} - known) | |
| self.assertEqual(unknown, []) | |
| def test_every_group_has_at_least_one_demo(self): | |
| used = {d["group"] for d in DEMOS} | |
| empty = [g["id"] for g in GROUPS if g["id"] not in used] | |
| self.assertEqual(empty, []) | |
| def test_only_agent_decision_groups_remain(self): | |
| self.assertEqual({g["family"] for g in GROUPS}, {"Agent decisions"}) | |
| self.assertEqual([g["id"] for g in GROUPS], ["triage", "routing", "loop", "guardrails", "agent_security"]) | |
| self.assertEqual(len(DEMOS), 47) | |
| def test_demos_are_ordered_by_group(self): | |
| order = [g["id"] for g in GROUPS] | |
| positions = [order.index(d["group"]) for d in DEMOS] | |
| self.assertEqual(positions, sorted(positions)) | |
| def test_every_demo_passes_request_validation(self): | |
| for d in DEMOS: | |
| with self.subTest(demo=d["id"]): | |
| validate_questions(d["questions"], max_options=40) | |
| def test_every_scored_demo_has_a_reference_for_every_question(self): | |
| for d in DEMOS: | |
| if d["group"] == "agent_security": | |
| continue | |
| with self.subTest(demo=d["id"]): | |
| self.assertEqual(sorted(d["reference"]), sorted(d["questions"])) | |
| def test_nano_set_is_states_and_questions_only(self): | |
| """Operator ruling 2026-09-30: only the states and questions of decisionlab_nano_demos.json are added.""" | |
| nano = [d for d in DEMOS if d["group"] == "agent_security"] | |
| self.assertEqual(len(nano), 23) | |
| label = next(g["label"] for g in GROUPS if g["id"] == "agent_security") | |
| self.assertEqual(label, "Agent security") # operator, 2026-10-01: no "(nano set)" | |
| self.assertEqual(sum(len(d["questions"]) for d in nano), 34) | |
| for d in nano: | |
| with self.subTest(demo=d["id"]): | |
| self.assertEqual(set(d), {"id", "group", "title", "blurb", "state", "questions"}) | |
| self.assertTrue(d["id"].startswith("nano-")) | |
| def test_every_reference_is_a_valid_answer(self): | |
| for d in DEMOS: | |
| for name, q in d["questions"].items(): | |
| if name not in d.get("reference", {}): | |
| continue | |
| ref = d["reference"][name] | |
| with self.subTest(demo=d["id"], question=name): | |
| qtype = q.get("type", "choice") | |
| if qtype == "noul": | |
| self.assertIn(ref, ("true", "false")) | |
| elif qtype == "score": | |
| self.assertIn(ref, [str(i) for i in range(len(q["criteria"]))]) | |
| else: | |
| self.assertIn(ref, q["criteria"]) | |
| def test_stakes_name_existing_questions_and_only_high(self): | |
| for d in DEMOS: | |
| stakes = d.get("stakes", {}) | |
| with self.subTest(demo=d["id"]): | |
| self.assertLessEqual(set(stakes), set(d["questions"])) | |
| self.assertLessEqual(set(stakes.values()), {"high"}) | |
| if __name__ == "__main__": | |
| unittest.main() | |