"""Realistic science-prompt fixtures used by test_invariants.py. The list `REALISTIC_SCIENCE_PROMPTS` is hand-crafted to contain four classes of items so tests can pin exact post-stage counts: Class A: byte-identical duplicates (3 copies) -> caught by Stage 1 (caller's exact-prompt reservoir dedup) Class B: formatting variants of one prompt (3 prompts: extra whitespace, upper-case differences, trailing newline) -> NOT caught by exact dedup; caught by Stage 2 (MinHash with normalize=True) Class C: near-duplicate paraphrases of one prompt (3 prompts: each is the same long base text with one single-word edit) -> NOT caught by exact or formatting normalization; caught by Stage 2 (MinHash with threshold=0.8) Class D: 7 distinct singleton prompts (math / biology / chemistry / physics) -> none of the dedup stages should touch these Total = 3 + 3 + 3 + 7 = 16 prompts. After exact dedup: 1 + 3 + 3 + 7 = 14 After MinHash: 1 + 1 + 1 + 7 = 10 Class C uses a ~140-word base text so that single-word edits keep Jaccard robustly above MinHash's threshold of 0.8. With n=140 tokens (≈136 5-grams) and a single-word edit affecting 5 5-grams, Jaccard = 131/141 ≈ 0.929 — well into LSH's high-detection regime, so the test is deterministic. """ from __future__ import annotations # ── Class A: byte-identical ───────────────────────────────────────────────── _EXACT_DUP = ( "Solve the equation x squared minus four equals zero. Show all steps " "clearly." ) # ── Class B: formatting variants of one prompt ────────────────────────────── _FMT_BASE = "Calculate the derivative of f(x) = x^3 + 2x with respect to x." _FORMATTING_VARIANTS = [ _FMT_BASE, " Calculate the derivative of f(x) = x^3 + 2x with respect to x. ", "Calculate the DERIVATIVE of f(x) = x^3 + 2x with respect to x.\n", ] # ── Class C: near-duplicate paraphrases (single-word edits over ~80 words) ── _NEAR_DUP_BASE = ( "Explain in detail the process of photosynthesis in green plants. " "Describe step by step how light energy is captured by chlorophyll " "molecules in the thylakoid membrane, how electrons are transported " "through photosystems II and I, how ATP and NADPH are generated through " "photophosphorylation, and how the Calvin cycle uses these energy " "carriers to fix carbon dioxide into glucose. Include the role of stomata " "in gas exchange and discuss how environmental factors like light " "intensity, water availability, soil nutrients, atmospheric carbon " "dioxide concentration, and temperature affect the overall rate of " "photosynthesis. Discuss the differences between C3, C4, and CAM " "photosynthetic pathways and their evolutionary adaptations to different " "environmental conditions including drought stress and high light " "intensity. Finally explain how mitochondrial respiration interacts with " "chloroplast metabolism in the cells of plants and how this energy " "balance is maintained throughout the day and night cycle of plant " "tissues across different seasons in temperate and tropical biomes." ) _NEAR_DUP_VARIANTS = [ _NEAR_DUP_BASE, _NEAR_DUP_BASE.replace("process", "mechanism", 1), # 1-word edit near start _NEAR_DUP_BASE.replace("glucose", "sugar", 1), # 1-word edit in middle ] # ── Class D: distinct singletons ──────────────────────────────────────────── _SINGLETONS = [ "Compute the integral of sin(x) cos(x) from 0 to pi/2 using substitution.", "Describe the structure of a eukaryotic cell and its main organelles in detail.", "What is the wavelength of light with frequency 5 x 10^14 Hz in vacuum?", "Find the eigenvalues of the 2x2 matrix [[2, 1], [1, 2]] step by step.", "Explain how DNA replication occurs during cell division and list the enzymes involved.", "Balance the chemical equation H2 + O2 produces H2O and explain conservation of atoms.", "Determine the molarity of a solution containing 5 grams of NaCl in 250 mL of water.", ] REALISTIC_SCIENCE_PROMPTS: list[str] = ( [_EXACT_DUP, _EXACT_DUP, _EXACT_DUP] + _FORMATTING_VARIANTS + _NEAR_DUP_VARIANTS + _SINGLETONS ) # Index ranges for assertions in tests. EXACT_DUP_RANGE = range(0, 3) # indices 0, 1, 2 FORMATTING_RANGE = range(3, 6) # indices 3, 4, 5 NEAR_DUP_RANGE = range(6, 9) # indices 6, 7, 8 SINGLETON_RANGE = range(9, 16) # indices 9..15