Download source.py from VALOR0316/koth-miner-92: direct link, hf CLI and curl.
- Browser
- Download file 55.7 kB
-
https://huggingface.co/VALOR0316/koth-miner-92/resolve/main/source.py
- Command line
-
hf download hf://VALOR0316/koth-miner-92/source.py
-
curl -L -o source.py https://huggingface.co/VALOR0316/koth-miner-92/resolve/main/source.py
55.7 kB
| """Cost-bounded UID 92 consensus generalist v7.17 research candidate. | |
| Independent candidates are selected by deterministic sample, execution, and consensus evidence. | |
| Sparse or contradictory evidence escalates. The policy has no task IDs, prompt fingerprints, | |
| lookup tables, stored solutions, or benchmark-trained answer selector. | |
| """ | |
| import ast | |
| import json | |
| import math | |
| import re | |
| import subprocess | |
| import sys | |
| import time | |
| _MODELS = ( | |
| "qwen/qwen3.7-flash", | |
| "deepseek/deepseek-v4-flash", | |
| "deepseek/deepseek-v4-pro", | |
| "z-ai/glm-5.2", | |
| "openai/gpt-5.6-luna", | |
| "google/gemini-3.6-flash", | |
| "moonshotai/kimi-k3", | |
| ) | |
| _KIND = "valor-uid92-consensus-generalist-v7.17" | |
| _SAMPLE_MARK = re.compile(r"^Sample (Input|Output) (\d+)\s*$", re.M) | |
| _CODE_WORD = re.compile(r"\b(?:input|print|sys|def|import|from)\b") | |
| _NUMBER = re.compile(r"[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?\Z") | |
| _FLOAT_JUDGE = re.compile( | |
| r"(?:absolute\s+or\s+relative|relative\s+or\s+absolute)\s+(?:error|difference)|" | |
| r"(?:absolute|relative)\s+error|error[^\n]{0,80}(?:10\^|1e-)", re.I | |
| ) | |
| _AMBIGUOUS_JUDGE = re.compile( | |
| r"(?:print|output|return)\s+any\b|any\s+(?:valid\s+)?(?:answer|solution)\b|" | |
| r"multiple\s+(?:answers|solutions)[^\n]{0,100}(?:accepted|print any)", re.I | |
| ) | |
| _SEQUENTIAL_RISK = re.compile( | |
| r"(?:in|chronological)\s+order[\s\S]{0,300}(?:replace|overwrite|update)|" | |
| r"(?:replace|overwrite|update)[\s\S]{0,300}(?:in|chronological)\s+order|" | |
| r"for\s+each[\s\S]{0,180}(?:in\s+this\s+order|in\s+order)[\s\S]{0,250}" | |
| r"(?:operations?|append|delete|replace|update)", | |
| re.I, | |
| ) | |
| _MUTABLE_GRAPH_RISK = re.compile( | |
| r"(?:becomes?|turns?).{0,80}(?:into|from).{0,80}(?:road|open|available)|" | |
| r"(?:destroy|remove|unlock|open).{0,100}(?:wall|edge|cell|node)", re.I | re.S | |
| ) | |
| _MANY_OUTPUT_RISK = re.compile( | |
| r"(?:for\s+each|for\s+every).{0,100}(?:integer|value|number)|" | |
| r"(?:all|many).{0,80}(?:coefficients?|answers?|values?)", re.I | re.S | |
| ) | |
| _MODULAR_RISK = re.compile( | |
| r"(?:prime|modulo|modulus).{0,100}\bP\b|\bP\b.{0,100}(?:prime|modulo|modulus)", | |
| re.I | re.S, | |
| ) | |
| _COUNTING_RISK = re.compile( | |
| r"(?:number\s+of|count).{0,120}(?:graphs?|ways?|configurations?|sequences?|subsets?)", | |
| re.I | re.S, | |
| ) | |
| _CHOICE_LETTER = re.compile(r"(?<![A-Za-z])([A-D])(?![A-Za-z])", re.I) | |
| _NAMED_BLOCK = re.compile(r"```(candidate|generator|reference)\s*\n(.*?)```", re.I | re.S) | |
| _INPUT_BLOCK = re.compile(r"```(?:input|text|txt)?\s*\n(.*?)```", re.I | re.S) | |
| _FLOAT_TOL = 1e-9 | |
| _CASE_TIMEOUT = 3.0 | |
| _GEN_TIMEOUT = 1.0 | |
| _STRESS_TIMEOUT = 2.0 | |
| _CANDIDATE_TIMEOUT = 6.0 | |
| _CANDIDATE_GRACE_TIMEOUT = 9.0 | |
| _STRESS_SECONDS = 18.0 | |
| _STRESS_SIZES = (1, 2, 3, 5, 8, 12, 24, 48) | |
| _SOLVE_A = ( | |
| "Derive the algorithm only from the statement and constraints. The checker compares stdout " | |
| "tokens exactly, so match the required token count, order, and formatting. Use Python 3 and " | |
| "buffered whole-input parsing and output assembly. " | |
| ) | |
| _SOLVE_B = ( | |
| "Instantiate the largest stated constraints and derive concrete time and memory bounds before " | |
| "coding. Reject repeated rescans, slicing, convolutions, or state tables that exceed them. The " | |
| "largest case must retain substantial runtime margin rather than merely finish at the limit. " | |
| ) | |
| _SOLVE_C = ( | |
| "Check indexing, duplicates, boundaries, precision, and all statement samples. During pairwise " | |
| "loops keep operands in fresh local names, and test compressed or greedy invariants against " | |
| "literal exhaustive reasoning on the smallest legal instances." | |
| ) | |
| _SOLVE_D = ( | |
| "When a counting computation needs many coefficients of a bounded-degree polynomial, compare " | |
| "coefficient-space transitions with value-space evaluation followed by interpolation, and use " | |
| "the representation with the smaller proved operation count." | |
| ) | |
| _SOLVE = _SOLVE_A + _SOLVE_B + _SOLVE_C | |
| _PRIMARY_FORMAT = ( | |
| "First instantiate the largest stated constraints and derive a concrete time and memory bound. " | |
| "Reject any recurrence, convolution, state table, or repeated scan that cannot meet that bound. " | |
| "Then return only one complete efficient Python 3 program as raw source: no prose, fences, " | |
| "reference implementation, or sample-specific branch." | |
| ) | |
| _SEQUENTIAL_A = ( | |
| "For mandatory sequential overwrite operations, every incoming value is consumed exactly once, " | |
| "but an early incoming value may be deliberately erased by a later write to the same target. " | |
| "Separate values that survive in the final state from operations sacrificed by later overwrites. " | |
| ) | |
| _SEQUENTIAL_B = ( | |
| "Do not treat all incoming values as freely placeable or refundable. Prove any greedy choice by an " | |
| "exchange argument, preserve chronological last-write semantics, and exhaustively enumerate all " | |
| "target sequences for tiny state and operation counts before returning code." | |
| ) | |
| _MUTABLE_GRAPH = ( | |
| "During shortest-path search, never mutate a graph, grid, or availability table shared by all " | |
| "queued paths. A change discovered at higher cost must not leak into lower-cost states. Attach " | |
| "persistent changes to the search state, or prove an equivalent weighted transition on an " | |
| "immutable graph before using Dijkstra or 0-1 BFS." | |
| ) | |
| _REPAIR = ( | |
| "Re-derive independently; use the failure as evidence of a general defect, never a special " | |
| "case. Test greedy, compressed-state, and sequential-update invariants against a literal " | |
| "exhaustive solver on the smallest legal instances. Trace every statement sample through input " | |
| "parsing, state changes, and output length. Return only complete raw Python 3 source." | |
| ) | |
| _REVIEW = ( | |
| "Solve independently, then audit both candidates for logic, boundaries, complexity, and format. " | |
| "Test greedy or compressed-state invariants against a literal exhaustive solver on the smallest " | |
| "legal instances, including duplicates, ordering, and overwrites. Trace every statement sample " | |
| "through parsing and output length. Return one corrected raw Python 3 program without prose." | |
| ) | |
| _MUST_PASS = ( | |
| "Every statement sample is an executable acceptance condition. Re-derive the algorithm rather " | |
| "than fitting sample outputs, mentally execute the complete program on every stated sample, and " | |
| "return code only after all samples agree. If prior candidates share an invariant, actively seek " | |
| "a counterexample to that invariant before reusing it." | |
| ) | |
| _CHALLENGE_A = ( | |
| "Solve independently without copying the prior candidate. Return exactly three fenced blocks and " | |
| "no prose. The candidate block must contain a complete Python 3 solution. " | |
| ) | |
| _CHALLENGE_B = ( | |
| "The generator block must contain a Python program accepting integer seed and size command-line " | |
| "arguments, calling random.seed(seed), and printing one small legal input derived only from the " | |
| "statement. Vary ties, duplicates, boundaries, operation order, and branch combinations. " | |
| ) | |
| _CHALLENGE_SCALE = ( | |
| "For size values up to 5, emit literal tiny cases suitable for the reference. For " | |
| "larger size values, grow at least one main input dimension toward the stated constraints so the " | |
| "efficient candidate receives a performance smoke test; do not cap every output at a toy size. " | |
| "The generated input must remain legal.\n" | |
| ) | |
| _CHALLENGE_FORMAT = ( | |
| "The reference block must be a literal exhaustive solver for only small legal inputs and must " | |
| "not reuse the candidate invariant. The first candidate fence must be labeled python so the " | |
| "validator extracts exactly the executable candidate used by the selector.\n" | |
| "```python\n<program>\n```\n```generator\n<program>\n```" | |
| "\n```reference\n<program>\n```" | |
| ) | |
| _CHALLENGE = _CHALLENGE_A + _CHALLENGE_B + _CHALLENGE_SCALE + _CHALLENGE_FORMAT | |
| _GENERATOR_ONLY_A = ( | |
| "Return exactly two fenced blocks and no prose. Generator accepts integer seed and size command-" | |
| "line arguments, calls random.seed(seed), and prints one legal input. Reference is a literal " | |
| "exhaustive solver for small legal inputs and must not use an optimized invariant.\n" | |
| ) | |
| _GENERATOR_ONLY_B = ( | |
| "Use literal tiny inputs when " | |
| "size is at most 5, but for larger values grow a main dimension toward the statement constraints " | |
| "instead of capping every case at toy size. The reference may reject larger smoke-test inputs.\n" | |
| ) | |
| _GENERATOR_ONLY_C = ( | |
| "Construct and verify every promised input property inside the generator instead of guessing " | |
| "values. In particular validate primality when a field must be prime, construct permutations " | |
| "by shuffling a complete range, and construct required graph properties by design.\n" | |
| "```generator\n<program>\n```\n```reference\n<program>\n```" | |
| ) | |
| _GENERATOR_ONLY = _GENERATOR_ONLY_A + _GENERATOR_ONLY_B + _GENERATOR_ONLY_C | |
| _LARGE_INPUT_ONLY = ( | |
| "Performance-smoke-input protocol: derive one legal test case only from the statement. Make a " | |
| "main dimension close to its maximum constraint while keeping the serialized input under 12000 " | |
| "characters. Return exactly one fenced block labeled input, with no program, answer, or prose.\n" | |
| "```input\n<one legal input>\n```" | |
| ) | |
| _OPTIMIZE = ( | |
| "The program below passed every statement sample and the available independent small-reference " | |
| "checks, but exceeded the required runtime margin on a generated maximum-scale legal input. " | |
| "Preserve its exact semantics while replacing the bottleneck with a lower-complexity algorithm. " | |
| ) | |
| _OPTIMIZE_FORMAT = ( | |
| "Re-derive the time and memory bound, do not special-case the shown input, and return only one " | |
| "complete raw Python 3 program with no prose or fences.\n\nSlow program:\n" | |
| ) | |
| _FINAL_DERIVE = ( | |
| "Prior attempts failed deterministic sample, independent-reference, or maximum-scale checks. " | |
| "Derive one final solution from the statement from first principles. Use the execution evidence " | |
| "only to reject broken invariants; never add an input-specific branch or memorize an output. " | |
| ) | |
| _FLOAT_FORMAT = ( | |
| "The statement permits floating-point tolerance. Print each real-valued result in fixed " | |
| "decimal notation with exactly 15 digits after the decimal point so exact-token benchmark " | |
| "harnesses retain the precision shown by standard contest reference outputs." | |
| ) | |
| def _load(weights): | |
| try: | |
| data = json.loads(bytes(weights).decode("utf-8")) | |
| except Exception as exc: | |
| raise ValueError("UID 92 clean settings are not valid JSON") from exc | |
| required = { | |
| "schema", "kind", "primary", "challenger", "generator", "reviewer", "rescue", | |
| "mcq_tiebreaker", | |
| "primary_max_tokens", "challenge_max_tokens", "review_max_tokens", | |
| "generator_max_tokens", "rescue_max_tokens", "sample_limit", "stress_cases", | |
| "min_consensus_cases", "critic", | |
| "temperature", "primary_reasoning", "challenge_reasoning", "review_reasoning", | |
| "generator_reasoning", "rescue_reasoning", | |
| } | |
| if not isinstance(data, dict) or set(data) != required: | |
| raise ValueError("UID 92 clean settings have an invalid schema") | |
| if data["schema"] != 3 or data["kind"] != _KIND: | |
| raise ValueError("UID 92 clean settings do not match this source") | |
| for field in ("primary", "challenger", "generator", "reviewer", "rescue", | |
| "mcq_tiebreaker"): | |
| if type(data[field]) is not int or not 0 <= data[field] < len(_MODELS): | |
| raise ValueError("UID 92 clean model index is invalid") | |
| if (type(data["primary_max_tokens"]) is not int | |
| or not 4096 <= data["primary_max_tokens"] <= 24576): | |
| raise ValueError("UID 92 clean primary token limit is invalid") | |
| for field in ("challenge_max_tokens", "generator_max_tokens", "review_max_tokens"): | |
| if type(data[field]) is not int or not 2048 <= data[field] <= 16384: | |
| raise ValueError("UID 92 clean fallback token limit is invalid") | |
| if (type(data["rescue_max_tokens"]) is not int | |
| or not 4096 <= data["rescue_max_tokens"] <= 24576): | |
| raise ValueError("UID 92 clean rescue token limit is invalid") | |
| if type(data["sample_limit"]) is not int or not 1 <= data["sample_limit"] <= 4: | |
| raise ValueError("UID 92 clean sample limit is invalid") | |
| if type(data["stress_cases"]) is not int or not 4 <= data["stress_cases"] <= 24: | |
| raise ValueError("UID 92 clean stress case limit is invalid") | |
| if (type(data["min_consensus_cases"]) is not int | |
| or not 2 <= data["min_consensus_cases"] <= data["stress_cases"]): | |
| raise ValueError("UID 92 clean consensus threshold is invalid") | |
| critic = data["critic"] | |
| if (not isinstance(critic, dict) | |
| or set(critic) != {"schema", "mean", "scale", "coef", "intercept"} | |
| or critic["schema"] != 1): | |
| raise ValueError("UID 92 critic schema is invalid") | |
| for field in ("mean", "scale", "coef"): | |
| if (not isinstance(critic[field], list) or len(critic[field]) != 18 | |
| or any(type(value) not in (int, float) or not math.isfinite(value) | |
| for value in critic[field])): | |
| raise ValueError("UID 92 critic vector is invalid") | |
| if (any(value <= 0 for value in critic["scale"]) | |
| or type(critic["intercept"]) not in (int, float) | |
| or not math.isfinite(critic["intercept"])): | |
| raise ValueError("UID 92 critic normalization is invalid") | |
| if type(data["temperature"]) not in (int, float) or data["temperature"] != 0: | |
| raise ValueError("UID 92 clean temperature must be deterministic") | |
| for field in ("primary_reasoning", "challenge_reasoning", "generator_reasoning", | |
| "review_reasoning", "rescue_reasoning"): | |
| if data[field] not in ("none", "low", "medium", "high"): | |
| raise ValueError("UID 92 clean reasoning effort is invalid") | |
| return data | |
| def _is_code(prompt): | |
| text = str(prompt) | |
| return ("Write a complete Python 3 program" in text | |
| and "standard input" in text and "standard output" in text) | |
| def _is_mcq(prompt): | |
| text = "\n" + str(prompt) | |
| return all("\n" + option in text for option in ("A)", "B)", "C)", "D)")) | |
| def _non_code_request(prompt): | |
| text = str(prompt) | |
| if _is_mcq(text): | |
| return text | |
| return text + "\n\nDerive the answer strictly from the question and return only the answer." | |
| def _choice(answer): | |
| hits = _CHOICE_LETTER.findall(str(answer)) | |
| return hits[-1].upper() if hits else "" | |
| def _mcq_audit_request(prompt): | |
| return ( | |
| str(prompt) | |
| + "\n\nIndependently solve the question from its stated facts and options. Recompute any " | |
| "arithmetic or definition instead of trusting another solver. Return exactly one option " | |
| "letter A, B, C, or D and no other text." | |
| ) | |
| def _samples(prompt, limit): | |
| text = str(prompt).replace("\r\n", "\n").replace("\r", "\n") | |
| if _AMBIGUOUS_JUDGE.search(text): | |
| return [] | |
| marks = [(m.start(), m.end(), m.group(1), int(m.group(2))) | |
| for m in _SAMPLE_MARK.finditer(text)] | |
| blocks = {} | |
| for index, (_start, end, kind, number) in enumerate(marks): | |
| stop = marks[index + 1][0] if index + 1 < len(marks) else len(text) | |
| lines = text[end:stop].split("\n") | |
| while lines and not lines[0].strip(): | |
| lines.pop(0) | |
| kept = [] | |
| for line in lines: | |
| if not line.strip(): | |
| break | |
| kept.append(line) | |
| blocks[(kind, number)] = "\n".join(kept) | |
| pairs = [] | |
| for number in sorted({number for _kind, number in blocks}): | |
| stdin = blocks.get(("Input", number)) | |
| stdout = blocks.get(("Output", number)) | |
| if stdin and stdout: | |
| pairs.append((stdin, stdout)) | |
| return pairs[:limit] | |
| def _source(answer): | |
| """Mirror the validator's first-code extraction for every returnable candidate.""" | |
| # Selecting a later named/longer block is unsafe even when it came from the same pool response: | |
| # the validator compares only its first qualifying extraction. Ranking may use local execution | |
| # evidence, but every candidate enters here and is returned without rewriting. | |
| text = str(answer or "") | |
| if "```" in text: | |
| for block in (value for value in text.split("```") if value.strip()): | |
| block = block[len("python"):] if block.lstrip().lower().startswith("python") else block | |
| if "input" in block or "print" in block: | |
| return block.strip() + "\n" | |
| return text.strip() + "\n" | |
| def _tokens_match(observed, expected, numeric): | |
| got = str(observed).split() | |
| want = str(expected).split() | |
| if got == want: | |
| return True | |
| if not numeric or len(got) != len(want) or not got: | |
| return False | |
| for left, right in zip(got, want): | |
| if _NUMBER.fullmatch(left) is None or _NUMBER.fullmatch(right) is None: | |
| return False | |
| try: | |
| a, b = float(left), float(right) | |
| except ValueError: | |
| return False | |
| if not math.isfinite(a) or not math.isfinite(b): | |
| return False | |
| if abs(a - b) > _FLOAT_TOL * max(1.0, abs(b)): | |
| return False | |
| return True | |
| def _run_program_detailed(code, stdin_text, timeout, argv=()): | |
| try: | |
| run = subprocess.run( | |
| [sys.executable, "-I", "-c", code, *[str(value) for value in argv]], | |
| input=str(stdin_text), | |
| capture_output=True, text=True, timeout=timeout, | |
| ) | |
| except subprocess.TimeoutExpired: | |
| return None, "runtime timeout after " + str(timeout) + " seconds" | |
| except (OSError, ValueError) as exc: | |
| return None, type(exc).__name__ + ": " + str(exc)[:600] | |
| if run.returncode == 0: | |
| return run.stdout, "" | |
| detail = (run.stderr or "").strip() | |
| if not detail: | |
| detail = "process exited with status " + str(run.returncode) | |
| return None, detail[-1200:] | |
| def _run_program(code, stdin_text, timeout, argv=()): | |
| return _run_program_detailed(code, stdin_text, timeout, argv)[0] | |
| def _check(code, samples, numeric): | |
| if not code.strip(): | |
| return 0, max(1, len(samples)), ( | |
| samples[0][0] if samples else "", "<empty>", | |
| samples[0][1] if samples else "valid Python", | |
| ) | |
| try: | |
| compile(code, "<candidate>", "exec") | |
| except (SyntaxError, ValueError, TypeError) as exc: | |
| return 0, max(1, len(samples)), ( | |
| samples[0][0] if samples else "", | |
| type(exc).__name__ + ": " + str(exc)[:600], | |
| samples[0][1] if samples else "valid Python", | |
| ) | |
| passed = 0 | |
| first_bad = None | |
| for stdin, expected in samples: | |
| observed, runtime_error = _run_program_detailed(code, stdin, _CASE_TIMEOUT) | |
| good = observed is not None and _tokens_match(observed, expected, numeric) | |
| if good: | |
| passed += 1 | |
| elif first_bad is None: | |
| if observed is not None: | |
| evidence = observed | |
| elif runtime_error: | |
| evidence = "<runtime failure>\n" + runtime_error | |
| else: | |
| evidence = "<no output>" | |
| first_bad = (stdin, evidence, expected) | |
| return passed, len(samples), first_bad | |
| def _failure_note(passed, total, failure): | |
| if failure is None: | |
| return "sample verification: " + str(passed) + "/" + str(total) | |
| stdin, observed, expected = failure | |
| return ( | |
| "sample verification: " + str(passed) + "/" + str(total) | |
| + "\nFailing sample input:\n" + str(stdin)[:500] | |
| + "\nCandidate output:\n" + str(observed)[:500] | |
| + "\nExpected output:\n" + str(expected)[:500] | |
| ) | |
| def _sample_failure_report(candidates, samples, numeric): | |
| notes = [] | |
| for index, code in enumerate(candidates): | |
| passed, total, failure = _check(code, samples, numeric) | |
| if failure is None and passed == total: | |
| continue | |
| notes.append( | |
| "Candidate " + str(index + 1) + " deterministic diagnostics:\n" | |
| + _failure_note(passed, total, failure) | |
| ) | |
| return "\n\n".join(notes)[:6000] | |
| def _named_programs(answer): | |
| programs = {} | |
| for name, body in _NAMED_BLOCK.findall(str(answer or "")): | |
| lines = [line for line in body.strip().splitlines() | |
| if line.strip().lower() not in ("<program>", "program")] | |
| code = "\n".join(lines).strip() | |
| if code: | |
| programs[name.lower()] = code + "\n" | |
| if programs: | |
| # Challenge replies use a python-labeled first fence for the efficient candidate so the | |
| # validator's first-code extraction and our candidate are identical. Tooling remains in | |
| # named generator/reference fences and can never become a returned answer. | |
| python_blocks = re.findall( | |
| r"```(?:python|python3|py)\s*\n(.*?)```", str(answer or ""), re.I | re.S) | |
| if python_blocks: | |
| candidate = python_blocks[0].strip() | |
| if candidate: | |
| programs.setdefault("candidate", candidate + "\n") | |
| else: | |
| blocks = [] | |
| for raw in re.findall(r"```(?:python|python3|py)?\s*\n(.*?)```", str(answer or ""), | |
| re.I | re.S): | |
| code = raw.strip() | |
| if code: | |
| blocks.append(code + "\n") | |
| if len(blocks) == 3: | |
| programs = dict(zip(("candidate", "generator", "reference"), blocks)) | |
| return programs | |
| def _plain_program_blocks(answer): | |
| blocks = [] | |
| for raw in re.findall(r"```(?:python|python3|py)?\s*\n(.*?)```", str(answer or ""), | |
| re.I | re.S): | |
| code = raw.strip() | |
| if code: | |
| blocks.append(code + "\n") | |
| return blocks | |
| def _large_input(answer): | |
| blocks = _INPUT_BLOCK.findall(str(answer or "")) | |
| if len(blocks) != 1: | |
| return "" | |
| value = blocks[0].strip() | |
| if not value or len(value) > 12000: | |
| return "" | |
| return value + "\n" | |
| def _deep_code_risk(text): | |
| """Select effort by broad statement structure, never by identity or stored prompt similarity.""" | |
| value = str(text) | |
| signals = sum(bool(pattern.search(value)) for pattern in ( | |
| _MANY_OUTPUT_RISK, _MODULAR_RISK, _COUNTING_RISK, | |
| )) | |
| return signals >= 2 | |
| def _generated_inputs(generator, case_limit, require_scale=True): | |
| if not generator or not generator.strip(): | |
| return [] | |
| # A generator that ignores the supplied scale only exercises toy inputs and can hide | |
| # asymptotic failures. This is a task-independent interface requirement, not prompt dispatch. | |
| direct_size = re.search(r"(?:sys\.)?argv\s*\[\s*2\s*\]", generator) | |
| sliced_args = (re.search(r"(?:sys\.)?argv\s*\[\s*1\s*:\s*\]", generator) | |
| and re.search(r"\[\s*1\s*\]", generator)) | |
| if direct_size is None and not sliced_args: | |
| return [] | |
| try: | |
| compile(generator, "<generator>", "exec") | |
| except (SyntaxError, ValueError, TypeError): | |
| return [] | |
| deadline = time.monotonic() + _STRESS_SECONDS | |
| cases = [] | |
| observations = [] | |
| if case_limit <= 1: | |
| size_schedule = _STRESS_SIZES[:case_limit] | |
| elif case_limit < len(_STRESS_SIZES): | |
| last = len(_STRESS_SIZES) - 1 | |
| size_schedule = tuple( | |
| _STRESS_SIZES[round(index * last / (case_limit - 1))] | |
| for index in range(case_limit) | |
| ) | |
| else: | |
| size_schedule = tuple( | |
| _STRESS_SIZES[index % len(_STRESS_SIZES)] | |
| for index in range(case_limit) | |
| ) | |
| for index, size in enumerate(size_schedule): | |
| if time.monotonic() >= deadline: | |
| break | |
| stdin = _run_program(generator, "", _GEN_TIMEOUT, (index + 1, size)) | |
| if not stdin or not stdin.strip() or len(stdin) > 20000: | |
| continue | |
| observations.append((size, stdin)) | |
| if stdin not in cases: | |
| cases.append(stdin) | |
| # Reject generators that merely mention argv[2] while capping all large requests at the same | |
| # toy shape. Compare text size, token count, and corresponding early integer fields; constants | |
| # such as a modulus stay equal and therefore cannot fake growth. | |
| small = [text for size, text in observations if size <= 5] | |
| large = [text for size, text in observations if size >= 24] | |
| if require_scale and (not small or not large): | |
| return [] | |
| if small and large: | |
| def shape(text): | |
| tokens = str(text).split() | |
| integers = [] | |
| for token in tokens[:32]: | |
| try: | |
| integers.append(abs(int(token))) | |
| except ValueError: | |
| integers.append(0) | |
| return len(str(text)), len(tokens), integers | |
| small_shapes = [shape(text) for text in small] | |
| large_shapes = [shape(text) for text in large] | |
| small_chars = max(row[0] for row in small_shapes) | |
| small_tokens = max(row[1] for row in small_shapes) | |
| grew = (max(row[0] for row in large_shapes) >= max(8, int(small_chars * 1.25)) | |
| or max(row[1] for row in large_shapes) >= max(3, int(small_tokens * 1.25))) | |
| width = max((len(row[2]) for row in small_shapes + large_shapes), default=0) | |
| for position in range(width): | |
| low = max((row[2][position] for row in small_shapes | |
| if position < len(row[2])), default=0) | |
| high = max((row[2][position] for row in large_shapes | |
| if position < len(row[2])), default=0) | |
| if high > low and high >= max(2, int(low * 1.25)): | |
| grew = True | |
| break | |
| if require_scale and not grew: | |
| return [] | |
| return cases | |
| def _quality(code): | |
| """Return a small task-independent static safety score; never infer a task identity.""" | |
| try: | |
| tree = ast.parse(code) | |
| except (SyntaxError, ValueError, TypeError): | |
| return -100 | |
| score = 0 | |
| nodes = list(ast.walk(tree)) | |
| score += int(any(isinstance(node, (ast.For, ast.While)) for node in nodes)) | |
| score += int(any(isinstance(node, ast.FunctionDef) for node in nodes)) | |
| score += int(40 <= len(code) <= 12000) | |
| dangerous = {"eval", "exec", "compile", "__import__"} | |
| score -= 4 * sum( | |
| isinstance(node, ast.Call) and isinstance(node.func, ast.Name) | |
| and node.func.id in dangerous for node in nodes | |
| ) | |
| score -= int(len(code) > 20000) | |
| return score | |
| def _ast_depth(node): | |
| children = list(ast.iter_child_nodes(node)) | |
| return 1 + max((_ast_depth(child) for child in children), default=0) | |
| def _critic_features(code, sample_ratio, completion_ratio, consensus_ratio): | |
| try: | |
| tree = ast.parse(code) | |
| except (SyntaxError, ValueError, TypeError): | |
| return [0.0] * 15 + [float(sample_ratio), float(completion_ratio), | |
| float(consensus_ratio)] | |
| nodes = list(ast.walk(tree)) | |
| dangerous = {"eval", "exec", "compile", "__import__"} | |
| return [ | |
| math.log1p(len(code)), math.log1p(code.count("\n") + 1), math.log1p(len(nodes)), | |
| float(sum(isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)) for node in nodes)), | |
| float(sum(isinstance(node, (ast.For, ast.AsyncFor, ast.While)) for node in nodes)), | |
| float(sum(isinstance(node, (ast.If, ast.IfExp, ast.Match)) for node in nodes)), | |
| float(sum(isinstance(node, ast.Call) for node in nodes)), | |
| float(sum(isinstance(node, ast.Subscript) for node in nodes)), | |
| float(sum(isinstance(node, (ast.ListComp, ast.SetComp, ast.DictComp, ast.GeneratorExp)) | |
| for node in nodes)), | |
| float(sum(isinstance(node, (ast.Try, ast.Raise)) for node in nodes)), | |
| float(sum(isinstance(node, (ast.Import, ast.ImportFrom)) for node in nodes)), | |
| float(sum(isinstance(node, ast.Return) for node in nodes)), | |
| float(sum(isinstance(node, ast.Constant) and type(node.value) is int for node in nodes)), | |
| float(_ast_depth(tree)), | |
| float(sum(isinstance(node, ast.Call) and isinstance(node.func, ast.Name) | |
| and node.func.id in dangerous for node in nodes)), | |
| float(sample_ratio), float(completion_ratio), float(consensus_ratio), | |
| ] | |
| def _critic_score(code, sample_ratio, completion_ratio, consensus_ratio, critic): | |
| if not critic: | |
| return 0.0 | |
| values = _critic_features(code, sample_ratio, completion_ratio, consensus_ratio) | |
| logit = float(critic["intercept"]) | |
| for value, mean, scale, coefficient in zip( | |
| values, critic["mean"], critic["scale"], critic["coef"]): | |
| logit += ((value - mean) / scale) * coefficient | |
| return 1.0 / (1.0 + math.exp(-max(-30.0, min(30.0, logit)))) | |
| def _reference_valid(reference, samples, numeric): | |
| if not reference or not reference.strip(): | |
| return False | |
| try: | |
| compile(reference, "<reference>", "exec") | |
| except (SyntaxError, ValueError, TypeError): | |
| return False | |
| agreed = 0 | |
| for stdin, expected in samples: | |
| observed = _run_program(reference, stdin, _STRESS_TIMEOUT) | |
| if observed is None: | |
| continue | |
| if not _tokens_match(observed, expected, numeric): | |
| return False | |
| agreed += 1 | |
| return agreed >= min(2, len(samples)) | |
| def _trusted_oracle(reference_a, reference_b, samples, generated, numeric, minimum): | |
| if (not _reference_valid(reference_a, samples, numeric) | |
| or not _reference_valid(reference_b, samples, numeric)): | |
| return [] | |
| agreed = [] | |
| for stdin in generated: | |
| left = _run_program(reference_a, stdin, _STRESS_TIMEOUT) | |
| right = _run_program(reference_b, stdin, _STRESS_TIMEOUT) | |
| # Literal references are intentionally allowed to be exponential. A timeout or a | |
| # disagreement means that case has no trusted oracle; neither can invalidate separate | |
| # inputs on which the independent constructions do agree. | |
| if left is None or right is None: | |
| continue | |
| if not _tokens_match(left, right, numeric): | |
| continue | |
| agreed.append((stdin, left)) | |
| return agreed if len(agreed) >= minimum else [] | |
| def _candidate_evidence(candidates, samples, generated, numeric, oracle_cases=(), critic=None, | |
| oracle_groups=(), execution_cache=None): | |
| cache = execution_cache if execution_cache is not None else {} | |
| def execute(code, stdin): | |
| key = (code, stdin) | |
| if key not in cache: | |
| cache[key] = _run_program(code, stdin, _CANDIDATE_TIMEOUT) | |
| return cache[key] | |
| rows = [] | |
| outputs = [] | |
| for index, code in enumerate(candidates): | |
| passed, total, failure = _check(code, samples, numeric) | |
| values = [execute(code, stdin) for stdin in generated] | |
| outputs.append(values) | |
| rows.append({ | |
| "index": index, | |
| "sample_passed": passed, | |
| "sample_total": total, | |
| "sample_valid": bool(code.strip() and failure is None and passed == total), | |
| "completed": sum(value is not None for value in values), | |
| "consensus": 0, | |
| "oracle": 0, | |
| "oracle_total": len(oracle_cases), | |
| "oracle_groups": [], | |
| "critic": 0.0, | |
| "quality": _quality(code), | |
| }) | |
| for case_index in range(len(generated)): | |
| groups = [] | |
| for candidate_index, values in enumerate(outputs): | |
| value = values[case_index] | |
| if value is None: | |
| continue | |
| placed = False | |
| for representative, members in groups: | |
| if _tokens_match(value, representative, numeric): | |
| members.append(candidate_index) | |
| placed = True | |
| break | |
| if not placed: | |
| groups.append((value, [candidate_index])) | |
| if not groups: | |
| continue | |
| largest = max(len(members) for _value, members in groups) | |
| if largest < 2: | |
| continue | |
| winners = [members for _value, members in groups if len(members) == largest] | |
| if len(winners) == 1: | |
| for candidate_index in winners[0]: | |
| rows[candidate_index]["consensus"] += 1 | |
| for row, code in zip(rows, candidates): | |
| for stdin, expected in oracle_cases: | |
| observed = execute(code, stdin) | |
| if observed is not None and _tokens_match(observed, expected, numeric): | |
| row["oracle"] += 1 | |
| for group in oracle_groups: | |
| matches = 0 | |
| for stdin, expected in group: | |
| observed = execute(code, stdin) | |
| if observed is not None and _tokens_match(observed, expected, numeric): | |
| matches += 1 | |
| row["oracle_groups"].append((matches, len(group))) | |
| row["critic"] = _critic_score( | |
| code, | |
| row["sample_passed"] / max(1, row["sample_total"]), | |
| row["completed"] / max(1, len(generated)), | |
| row["consensus"] / max(1, len(generated)), | |
| critic, | |
| ) | |
| return rows, outputs | |
| def _retry_oracle_timeouts(candidates, rows, outputs, samples, generated, numeric, | |
| oracle_cases, critic, oracle_groups, execution_cache, | |
| minimum, attempted): | |
| """Retry only transient scale timeouts backed by complete independent oracle evidence.""" | |
| if not oracle_cases or not generated: | |
| return rows, outputs | |
| changed = False | |
| for code, row, values in zip(candidates, rows, outputs): | |
| if not row["sample_valid"] or not _oracle_supported(row, minimum): | |
| continue | |
| for stdin, output in zip(generated, values): | |
| key = (code, stdin) | |
| if output is not None or key in attempted: | |
| continue | |
| attempted.add(key) | |
| execution_cache[key] = _run_program(code, stdin, _CANDIDATE_GRACE_TIMEOUT) | |
| changed = True | |
| if not changed: | |
| return rows, outputs | |
| return _candidate_evidence( | |
| candidates, samples, generated, numeric, oracle_cases, critic, | |
| oracle_groups, execution_cache, | |
| ) | |
| def _best_candidate(candidates, rows): | |
| viable = [row for row in rows if row["sample_valid"]] | |
| pool = viable if viable else rows | |
| best = max(pool, key=lambda row: ( | |
| row["sample_valid"], row["sample_passed"], | |
| bool(row["oracle_total"] and row["oracle"] == row["oracle_total"]), | |
| row["oracle"], row["completed"], row["consensus"], | |
| row["critic"], row["quality"], -row["index"], | |
| )) | |
| return candidates[best["index"]], best | |
| def _oracle_supported(row, minimum): | |
| """Require one clean generator family and no repeated contradiction from another.""" | |
| groups = row.get("oracle_groups") or [(row["oracle"], row["oracle_total"])] | |
| return (any(total >= minimum and matches == total for matches, total in groups) | |
| and all(total - matches < 2 for matches, total in groups if total >= minimum)) | |
| def _supported_candidate(row, candidates, generated, minimum, oracle_cases, disagreement): | |
| """Require sample validity, full performance completion, and independent evidence.""" | |
| if (not row["sample_valid"] | |
| or (generated and row["completed"] != len(generated))): | |
| return False | |
| if oracle_cases: | |
| # A model-written generator can accidentally violate a semantic input constraint. Require | |
| # one independently generated family with complete agreement, and reject a candidate only | |
| # when another family supplies repeated rather than isolated contradictory evidence. | |
| return _oracle_supported(row, minimum) | |
| return (len(candidates) >= 2 | |
| and len(generated) >= minimum | |
| and not disagreement | |
| and row["consensus"] >= minimum) | |
| def _disagreement_note(generated, outputs): | |
| notes = [] | |
| for case_index, stdin in enumerate(generated): | |
| values = [rows[case_index] for rows in outputs] | |
| completed = [value for value in values if value is not None] | |
| if len(completed) < 2 or all(value.split() == completed[0].split() | |
| for value in completed[1:]): | |
| continue | |
| notes.append( | |
| "Input:\n" + stdin[:800] + "\nCandidate outputs:\n" | |
| + "\n---\n".join((value if value is not None else "<no output>")[:800] | |
| for value in values) | |
| ) | |
| if len(notes) >= 2: | |
| break | |
| return "\n\n".join(notes) | |
| def _execution_failure_report(candidates, generated, outputs): | |
| notes = [] | |
| for candidate_index, values in enumerate(outputs): | |
| for case_index, value in enumerate(values): | |
| if value is not None: | |
| continue | |
| notes.append( | |
| "Candidate " + str(candidate_index + 1) | |
| + " failed to complete this generated legal input within the local " | |
| + str(_STRESS_TIMEOUT) + " second smoke-test limit:\n" | |
| + generated[case_index][:1200] | |
| ) | |
| break | |
| if len(notes) >= 3: | |
| break | |
| return "\n\n".join(notes) | |
| def build_agent(weights): | |
| config = _load(weights) | |
| def invoke(call_model, model_index, prompt, max_tokens, reasoning): | |
| params = { | |
| "max_tokens": max_tokens, | |
| "reasoning": {"effort": reasoning}, | |
| "temperature": config["temperature"], | |
| } | |
| try: | |
| return call_model( | |
| _MODELS[model_index], [{"role": "user", "content": prompt}], params | |
| ) | |
| except Exception: | |
| # A transient provider/transport failure is an invalid candidate, not a reason to abort | |
| # the whole six-task proof. The bounded next stage still observes the same statement. | |
| return "" | |
| def agent(prompt, call_model): | |
| original = str(prompt) | |
| if not _is_code(original): | |
| answer = invoke( | |
| call_model, config["primary"], _non_code_request(original), | |
| config["primary_max_tokens"], config["primary_reasoning"], | |
| ) | |
| if _is_mcq(original): | |
| first = _choice(answer) | |
| review = invoke( | |
| call_model, config["reviewer"], _mcq_audit_request(original), | |
| config["review_max_tokens"], config["review_reasoning"], | |
| ) | |
| second = _choice(review) | |
| if first and first == second: | |
| return first | |
| if not first and second: | |
| return second | |
| if first and not second: | |
| return first | |
| # A disagreement is decided by a third independent family. This routing depends | |
| # only on the candidates' A-D outputs, never on a question identity. | |
| tie = invoke( | |
| call_model, config["mcq_tiebreaker"], _mcq_audit_request(original), | |
| config["generator_max_tokens"], config["generator_reasoning"], | |
| ) | |
| third = _choice(tie) | |
| for value in (first, second, third): | |
| if value and (first, second, third).count(value) >= 2: | |
| return value | |
| return third or second or first or str(tie).strip() or str(review).strip() | |
| if str(answer).strip(): | |
| return answer | |
| # A parent watchdog turns a stalled provider call into an empty response. Returning | |
| # that value would preserve the proof but can turn a recoverable transport failure into | |
| # an MMLU/math floor DQ, so retry once through the pinned orthogonal family. | |
| return invoke( | |
| call_model, config["reviewer"], _non_code_request(original), | |
| config["review_max_tokens"], config["review_reasoning"], | |
| ) | |
| samples = _samples(original, config["sample_limit"]) | |
| numeric = bool(_FLOAT_JUDGE.search(original)) | |
| sequential_risk = bool(_SEQUENTIAL_RISK.search(original)) | |
| mutable_graph_risk = bool(_MUTABLE_GRAPH_RISK.search(original)) | |
| deep_risk = _deep_code_risk(original) | |
| request = original + "\n\nGeneral reliability protocol: " + _SOLVE | |
| if deep_risk: | |
| request += "\n\nHigh-complexity representation protocol: " + _SOLVE_D | |
| if numeric: | |
| request += "\n\nNumeric-output protocol: " + _FLOAT_FORMAT | |
| if sequential_risk: | |
| request += "\n\nSequential-update protocol: " + _SEQUENTIAL_A + _SEQUENTIAL_B | |
| if mutable_graph_risk: | |
| request += "\n\nMutable-state shortest-path protocol: " + _MUTABLE_GRAPH | |
| primary_answer = invoke( | |
| call_model, config["primary"], request + "\n\nOutput protocol:\n" + _PRIMARY_FORMAT, | |
| config["primary_max_tokens"], | |
| "high" if deep_risk else config["primary_reasoning"], | |
| ) | |
| primary_programs = _named_programs(primary_answer) | |
| first = _source(primary_answer) | |
| references = [] | |
| if primary_programs.get("reference"): | |
| references.append(primary_programs["reference"]) | |
| tooling_request = ( | |
| request + "\n\nEfficient program under test:\n" + first | |
| + "\n\nBuild independent executable test tooling; do not rewrite the efficient " | |
| "program or assume its invariant.\n" + _GENERATOR_ONLY | |
| ) | |
| challenge_answer = invoke( | |
| call_model, config["challenger"], tooling_request, | |
| config["challenge_max_tokens"], config["challenge_reasoning"], | |
| ) | |
| programs = _named_programs(challenge_answer) | |
| generators = [] | |
| if programs.get("generator"): | |
| generators.append(programs["generator"]) | |
| if programs.get("reference"): | |
| references.append(programs["reference"]) | |
| candidates = [first] | |
| # A different provider builds a second literal reference. Only outputs on which both | |
| # independently written references agree can become trusted oracle evidence. | |
| cross_request = ( | |
| request + "\n\nA separate solver proposed this program:\n" + first | |
| + "\n\nIndependently derive a competing efficient solution and executable test " | |
| "tooling. Do not copy its invariant.\n" + _CHALLENGE | |
| ) | |
| cross_answer = invoke( | |
| call_model, config["reviewer"], cross_request, | |
| config["review_max_tokens"], config["review_reasoning"], | |
| ) | |
| cross_programs = _named_programs(cross_answer) | |
| second = _source(cross_answer) | |
| if second and second.strip() and second != first: | |
| candidates.append(second) | |
| if (cross_programs.get("generator") | |
| and cross_programs["generator"] not in generators): | |
| generators.append(cross_programs["generator"]) | |
| if cross_programs.get("reference"): | |
| references.append(cross_programs["reference"]) | |
| sample_failures = _sample_failure_report(candidates, samples, numeric) | |
| # Preserve challenge-generated cases even when the first programs fail samples: the same | |
| # task-independent cases must smoke-test the reviewer and rescue candidates after repair. | |
| # Executing a generator already present in the response adds no provider call. | |
| generated = [] | |
| generated_groups = [] | |
| scale_present = False | |
| per_generator = config["stress_cases"] | |
| for generator in generators[:2]: | |
| group = _generated_inputs(generator, per_generator, require_scale=True) | |
| if group: | |
| scale_present = True | |
| else: | |
| group = _generated_inputs(generator, per_generator, require_scale=False) | |
| group = [stdin for stdin in group if stdin not in generated] | |
| if group: | |
| generated_groups.append(group) | |
| generated.extend(group) | |
| if not scale_present: | |
| # Retain useful small differential cases even if proposed generators cap their scale, | |
| # then add one independently requested large legal case for an asymptotic smoke test. | |
| large_input_answer = invoke( | |
| call_model, config["generator"], original + "\n\n" + _LARGE_INPUT_ONLY, | |
| config["generator_max_tokens"], config["generator_reasoning"], | |
| ) | |
| large_case = _large_input(large_input_answer) | |
| if large_case and large_case not in generated: | |
| generated.append(large_case) | |
| oracle_cases = [] | |
| oracle_groups = [] | |
| for left_index in range(len(references)): | |
| for right_index in range(left_index + 1, len(references)): | |
| groups = [] | |
| for generated_group in generated_groups: | |
| agreed = _trusted_oracle( | |
| references[left_index], references[right_index], samples, | |
| generated_group, numeric, config["min_consensus_cases"], | |
| ) | |
| if agreed: | |
| groups.append(agreed) | |
| if groups: | |
| oracle_groups = groups | |
| oracle_cases = [] | |
| for group in groups: | |
| for case in group: | |
| if case not in oracle_cases: | |
| oracle_cases.append(case) | |
| break | |
| if oracle_cases: | |
| break | |
| execution_cache = {} | |
| grace_attempted = set() | |
| rows, outputs = _candidate_evidence( | |
| candidates, samples, generated, numeric, oracle_cases, config["critic"], | |
| oracle_groups, execution_cache, | |
| ) | |
| rows, outputs = _retry_oracle_timeouts( | |
| candidates, rows, outputs, samples, generated, numeric, oracle_cases, | |
| config["critic"], oracle_groups, execution_cache, | |
| config["min_consensus_cases"], grace_attempted, | |
| ) | |
| best, best_row = _best_candidate(candidates, rows) | |
| disagreement = _disagreement_note(generated, outputs) | |
| execution_failures = _execution_failure_report(candidates, generated, outputs) | |
| if _supported_candidate( | |
| best_row, candidates, generated, config["min_consensus_cases"], | |
| oracle_cases, disagreement): | |
| return best | |
| candidate_text = "\n\n".join( | |
| "Candidate " + str(index + 1) + ":\n" + code | |
| for index, code in enumerate(candidates) | |
| ) | |
| review_request = ( | |
| request + "\n\nIndependent review protocol: " + _REVIEW | |
| + "\n\nStatement-sample acceptance protocol: " + _MUST_PASS | |
| + "\n\nPrograms under review:\n" + candidate_text | |
| ) | |
| if sample_failures: | |
| review_request += ( | |
| "\n\nTrusted local execution diagnostics follow. Treat runtime errors and output " | |
| "mismatches as hard counterexamples; do not claim a candidate passes without " | |
| "correcting them.\n" + sample_failures | |
| ) | |
| if disagreement: | |
| review_request += ( | |
| "\n\nThe programs disagreed on small generated inputs. These outputs are evidence " | |
| "of disagreement only; none is a trusted oracle.\n" + disagreement | |
| ) | |
| if execution_failures: | |
| review_request += ( | |
| "\n\nTrusted local performance diagnostics follow. A candidate that does not " | |
| "complete a legal smoke test is rejected even if it matches every sample. Replace " | |
| "the asymptotically slow state or convolution rather than special-casing the input.\n" | |
| + execution_failures | |
| ) | |
| if oracle_cases: | |
| counterexamples = [] | |
| for stdin, expected in oracle_cases: | |
| observed = _run_program(best, stdin, _CANDIDATE_TIMEOUT) | |
| if observed is None or not _tokens_match(observed, expected, numeric): | |
| counterexamples.append( | |
| "Input:\n" + stdin[:800] + "\nTrusted dual-reference output:\n" | |
| + expected[:800] + "\nSelected-candidate output:\n" | |
| + (observed if observed is not None else "<no output>")[:800] | |
| ) | |
| if len(counterexamples) >= 2: | |
| break | |
| if counterexamples: | |
| review_request += ( | |
| "\n\nTwo independent small references agreed on these counterexamples:\n" | |
| + "\n\n".join(counterexamples) | |
| ) | |
| third = _source(invoke( | |
| call_model, config["reviewer"], review_request, config["review_max_tokens"], | |
| config["review_reasoning"], | |
| )) | |
| if third.strip(): | |
| candidates.append(third) | |
| rows, outputs = _candidate_evidence( | |
| candidates, samples, generated, numeric, oracle_cases, config["critic"], | |
| oracle_groups, execution_cache, | |
| ) | |
| rows, outputs = _retry_oracle_timeouts( | |
| candidates, rows, outputs, samples, generated, numeric, oracle_cases, | |
| config["critic"], oracle_groups, execution_cache, | |
| config["min_consensus_cases"], grace_attempted, | |
| ) | |
| best, best_row = _best_candidate(candidates, rows) | |
| disagreement = _disagreement_note(generated, outputs) | |
| if _supported_candidate( | |
| best_row, candidates, generated, config["min_consensus_cases"], | |
| oracle_cases, disagreement): | |
| return best | |
| # A slow emergency model cannot improve a candidate that already satisfies every | |
| # statement sample without concrete contradictory execution evidence. Preserve the best | |
| # valid candidate and reserve rescue for syntax/sample failure only. | |
| if (best_row["sample_valid"] | |
| and (not oracle_cases or best_row["oracle"] == best_row["oracle_total"]) | |
| and (not generated or best_row["completed"] == len(generated))): | |
| return best | |
| passed, total, failure = _check(best, samples, numeric) | |
| all_failures = _sample_failure_report(candidates, samples, numeric) | |
| rescue_request = ( | |
| request + "\n\nEmergency independent derivation: " + _REPAIR + _FINAL_DERIVE | |
| + "\n\nStatement-sample acceptance protocol: " + _MUST_PASS | |
| + "\n\nThe deterministic selector found insufficient execution consensus. " | |
| "This is the only expensive re-derivation: do not vote between prior programs; derive " | |
| "and return one complete raw program that meets the largest stated constraints." | |
| + "\n\nBest prior candidate diagnostics:\n" | |
| + _failure_note(passed, total, failure) | |
| + "\n\nBest prior candidate:\n" + best | |
| ) | |
| if all_failures: | |
| rescue_request += ( | |
| "\n\nTrusted diagnostics for every rejected candidate:\n" + all_failures | |
| ) | |
| final_execution_failures = _execution_failure_report(candidates, generated, outputs) | |
| if final_execution_failures: | |
| rescue_request += ( | |
| "\n\nTrusted performance counterexamples for rejected candidates:\n" | |
| + final_execution_failures | |
| ) | |
| rescued = _source(invoke( | |
| call_model, config["rescue"], rescue_request, config["rescue_max_tokens"], | |
| config["rescue_reasoning"], | |
| )) | |
| if rescued.strip(): | |
| candidates.append(rescued) | |
| rows, final_outputs = _candidate_evidence( | |
| candidates, samples, generated, numeric, oracle_cases, config["critic"], | |
| oracle_groups, execution_cache, | |
| ) | |
| rows, final_outputs = _retry_oracle_timeouts( | |
| candidates, rows, final_outputs, samples, generated, numeric, oracle_cases, | |
| config["critic"], oracle_groups, execution_cache, | |
| config["min_consensus_cases"], grace_attempted, | |
| ) | |
| rescued_row = rows[-1] if rescued.strip() else None | |
| if (rescued_row is not None and ( | |
| _supported_candidate( | |
| rescued_row, candidates, generated, config["min_consensus_cases"], | |
| oracle_cases, "") | |
| or (not oracle_cases and rescued_row["sample_valid"] | |
| and (not generated | |
| or rescued_row["completed"] == len(generated))))): | |
| return rescued | |
| if (rescued_row is not None and oracle_cases | |
| and rescued_row["sample_valid"] | |
| and _oracle_supported(rescued_row, config["min_consensus_cases"]) | |
| and generated and rescued_row["completed"] != len(generated)): | |
| failed_input = next( | |
| (stdin for stdin, output in zip(generated, final_outputs[-1]) | |
| if output is None), "" | |
| ) | |
| optimize_request = ( | |
| request + "\n\nPerformance repair protocol: " + _OPTIMIZE | |
| + "One triggering input follows:\n" + failed_input[:1200] | |
| + "\n\n" + _OPTIMIZE_FORMAT + rescued | |
| ) | |
| optimized = _source(invoke( | |
| call_model, config["reviewer"], optimize_request, | |
| config["review_max_tokens"], config["review_reasoning"], | |
| )) | |
| if optimized.strip(): | |
| candidates.append(optimized) | |
| rows, _final_outputs = _candidate_evidence( | |
| candidates, samples, generated, numeric, oracle_cases, config["critic"], | |
| oracle_groups, execution_cache, | |
| ) | |
| optimized_row = rows[-1] | |
| if _supported_candidate( | |
| optimized_row, candidates, generated, | |
| config["min_consensus_cases"], oracle_cases, ""): | |
| return optimized | |
| # The production judge permits ten seconds. The normal six-second gate keeps a | |
| # safety margin; after a semantically verified optimizer, spend that margin once on | |
| # the exact missing smoke cases instead of issuing a second expensive derivation. | |
| if (optimized_row["sample_valid"] and oracle_cases | |
| and _oracle_supported(optimized_row, config["min_consensus_cases"])): | |
| missing = [ | |
| stdin for stdin, output in zip(generated, _final_outputs[-1]) | |
| if output is None | |
| ] | |
| for stdin in missing: | |
| execution_cache[(optimized, stdin)] = _run_program( | |
| optimized, stdin, _CANDIDATE_GRACE_TIMEOUT) | |
| rows, _final_outputs = _candidate_evidence( | |
| candidates, samples, generated, numeric, oracle_cases, config["critic"], | |
| oracle_groups, execution_cache, | |
| ) | |
| if _supported_candidate( | |
| rows[-1], candidates, generated, config["min_consensus_cases"], | |
| oracle_cases, ""): | |
| return optimized | |
| # Never stack a second high-effort derivation after rescue. A single task must leave enough | |
| # of the shared epoch budget for the other five tasks and proof emission. | |
| return _best_candidate(candidates, rows)[0] | |
| return agent | |