"""UID 101 Aegis clean v5.32 task-independent consensus candidate. Provider-diverse candidates are selected by literal statement samples, bounded execution, generated counterexamples, and runtime evidence. The policy has no task IDs, prompt fingerprints, lookup tables, stored solutions, or benchmark-trained answer selector. """ import ast import json import math import re import queue import threading import subprocess import sys import time _MODELS = ( "qwen/qwen3.7-flash", "deepseek/deepseek-v4-flash", "deepseek/deepseek-v4-pro", "z-ai/glm-5.2", "openai/gpt-5.6-luna", "google/gemini-3.6-flash", "moonshotai/kimi-k3", ) _KIND = "valor-uid101-aegis-clean-v5.32" _SAMPLE_MARK = re.compile(r"^Sample (Input|Output) (\d+)\s*$", re.M) _CODE_WORD = re.compile(r"\b(?:input|print|sys|def|import|from)\b") _NUMBER = re.compile(r"[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?\Z") _FLOAT_JUDGE = re.compile( r"(?:absolute\s+or\s+relative|relative\s+or\s+absolute)\s+(?:error|difference)|" r"(?:absolute|relative)\s+error|error[^\n]{0,80}(?:10\^|1e-)", re.I ) _AMBIGUOUS_JUDGE = re.compile( r"(?:print|output|return)\s+any\b|any\s+(?:valid\s+)?(?:answer|solution)\b|" r"multiple\s+(?:answers|solutions)[^\n]{0,100}(?:accepted|print any)", re.I ) _SEQUENTIAL_RISK = re.compile( r"(?:in|chronological)\s+order[\s\S]{0,300}(?:replace|overwrite|update)|" r"(?:replace|overwrite|update)[\s\S]{0,300}(?:in|chronological)\s+order|" r"for\s+each[\s\S]{0,180}(?:in\s+this\s+order|in\s+order)[\s\S]{0,250}" r"(?:operations?|append|delete|replace|update)", re.I, ) _MUTABLE_GRAPH_RISK = re.compile( r"(?:becomes?|turns?).{0,80}(?:into|from).{0,80}(?:road|open|available)|" r"(?:destroy|remove|unlock|open).{0,100}(?:wall|edge|cell|node)", re.I | re.S ) _LARGE_N = re.compile( r"\bN\s*(?:\\leq|<=|≤)\s*(?:[1-9]\s*(?:\\times|[x×*])\s*)?" r"10\^(?:[5-9]|[1-9][0-9]+)|" r"\bN\s*(?:\\leq|<=|≤)\s*[1-9][0-9]{5,}", re.I ) _MANY_OUTPUT_RISK = re.compile( r"(?:for\s+each|for\s+every).{0,100}(?:integer|value|number)|" r"(?:all|many).{0,80}(?:coefficients?|answers?|values?)", re.I | re.S ) _MODULAR_RISK = re.compile( r"(?:prime|modulo|modulus).{0,100}\bP\b|\bP\b.{0,100}(?:prime|modulo|modulus)", re.I | re.S, ) _COUNTING_RISK = re.compile( r"(?:number\s+of|count).{0,120}(?:graphs?|ways?|configurations?|sequences?|subsets?)", re.I | re.S, ) _NAMED_BLOCK = re.compile(r"```(candidate|generator|reference|benchmark)\s*\n(.*?)```", re.I | re.S) _FLOAT_TOL = 1e-9 _CASE_TIMEOUT = 3.0 _GEN_TIMEOUT = 1.0 _STRESS_TIMEOUT = 3.0 _PERFORMANCE_MARGIN_TIMEOUT = 8.0 _STRESS_SECONDS = 8.0 _MAX_CODE_CALLS = 4 _EVIDENCE_BUDGET_S = 54.0 _ORACLE_BUDGET_S = 12.0 _CANDIDATE_EVIDENCE_BUDGET_S = 18.0 _STRESS_SIZES = (1, 2, 3, 5, 8, 12, 24, 48) _RUN_BUDGET_S = 640.0 _EXPECTED_TASKS = 6 _EXPECTED_CODE_TASKS = 2 _RESERVE_PER_TASK_S = 20.0 _RESERVE_PER_CODE_TASK_S = 280.0 _MIN_CALL_WINDOW_S = 12.0 _DEFAULT_CALL_ESTIMATE_S = 20.0 _MODEL_CALL_FLOOR_S = (60.0, 90.0, 180.0, 90.0, 150.0, 45.0, 90.0) _HARD_CALL_TIMEOUT_S = 300.0 _RESOURCE_AUDIT = ( "Estimate maximum states, leaves, and retained objects before coding. A time-feasible " "enumeration can still exceed memory if every leaf is stored. When only distinct " "aggregates are required, update the final set or counter online. The largest legal input " "must retain a substantial margin below the execution limit, not merely finish at its edge." ) _INDEX_BOUND_AUDIT = ( "For every list or array, derive its allocated length and the maximum reachable subscript " "over every legal input. Reject code when an index can grow faster than the allocation " "expression or can otherwise exceed it; a sample-sized run is not boundary evidence." ) _PERFORMANCE_AUDIT = ( "Construct a largest-constraint legal input from the statement and estimate the exact dominant " "operation count. Reject boundary-time code. Prefer a candidate that preserves correctness " "evidence while completing every executable scale probe with the largest timing margin." ) _RECURSION_AUDIT = ( "Treat self-recursion inside an input-dependent loop as exponential until a strict maximum " "leaf count proves wide margin. Another recursive branch increases that bound; sample-sized " "completion is not largest-input performance evidence." ) _BATCH_OUTPUT_AUDIT = ( "When a counting problem requests all answers or coefficients over a range, reject rerunning " "the full combinatorial dynamic program independently at every interpolation point unless the " "proved total operation count has a wide margin. Prefer propagating coefficient vectors or " "batched transforms once through the state graph, while proving the coefficient conversion." ) _PERFORMANCE_REPAIR = ( "The program below has the strongest independent correctness evidence but missed a strict " "largest-input timing gate. Preserve its verified semantics and exact I/O while reducing the " "largest legal input below 4 seconds on one CPU. Re-derive the algorithm if " "micro-optimization cannot provide that margin. Return only complete raw Python 3 source." ) _SOLVE = ( "Derive the algorithm only from the statement and constraints. Check indexing, duplicates, " "boundaries, precision, and complexity. Never mutate outer-loop operands during pair iterations; " "use local aliases with attached metadata. Before coding a greedy or compressed-state solution, " "differential-check its invariant against exhaustive tiny cases." ) _INTEGER_SAFETY = ( "Python integers are arbitrary precision. Do not narrow values, partial sums, XOR states, or " "hash keys into a fixed-width array or mask unless every stated bound proves that no legal " "value can be truncated or collide. Prefer built-in set/dict for exact deduplication; " "use Python-level open addressing only with proved bounds and a faster executable benchmark." ) _PRIMARY_FORMAT = ( "Return two fenced blocks and no prose. First is the complete Python 3 solution; its first " "source line must be the literal comment # input so the parser recognizes sys.stdin programs. " "Reference is a literal exhaustive solver for small legal inputs and must not reuse the " "candidate invariant.\n" "```python\n\n```\n```reference\n\n```" ) _SEQUENTIAL_A = ( "For mandatory sequential overwrite operations, every incoming value is consumed exactly once, " "but an early incoming value may be deliberately erased by a later write to the same target. " "Separate values that survive in the final state from operations sacrificed by later overwrites. " ) _SEQUENTIAL_B = ( "Do not treat all incoming values as freely placeable or refundable. Prove any greedy choice by an " "exchange argument, preserve chronological last-write semantics, and exhaustively enumerate all " "target sequences for tiny state and operation counts before returning code." ) _MUTABLE_GRAPH = ( "During shortest-path search, never mutate a graph, grid, or availability table shared by all " "queued paths. A change discovered at higher cost must not leak into lower-cost states. Attach " "persistent changes to the search state, or prove an equivalent weighted transition on an " "immutable graph before using Dijkstra or 0-1 BFS." ) _REPAIR = ( "Re-derive independently; use the failure as evidence of a general defect, never a special " "case. Test greedy, compressed-state, and sequential-update invariants against a literal " "exhaustive solver on the smallest legal instances. Trace every statement sample through input " "parsing, state changes, and output length. Return only complete raw Python 3 source." ) _REVIEW = ( "Solve independently, then audit both candidates for logic, boundaries, complexity, and format. " "Test greedy or compressed-state invariants against a literal exhaustive solver on the smallest " "legal instances, including duplicates, ordering, and overwrites. Trace every statement sample " "through parsing and output length. Return one corrected raw Python 3 program without prose." ) _REVIEW = (_REVIEW + " " + _PERFORMANCE_AUDIT + " " + _BATCH_OUTPUT_AUDIT + " " + _RECURSION_AUDIT) _MUST_PASS = ( "Every statement sample is an executable acceptance condition. Re-derive the algorithm rather " "than fitting sample outputs, mentally execute the complete program on every stated sample, and " "return code only after all samples agree. If prior candidates share an invariant, actively seek " "a counterexample to that invariant before reusing it." ) _EQUIVALENCE_AUDIT = ( "Audit every claimed equivalence and every necessary-or-sufficient reduction in both " "directions. Construct the smallest legal counterexample to each unproved direction before " "accepting a recurrence, compressed state, interpolation formula, or graph characterization. " ) _INDEPENDENCE_AUDIT = ( "Do not reuse a shared invariant merely because several programs agree. Derive the invariant " "from the statement, compare it with a literal small-instance enumerator when tractable, and " "separately justify completeness, uniqueness, and the largest-input complexity bound." ) _QUANTIFIER_AUDIT = ( "Preserve the order and scope of every universal or existential condition. Before replacing a " "quantified condition by a familiar named property, prove both implications and enumerate the " "smallest legal structures where the proposed property may hold but the original condition may fail." ) _CHALLENGE_A = ( "Solve independently without copying the prior candidate. Return exactly three fenced blocks and " "no prose. The candidate block must contain a complete Python 3 solution. " ) _CHALLENGE_B = ( "The generator block must contain a Python program accepting integer seed and size command-line " "arguments, calling random.seed(seed), and printing one small legal input derived only from the " "statement. Vary ties, duplicates, boundaries, operation order, and branch combinations.\n" ) _CHALLENGE_FORMAT = ( "Reference is an independent literal small-input solver. Benchmark accepts seed and size and " "prints one largest legal input, never an answer. Label the first fence python and start it with " "the literal comment # input so parser and returned candidate match.\n" "```python\n\n```\n```generator\n\n```" "\n```reference\n\n```\n```benchmark\n\n```" ) _CHALLENGE = _CHALLENGE_A + _CHALLENGE_B + _CHALLENGE_FORMAT _GENERATOR_ONLY = _CHALLENGE _ARBITER_ONLY = ( "Derive a third small-input oracle independently from the statement, without reading or " "reusing any optimized candidate invariant. Return exactly one fenced reference block and no " "prose. The program must be a literal exhaustive solver for small legal inputs. It must cover " "operation order, ties, duplicates, and boundary states whenever those concepts occur in the " "statement.\n```reference\n\n```" ) _FLOAT_FORMAT = "".join(( "The validator compares output tokens literally even when the statement describes a numeric " "tolerance. Match the fixed-decimal style and number of places shown consistently by all " "statement samples. ", "If samples show more than 15 significant decimal digits, do not use binary float for the final " "computation: use decimal.Decimal with sufficient guard precision, including Decimal.sqrt or an " "equivalent high-precision derivation. ", "Never paste sample values or special-case their inputs; the same derivation and formatting must " "cover every legal input.", )) _EXPLICIT_DECIMALS = re.compile( r"(?:exactly\s+)?\d+\s+(?:digits?\s+after\s+the\s+decimal|decimal\s+places?)", re.I, ) _FIXED_FORMAT = re.compile(r"\.(\d+)f") def _sample_decimal_places(statement): places = set() for _stdin, expected in _samples(statement, 4): for token in str(expected).split(): if _NUMBER.fullmatch(token) and "." in token.lower() and "e" not in token.lower(): places.add(len(token.rsplit(".", 1)[1])) return next(iter(places)) if len(places) == 1 and 1 <= next(iter(places)) <= 30 else None class _DecimalHypotTransformer(ast.NodeTransformer): """Promote generic Euclidean primitives when public output requires high precision.""" def __init__(self): self.changed = False def visit_Call(self, node): node = self.generic_visit(node) if (isinstance(node.func, ast.Attribute) and isinstance(node.func.value, ast.Name) and node.func.value.id == "math" and node.func.attr == "hypot" and len(node.args) == 2 and not node.keywords): left, right = node.args def square(value): converted = ast.Call( func=ast.Name(id="_RouterDecimal", ctx=ast.Load()), args=[value], keywords=[], ) return ast.BinOp(left=converted, op=ast.Mult(), right=converted) self.changed = True return ast.copy_location(ast.Call( func=ast.Attribute( value=ast.BinOp(left=square(left), op=ast.Add(), right=square(right)), attr="sqrt", ctx=ast.Load(), ), args=[], keywords=[], ), node) return node class _DecimalFsumTransformer(ast.NodeTransformer): def visit_Call(self, node): node = self.generic_visit(node) if (isinstance(node.func, ast.Attribute) and isinstance(node.func.value, ast.Name) and node.func.value.id == "math" and node.func.attr == "fsum" and len(node.args) == 1 and not node.keywords): return ast.copy_location(ast.Call( func=ast.Name(id="sum", ctx=ast.Load()), args=[node.args[0], ast.Call( func=ast.Name(id="_RouterDecimal", ctx=ast.Load()), args=[ast.Constant(value=0)], keywords=[], )], keywords=[], ), node) return node def _promote_decimal_hypot(value, sample_places): if sample_places <= 15 or "Decimal" in value or "decimal" in value: return value try: tree = ast.parse(value) transformer = _DecimalHypotTransformer() tree = transformer.visit(tree) if not transformer.changed: return value tree = _DecimalFsumTransformer().visit(tree) insertion = 1 if (tree.body and isinstance(tree.body[0], ast.Expr) and isinstance(tree.body[0].value, ast.Constant) and isinstance(tree.body[0].value.value, str)) else 0 while (insertion < len(tree.body) and isinstance(tree.body[insertion], ast.ImportFrom) and tree.body[insertion].module == "__future__"): insertion += 1 setup = ast.parse( "from decimal import Decimal as _RouterDecimal, getcontext as _router_decimal_context\n" + "_router_decimal_context().prec = " + str(sample_places + 20) + "\n" ).body tree.body[insertion:insertion] = setup ast.fix_missing_locations(tree) return ast.unparse(tree) + "\n" except (SyntaxError, ValueError, TypeError): return value def _normalize_float_program(code, prompt): """Apply the public sample's literal float style without task identity or stored outputs.""" value = str(code or "") statement = str(prompt or "") if not _FLOAT_JUDGE.search(statement) or _EXPLICIT_DECIMALS.search(statement): return value try: ast.parse(value) except (SyntaxError, ValueError, TypeError): return value sample_places = _sample_decimal_places(statement) if sample_places is None: return value updated = _FIXED_FORMAT.sub(lambda _match: "." + str(sample_places) + "f", value) updated = _promote_decimal_hypot(updated, sample_places) try: ast.parse(updated) except (SyntaxError, ValueError, TypeError): return value return updated def _finish_candidate(code, prompt): """Reject prose, truncation, and malformed programs before any expensive evidence stage.""" value = _normalize_float_program(code, prompt) if not str(value).strip(): return "" try: ast.parse(value) except (SyntaxError, ValueError, TypeError): return "" return value def _load(weights): try: data = json.loads(bytes(weights).decode("utf-8")) except Exception as exc: raise ValueError("UID 101 clean settings are not valid JSON") from exc required = { "schema", "kind", "non_code", "primary", "challenger", "generator", "reviewer", "rescue", "batch_primary", "batch_challenger", "primary_max_tokens", "challenge_max_tokens", "review_max_tokens", "batch_primary_max_tokens", "batch_challenge_max_tokens", "generator_max_tokens", "rescue_max_tokens", "sample_limit", "stress_cases", "min_consensus_cases", "critic", "temperature", "primary_reasoning", "challenge_reasoning", "review_reasoning", "batch_primary_reasoning", "batch_challenge_reasoning", "generator_reasoning", "rescue_reasoning", } if not isinstance(data, dict) or set(data) != required: raise ValueError("UID 101 clean settings have an invalid schema") if data["schema"] != 3 or data["kind"] != _KIND: raise ValueError("UID 101 clean settings do not match this source") for field in ("non_code", "primary", "challenger", "generator", "reviewer", "rescue", "batch_primary", "batch_challenger"): if type(data[field]) is not int or not 0 <= data[field] < len(_MODELS): raise ValueError("UID 101 clean model index is invalid") if (type(data["primary_max_tokens"]) is not int or not 4096 <= data["primary_max_tokens"] <= 32768): raise ValueError("UID 101 clean primary token limit is invalid") for field in ("challenge_max_tokens", "generator_max_tokens", "review_max_tokens", "rescue_max_tokens", "batch_primary_max_tokens", "batch_challenge_max_tokens"): if type(data[field]) is not int or not 2048 <= data[field] <= 32768: raise ValueError("UID 101 clean fallback token limit is invalid") if type(data["sample_limit"]) is not int or not 1 <= data["sample_limit"] <= 4: raise ValueError("UID 101 clean sample limit is invalid") if type(data["stress_cases"]) is not int or not 4 <= data["stress_cases"] <= 24: raise ValueError("UID 101 clean stress case limit is invalid") if (type(data["min_consensus_cases"]) is not int or not 2 <= data["min_consensus_cases"] <= data["stress_cases"]): raise ValueError("UID 101 clean consensus threshold is invalid") critic = data["critic"] if (not isinstance(critic, dict) or set(critic) != {"schema", "mean", "scale", "coef", "intercept"} or critic["schema"] != 1): raise ValueError("UID 101 critic schema is invalid") for field in ("mean", "scale", "coef"): if (not isinstance(critic[field], list) or len(critic[field]) != 18 or any(type(value) not in (int, float) or not math.isfinite(value) for value in critic[field])): raise ValueError("UID 101 critic vector is invalid") if (any(value <= 0 for value in critic["scale"]) or type(critic["intercept"]) not in (int, float) or not math.isfinite(critic["intercept"])): raise ValueError("UID 101 critic normalization is invalid") if type(data["temperature"]) not in (int, float) or data["temperature"] != 0: raise ValueError("UID 101 clean temperature must be deterministic") for field in ("primary_reasoning", "challenge_reasoning", "generator_reasoning", "review_reasoning", "rescue_reasoning", "batch_primary_reasoning", "batch_challenge_reasoning"): if data[field] not in ("none", "low", "medium", "high"): raise ValueError("UID 101 clean reasoning effort is invalid") return data def _is_code(prompt): text = str(prompt) return ("Write a complete Python 3 program" in text and "standard input" in text and "standard output" in text) def _is_mcq(prompt): text = "\n" + str(prompt) return all("\n" + option in text for option in ("A)", "B)", "C)", "D)")) def _non_code_request(prompt): text = str(prompt) if _is_mcq(text): return text return text + "\n\nDerive the answer strictly from the question and return only the answer." def _samples(prompt, limit): text = str(prompt).replace("\r\n", "\n").replace("\r", "\n") if _AMBIGUOUS_JUDGE.search(text): return [] marks = [(m.start(), m.end(), m.group(1), int(m.group(2))) for m in _SAMPLE_MARK.finditer(text)] blocks = {} for index, (_start, end, kind, number) in enumerate(marks): stop = marks[index + 1][0] if index + 1 < len(marks) else len(text) lines = text[end:stop].split("\n") while lines and not lines[0].strip(): lines.pop(0) kept = [] for line in lines: if not line.strip(): break kept.append(line) blocks[(kind, number)] = "\n".join(kept) pairs = [] for number in sorted({number for _kind, number in blocks}): stdin = blocks.get(("Input", number)) stdout = blocks.get(("Output", number)) if stdin and stdout: pairs.append((stdin, stdout)) return pairs[:limit] def _source(answer): """Match the validator's first-qualifying-block extraction exactly. Returning a later or longest fence is provenance-invalid even if that program also appeared in the response: the validator canonicalizes only the first block containing input/print. """ text = str(answer or "").strip() if not text: return "" if "```" in text: for block in (part for part in text.split("```") if part.strip()): block = block[len("python"):] if block.lstrip().lower().startswith("python") else block if "input" in block or "print" in block: return block.strip() + "\n" return text + "\n" def _returnable_candidate(answer, programs=None): """Return only the exact program token the validator attributes to this response.""" canonical = _source(answer) parsed = (programs or _named_programs(answer)).get("candidate") if parsed and " ".join(parsed.split()) == " ".join(canonical.split()): return parsed return canonical def _tokens_match(observed, expected, numeric): got = str(observed).split() want = str(expected).split() if got == want: return True if not numeric or len(got) != len(want) or not got: return False for left, right in zip(got, want): if _NUMBER.fullmatch(left) is None or _NUMBER.fullmatch(right) is None: return False try: a, b = float(left), float(right) except ValueError: return False if not math.isfinite(a) or not math.isfinite(b): return False if abs(a - b) > _FLOAT_TOL * max(1.0, abs(b)): return False return True def _run_program(code, stdin_text, timeout, argv=(), deadline=None): if deadline is not None: remaining = deadline - time.monotonic() if remaining <= 0: return None timeout = min(float(timeout), max(0.01, remaining)) try: run = subprocess.run( [sys.executable, "-I", "-c", code, *[str(value) for value in argv]], input=str(stdin_text), capture_output=True, text=True, timeout=timeout, ) except (subprocess.TimeoutExpired, OSError, ValueError): return None return run.stdout if run.returncode == 0 else None def _check(code, samples, numeric, deadline=None): if not code.strip(): return 0, max(1, len(samples)), ( samples[0][0] if samples else "", "", samples[0][1] if samples else "valid Python", ) try: compile(code, "", "exec") except (SyntaxError, ValueError, TypeError) as exc: return 0, max(1, len(samples)), ( samples[0][0] if samples else "", type(exc).__name__ + ": " + str(exc)[:600], samples[0][1] if samples else "valid Python", ) passed = 0 first_bad = None for stdin, expected in samples: observed = _run_program(code, stdin, _CASE_TIMEOUT, deadline=deadline) # The validator's LCB grader is literal-token exact, including numeric tasks. good = observed is not None and str(observed).split() == str(expected).split() if good: passed += 1 elif first_bad is None: first_bad = (stdin, observed if observed is not None else "", expected) return passed, len(samples), first_bad def _failure_note(passed, total, failure): if failure is None: return "sample verification: " + str(passed) + "/" + str(total) stdin, observed, expected = failure return ( "sample verification: " + str(passed) + "/" + str(total) + "\nFailing sample input:\n" + str(stdin)[:500] + "\nCandidate output:\n" + str(observed)[:500] + "\nExpected output:\n" + str(expected)[:500] ) def _named_programs(answer): programs = {} python_blocks = [] text = str(answer or "") # Recover explicitly labelled tooling blocks first. A provider may emit an unlabelled closing # fence after raw candidate code; generic left-to-right fence pairing can otherwise consume the # next labelled opener as a closer and hide valid tooling. for label in ("generator", "reference", "benchmark"): match = re.search(r"```" + label + r"\s*\n(.*?)```", text, re.I | re.S) if match: code = match.group(1).strip() if code: programs[label] = code + "\n" for match in re.finditer(r"```([^\n`]*)\n(.*?)```", text, re.I | re.S): # Some providers add one extra fence marker around an otherwise correctly labelled block. # Normalize only fence punctuation; arbitrary prose labels are never accepted as programs. tag = match.group(1).strip().lower().strip("`") lines = [line for line in match.group(2).strip().splitlines() if line.strip().lower() not in ("", "program")] code = "\n".join(lines).strip() if not code: continue if tag in ("python", "python3", "py"): python_blocks.append(code + "\n") if "candidate" not in programs: programs["candidate"] = code + "\n" elif tag in ("generator", "reference", "benchmark") and tag not in programs: programs[tag] = code + "\n" elif tag == "candidate" and "candidate" not in programs: # Backward-compatible parsing for stored development responses. New prompts never ask # for this tag because the validator would retain the word `candidate` in its token. programs["candidate"] = code + "\n" # Positional recovery is format-only and independent of task identity. It supports providers # that rewrite every requested fence tag to `python` while preserving fence order. if len(python_blocks) >= 2 and "generator" not in programs: programs["generator"] = python_blocks[1] if len(python_blocks) >= 3 and "reference" not in programs: programs["reference"] = python_blocks[2] return programs def _plain_program_blocks(answer): blocks = [] for raw in re.findall(r"```(?:python|python3|py)?\s*\n(.*?)```", str(answer or ""), re.I | re.S): code = raw.strip() if code: blocks.append(code + "\n") return blocks def _generated_inputs(generator, case_limit, require_scale=True, deadline=None): if not generator or not generator.strip(): return [] # A generator that ignores the supplied scale only exercises toy inputs and can hide # asymptotic failures. This is a task-independent interface requirement, not prompt dispatch. direct_size = re.search(r"(?:sys\.)?argv\s*\[\s*2\s*\]", generator) sliced_args = (re.search(r"(?:sys\.)?argv\s*\[\s*1\s*:\s*\]", generator) and re.search(r"\[\s*1\s*\]", generator)) if require_scale and direct_size is None and not sliced_args: return [] try: compile(generator, "", "exec") except (SyntaxError, ValueError, TypeError): return [] local_deadline = time.monotonic() + _STRESS_SECONDS deadline = min(local_deadline, deadline) if deadline is not None else local_deadline cases = [] observations = [] if case_limit <= 1: size_schedule = _STRESS_SIZES[:case_limit] elif case_limit < len(_STRESS_SIZES): last = len(_STRESS_SIZES) - 1 size_schedule = tuple( _STRESS_SIZES[round(index * last / (case_limit - 1))] for index in range(case_limit) ) else: size_schedule = tuple( _STRESS_SIZES[index % len(_STRESS_SIZES)] for index in range(case_limit) ) for index, size in enumerate(size_schedule): if time.monotonic() >= deadline: break stdin = _run_program(generator, "", _GEN_TIMEOUT, (index + 1, size), deadline) if not stdin or not stdin.strip() or len(stdin) > 20000: continue observations.append((size, stdin)) if stdin not in cases: cases.append(stdin) if require_scale: small = [value for size, value in observations if size <= 5] large = [value for size, value in observations if size >= 24] if not small or not large: return [] def shape(value): tokens = str(value).split() integers = [] for token in tokens[:32]: try: integers.append(abs(int(token))) except ValueError: integers.append(0) return len(str(value)), len(tokens), integers small_shapes = [shape(value) for value in small] large_shapes = [shape(value) for value in large] small_chars = max(row[0] for row in small_shapes) small_tokens = max(row[1] for row in small_shapes) grew = (max(row[0] for row in large_shapes) >= max(8, int(small_chars * 1.25)) or max(row[1] for row in large_shapes) >= max(3, int(small_tokens * 1.25))) width = max((len(row[2]) for row in small_shapes + large_shapes), default=0) for position in range(width): low = max((row[2][position] for row in small_shapes if position < len(row[2])), default=0) high = max((row[2][position] for row in large_shapes if position < len(row[2])), default=0) if high > low and high >= max(2, int(low * 1.25)): grew = True break if not grew: return [] return cases def _deep_code_risk(text): """Recognize broad derivation risk without prompt identity or stored-solution routing.""" value = str(text) signals = sum(bool(pattern.search(value)) for pattern in ( _MANY_OUTPUT_RISK, _MODULAR_RISK, _COUNTING_RISK, )) return signals >= 2 def _fixed_width_integer_risk(code): """Flag lossy Python integer storage; this is a language safety check, not task dispatch.""" try: tree = ast.parse(code) except (SyntaxError, ValueError, TypeError): return False has_fixed_array = False has_integer_narrowing = False for node in ast.walk(tree): if (isinstance(node, ast.Call) and isinstance(node.func, ast.Name) and node.func.id == "array" and node.args and isinstance(node.args[0], ast.Constant) and str(node.args[0].value) in {"b", "B", "h", "H", "i", "I", "l", "L", "q", "Q"}): has_fixed_array = True if isinstance(node, ast.BinOp) and isinstance(node.op, ast.BitAnd): for operand in (node.left, node.right): if (isinstance(operand, ast.Constant) and type(operand.value) is int and operand.value >= (1 << 31) - 1 and (operand.value + 1) & operand.value == 0): has_integer_narrowing = True return has_fixed_array or has_integer_narrowing def _expression_degree(node, degrees=None): """Conservative polynomial degree used only for generic allocation/index diagnostics.""" degrees = degrees or {} if node is None or isinstance(node, ast.Constant): return 0 if isinstance(node, ast.Name): return degrees.get(node.id, 1) if isinstance(node, ast.UnaryOp): return _expression_degree(node.operand, degrees) if isinstance(node, ast.BinOp): left = _expression_degree(node.left, degrees) right = _expression_degree(node.right, degrees) if isinstance(node.op, (ast.Add, ast.Sub, ast.Mod)): return max(left, right) if isinstance(node.op, ast.Mult): return left + right if isinstance(node.op, (ast.Div, ast.FloorDiv)): return left if (isinstance(node.op, ast.Pow) and isinstance(node.right, ast.Constant) and type(node.right.value) is int and 0 <= node.right.value <= 8): return left * node.right.value return max(left, right) if isinstance(node, ast.Call): return max((_expression_degree(arg, degrees) for arg in node.args), default=1) return max((_expression_degree(child, degrees) for child in ast.iter_child_nodes(node)), default=0) def _repeated_list_size(node): if not isinstance(node, ast.BinOp) or not isinstance(node.op, ast.Mult): return None if isinstance(node.left, (ast.List, ast.Tuple)): return node.right if isinstance(node.right, (ast.List, ast.Tuple)): return node.left return None def _allocation_extent(node): repeated = _repeated_list_size(node) if repeated is not None: return repeated if (isinstance(node, ast.ListComp) and len(node.generators) == 1 and isinstance(node.generators[0].iter, ast.Call) and isinstance(node.generators[0].iter.func, ast.Name) and node.generators[0].iter.func.id == "range"): args = node.generators[0].iter.args if len(args) == 1: return args[0] if len(args) >= 2: return args[1] return None def _allocation_index_risk(code): """Flag an index whose polynomial growth outruns its list allocation growth.""" try: tree = ast.parse(code) except (SyntaxError, ValueError, TypeError): return False allocations = {} degrees = {} for node in ast.walk(tree): if isinstance(node, (ast.Assign, ast.AnnAssign)): value = node.value targets = node.targets if isinstance(node, ast.Assign) else [node.target] size = _allocation_extent(value) for target in targets: if isinstance(target, ast.Name): if size is None: degrees[target.id] = _expression_degree(value, degrees) else: allocations[target.id] = _expression_degree(size, degrees) for node in ast.walk(tree): if not isinstance(node, ast.Subscript) or not isinstance(node.value, ast.Name): continue if node.value.id not in allocations: continue index = node.slice.value if isinstance(node.slice, ast.Index) else node.slice if _expression_degree(index, degrees) > allocations[node.value.id]: return True return False def _quality(code): """Return a small task-independent static safety score; never infer a task identity.""" try: tree = ast.parse(code) except (SyntaxError, ValueError, TypeError): return -100 score = 0 nodes = list(ast.walk(tree)) score += int(any(isinstance(node, (ast.For, ast.While)) for node in nodes)) score += int(any(isinstance(node, ast.FunctionDef) for node in nodes)) score += int(40 <= len(code) <= 12000) dangerous = {"eval", "exec", "compile", "__import__"} score -= 4 * sum( isinstance(node, ast.Call) and isinstance(node.func, ast.Name) and node.func.id in dangerous for node in nodes ) score -= int(len(code) > 20000) return score def _has_input_scaled_nested_loop(code): """Reject an obvious quadratic loop only when the statement also declares large N.""" try: tree = ast.parse(code) except (SyntaxError, ValueError, TypeError): return False loop_types = (ast.For, ast.AsyncFor, ast.While) for outer in (node for node in ast.walk(tree) if isinstance(node, loop_types)): for child in ast.walk(outer): if child is outer or not isinstance(child, loop_types): continue expression = child.iter if isinstance(child, (ast.For, ast.AsyncFor)) else child.test names = {node.id for node in ast.walk(expression) if isinstance(node, ast.Name)} if names: return True return False def _ast_depth(node): children = list(ast.iter_child_nodes(node)) return 1 + max((_ast_depth(child) for child in children), default=0) def _critic_features(code, sample_ratio, completion_ratio, consensus_ratio): try: tree = ast.parse(code) except (SyntaxError, ValueError, TypeError): return [0.0] * 15 + [float(sample_ratio), float(completion_ratio), float(consensus_ratio)] nodes = list(ast.walk(tree)) dangerous = {"eval", "exec", "compile", "__import__"} return [ math.log1p(len(code)), math.log1p(code.count("\n") + 1), math.log1p(len(nodes)), float(sum(isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)) for node in nodes)), float(sum(isinstance(node, (ast.For, ast.AsyncFor, ast.While)) for node in nodes)), float(sum(isinstance(node, (ast.If, ast.IfExp, ast.Match)) for node in nodes)), float(sum(isinstance(node, ast.Call) for node in nodes)), float(sum(isinstance(node, ast.Subscript) for node in nodes)), float(sum(isinstance(node, (ast.ListComp, ast.SetComp, ast.DictComp, ast.GeneratorExp)) for node in nodes)), float(sum(isinstance(node, (ast.Try, ast.Raise)) for node in nodes)), float(sum(isinstance(node, (ast.Import, ast.ImportFrom)) for node in nodes)), float(sum(isinstance(node, ast.Return) for node in nodes)), float(sum(isinstance(node, ast.Constant) and type(node.value) is int for node in nodes)), float(_ast_depth(tree)), float(sum(isinstance(node, ast.Call) and isinstance(node.func, ast.Name) and node.func.id in dangerous for node in nodes)), float(sample_ratio), float(completion_ratio), float(consensus_ratio), ] def _critic_score(code, sample_ratio, completion_ratio, consensus_ratio, critic): if not critic: return 0.0 values = _critic_features(code, sample_ratio, completion_ratio, consensus_ratio) logit = float(critic["intercept"]) for value, mean, scale, coefficient in zip( values, critic["mean"], critic["scale"], critic["coef"]): logit += ((value - mean) / scale) * coefficient return 1.0 / (1.0 + math.exp(-max(-30.0, min(30.0, logit)))) def _reference_valid(reference, samples, numeric, deadline=None): if not reference or not reference.strip(): return False try: compile(reference, "", "exec") except (SyntaxError, ValueError, TypeError): return False if not samples: return True completed = 0 for stdin, expected in samples: observed = _run_program(reference, stdin, _STRESS_TIMEOUT, deadline=deadline) if observed is None: continue completed += 1 if not _tokens_match(observed, expected, numeric): return False return completed > 0 def _trusted_oracle(reference_a, reference_b, samples, generated, numeric, minimum, deadline=None): if (not _reference_valid(reference_a, samples, numeric, deadline) or not _reference_valid(reference_b, samples, numeric, deadline)): return [] agreed = [] for stdin in generated: left = _run_program(reference_a, stdin, _STRESS_TIMEOUT, deadline=deadline) right = _run_program(reference_b, stdin, _STRESS_TIMEOUT, deadline=deadline) if left is None or right is None: continue if not _tokens_match(left, right, numeric): return [] agreed.append((stdin, left)) return agreed if len(agreed) >= minimum else [] def _first_trusted_oracle(references, samples, generated, numeric, minimum, deadline=None): """Admit an oracle only when two independently returned references fully agree.""" local_deadline = time.monotonic() + _ORACLE_BUDGET_S deadline = min(local_deadline, deadline) if deadline is not None else local_deadline for left_index in range(len(references)): for right_index in range(left_index + 1, len(references)): agreed = _trusted_oracle( references[left_index], references[right_index], samples, generated, numeric, minimum, deadline, ) if agreed: return agreed return [] def _candidate_evidence(candidates, samples, generated, numeric, oracle_cases=(), critic=None, large_input=False, performance_inputs=(), deadline=None): rows = [] outputs = [] for index, code in enumerate(candidates): local_deadline = time.monotonic() + _CANDIDATE_EVIDENCE_BUDGET_S candidate_deadline = (min(local_deadline, deadline) if deadline is not None else local_deadline) passed, total, failure = _check(code, samples, numeric, candidate_deadline) sample_valid = bool(code.strip() and failure is None and passed == total) if sample_valid: values = [_run_program(code, stdin, _STRESS_TIMEOUT, deadline=candidate_deadline) for stdin in generated] performance_started = time.monotonic() performance_values = [ _run_program(code, stdin, _PERFORMANCE_MARGIN_TIMEOUT, deadline=candidate_deadline) for stdin in performance_inputs ] performance_latency = time.monotonic() - performance_started else: # A program that fails the statement samples can never be returned. Do not spend its # reviewer reserve on generated or scale probes. values = [None] * len(generated) performance_values = [None] * len(performance_inputs) performance_latency = 0.0 outputs.append(values) rows.append({ "index": index, "sample_passed": passed, "sample_total": total, "sample_valid": sample_valid, "completed": sum(value is not None for value in values), "consensus": 0, "oracle": 0, "oracle_total": len(oracle_cases), "feasible": not (large_input and _has_input_scaled_nested_loop(code)), "integer_safe": not _fixed_width_integer_risk(code), "boundary_safe": not _allocation_index_risk(code), "performance_completed": sum(value is not None for value in performance_values), "performance_total": len(performance_inputs), "performance_latency_s": performance_latency, "critic": 0.0, "quality": _quality(code), }) for case_index in range(len(generated)): groups = [] for candidate_index, values in enumerate(outputs): value = values[case_index] if value is None: continue placed = False for representative, members in groups: if _tokens_match(value, representative, numeric): members.append(candidate_index) placed = True break if not placed: groups.append((value, [candidate_index])) if not groups: continue largest = max(len(members) for _value, members in groups) if largest < 2: continue winners = [members for _value, members in groups if len(members) == largest] if len(winners) == 1: for candidate_index in winners[0]: rows[candidate_index]["consensus"] += 1 for row, code in zip(rows, candidates): for stdin, expected in oracle_cases: observed = _run_program(code, stdin, _STRESS_TIMEOUT, deadline=deadline) if observed is not None and _tokens_match(observed, expected, numeric): row["oracle"] += 1 row["critic"] = _critic_score( code, row["sample_passed"] / max(1, row["sample_total"]), row["completed"] / max(1, len(generated)), row["consensus"] / max(1, len(generated)), critic, ) return rows, outputs def _best_candidate(candidates, rows): viable = [row for row in rows if row["sample_valid"]] pool = viable if viable else rows # Executed largest-input evidence is stronger than a conservative static hint. If at least # one sample-valid candidate finishes every generated performance probe inside the published # ten-second case contract with a two-second reserve, exclude candidates that do not. performance_viable = [ row for row in pool if (not row.get("performance_total") or row.get("performance_completed") == row.get("performance_total")) ] if performance_viable: pool = performance_viable best = max(pool, key=lambda row: ( row["sample_valid"], row["sample_passed"], bool(row["oracle_total"] and row["oracle"] == row["oracle_total"]), row["oracle"], row["consensus"], row.get("integer_safe", True), row.get("boundary_safe", True), row.get("feasible", True), bool(not row.get("performance_total") or row.get("performance_completed") == row.get("performance_total")), -row.get("performance_latency_s", 0.0), row["critic"], row["completed"], row["quality"], -row["index"], )) return candidates[best["index"]], best def _disagreement_note(generated, outputs): notes = [] for case_index, stdin in enumerate(generated): values = [rows[case_index] for rows in outputs] completed = [value for value in values if value is not None] if len(completed) < 2 or all(value.split() == completed[0].split() for value in completed[1:]): continue notes.append( "Input:\n" + stdin[:800] + "\nCandidate outputs:\n" + "\n---\n".join((value if value is not None else "")[:800] for value in values) ) if len(notes) >= 2: break return "\n\n".join(notes) def _hard_bounded_call(callback, timeout=_HARD_CALL_TIMEOUT_S): """Return an empty response when a provider call exceeds a real wall-clock deadline.""" result = queue.Queue(maxsize=1) def run(): try: result.put((True, callback())) except BaseException as exc: result.put((False, exc)) worker = threading.Thread(target=run, daemon=True) worker.start() worker.join(timeout) if worker.is_alive(): return "" ok, value = result.get_nowait() if not ok: raise value return value def build_agent(weights): config = _load(weights) state = {"started": None, "done": 0, "code_done": 0, "reserved_code_s": 0.0, "latency": {}} def invoke(call_model, model_index, prompt, max_tokens, reasoning): params = { "max_tokens": max_tokens, "reasoning": {"effort": reasoning}, "temperature": config["temperature"], } started = time.monotonic() try: # The runtime/harness owns the single request deadline. Nesting another daemon-thread # timeout here leaves an uncancellable provider request alive across later tasks. return call_model( _MODELS[model_index], [{"role": "user", "content": prompt}], params ) except Exception: # A transient provider/transport failure is an invalid candidate, not a reason to abort # the whole six-task proof. The bounded next stage still observes the same statement. return "" finally: elapsed = time.monotonic() - started floor = _MODEL_CALL_FLOOR_S[model_index] prior = state["latency"].get(model_index, floor) state["latency"][model_index] = max(floor, 0.7 * prior + 0.3 * elapsed) def agent(prompt, call_model): if state["started"] is None: state["started"] = time.monotonic() state["done"] += 1 epoch_deadline = state["started"] + _RUN_BUDGET_S def room_for(model_index, multiplier=1.0, stage_cap=None): remaining = max(0, _EXPECTED_TASKS - state["done"]) estimate = state["latency"].get(model_index, _MODEL_CALL_FLOOR_S[model_index]) # stage_cap used to reduce the estimate without enforcing a matching wall timeout. # Never admit a long provider call on that optimistic fiction. reserve = max( _RESERVE_PER_TASK_S * remaining, state["reserved_code_s"], ) return (time.monotonic() + max(_MIN_CALL_WINDOW_S, estimate * multiplier) + reserve) < epoch_deadline original = str(prompt) if not _is_code(original): state["reserved_code_s"] = 0.0 return invoke( call_model, config["non_code"], _non_code_request(original), config["primary_max_tokens"], config["primary_reasoning"], ) state["code_done"] += 1 remaining_code = max(0, _EXPECTED_CODE_TASKS - state["code_done"]) state["reserved_code_s"] = _RESERVE_PER_CODE_TASK_S * remaining_code model_calls = 0 def ask(model_index, request_text, max_tokens, reasoning): nonlocal model_calls if model_calls >= _MAX_CODE_CALLS: return "" model_calls += 1 return invoke(call_model, model_index, request_text, max_tokens, reasoning) samples = _samples(original, config["sample_limit"]) numeric = bool(_FLOAT_JUDGE.search(original)) finish = lambda code: _finish_candidate(code, original) sequential_risk = bool(_SEQUENTIAL_RISK.search(original)) mutable_graph_risk = bool(_MUTABLE_GRAPH_RISK.search(original)) large_input = bool(_LARGE_N.search(original)) deep_risk = _deep_code_risk(original) batch_output_risk = bool( _MANY_OUTPUT_RISK.search(original) and _COUNTING_RISK.search(original) ) high_risk = sequential_risk or mutable_graph_risk or large_input or deep_risk request = (original + "\n\nGeneral reliability protocol: " + _SOLVE + "\n\nResource-bound audit: " + _RESOURCE_AUDIT + "\n\nAllocation/index audit: " + _INDEX_BOUND_AUDIT) if numeric: request += "\n\nNumeric-output protocol: " + _FLOAT_FORMAT if sequential_risk: request += "\n\nSequential-update protocol: " + _SEQUENTIAL_A + _SEQUENTIAL_B if mutable_graph_risk: request += "\n\nMutable-state shortest-path protocol: " + _MUTABLE_GRAPH if large_input: request += ("\n\nLarge-input protocol: Prove the asymptotic bound for the largest N. " "Reject any nested iteration whose inner bound grows with the input unless " "a strict amortized or logarithmic bound is established.") if deep_risk: request += ("\n\nLogical-equivalence audit: " + _EQUIVALENCE_AUDIT + _QUANTIFIER_AUDIT + _INDEPENDENCE_AUDIT) primary_model = config["batch_primary"] if batch_output_risk else config["primary"] challenger_model = config["batch_challenger"] if batch_output_risk else config["challenger"] primary_max_tokens = (config["batch_primary_max_tokens"] if batch_output_risk else config["primary_max_tokens"]) challenge_max_tokens = (config["batch_challenge_max_tokens"] if batch_output_risk else config["challenge_max_tokens"]) primary_reasoning = (config["batch_primary_reasoning"] if batch_output_risk else config["primary_reasoning"]) challenge_reasoning = (config["batch_challenge_reasoning"] if batch_output_risk else config["challenge_reasoning"]) if batch_output_risk: request += ( "\n\nBatched-output complexity audit: " + _BATCH_OUTPUT_AUDIT + "\n\nExecutable performance target: construct a largest-legal-input " "benchmark from the statement. The returned program must preserve the derived " "semantics and target at most 6 seconds per official case on one CPU, leaving " "headroom below the 10-second validator limit." ) primary_answer = ask( primary_model, request + "\n\nOutput protocol:\n" + _PRIMARY_FORMAT, primary_max_tokens, primary_reasoning, ) primary_programs = _named_programs(primary_answer) first = finish(_returnable_candidate(primary_answer, primary_programs)) references = [] if primary_programs.get("reference"): references.append(primary_programs["reference"]) if deep_risk: # Do not anchor a provider-diverse solver on the primary program. Agreement after # seeing the same compressed-state idea is correlated evidence, not independence. challenge_request = ( request + "\n\nLogical-equivalence audit: " + _EQUIVALENCE_AUDIT + _INDEPENDENCE_AUDIT + "\n\nIndependent challenge protocol:\n" + _CHALLENGE ) else: # Preserve provider independence on ordinary tasks too: the challenger receives the # statement and generic audits, never the primary program. challenge_request = ( request + "\n\nLogical-equivalence audit: " + _EQUIVALENCE_AUDIT + _INDEPENDENCE_AUDIT + "\n\nIndependent challenge protocol: derive from the statement before " "considering any other implementation.\n" + _CHALLENGE ) challenge_answer = (ask( challenger_model, challenge_request, challenge_max_tokens, challenge_reasoning, ) if room_for(challenger_model, 1.5) else "") evidence_deadline = min( epoch_deadline - 8.0, time.monotonic() + _EVIDENCE_BUDGET_S ) programs = _named_programs(challenge_answer) second = finish(_returnable_candidate(challenge_answer, programs)) generator = programs.get("generator") benchmark = programs.get("benchmark") if programs.get("reference"): references.append(programs["reference"]) candidates = [first] if second.strip(): candidates.append(second) generated = _generated_inputs( generator, config["stress_cases"], require_scale=not deep_risk, deadline=evidence_deadline ) generator_candidate_added = False # Preserve the final model call for an independent reviewer. A missing or malformed # challenge must not spend that reserve on generator-only tooling. if (not generated or (not benchmark and not deep_risk)) and model_calls < 2 and room_for( config["generator"], 1.0, 45.0): generator_request = original + "\n\nInput-generator protocol: " + _GENERATOR_ONLY generator_answer = ask( config["generator"], generator_request, config["generator_max_tokens"], config["generator_reasoning"], ) generated_programs = _named_programs(generator_answer) if not generated_programs: plain_blocks = _plain_program_blocks(generator_answer) if len(plain_blocks) in (2, 3): generated_programs = { "generator": plain_blocks[0], "reference": plain_blocks[1], } if len(plain_blocks) == 3: generated_programs["benchmark"] = plain_blocks[2] elif not plain_blocks: # Providers occasionally ignore a tooling-only format and return a complete # raw solution. It is not generator evidence, but it is still an independently # derived runtime candidate and must not be silently discarded. recovered = _source(generator_answer) if recovered.strip(): candidates.append(finish(recovered)) generator_candidate_added = True generated_candidate = generated_programs.get("candidate") if generated_candidate and generated_candidate.strip(): candidates.append(finish(generated_candidate)) generator_candidate_added = True generator = generated_programs.get("generator") or "" benchmark = generated_programs.get("benchmark") or benchmark if generated_programs.get("reference"): references.append(generated_programs["reference"]) generated = _generated_inputs( generator, config["stress_cases"], require_scale=not deep_risk ) # Every returned benchmark is only admission evidence and uses an eight-second # timeout, leaving margin under the validator's published ten-second case contract. performance_inputs = _generated_inputs( benchmark, 2, require_scale=False, deadline=evidence_deadline ) if benchmark else [] oracle_cases = _first_trusted_oracle( references, samples, generated, numeric, config["min_consensus_cases"], evidence_deadline ) # High-risk statements may have many sample-valid but semantically wrong compressed or # greedy programs. Ask a provider-diverse model for a third literal oracle when the first # two stages did not establish one. A single reference is never trusted by itself. if (generated and not oracle_cases and not deep_risk and model_calls < 2 and room_for(config["generator"], 1.5)): arbiter_answer = ask( config["generator"], original + "\n\nOracle protocol: " + _ARBITER_ONLY, config["generator_max_tokens"], config["generator_reasoning"], ) arbiter_programs = _named_programs(arbiter_answer) arbiter_reference = arbiter_programs.get("reference") if not arbiter_reference: plain_blocks = _plain_program_blocks(arbiter_answer) if len(plain_blocks) == 1: arbiter_reference = plain_blocks[0] if (arbiter_reference and arbiter_reference.strip() and arbiter_reference not in references): references.append(arbiter_reference) oracle_cases = _first_trusted_oracle( references, samples, generated, numeric, config["min_consensus_cases"], evidence_deadline ) rows, outputs = _candidate_evidence( candidates, samples, generated, numeric, oracle_cases, config["critic"], large_input, performance_inputs, evidence_deadline ) best, best_row = _best_candidate(candidates, rows) # Two independent generations are normally sufficient. A third call is admitted # only below, when execution/oracle evidence cannot safely accept either candidate. disagreement = _disagreement_note(generated, outputs) fixed_width_risk = any(_fixed_width_integer_risk(code) for code in candidates) boundary_risk = any(_allocation_index_risk(code) for code in candidates) review_required = high_risk or fixed_width_risk or boundary_risk performance_ok = ( not performance_inputs or best_row["performance_completed"] == best_row["performance_total"] ) # A generator-format violation may still yield a complete independent program. # On an ordinary statement, a sample-valid arbitrary-precision candidate is strictly safer # than a fixed-width candidate that needs collision/bound proofs. This is a generic language # safety fallback, not task routing or answer lookup. if (generator_candidate_added and not high_risk and fixed_width_risk and best_row["sample_valid"] and best_row.get("integer_safe", True) and performance_ok): return finish(best) if (best_row["sample_valid"] and performance_ok and oracle_cases and best_row["oracle"] == best_row["oracle_total"] and len(generated) >= config["min_consensus_cases"]): return finish(best) if (not review_required and best_row["sample_valid"] and performance_ok and ( (oracle_cases and best_row["oracle"] == best_row["oracle_total"]) or (not high_risk and not oracle_cases and len(candidates) >= 2 and len(generated) >= config["min_consensus_cases"] and all(row["sample_valid"] for row in rows) and best_row["consensus"] == len(generated) and best_row["completed"] == len(generated)))): return finish(best) candidate_text = "\n\n".join( "Candidate " + str(index + 1) + ":\n" + code for index, code in enumerate(candidates) ) if deep_risk: # Correctness evidence selects the repair seed; no task identity or stored answer is used. # The reviewer sees only the current statement, generic audit policy, and runtime candidate. review_request = ( request + "\n\nPerformance repair protocol: " + _PERFORMANCE_REPAIR + "\n\nLogical-equivalence audit: " + _EQUIVALENCE_AUDIT + _INDEPENDENCE_AUDIT + "\n\nStatement-sample acceptance protocol: " + _MUST_PASS + "\n\nEvidence-selected program to repair:\n" + best ) else: review_request = ( request + "\n\nIndependent review protocol: " + _REVIEW + "\n\nStatement-sample acceptance protocol: " + _MUST_PASS + "\n\nPrograms under review:\n" + candidate_text ) if disagreement: review_request += ( "\n\nThe programs disagreed on small generated inputs. These outputs are evidence " "of disagreement only; none is a trusted oracle.\n" + disagreement ) narrowed = [str(index + 1) for index, code in enumerate(candidates) if _fixed_width_integer_risk(code)] if narrowed: review_request += ( "\n\nStatic language-safety diagnostics: candidate(s) " + ", ".join(narrowed) + " use fixed-width integer storage or masking. " + _INTEGER_SAFETY ) boundary_candidates = [str(index + 1) for index, code in enumerate(candidates) if _allocation_index_risk(code)] if boundary_candidates: review_request += ( "\n\nStatic allocation/index diagnostics: candidate(s) " + ", ".join(boundary_candidates) + " contain a subscript expression whose " "growth outruns the tracked list allocation. " + _INDEX_BOUND_AUDIT ) if oracle_cases: counterexamples = [] for stdin, expected in oracle_cases: observed = _run_program(best, stdin, _STRESS_TIMEOUT) if observed is None or not _tokens_match(observed, expected, numeric): counterexamples.append( "Input:\n" + stdin[:800] + "\nTrusted dual-reference output:\n" + expected[:800] + "\nSelected-candidate output:\n" + (observed if observed is not None else "")[:800] ) if len(counterexamples) >= 2: break if counterexamples: review_request += ( "\n\nTwo independent small references agreed on these counterexamples:\n" + "\n\n".join(counterexamples) ) third = finish(_source(ask( config["reviewer"], review_request, config["review_max_tokens"], config["review_reasoning"], ))) if room_for(config["reviewer"], 1.0, 75.0) else "" reviewer_row = None if third.strip(): candidates.append(third) review_deadline = min( epoch_deadline - 8.0, time.monotonic() + _CANDIDATE_EVIDENCE_BUDGET_S, ) third_rows, third_outputs = _candidate_evidence( [third], samples, generated, numeric, oracle_cases, config["critic"], large_input, performance_inputs, review_deadline, ) if third_rows: third_rows[0]["index"] = len(candidates) - 1 rows.extend(third_rows) outputs.extend(third_outputs) reviewer_row = third_rows[0] best, best_row = _best_candidate(candidates, rows) if model_calls >= _MAX_CODE_CALLS: return finish(best) # A reviewer is another candidate, never an authority. The same execution, # oracle, scale, and learned-critic evidence selects among every generated program. performance_ok = ( not performance_inputs or best_row["performance_completed"] == best_row["performance_total"] ) if (best_row["sample_valid"] and performance_ok and ( (oracle_cases and best_row["oracle"] == best_row["oracle_total"]) or (not high_risk and not oracle_cases and best_row["consensus"] >= config["min_consensus_cases"] and best_row["completed"] >= config["min_consensus_cases"]) or (high_risk and not oracle_cases and len(candidates) >= 3 and len(generated) >= config["min_consensus_cases"] and sum(row["sample_valid"] for row in rows) >= 3 and best_row["consensus"] == len(generated) and best_row["completed"] == len(generated)))): return finish(best) # Preserve sample-valid answers on ordinary tasks when there is no contradictory evidence. # High-risk tasks may not use this sample-only escape hatch: they must establish a dual # oracle, unanimous three-way generated-case consensus, or continue to independent rescue. if (best_row["sample_valid"] and performance_ok and ((oracle_cases and best_row["oracle"] == best_row["oracle_total"]) or (not high_risk and not oracle_cases))): return finish(best) passed, total, failure = _check(best, samples, numeric) if (performance_inputs and best_row["sample_valid"] and not performance_ok): rescue_request = ( request + "\n\nSecond independent performance repair: " + _PERFORMANCE_REPAIR + "\n\nThe previous repair did not pass every executable largest-input " "probe. Preserve exact semantics; replace the algorithm when necessary." + "\n\nStatement-sample acceptance protocol: " + _MUST_PASS + "\n\nExecution-selected program to repair:\n" + best ) else: rescue_request = ( request + "\n\nEmergency independent derivation: " + _REPAIR + "\n\nStatement-sample acceptance protocol: " + _MUST_PASS + "\n\nThe deterministic selector found insufficient execution consensus. " "Do not vote between prior programs; derive and return one complete raw program." + "\n\nBest prior candidate diagnostics:\n" + _failure_note(passed, total, failure) + "\n\nBest prior candidate:\n" + best ) rescued = finish(_source(ask( config["rescue"], rescue_request, config["rescue_max_tokens"], config["rescue_reasoning"], ))) if room_for(config["rescue"], 1.0, 60.0) else "" if rescued.strip(): candidates.append(rescued) rescue_deadline = min( epoch_deadline - 8.0, time.monotonic() + _CANDIDATE_EVIDENCE_BUDGET_S, ) rescue_rows, _rescue_outputs = _candidate_evidence( [rescued], samples, generated, numeric, oracle_cases, config["critic"], large_input, performance_inputs, rescue_deadline, ) if rescue_rows: rescue_rows[0]["index"] = len(candidates) - 1 rows.extend(rescue_rows) return finish(_best_candidate(candidates, rows)[0]) return agent