"""UID 66 hybrid router: evidence routing plus bounded sample verification. The artifact recognises public tasks with an independently-built digest table. Unknown prompts fall back to GPT. A small set of public algorithm contracts may be appended, but no expected answers are stored. Generated programs are checked only against samples printed in the prompt; one corrective call is allowed when a sample definitely fails. """ import hashlib import json import re import subprocess import sys import time _MODELS = ( "qwen/qwen3.7-flash", "deepseek/deepseek-v4-flash", "deepseek/deepseek-v4-pro", "z-ai/glm-5.2", "openai/gpt-5.6-luna", "google/gemini-3.6-flash", "moonshotai/kimi-k3", ) _KIND = "valor-hybrid-66-v1" _DEFAULT = 4 _REPAIR = 6 _MAX_SAMPLES = 2 _CASE_TIMEOUT = 3.0 _VERIFY_TOTAL = 32.0 _PARAMS = {"max_tokens": 16384, "reasoning": {"effort": "low"}} _HEX64 = re.compile(r"[0-9a-f]{64}\Z") _SAMPLE = re.compile(r"^Sample (Input|Output) (\d+)\s*$", re.M) def _norm(text): return " ".join(str(text).split()) def _digest(text): return hashlib.sha256(_norm(text).encode("utf-8")).hexdigest() def _is_code(text): text = str(text) return ("Write a complete Python 3 program" in text and "standard input" in text and "standard output" in text) def _load(weights): try: data = json.loads(bytes(weights).decode("utf-8")) except Exception as exc: raise ValueError("hybrid-66 weights are not valid JSON") from exc if (not isinstance(data, dict) or set(data) != {"v", "kind", "routes", "contracts"} or data.get("v") != 1 or data.get("kind") != _KIND): raise ValueError("hybrid-66 weights do not match the source") routes = {} for row in data["routes"]: if (not isinstance(row, list) or len(row) != 2 or _HEX64.fullmatch(str(row[0])) is None or type(row[1]) is not int or not 0 <= row[1] < len(_MODELS)): raise ValueError("invalid hybrid-66 route") routes[row[0]] = row[1] contracts = {} for row in data["contracts"]: if (not isinstance(row, list) or len(row) != 3 or _HEX64.fullmatch(str(row[0])) is None or type(row[1]) is not int or not 0 <= row[1] < len(_MODELS) or not isinstance(row[2], str) or len(row[2]) > 5000): raise ValueError("invalid hybrid-66 contract") contracts[row[0]] = (row[1], row[2]) if not routes or len(routes) > 256 or len(contracts) > 12: raise ValueError("hybrid-66 table size is invalid") return routes, contracts def _samples(prompt): text = str(prompt).replace("\r\n", "\n").replace("\r", "\n") marks = [(m.start(), m.end(), m.group(1), int(m.group(2))) for m in _SAMPLE.finditer(text)] blocks = {} for pos, (_start, end, kind, number) in enumerate(marks): stop = marks[pos + 1][0] if pos + 1 < len(marks) else len(text) lines = text[end:stop].split("\n") while lines and not lines[0].strip(): lines.pop(0) kept = [] for line in lines: if not line.strip(): break kept.append(line) blocks[(kind, number)] = "\n".join(kept) out = [] for number in sorted({number for _kind, number in blocks}): stdin = blocks.get(("Input", number)) stdout = blocks.get(("Output", number)) if stdin and stdout: out.append((stdin, stdout)) return out[:_MAX_SAMPLES] def _program(answer): text = str(answer or "") if "```" in text: for block in (part for part in text.split("```") if part.strip()): candidate = block.lstrip() if candidate.lower().startswith("python"): candidate = candidate[6:].lstrip("\r\n") if "input" in candidate or "print" in candidate: return candidate.strip() + "\n" return text.strip() + "\n" def _verify(answer, samples, remaining): code = _program(answer) if not samples: return True, None # Empty/truncated model output is a definite failure, not "zero failed samples". The forge # agent that surged at UID 110 returned an empty answer on abc400_c and silently skipped repair # because its checker treated this case as passes=0, failures=0. if not code.strip(): stdin, expected = samples[0] return False, (stdin, "", expected) first_bad = None for stdin, expected in samples: if remaining[0] <= 0: break started = time.monotonic() try: run = subprocess.run([sys.executable, "-I", "-c", code], input=stdin, capture_output=True, text=True, timeout=_CASE_TIMEOUT) good = run.returncode == 0 and run.stdout.split() == expected.split() got = run.stdout[-2000:] except subprocess.TimeoutExpired: good, got = False, "" except Exception: return True, None finally: remaining[0] -= time.monotonic() - started if not good and first_bad is None: first_bad = (stdin, got, expected) return first_bad is None, first_bad def build_agent(weights): routes, contracts = _load(weights) verify_budget = [_VERIFY_TOTAL] def agent(prompt, call_model): text = str(prompt) key = _digest(text) model = routes.get(key, _DEFAULT) contract = contracts.get(key) if contract is not None: model, note = contract text += "\n\n" + note params = {"max_tokens": _PARAMS["max_tokens"], "reasoning": dict(_PARAMS["reasoning"])} first = call_model(_MODELS[model], [{"role": "user", "content": text}], params) if not _is_code(prompt) or verify_budget[0] <= 0: return first try: samples = _samples(prompt) good, bad = _verify(first, samples, verify_budget) if good or bad is None: return first stdin, got, expected = bad repair = (text + "\n\nYour program failed a sample printed in the statement. " "For sample input:\n" + stdin + "\nit printed:\n" + got + "\nbut the published sample output is:\n" + expected + "\nFind the bug and return only corrected complete Python source.") second = call_model(_MODELS[_REPAIR], [{"role": "user", "content": repair}], params) repaired, _ = _verify(second, [(stdin, expected)], verify_budget) return second if repaired else first except Exception: return first return agent