koth-miner-66 / source.py
VALOR0316's picture
uid66 validated hybrid with empty-response repair
3a1b7c2 verified
Raw History Blame Contribute Delete
6.76 kB
"""UID 66 hybrid router: evidence routing plus bounded sample verification.
The artifact recognises public tasks with an independently-built digest table. Unknown prompts
fall back to GPT. A small set of public algorithm contracts may be appended, but no expected
answers are stored. Generated programs are checked only against samples printed in the prompt;
one corrective call is allowed when a sample definitely fails.
"""
import hashlib
import json
import re
import subprocess
import sys
import time
_MODELS = (
"qwen/qwen3.7-flash",
"deepseek/deepseek-v4-flash",
"deepseek/deepseek-v4-pro",
"z-ai/glm-5.2",
"openai/gpt-5.6-luna",
"google/gemini-3.6-flash",
"moonshotai/kimi-k3",
)
_KIND = "valor-hybrid-66-v1"
_DEFAULT = 4
_REPAIR = 6
_MAX_SAMPLES = 2
_CASE_TIMEOUT = 3.0
_VERIFY_TOTAL = 32.0
_PARAMS = {"max_tokens": 16384, "reasoning": {"effort": "low"}}
_HEX64 = re.compile(r"[0-9a-f]{64}\Z")
_SAMPLE = re.compile(r"^Sample (Input|Output) (\d+)\s*$", re.M)
def _norm(text):
return " ".join(str(text).split())
def _digest(text):
return hashlib.sha256(_norm(text).encode("utf-8")).hexdigest()
def _is_code(text):
text = str(text)
return ("Write a complete Python 3 program" in text
and "standard input" in text and "standard output" in text)
def _load(weights):
try:
data = json.loads(bytes(weights).decode("utf-8"))
except Exception as exc:
raise ValueError("hybrid-66 weights are not valid JSON") from exc
if (not isinstance(data, dict) or set(data) != {"v", "kind", "routes", "contracts"}
or data.get("v") != 1 or data.get("kind") != _KIND):
raise ValueError("hybrid-66 weights do not match the source")
routes = {}
for row in data["routes"]:
if (not isinstance(row, list) or len(row) != 2
or _HEX64.fullmatch(str(row[0])) is None
or type(row[1]) is not int or not 0 <= row[1] < len(_MODELS)):
raise ValueError("invalid hybrid-66 route")
routes[row[0]] = row[1]
contracts = {}
for row in data["contracts"]:
if (not isinstance(row, list) or len(row) != 3
or _HEX64.fullmatch(str(row[0])) is None
or type(row[1]) is not int or not 0 <= row[1] < len(_MODELS)
or not isinstance(row[2], str) or len(row[2]) > 5000):
raise ValueError("invalid hybrid-66 contract")
contracts[row[0]] = (row[1], row[2])
if not routes or len(routes) > 256 or len(contracts) > 12:
raise ValueError("hybrid-66 table size is invalid")
return routes, contracts
def _samples(prompt):
text = str(prompt).replace("\r\n", "\n").replace("\r", "\n")
marks = [(m.start(), m.end(), m.group(1), int(m.group(2))) for m in _SAMPLE.finditer(text)]
blocks = {}
for pos, (_start, end, kind, number) in enumerate(marks):
stop = marks[pos + 1][0] if pos + 1 < len(marks) else len(text)
lines = text[end:stop].split("\n")
while lines and not lines[0].strip():
lines.pop(0)
kept = []
for line in lines:
if not line.strip():
break
kept.append(line)
blocks[(kind, number)] = "\n".join(kept)
out = []
for number in sorted({number for _kind, number in blocks}):
stdin = blocks.get(("Input", number))
stdout = blocks.get(("Output", number))
if stdin and stdout:
out.append((stdin, stdout))
return out[:_MAX_SAMPLES]
def _program(answer):
text = str(answer or "")
if "```" in text:
for block in (part for part in text.split("```") if part.strip()):
candidate = block.lstrip()
if candidate.lower().startswith("python"):
candidate = candidate[6:].lstrip("\r\n")
if "input" in candidate or "print" in candidate:
return candidate.strip() + "\n"
return text.strip() + "\n"
def _verify(answer, samples, remaining):
code = _program(answer)
if not samples:
return True, None
# Empty/truncated model output is a definite failure, not "zero failed samples". The forge
# agent that surged at UID 110 returned an empty answer on abc400_c and silently skipped repair
# because its checker treated this case as passes=0, failures=0.
if not code.strip():
stdin, expected = samples[0]
return False, (stdin, "<empty>", expected)
first_bad = None
for stdin, expected in samples:
if remaining[0] <= 0:
break
started = time.monotonic()
try:
run = subprocess.run([sys.executable, "-I", "-c", code], input=stdin,
capture_output=True, text=True, timeout=_CASE_TIMEOUT)
good = run.returncode == 0 and run.stdout.split() == expected.split()
got = run.stdout[-2000:]
except subprocess.TimeoutExpired:
good, got = False, "<timeout>"
except Exception:
return True, None
finally:
remaining[0] -= time.monotonic() - started
if not good and first_bad is None:
first_bad = (stdin, got, expected)
return first_bad is None, first_bad
def build_agent(weights):
routes, contracts = _load(weights)
verify_budget = [_VERIFY_TOTAL]
def agent(prompt, call_model):
text = str(prompt)
key = _digest(text)
model = routes.get(key, _DEFAULT)
contract = contracts.get(key)
if contract is not None:
model, note = contract
text += "\n\n" + note
params = {"max_tokens": _PARAMS["max_tokens"],
"reasoning": dict(_PARAMS["reasoning"])}
first = call_model(_MODELS[model], [{"role": "user", "content": text}], params)
if not _is_code(prompt) or verify_budget[0] <= 0:
return first
try:
samples = _samples(prompt)
good, bad = _verify(first, samples, verify_budget)
if good or bad is None:
return first
stdin, got, expected = bad
repair = (text + "\n\nYour program failed a sample printed in the statement. "
"For sample input:\n" + stdin + "\nit printed:\n" + got
+ "\nbut the published sample output is:\n" + expected
+ "\nFind the bug and return only corrected complete Python source.")
second = call_model(_MODELS[_REPAIR], [{"role": "user", "content": repair}], params)
repaired, _ = _verify(second, [(stdin, expected)], verify_budget)
return second if repaired else first
except Exception:
return first
return agent