File size: 7,423 Bytes
b74affc | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 | """
IOL-AI Challenge 2026 — submission script (OFFLINE / Mode B).
Runtime facts (Space Submission tab):
* T4 medium, 16 GB VRAM, Python 3.10, 30-min wall clock.
* NO internet: cannot pip install or download anything. Model weights must be
committed into THIS repo (the working dir) and loaded from ".". Only the
pre-installed libraries/versions are available (torch 2.4.0, transformers
4.44.1, accelerate 0.34.2, bitsandbytes 0.43.3, autoawq 0.2.7, pandas 2.2.2,
numpy 2.1.3, ...). Do NOT pin different majors of torch/transformers/numpy.
* Read hidden test set from /tmp/data/test.csv; write submission.csv here.
* pred = JSON list, one entry per numbered item, in query order.
Ship the model in the repo with build_repo.py. This script loads it from "." with
bitsandbytes 4-bit by default so a ~7B fits 16 GB. T4 has no bf16 -> use float16.
Local dev: set IOL_TEST_CSV to a mock file. Quantization auto-disables if there's
no CUDA so the plumbing can be exercised on CPU with a tiny model.
"""
import os
os.environ.setdefault("HF_HUB_OFFLINE", "1")
os.environ.setdefault("TRANSFORMERS_OFFLINE", "1")
import re
import csv
import json
MODEL_DIR = os.environ.get("IOL_MODEL_DIR", ".") # weights live in the repo
TEST_CSV = os.environ.get("IOL_TEST_CSV", "/tmp/data/test.csv")
OUT_CSV = os.environ.get("IOL_OUT_CSV", "submission.csv")
MAX_NEW_TOKENS = int(os.environ.get("IOL_MAX_NEW_TOKENS", "1024"))
# "4bit" (bitsandbytes), "awq" (weights already AWQ-quantized), or "fp16".
QUANT = os.environ.get("IOL_QUANT", "4bit")
ANSWER_MARKER = "###ANSWERS###"
SYSTEM_PROMPT = (
"You are an expert competitor at the International Linguistics Olympiad. "
"Each problem gives data from a language you have never seen; deduce its "
"grammar and vocabulary using ONLY the data and hints in the problem. "
"Work through it briefly, then give your final answers.\n\n"
"You MUST end your reply with the answers in EXACTLY this format and write "
"nothing after it:\n"
f"{ANSWER_MARKER}\n"
"1. <answer to item 1>\n"
"2. <answer to item 2>\n"
"...(one numbered line per item, in order)\n\n"
"Each answer must contain ONLY the requested form and nothing else: a single "
"word, phrase, number, or letter. Do NOT restate the question, explain, or add "
"commentary after the answer. For letter-matching items give just the letter "
"(e.g. B). For number items give the digits or written-out number as asked. "
"Give exactly one answer for every numbered item — never leave one blank."
)
def count_items(query):
"""Number of numbered items in a query, e.g. '17. .. 18. ..' -> 2."""
nums = re.findall(r"(?m)^\s*(\d+)[\.\)]", query)
return len(nums) if nums else 1
def _clean_answer(s):
"""Strip list markers, common 'Answer:' labels, and surrounding quotes."""
s = re.sub(r"^\s*(?:\d+[\.\):]|[-*•])\s*", "", s).strip()
s = re.sub(r"^(?:answer|ans|translation|result)\s*[:\-]\s*", "", s, flags=re.I).strip()
return s.strip("\"'“”‘’` ").strip()
def _numbered_map(segment, n_items):
"""Collect 'n. text' / 'n) text' lines into {index: answer}. Answers are the
reliable anchor: even if reasoning is interleaved, the trailing numbered list
is what we want, so a later line for the same index overwrites an earlier one."""
out = {}
for m in re.finditer(r"(?m)^\s*(\d+)[\.\)]\s*(.+?)\s*$", segment):
idx = int(m.group(1))
if 1 <= idx <= n_items:
out[idx] = _clean_answer(m.group(2))
return out
def parse_answers(text, n_items):
"""Extract exactly n_items answers. Prefer the marked block; anchor on the
numbered list; fall back to the LAST n non-empty lines (answers come last)."""
seg = text.rsplit(ANSWER_MARKER, 1)[1] if ANSWER_MARKER in text else text
numbered = _numbered_map(seg, n_items)
if len(numbered) >= n_items or (numbered and ANSWER_MARKER in text):
answers = [numbered.get(i, "") for i in range(1, n_items + 1)]
else:
# No usable numbered list: take the last n_items non-empty lines.
lines = [_clean_answer(ln) for ln in seg.splitlines() if ln.strip()]
lines = [ln for ln in lines if ln]
answers = lines[-n_items:] if len(lines) >= n_items else lines
# Guarantee exactly n_items, never blank (fall back to last good answer).
last_good = next((a for a in reversed(answers) if a), "")
answers = [a if a else last_good for a in answers]
if len(answers) < n_items:
answers += [last_good] * (n_items - len(answers))
return answers[:n_items]
def _already_quantized(model_dir):
"""True if the shipped weights are pre-quantized (e.g. AWQ) — then transformers
auto-detects the config and we must NOT stack bitsandbytes on top."""
cfg = os.path.join(model_dir, "config.json")
try:
with open(cfg, encoding="utf-8") as f:
return "quantization_config" in json.load(f)
except Exception:
return False
def load_model():
import torch
from transformers import AutoTokenizer, AutoModelForCausalLM
tok = AutoTokenizer.from_pretrained(MODEL_DIR)
if not torch.cuda.is_available():
model = AutoModelForCausalLM.from_pretrained(
MODEL_DIR, torch_dtype=torch.float32).eval() # CPU dev fallback
return tok, model
kwargs = dict(torch_dtype=torch.float16, device_map="auto") # T4 has no bf16
if _already_quantized(MODEL_DIR):
pass # AWQ/pre-quant: transformers reads quantization_config from config.json
elif QUANT == "4bit":
from transformers import BitsAndBytesConfig
kwargs["quantization_config"] = BitsAndBytesConfig(
load_in_4bit=True,
bnb_4bit_compute_dtype=torch.float16,
bnb_4bit_quant_type="nf4",
bnb_4bit_use_double_quant=True,
)
model = AutoModelForCausalLM.from_pretrained(MODEL_DIR, **kwargs).eval()
return tok, model
def main():
import torch
tok, model = load_model()
with open(TEST_CSV, newline="", encoding="utf-8") as f:
rows = list(csv.DictReader(f))
dev = model.device if hasattr(model, "device") else "cpu"
out = []
for i, r in enumerate(rows):
context = (r.get("context") or "").strip()
query = (r.get("query") or "").strip()
n_items = count_items(query)
messages = [
{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user", "content": context + "\n\n" + query},
]
ids = tok.apply_chat_template(
messages, add_generation_prompt=True, return_tensors="pt"
).to(dev)
with torch.no_grad():
gen = model.generate(
ids, max_new_tokens=MAX_NEW_TOKENS, do_sample=False,
pad_token_id=tok.eos_token_id,
)
text = tok.decode(gen[0][ids.shape[-1]:], skip_special_tokens=True).strip()
answers = parse_answers(text, n_items)
out.append({"id": r["id"], "pred": json.dumps(answers, ensure_ascii=False)})
print("%d/%d done" % (i + 1, len(rows)), flush=True)
with open(OUT_CSV, "w", newline="", encoding="utf-8") as f:
w = csv.DictWriter(f, fieldnames=["id", "pred"])
w.writeheader()
w.writerows(out)
print("wrote %s (%d rows)" % (OUT_CSV, len(out)), flush=True)
if __name__ == "__main__":
main()
|