Download training/code/prepare_public.py from Endikavi/Ines-1: direct link, hf CLI and curl.
- Browser
- Download file 9.77 kB
-
https://huggingface.co/Endikavi/Ines-1/resolve/main/training/code/prepare_public.py
- Command line
-
hf download hf://Endikavi/Ines-1/training/code/prepare_public.py
-
curl -L -o prepare_public.py https://huggingface.co/Endikavi/Ines-1/resolve/main/training/code/prepare_public.py
9.77 kB
| """Public decision datasets -> common case format (see decisions.py), plus a Spanish version. | |
| python scripts/decisions/prepare_public.py --out $DATA_ROOT/decisions/public \ | |
| --translator URL/v1/chat/completions --translator-model NAME [--jev-sample 10000] | |
| typed-decisions (LocalLLaMA, Apache-2.0), pinned to c76749ec (the repo has moved since): | |
| train 1,200 cases / 6,000 questions, test 400 / 2,000; choice, score and noul. Gold carries a | |
| distribution (a ~4B teacher sampled 3 times) -> used as the soft target. | |
| Validation = 10 % of train by id hash (same ids on the EN and ES sides). | |
| jev-decisions-v1 (samatv256, CC-BY-4.0: keep attribution), config general-clean-50k: | |
| agent tool selection, choice only, hard target. English only (translating would break tool | |
| names and schemas). Up to 26 options; the long system policy is dropped, the user goal and | |
| the last turns are kept. Random sample of --jev-sample, 5 % of it held out for validation. | |
| Spanish: typed-decisions train and test translated by an LLM behind an OpenAI-compatible endpoint | |
| (--translator, --translator-model; thinking disabled, temperature 0.3, so a rerun does not reproduce the | |
| released translation byte for byte): only text values; keys, option ids and structure unchanged, checked, | |
| else retried/dropped. The translator actually used is recorded in training/TRANSLATION_PROVENANCE.md. | |
| Outputs (JSONL): typed_{train,val,test}_{en,es}.jsonl, jev_{train,val}.jsonl, SOURCES.md | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import hashlib | |
| import json | |
| import random | |
| import re | |
| import sys | |
| import threading | |
| import urllib.request | |
| from concurrent.futures import ThreadPoolExecutor | |
| from pathlib import Path | |
| TYPED = ("LocalLLaMA/typed-decisions", "c76749ec58bd8c3d2ea706b31c333a9059c38f90") | |
| JEV = ("samatv256/jev-decisions-v1", None) | |
| def val_side(key, frac=0.1): | |
| return int(hashlib.sha256(str(key).encode()).hexdigest(), 16) % 1000 < frac * 1000 | |
| def js(x): | |
| return json.loads(x) if isinstance(x, str) else x | |
| def typed_cases(parquet, split): | |
| import pyarrow.parquet as pq | |
| out = [] | |
| for r in pq.read_table(parquet).to_pylist(): | |
| qs, gold = js(r["questions"]), js(r["gold"]) | |
| g, soft = {}, {} | |
| for qid, q in qs.items(): | |
| gq = gold[qid] | |
| lab = gq["label"] | |
| if q["type"] == "noul": | |
| lab = str(lab).lower() | |
| g[qid] = [str(lab)] | |
| probs = gq.get("probabilities") | |
| if probs: | |
| soft[qid] = {str(k).lower() if q["type"] == "noul" else str(k): float(v) for k, v in probs.items()} | |
| out.append({"id": r["id"], "grupo": "typed/" + r["workflow"], "lang": "en", "state": js(r["state"]), | |
| "questions": qs, "gold": g, "suave": soft, "split": split}) | |
| return out | |
| def jev_cases(files, sample, rng): | |
| import pyarrow.parquet as pq | |
| rows = [] | |
| for f in files: | |
| rows += pq.read_table(f, columns=["id", "source", "state", "question", "answer_options", "target_index"]).to_pylist() | |
| rng.shuffle(rows) | |
| out = [] | |
| for r in rows: | |
| opts = r["answer_options"] or [] | |
| if not 2 <= len(opts) <= 26 or r["target_index"] is None or r["target_index"] >= len(opts): | |
| continue | |
| st = r["state"] or {} | |
| hist = st.get("history") or [] | |
| turns = [] | |
| for h in hist[-6:]: | |
| txt = h.get("content") or h.get("text") or json.dumps({k: v for k, v in h.items() if k != "role"}, ensure_ascii=False) | |
| turns.append("%s: %s" % (h.get("role", "?"), str(txt)[:600])) | |
| state = {"user_goal": st.get("user_goal") or "", "recent_history": "\n".join(turns)} | |
| crit, keys = {}, [] | |
| for i, o in enumerate(opts): | |
| k = "%s_%d" % (re.sub(r"[^a-z0-9]+", "_", str(o.get("label") or "opt").lower()).strip("_")[:40], i) | |
| crit[k] = "%s: %s" % (o.get("label"), (o.get("description") or "")[:300]) | |
| keys.append(k) | |
| out.append({"id": "jev-" + str(r["id"]), "grupo": "jev/" + str(r["source"]).split("/")[-1][:40], "lang": "en", | |
| "state": state, "questions": {"q": {"type": "choice", "instructions": r["question"], "criteria": crit}}, | |
| "gold": {"q": [keys[r["target_index"]]]}}) | |
| if len(out) >= sample: | |
| break | |
| return out | |
| # ------------------------------------------------------------------ translation | |
| def same_shape(a, b): | |
| """Same keys everywhere, same non-string leaves (numbers/bools/None), strings may change.""" | |
| if isinstance(a, dict): | |
| return isinstance(b, dict) and set(a) == set(b) and all(same_shape(a[k], b[k]) for k in a) | |
| if isinstance(a, list): | |
| return isinstance(b, list) and len(a) == len(b) and all(same_shape(x, y) for x, y in zip(a, b)) | |
| if isinstance(a, str): | |
| return isinstance(b, str) | |
| return a == b | |
| def translate_case(url, model, case, tries=3): | |
| # noul questions may come without criteria (literal true/false): only what exists is sent | |
| payload = {"state": case["state"], | |
| "questions": {qid: {k: q[k] for k in ("instructions", "criteria") if q.get(k) is not None} | |
| for qid, q in case["questions"].items()}} | |
| prompt = ("Traduce al español TODOS los textos de este JSON (valores de tipo texto), con redacción natural " | |
| "de España y parafraseando si suena mejor. NO cambies ninguna clave, ni los números, ni los " | |
| "booleanos, ni la estructura, ni los identificadores técnicos (códigos, ids, emails, rutas). " | |
| "Devuelve SOLO el JSON traducido.\n\n" + json.dumps(payload, ensure_ascii=False)) | |
| body = {"model": model, "messages": [{"role": "user", "content": prompt}], "max_tokens": 4000, | |
| "temperature": 0.3, "chat_template_kwargs": {"enable_thinking": False}} | |
| for _ in range(tries): | |
| try: | |
| req = urllib.request.Request(url, json.dumps(body).encode(), {"Content-Type": "application/json"}) | |
| with urllib.request.urlopen(req, timeout=300) as r: | |
| txt = json.loads(r.read())["choices"][0]["message"]["content"] or "" | |
| m = re.search(r"\{.*\}", txt, re.S) | |
| tr = json.loads(m.group(0)) if m else None | |
| if tr is not None and same_shape(payload, tr): | |
| es = dict(case, lang="es", id=case["id"] + "-es", state=tr["state"], questions={ | |
| qid: dict(q, **tr["questions"][qid]) for qid, q in case["questions"].items()}) | |
| return es | |
| except Exception: | |
| pass | |
| return None | |
| def write(path, cases): | |
| with open(path, "w", encoding="utf-8") as f: | |
| for c in cases: | |
| f.write(json.dumps(c, ensure_ascii=False) + "\n") | |
| print(" %-26s %6d cases, %6d questions" % (Path(path).name, len(cases), sum(len(c["questions"]) for c in cases))) | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--out", type=Path, required=True) | |
| ap.add_argument("--translator", required=True, help="OpenAI-compatible chat completions URL") | |
| ap.add_argument("--translator-model", required=True, | |
| help="model name sent to --translator (historical value: see TRANSLATION_PROVENANCE.md)") | |
| ap.add_argument("--threads", type=int, default=16) | |
| ap.add_argument("--jev-sample", type=int, default=10000) | |
| ap.add_argument("--seed", type=int, default=13) | |
| a = ap.parse_args() | |
| a.out.mkdir(parents=True, exist_ok=True) | |
| rng = random.Random(a.seed) | |
| from huggingface_hub import hf_hub_download, list_repo_files | |
| tr_p = hf_hub_download(TYPED[0], "all/train-00000-of-00001.parquet", repo_type="dataset", revision=TYPED[1]) | |
| te_p = hf_hub_download(TYPED[0], "all/test-00000-of-00001.parquet", repo_type="dataset", revision=TYPED[1]) | |
| train_all, test = typed_cases(tr_p, "train"), typed_cases(te_p, "test") | |
| train = [c for c in train_all if not val_side(c["id"])] | |
| val = [c for c in train_all if val_side(c["id"])] | |
| print("typed-decisions @%s:" % TYPED[1][:8]) | |
| write(a.out / "typed_train_en.jsonl", train) | |
| write(a.out / "typed_val_en.jsonl", val) | |
| write(a.out / "typed_test_en.jsonl", test) | |
| jev_files = [f for f in list_repo_files(JEV[0], repo_type="dataset") if f.startswith("clean50k/data/") and f.endswith(".parquet")] | |
| local = [hf_hub_download(JEV[0], f, repo_type="dataset") for f in sorted(jev_files)] | |
| jev = jev_cases(local, a.jev_sample, rng) | |
| print("jev-decisions-v1 (general-clean-50k):") | |
| write(a.out / "jev_train.jsonl", [c for c in jev if not val_side(c["id"], 0.05)]) | |
| write(a.out / "jev_val.jsonl", [c for c in jev if val_side(c["id"], 0.05)]) | |
| print("translating typed-decisions to Spanish with %s ..." % a.translator, flush=True) | |
| for name, cases in (("train", train), ("val", val), ("test", test)): | |
| done, lock = [], threading.Lock() | |
| with ThreadPoolExecutor(a.threads) as ex: | |
| for es in ex.map(lambda c: translate_case(a.translator, a.translator_model, c), cases): | |
| if es is not None: | |
| with lock: | |
| done.append(es) | |
| print(" %s: %d of %d translated with identical structure" % (name, len(done), len(cases)), flush=True) | |
| write(a.out / ("typed_%s_es.jsonl" % name), done) | |
| (a.out / "SOURCES.md").write_text( | |
| "# Sources\n\n" | |
| "- LocalLLaMA/typed-decisions @%s (Apache-2.0). Spanish files: machine translation by %s.\n" | |
| "- samatv256/jev-decisions-v1, config general-clean-50k (CC-BY-4.0, see its SOURCE_LICENSES.md;\n" | |
| " upstream NVIDIA Nemotron agentic datasets). Sample of %d, reformatted.\n" % (TYPED[1], a.translator_model, a.jev_sample)) | |
| if __name__ == "__main__": | |
| sys.exit(main()) | |