#!/bin/bash # Audit: per-item outputs (full probabilities) of each model on the four public tests, Engram on and off, plus the # full-context stress tests. Usage: audit_eval.sh GPU name:checkpoint_dir ... (output in $DATA_ROOT/decisiones/auditoria) set -u . "$(dirname "$0")/_env.sh" GPU=$1; shift O=$DATA_ROOT/decisiones/auditoria T="--test typed_en=$PUBLIC/typed_test_en.jsonl --test typed_es=$PUBLIC/typed_test_es.jsonl --test telepatia_es=$MIX/test_telepatia_es.jsonl --test tasksource=$MIX/test_tasksource.jsonl" cd "$REPO" for spec in "$@"; do n=${spec%%:*}; ck=${spec#*:} for off in "" "--engram-off"; do CUDA_VISIBLE_DEVICES=$GPU PYTHONPATH=$REPO PYTHONUNBUFFERED=1 $PYTHON scripts/decisions/eval_items.py --checkpoint $ck --name $n $T --out $O/items $off done for mode in context joint; do CUDA_VISIBLE_DEVICES=$GPU PYTHONPATH=$REPO PYTHONUNBUFFERED=1 $PYTHON scripts/decisions/eval_fullcase.py --checkpoint $ck --name $n \ --test typed_en=$PUBLIC/typed_test_en.jsonl --test typed_es=$PUBLIC/typed_test_es.jsonl --test telepatia_es=$MIX/test_telepatia_es.jsonl \ --mode $mode --out $O/fullcase done done echo END