Ines-1 / training /code /launch /audit_eval.sh
Endikavi's picture
Ines-1 RC1 (private staging; release commit b7f5644)
61b6fb9 verified
Raw History Blame Contribute Delete
1.15 kB
#!/bin/bash
# Audit: per-item outputs (full probabilities) of each model on the four public tests, Engram on and off, plus the
# full-context stress tests. Usage: audit_eval.sh GPU name:checkpoint_dir ... (output in $DATA_ROOT/decisiones/auditoria)
set -u
. "$(dirname "$0")/_env.sh"
GPU=$1; shift
O=$DATA_ROOT/decisiones/auditoria
T="--test typed_en=$PUBLIC/typed_test_en.jsonl --test typed_es=$PUBLIC/typed_test_es.jsonl --test telepatia_es=$MIX/test_telepatia_es.jsonl --test tasksource=$MIX/test_tasksource.jsonl"
cd "$REPO"
for spec in "$@"; do
n=${spec%%:*}; ck=${spec#*:}
for off in "" "--engram-off"; do
CUDA_VISIBLE_DEVICES=$GPU PYTHONPATH=$REPO PYTHONUNBUFFERED=1 $PYTHON scripts/decisions/eval_items.py --checkpoint $ck --name $n $T --out $O/items $off
done
for mode in context joint; do
CUDA_VISIBLE_DEVICES=$GPU PYTHONPATH=$REPO PYTHONUNBUFFERED=1 $PYTHON scripts/decisions/eval_fullcase.py --checkpoint $ck --name $n \
--test typed_en=$PUBLIC/typed_test_en.jsonl --test typed_es=$PUBLIC/typed_test_es.jsonl --test telepatia_es=$MIX/test_telepatia_es.jsonl \
--mode $mode --out $O/fullcase
done
done
echo END