File size: 1,154 Bytes
61b6fb9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
#!/bin/bash
# Audit: per-item outputs (full probabilities) of each model on the four public tests, Engram on and off, plus the
# full-context stress tests. Usage: audit_eval.sh GPU name:checkpoint_dir ...   (output in $DATA_ROOT/decisiones/auditoria)
set -u
. "$(dirname "$0")/_env.sh"
GPU=$1; shift
O=$DATA_ROOT/decisiones/auditoria
T="--test typed_en=$PUBLIC/typed_test_en.jsonl --test typed_es=$PUBLIC/typed_test_es.jsonl --test telepatia_es=$MIX/test_telepatia_es.jsonl --test tasksource=$MIX/test_tasksource.jsonl"
cd "$REPO"
for spec in "$@"; do
  n=${spec%%:*}; ck=${spec#*:}
  for off in "" "--engram-off"; do
    CUDA_VISIBLE_DEVICES=$GPU PYTHONPATH=$REPO PYTHONUNBUFFERED=1 $PYTHON scripts/decisions/eval_items.py --checkpoint $ck --name $n $T --out $O/items $off
  done
  for mode in context joint; do
    CUDA_VISIBLE_DEVICES=$GPU PYTHONPATH=$REPO PYTHONUNBUFFERED=1 $PYTHON scripts/decisions/eval_fullcase.py --checkpoint $ck --name $n \
      --test typed_en=$PUBLIC/typed_test_en.jsonl --test typed_es=$PUBLIC/typed_test_es.jsonl --test telepatia_es=$MIX/test_telepatia_es.jsonl \
      --mode $mode --out $O/fullcase
  done
done
echo END