File size: 2,130 Bytes
b296ad4 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 | """Conservatively exclude training template variants flagged by the teacher audit."""
import argparse
from collections import Counter
import json
from pathlib import Path
def main():
p=argparse.ArgumentParser();p.add_argument('--data',required=True);p.add_argument('--audit',required=True)
args=p.parse_args();base=Path(args.data);bad=set();accepted=0;jobs=0
for line in Path(args.audit).read_text().splitlines():
row=json.loads(line);jobs+=1;accepted+=len(row['accepted'])
for record in row['rejected']:
# IDs are audit_<operation>_<index>_<language>; both names may contain underscores.
tail=record['id'][6:]
import re
match=re.fullmatch(r'(.+)_(\d+)_(en|noisy_en|hi|hinglish)',tail)
assert match,record['id']
op,index,language=match.groups();bad.add((op,language,int(index)))
removed=Counter();kept=0;source=base/'train.jsonl';temp=base/'train.filtered.jsonl'
with source.open() as inp,temp.open('w') as out:
for line in inp:
row=json.loads(line);key=(row['operation'],row['language'],row.get('template_index'))
if key in bad and row['provenance'].startswith('programmatic semantics + Qwen3.8-27B-FP8 language template'):
removed['/'.join(map(str,key))]+=1
else:out.write(line);kept+=1
temp.replace(source)
report={'audit_jobs':jobs,'accepted_template_instances':accepted,'flagged_template_variants':len(bad),
'excluded_training_rows':sum(removed.values()),'remaining_training_rows':kept,'exclusions':dict(removed),
'policy':'Conservative exclusion after SQL round-trip disagreement; disagreement alone does not prove the teacher was correct. Validation, test, manual, direct verified paraphrases and separately authored contrasts remain unchanged.',
'limitation':'Earlier curriculum stages trained on these rows before the audit; this filtering does not undo earlier exposure.'}
(base/'template-filter.json').write_text(json.dumps(report,indent=2));print(json.dumps(report))
if __name__=='__main__':main()
|