ouroboros / scripts /export_annotation_stats.py
REXWind's picture
Publish dual-learning showcase with live inference and sourced annotation cases
a74f603 verified
Raw History Blame Contribute Delete
2.83 kB
"""Export the supplied annotation-key CSV without conflating counts with unique records."""
import csv, hashlib, json, shutil
from pathlib import Path
import sys
root=Path(__file__).resolve().parents[1]
source=Path(sys.argv[1]) if len(sys.argv)>1 else root/'public/data/annotation_key_counts.csv'
rows=[]
for r in csv.DictReader(source.open(encoding='utf-8-sig')):
rows.append({'key':r['key'],'count':int(r['record_count']),'coverage':float(r['coverage_percent']),'share':float(r['annotation_share_percent'])})
assert len({r['key'] for r in rows})==len(rows)
total=sum(r['count'] for r in rows)
assert all(abs(r['count']/total*100-r['share'])<.000001 for r in rows)
base=round(rows[0]['count']*100/rows[0]['coverage'])
assert all(abs(r['count']/base*100-r['coverage'])<.000001 for r in rows)
examples_path=root/'public/data/annotation-examples.json'
examples=json.loads(examples_path.read_text()) if examples_path.exists() else None
if examples:
assert set(examples['examples'])=={r['key'] for r in rows}
for row in rows:
row['example']=examples['examples'][row['key']]
data={'scope':'Dataset annotation statistics','scopeConfirmed':False,'keys':len(rows),'recordKeyPairs':total,'inferredRecords':base,'recordsInference':'Inferred from record_count / coverage_percent, consistent with supplied rounding; not labeled as a confirmed training-set size.','beyondTopFiveShare':sum(r['count'] for r in rows[5:])/total*100,'rows':rows,'source':{'filename':source.name,'sha256':hashlib.sha256(source.read_bytes()).hexdigest()},'notes':['Keys overlap: one record may contribute to multiple keys.','record_count counts records carrying a key, not individual text snippets.','Annotation breadth does not establish sequence, species, or structural diversity.','No cross-dataset comparison is supplied.']}
if examples:
data['exampleSource']=examples['source']
data['verifiedRecords']=examples['source']['records']
data['recordsInference']='Verified against the supplied annotation JSON; membership in a training split is not confirmed.'
(root/'public/data/annotation-stats.json').write_text(json.dumps(data,indent=2)+'\n')
target=root/'public/data/annotation_key_counts.csv'
if source.resolve()!=target.resolve():shutil.copyfile(source,target)
provenance_path=root/'public/data/provenance.json'
provenance=json.loads(provenance_path.read_text())
provenance['annotations']={'source':data['source'],'keys':data['keys'],'recordKeyPairs':total,'scope':data['scope'],'scopeConfirmed':data['scopeConfirmed'],'definitions':data['notes']}
if examples:
provenance['annotations']['exampleSource']=examples['source']
provenance_path.write_text(json.dumps(provenance,ensure_ascii=False,indent=2)+'\n')
print({'keys':len(rows),'recordKeyPairs':total,'inferredRecords':base,'beyondTopFiveShare':data['beyondTopFiveShare']})