Download scripts/export_annotation_stats.py from REXWind/ouroboros: direct link, hf CLI and curl.
- Browser
- Download file 2.83 kB
-
https://huggingface.co/spaces/REXWind/ouroboros/resolve/main/scripts/export_annotation_stats.py
- Command line
-
hf download hf://spaces/REXWind/ouroboros/scripts/export_annotation_stats.py
-
curl -L -o export_annotation_stats.py https://huggingface.co/spaces/REXWind/ouroboros/resolve/main/scripts/export_annotation_stats.py
2.83 kB
| """Export the supplied annotation-key CSV without conflating counts with unique records.""" | |
| import csv, hashlib, json, shutil | |
| from pathlib import Path | |
| import sys | |
| root=Path(__file__).resolve().parents[1] | |
| source=Path(sys.argv[1]) if len(sys.argv)>1 else root/'public/data/annotation_key_counts.csv' | |
| rows=[] | |
| for r in csv.DictReader(source.open(encoding='utf-8-sig')): | |
| rows.append({'key':r['key'],'count':int(r['record_count']),'coverage':float(r['coverage_percent']),'share':float(r['annotation_share_percent'])}) | |
| assert len({r['key'] for r in rows})==len(rows) | |
| total=sum(r['count'] for r in rows) | |
| assert all(abs(r['count']/total*100-r['share'])<.000001 for r in rows) | |
| base=round(rows[0]['count']*100/rows[0]['coverage']) | |
| assert all(abs(r['count']/base*100-r['coverage'])<.000001 for r in rows) | |
| examples_path=root/'public/data/annotation-examples.json' | |
| examples=json.loads(examples_path.read_text()) if examples_path.exists() else None | |
| if examples: | |
| assert set(examples['examples'])=={r['key'] for r in rows} | |
| for row in rows: | |
| row['example']=examples['examples'][row['key']] | |
| data={'scope':'Dataset annotation statistics','scopeConfirmed':False,'keys':len(rows),'recordKeyPairs':total,'inferredRecords':base,'recordsInference':'Inferred from record_count / coverage_percent, consistent with supplied rounding; not labeled as a confirmed training-set size.','beyondTopFiveShare':sum(r['count'] for r in rows[5:])/total*100,'rows':rows,'source':{'filename':source.name,'sha256':hashlib.sha256(source.read_bytes()).hexdigest()},'notes':['Keys overlap: one record may contribute to multiple keys.','record_count counts records carrying a key, not individual text snippets.','Annotation breadth does not establish sequence, species, or structural diversity.','No cross-dataset comparison is supplied.']} | |
| if examples: | |
| data['exampleSource']=examples['source'] | |
| data['verifiedRecords']=examples['source']['records'] | |
| data['recordsInference']='Verified against the supplied annotation JSON; membership in a training split is not confirmed.' | |
| (root/'public/data/annotation-stats.json').write_text(json.dumps(data,indent=2)+'\n') | |
| target=root/'public/data/annotation_key_counts.csv' | |
| if source.resolve()!=target.resolve():shutil.copyfile(source,target) | |
| provenance_path=root/'public/data/provenance.json' | |
| provenance=json.loads(provenance_path.read_text()) | |
| provenance['annotations']={'source':data['source'],'keys':data['keys'],'recordKeyPairs':total,'scope':data['scope'],'scopeConfirmed':data['scopeConfirmed'],'definitions':data['notes']} | |
| if examples: | |
| provenance['annotations']['exampleSource']=examples['source'] | |
| provenance_path.write_text(json.dumps(provenance,ensure_ascii=False,indent=2)+'\n') | |
| print({'keys':len(rows),'recordKeyPairs':total,'inferredRecords':base,'beyondTopFiveShare':data['beyondTopFiveShare']}) | |