"""Export the supplied annotation-key CSV without conflating counts with unique records.""" import csv, hashlib, json, shutil from pathlib import Path import sys root=Path(__file__).resolve().parents[1] source=Path(sys.argv[1]) if len(sys.argv)>1 else root/'public/data/annotation_key_counts.csv' rows=[] for r in csv.DictReader(source.open(encoding='utf-8-sig')): rows.append({'key':r['key'],'count':int(r['record_count']),'coverage':float(r['coverage_percent']),'share':float(r['annotation_share_percent'])}) assert len({r['key'] for r in rows})==len(rows) total=sum(r['count'] for r in rows) assert all(abs(r['count']/total*100-r['share'])<.000001 for r in rows) base=round(rows[0]['count']*100/rows[0]['coverage']) assert all(abs(r['count']/base*100-r['coverage'])<.000001 for r in rows) examples_path=root/'public/data/annotation-examples.json' examples=json.loads(examples_path.read_text()) if examples_path.exists() else None if examples: assert set(examples['examples'])=={r['key'] for r in rows} for row in rows: row['example']=examples['examples'][row['key']] data={'scope':'Dataset annotation statistics','scopeConfirmed':False,'keys':len(rows),'recordKeyPairs':total,'inferredRecords':base,'recordsInference':'Inferred from record_count / coverage_percent, consistent with supplied rounding; not labeled as a confirmed training-set size.','beyondTopFiveShare':sum(r['count'] for r in rows[5:])/total*100,'rows':rows,'source':{'filename':source.name,'sha256':hashlib.sha256(source.read_bytes()).hexdigest()},'notes':['Keys overlap: one record may contribute to multiple keys.','record_count counts records carrying a key, not individual text snippets.','Annotation breadth does not establish sequence, species, or structural diversity.','No cross-dataset comparison is supplied.']} if examples: data['exampleSource']=examples['source'] data['verifiedRecords']=examples['source']['records'] data['recordsInference']='Verified against the supplied annotation JSON; membership in a training split is not confirmed.' (root/'public/data/annotation-stats.json').write_text(json.dumps(data,indent=2)+'\n') target=root/'public/data/annotation_key_counts.csv' if source.resolve()!=target.resolve():shutil.copyfile(source,target) provenance_path=root/'public/data/provenance.json' provenance=json.loads(provenance_path.read_text()) provenance['annotations']={'source':data['source'],'keys':data['keys'],'recordKeyPairs':total,'scope':data['scope'],'scopeConfirmed':data['scopeConfirmed'],'definitions':data['notes']} if examples: provenance['annotations']['exampleSource']=examples['source'] provenance_path.write_text(json.dumps(provenance,ensure_ascii=False,indent=2)+'\n') print({'keys':len(rows),'recordKeyPairs':total,'inferredRecords':base,'beyondTopFiveShare':data['beyondTopFiveShare']})