"""Extract one verbatim, traceable annotation case per key from the supplied dataset.""" import csv import hashlib import json import sys from collections import Counter from pathlib import Path ROOT = Path(__file__).resolve().parents[1] source = Path(sys.argv[1]) if len(sys.argv)>1 else Path('/mnt/s3mount_2/shiboru/ouroboros_data/type_with_data_clean.json') with source.open() as f: records = json.load(f) keys = list(csv.DictReader((ROOT/'public/data/annotation_key_counts.csv').open(encoding='utf-8-sig'))) expected = {r['key']:int(r['record_count']) for r in keys} preferred = {'similarity':'A5IKJ0','function':'Q9REQ9','subunit':'B9LSR7','catalytic activity':'A5IKJ0','pathway':'A5IKJ0'} counts = Counter() selected = {} for index, row in enumerate(records): for key in expected.keys() & row.keys(): counts[key] += 1 value = row[key] snippets = value if isinstance(value,list) else [value] if not snippets or not all(isinstance(s,str) for s in snippets): continue text = ' '.join(snippets) if not text.strip(): continue # Prefer concise self-contained annotations; preserve the source verbatim. score = abs(len(text)-170) + max(0,len(text)-400)*2 if preferred.get(key)==row.get('uniprot_id'): score = -10000 if key not in selected or score < selected[key][0]: selected[key] = (score,index,row,text) assert counts == Counter(expected), {k:(counts[k],n) for k,n in expected.items() if counts[k]!=n} assert len(selected)==29, sorted(set(expected)-set(selected)) sha = hashlib.sha256() with source.open('rb') as f: for chunk in iter(lambda:f.read(1024*1024),b''):sha.update(chunk) source_info = {'filename':source.name,'sha256':sha.hexdigest(),'records':len(records),'keyCountsMatchSuppliedCSV':True} examples = {} for key in expected: _,index,row,text=selected[key] examples[key] = {'id':row['uniprot_id'],'text':text,'annotations':row[key],'source':'Supplied Swiss-Prot dataset ยท original annotation','origin':'project-dataset','url':'https://www.uniprot.org/uniprotkb/'+row['uniprot_id']+'/entry','recordIndex0':index,'sequenceLength':len(row['sequence']),'sequenceSha256':hashlib.sha256(row['sequence'].encode()).hexdigest()} data = {'source':source_info,'selection':'One concise illustrative record per key; source annotation text is unchanged. Not a representative performance sample.','examples':examples} (ROOT/'public/data/annotation-examples.json').write_text(json.dumps(data,ensure_ascii=False,indent=2)+'\n') print(json.dumps({'source':source_info,'examples':[(k,v['id'],len(v['text'])) for k,v in examples.items()]},indent=2),flush=True)