Download scripts/export_annotation_examples.py from REXWind/ouroboros: direct link, hf CLI and curl.
- Browser
- Download file 2.7 kB
-
https://huggingface.co/spaces/REXWind/ouroboros/resolve/main/scripts/export_annotation_examples.py
- Command line
-
hf download hf://spaces/REXWind/ouroboros/scripts/export_annotation_examples.py
-
curl -L -o export_annotation_examples.py https://huggingface.co/spaces/REXWind/ouroboros/resolve/main/scripts/export_annotation_examples.py
2.7 kB
| """Extract one verbatim, traceable annotation case per key from the supplied dataset.""" | |
| import csv | |
| import hashlib | |
| import json | |
| import sys | |
| from collections import Counter | |
| from pathlib import Path | |
| ROOT = Path(__file__).resolve().parents[1] | |
| source = Path(sys.argv[1]) if len(sys.argv)>1 else Path('/mnt/s3mount_2/shiboru/ouroboros_data/type_with_data_clean.json') | |
| with source.open() as f: | |
| records = json.load(f) | |
| keys = list(csv.DictReader((ROOT/'public/data/annotation_key_counts.csv').open(encoding='utf-8-sig'))) | |
| expected = {r['key']:int(r['record_count']) for r in keys} | |
| preferred = {'similarity':'A5IKJ0','function':'Q9REQ9','subunit':'B9LSR7','catalytic activity':'A5IKJ0','pathway':'A5IKJ0'} | |
| counts = Counter() | |
| selected = {} | |
| for index, row in enumerate(records): | |
| for key in expected.keys() & row.keys(): | |
| counts[key] += 1 | |
| value = row[key] | |
| snippets = value if isinstance(value,list) else [value] | |
| if not snippets or not all(isinstance(s,str) for s in snippets): | |
| continue | |
| text = ' '.join(snippets) | |
| if not text.strip(): | |
| continue | |
| # Prefer concise self-contained annotations; preserve the source verbatim. | |
| score = abs(len(text)-170) + max(0,len(text)-400)*2 | |
| if preferred.get(key)==row.get('uniprot_id'): | |
| score = -10000 | |
| if key not in selected or score < selected[key][0]: | |
| selected[key] = (score,index,row,text) | |
| assert counts == Counter(expected), {k:(counts[k],n) for k,n in expected.items() if counts[k]!=n} | |
| assert len(selected)==29, sorted(set(expected)-set(selected)) | |
| sha = hashlib.sha256() | |
| with source.open('rb') as f: | |
| for chunk in iter(lambda:f.read(1024*1024),b''):sha.update(chunk) | |
| source_info = {'filename':source.name,'sha256':sha.hexdigest(),'records':len(records),'keyCountsMatchSuppliedCSV':True} | |
| examples = {} | |
| for key in expected: | |
| _,index,row,text=selected[key] | |
| examples[key] = {'id':row['uniprot_id'],'text':text,'annotations':row[key],'source':'Supplied Swiss-Prot dataset · original annotation','origin':'project-dataset','url':'https://www.uniprot.org/uniprotkb/'+row['uniprot_id']+'/entry','recordIndex0':index,'sequenceLength':len(row['sequence']),'sequenceSha256':hashlib.sha256(row['sequence'].encode()).hexdigest()} | |
| data = {'source':source_info,'selection':'One concise illustrative record per key; source annotation text is unchanged. Not a representative performance sample.','examples':examples} | |
| (ROOT/'public/data/annotation-examples.json').write_text(json.dumps(data,ensure_ascii=False,indent=2)+'\n') | |
| print(json.dumps({'source':source_info,'examples':[(k,v['id'],len(v['text'])) for k,v in examples.items()]},indent=2),flush=True) | |