File size: 2,704 Bytes
a74f603 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 | """Extract one verbatim, traceable annotation case per key from the supplied dataset."""
import csv
import hashlib
import json
import sys
from collections import Counter
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
source = Path(sys.argv[1]) if len(sys.argv)>1 else Path('/mnt/s3mount_2/shiboru/ouroboros_data/type_with_data_clean.json')
with source.open() as f:
records = json.load(f)
keys = list(csv.DictReader((ROOT/'public/data/annotation_key_counts.csv').open(encoding='utf-8-sig')))
expected = {r['key']:int(r['record_count']) for r in keys}
preferred = {'similarity':'A5IKJ0','function':'Q9REQ9','subunit':'B9LSR7','catalytic activity':'A5IKJ0','pathway':'A5IKJ0'}
counts = Counter()
selected = {}
for index, row in enumerate(records):
for key in expected.keys() & row.keys():
counts[key] += 1
value = row[key]
snippets = value if isinstance(value,list) else [value]
if not snippets or not all(isinstance(s,str) for s in snippets):
continue
text = ' '.join(snippets)
if not text.strip():
continue
# Prefer concise self-contained annotations; preserve the source verbatim.
score = abs(len(text)-170) + max(0,len(text)-400)*2
if preferred.get(key)==row.get('uniprot_id'):
score = -10000
if key not in selected or score < selected[key][0]:
selected[key] = (score,index,row,text)
assert counts == Counter(expected), {k:(counts[k],n) for k,n in expected.items() if counts[k]!=n}
assert len(selected)==29, sorted(set(expected)-set(selected))
sha = hashlib.sha256()
with source.open('rb') as f:
for chunk in iter(lambda:f.read(1024*1024),b''):sha.update(chunk)
source_info = {'filename':source.name,'sha256':sha.hexdigest(),'records':len(records),'keyCountsMatchSuppliedCSV':True}
examples = {}
for key in expected:
_,index,row,text=selected[key]
examples[key] = {'id':row['uniprot_id'],'text':text,'annotations':row[key],'source':'Supplied Swiss-Prot dataset · original annotation','origin':'project-dataset','url':'https://www.uniprot.org/uniprotkb/'+row['uniprot_id']+'/entry','recordIndex0':index,'sequenceLength':len(row['sequence']),'sequenceSha256':hashlib.sha256(row['sequence'].encode()).hexdigest()}
data = {'source':source_info,'selection':'One concise illustrative record per key; source annotation text is unchanged. Not a representative performance sample.','examples':examples}
(ROOT/'public/data/annotation-examples.json').write_text(json.dumps(data,ensure_ascii=False,indent=2)+'\n')
print(json.dumps({'source':source_info,'examples':[(k,v['id'],len(v['text'])) for k,v in examples.items()]},indent=2),flush=True)
|