ouroboros / scripts /export_assets.py
REXWind's picture
Expand benchmark evidence with functional and inverse-folding metrics
f068b39 verified
Raw History Blame Contribute Delete
7.43 kB
"""Export traceable, precomputed examples. Does not run or modify any model."""
import csv
import hashlib
import json
import shutil
from pathlib import Path
from benchmark_metrics import augment
ROOT = Path(__file__).resolve().parents[1]
PROJECT = Path('/mnt/shared-storage-user/dnacoding/protein-text-dual-learn')
PAPER = Path('/mnt/shared-storage-user/dnacoding/sbr/Ouroboros_Thesis/prot_downstream')
AA = dict(zip('ALA ARG ASN ASP CYS GLN GLU GLY HIS ILE LEU LYS MET PHE PRO SER THR TRP TYR VAL UNK'.split(), 'ARNDCQEGHILKMFPSTWYVX'))
provenance = []
def read_json(path):
data = path.read_bytes()
provenance.append({'file': path.name, 'sha256': hashlib.sha256(data).hexdigest()})
return json.loads(data)
def csv_rows(path):
provenance.append({'file': path.name, 'sha256': hashlib.sha256(path.read_bytes()).hexdigest()})
with path.open() as f:
return list(csv.DictReader(f))
def save(name, value):
(ROOT / 'public/data' / name).write_text(json.dumps(value, ensure_ascii=False, indent=2) + '\n')
folder = PROJECT / 'data/petase/petase_ft'
records = read_json(folder / 'pred_seq_plddt_pae.json')
ranked = sorted([(row, p) for row in records for p in row['pred_seqs']], key=lambda rp: rp[1]['plddt'], reverse=True)
cases = []
for rank, (row, p) in enumerate(ranked[:3], 1):
path = folder / (p['protein_id'] + '.pdb')
ca = [s for s in path.read_text().splitlines() if s.startswith('ATOM ') and s[12:16].strip() == 'CA']
seq = ''.join(p['pred_seq'].split())
assert ''.join(AA.get(s[17:20], 'X') for s in ca) == seq, f'Sequence mismatch: {path}'
residues = [{'index': int(s[22:26]), 'chain': s[21].strip(), 'code': AA.get(s[17:20], 'X'), 'confidence': float(s[60:66])} for s in ca]
asset_name = 'petase-ft-' + p['protein_id']
shutil.copyfile(path, ROOT / 'public/structures' / (asset_name + '.pdb'))
(ROOT / 'public/structures' / (asset_name + '.fasta')).write_text('>' + asset_name + '\n' + seq + '\n')
cases.append({
'id': p['protein_id'], 'assetId': asset_name, 'title': 'PET hydrolase candidate',
'cohort': 'PETase · fine-tuned', 'rank': rank, 'cohortSize': len(ranked),
'sequence': seq, 'prompt': row['real_text'], 'residues': residues,
'plddt': p['plddt'], 'pae': p['pae'], 'length': len(seq),
'pdbUrl': '/structures/' + asset_name + '.pdb', 'fastaUrl': '/structures/' + asset_name + '.fasta',
'selection': 'Selected by mean pLDDT among 439 archived fine-tuned candidates. This is not a wet-lab activity ranking.',
'structureStatus': 'Precomputed predicted structure',
'experimentalStatus': 'Quantitative wet-lab results are not included in this release.',
'pdbSha256': hashlib.sha256(path.read_bytes()).hexdigest(),
'sequenceSha256': hashlib.sha256(seq.encode()).hexdigest(),
'provenance': 'Historical petase_ft cohort; exact adapter revision was not recorded in the source metadata. Distinct from lora_petase_3epoch (9,214 sequences).',
})
save('petase-archive.json', cases)
if not (ROOT / 'public/data/petase.json').exists():
save('petase.json', cases)
summary = csv_rows(PAPER / 'figures/text2prot/source_data/summary_statistics.csv')
text_rows = csv_rows(PAPER / 'figures/prot2text/main/source_data/text_distribution_statistics.csv')
metrics = []
for row in summary:
if row['metric'] in ['TM-score', 'Sequence recovery', 'pLDDT'] and row['model'] not in ['Random', 'Ground Truth']:
metrics.append({'task': 't2p', 'metric': row['metric'], 'split': row['split'], 'model': row['model'].replace('Ours (large)', 'Ouroboros'), 'n': int(row['n']), 'mean': float(row['mean'])})
for row in text_rows:
if row['metric'] in ['bleu_4', 'rouge_l']:
metrics.append({'task': 'p2t', 'metric': 'BLEU-4' if row['metric']=='bleu_4' else 'ROUGE-L', 'split': row['split'], 'model': row['label'].replace('Ours', 'Ouroboros'), 'n': int(row['n']), 'mean': float(row['mean'])})
scaling = csv_rows(PAPER / 'benchmarks/results/scaling_hard1_500/sample_main_aligned_summary.csv')
metrics = augment(metrics, PAPER, ROOT.parent, provenance)
save('evidence.json', {'metrics': metrics, 'scaling': [{'size': r['size'], 'parameters': int(r['parameters']), 'recovery': float(r['sequence_recovery_pct']), 'n': int(r['n'])} for r in scaling], 'date': '2026-10-07'})
predictions = read_json(PAPER / 'benchmarks/results/text2prot_generation/hard1/ours.json')
sequence_scores = read_json(ROOT.parent / 'prot_downstream/text2prot/results/sequence_scores/ours_hard1.json')['rows']
# Chosen from the archived hard1 outputs after checking both generated directions
# and excluding the long homopolymer loops present in the first qualifying rows.
example_indices = [385, 313, 232, 490, 476, 479, 497, 496, 259, 272]
example_functions = [
'superoxide_dismutase', 'capsule_biosynthesis', 'kallikrein_protease',
'fetal_hemoglobin', 'ribosomal_protein_us4', 'mitochondrial_complex_i',
'calmodulin', 'neurophysin_oxytocin', 'is1_transposase',
'udp_glucose_pyrophosphorylase',
]
assert len(predictions) == len(sequence_scores) and len(example_indices) == len(example_functions) == 10
examples = []
for i, function in zip(example_indices, example_functions):
row = predictions[i]
sequence = ''.join(row['pred_seq'].split())
reference = ''.join(row['real_seq'].split())
assert sequence_scores[i]['sequence_id'] == f'{i:04d}'
assert sequence_scores[i]['sequence_recovery'] >= 0.75
assert len(reference) <= 511
assert sequence and len(sequence) < 512 and max(sequence.count(aa) / len(sequence) for aa in set(sequence)) < 0.2
assert max(sequence.count(sequence[j:j + 4]) for j in range(len(sequence) - 3)) <= 3
assert row['bleu_4'] >= 0.55
name = f'hard_{i:04d}_{function}'
examples.append({'id': name, 'name': name, 'text': row['real_text'], 'protein': reference, 'generatedProtein': sequence, 'generatedText': row['pred_text'], 'mode': 'Precomputed · large-v5', 'note': 'Two independent directions: reference text → generated sequence; reference sequence → generated text. These are not round-trip outputs.'})
save('examples.json', examples)
benchmark_provenance = {'date': '2026-10-09', 'model': 'Ouroboros original large-v5 for benchmark and live generation', 'petase': 'Historical fine-tuned PETase cohort, selected by pLDDT. Adapter revision not recorded; not the newer 3-epoch cohort.', 'sources': provenance, 'notes': ['pLDDT is predicted structural confidence, not measured enzyme activity.', 'Hard1 is a temporally held-out, cluster-diverse benchmark; low homology is not established.', 'Text comparisons select best observed valid comparator configurations on each evaluation set.', 'Scaling uses matched references; Large historical sampling settings and seed are not fully verified. No fitted scaling law or training-seed confidence intervals.']}
previous_file = ROOT / 'public/data/provenance.json'
if previous_file.exists():
previous = json.loads(previous_file.read_text())
if 'wetlab' in previous:
benchmark_provenance.update(wetlab=previous['wetlab'], petase=previous['petase'])
if 'annotations' in previous:
benchmark_provenance['annotations'] = previous['annotations']
save('provenance.json', benchmark_provenance)
print(json.dumps({'selected': cases[0]['id'], 'plddt': cases[0]['plddt'], 'metric_rows': len(metrics), 'examples': [e['id'] for e in examples]}, indent=2))