"""Export traceable, precomputed examples. Does not run or modify any model.""" import csv import hashlib import json import shutil from pathlib import Path from benchmark_metrics import augment ROOT = Path(__file__).resolve().parents[1] PROJECT = Path('/mnt/shared-storage-user/dnacoding/protein-text-dual-learn') PAPER = Path('/mnt/shared-storage-user/dnacoding/sbr/Ouroboros_Thesis/prot_downstream') AA = dict(zip('ALA ARG ASN ASP CYS GLN GLU GLY HIS ILE LEU LYS MET PHE PRO SER THR TRP TYR VAL UNK'.split(), 'ARNDCQEGHILKMFPSTWYVX')) provenance = [] def read_json(path): data = path.read_bytes() provenance.append({'file': path.name, 'sha256': hashlib.sha256(data).hexdigest()}) return json.loads(data) def csv_rows(path): provenance.append({'file': path.name, 'sha256': hashlib.sha256(path.read_bytes()).hexdigest()}) with path.open() as f: return list(csv.DictReader(f)) def save(name, value): (ROOT / 'public/data' / name).write_text(json.dumps(value, ensure_ascii=False, indent=2) + '\n') folder = PROJECT / 'data/petase/petase_ft' records = read_json(folder / 'pred_seq_plddt_pae.json') ranked = sorted([(row, p) for row in records for p in row['pred_seqs']], key=lambda rp: rp[1]['plddt'], reverse=True) cases = [] for rank, (row, p) in enumerate(ranked[:3], 1): path = folder / (p['protein_id'] + '.pdb') ca = [s for s in path.read_text().splitlines() if s.startswith('ATOM ') and s[12:16].strip() == 'CA'] seq = ''.join(p['pred_seq'].split()) assert ''.join(AA.get(s[17:20], 'X') for s in ca) == seq, f'Sequence mismatch: {path}' residues = [{'index': int(s[22:26]), 'chain': s[21].strip(), 'code': AA.get(s[17:20], 'X'), 'confidence': float(s[60:66])} for s in ca] asset_name = 'petase-ft-' + p['protein_id'] shutil.copyfile(path, ROOT / 'public/structures' / (asset_name + '.pdb')) (ROOT / 'public/structures' / (asset_name + '.fasta')).write_text('>' + asset_name + '\n' + seq + '\n') cases.append({ 'id': p['protein_id'], 'assetId': asset_name, 'title': 'PET hydrolase candidate', 'cohort': 'PETase · fine-tuned', 'rank': rank, 'cohortSize': len(ranked), 'sequence': seq, 'prompt': row['real_text'], 'residues': residues, 'plddt': p['plddt'], 'pae': p['pae'], 'length': len(seq), 'pdbUrl': '/structures/' + asset_name + '.pdb', 'fastaUrl': '/structures/' + asset_name + '.fasta', 'selection': 'Selected by mean pLDDT among 439 archived fine-tuned candidates. This is not a wet-lab activity ranking.', 'structureStatus': 'Precomputed predicted structure', 'experimentalStatus': 'Quantitative wet-lab results are not included in this release.', 'pdbSha256': hashlib.sha256(path.read_bytes()).hexdigest(), 'sequenceSha256': hashlib.sha256(seq.encode()).hexdigest(), 'provenance': 'Historical petase_ft cohort; exact adapter revision was not recorded in the source metadata. Distinct from lora_petase_3epoch (9,214 sequences).', }) save('petase-archive.json', cases) if not (ROOT / 'public/data/petase.json').exists(): save('petase.json', cases) summary = csv_rows(PAPER / 'figures/text2prot/source_data/summary_statistics.csv') text_rows = csv_rows(PAPER / 'figures/prot2text/main/source_data/text_distribution_statistics.csv') metrics = [] for row in summary: if row['metric'] in ['TM-score', 'Sequence recovery', 'pLDDT'] and row['model'] not in ['Random', 'Ground Truth']: metrics.append({'task': 't2p', 'metric': row['metric'], 'split': row['split'], 'model': row['model'].replace('Ours (large)', 'Ouroboros'), 'n': int(row['n']), 'mean': float(row['mean'])}) for row in text_rows: if row['metric'] in ['bleu_4', 'rouge_l']: metrics.append({'task': 'p2t', 'metric': 'BLEU-4' if row['metric']=='bleu_4' else 'ROUGE-L', 'split': row['split'], 'model': row['label'].replace('Ours', 'Ouroboros'), 'n': int(row['n']), 'mean': float(row['mean'])}) scaling = csv_rows(PAPER / 'benchmarks/results/scaling_hard1_500/sample_main_aligned_summary.csv') metrics = augment(metrics, PAPER, ROOT.parent, provenance) save('evidence.json', {'metrics': metrics, 'scaling': [{'size': r['size'], 'parameters': int(r['parameters']), 'recovery': float(r['sequence_recovery_pct']), 'n': int(r['n'])} for r in scaling], 'date': '2026-10-07'}) predictions = read_json(PAPER / 'benchmarks/results/text2prot_generation/hard1/ours.json') sequence_scores = read_json(ROOT.parent / 'prot_downstream/text2prot/results/sequence_scores/ours_hard1.json')['rows'] # Chosen from the archived hard1 outputs after checking both generated directions # and excluding the long homopolymer loops present in the first qualifying rows. example_indices = [385, 313, 232, 490, 476, 479, 497, 496, 259, 272] example_functions = [ 'superoxide_dismutase', 'capsule_biosynthesis', 'kallikrein_protease', 'fetal_hemoglobin', 'ribosomal_protein_us4', 'mitochondrial_complex_i', 'calmodulin', 'neurophysin_oxytocin', 'is1_transposase', 'udp_glucose_pyrophosphorylase', ] assert len(predictions) == len(sequence_scores) and len(example_indices) == len(example_functions) == 10 examples = [] for i, function in zip(example_indices, example_functions): row = predictions[i] sequence = ''.join(row['pred_seq'].split()) reference = ''.join(row['real_seq'].split()) assert sequence_scores[i]['sequence_id'] == f'{i:04d}' assert sequence_scores[i]['sequence_recovery'] >= 0.75 assert len(reference) <= 511 assert sequence and len(sequence) < 512 and max(sequence.count(aa) / len(sequence) for aa in set(sequence)) < 0.2 assert max(sequence.count(sequence[j:j + 4]) for j in range(len(sequence) - 3)) <= 3 assert row['bleu_4'] >= 0.55 name = f'hard_{i:04d}_{function}' examples.append({'id': name, 'name': name, 'text': row['real_text'], 'protein': reference, 'generatedProtein': sequence, 'generatedText': row['pred_text'], 'mode': 'Precomputed · large-v5', 'note': 'Two independent directions: reference text → generated sequence; reference sequence → generated text. These are not round-trip outputs.'}) save('examples.json', examples) benchmark_provenance = {'date': '2026-10-09', 'model': 'Ouroboros original large-v5 for benchmark and live generation', 'petase': 'Historical fine-tuned PETase cohort, selected by pLDDT. Adapter revision not recorded; not the newer 3-epoch cohort.', 'sources': provenance, 'notes': ['pLDDT is predicted structural confidence, not measured enzyme activity.', 'Hard1 is a temporally held-out, cluster-diverse benchmark; low homology is not established.', 'Text comparisons select best observed valid comparator configurations on each evaluation set.', 'Scaling uses matched references; Large historical sampling settings and seed are not fully verified. No fitted scaling law or training-seed confidence intervals.']} previous_file = ROOT / 'public/data/provenance.json' if previous_file.exists(): previous = json.loads(previous_file.read_text()) if 'wetlab' in previous: benchmark_provenance.update(wetlab=previous['wetlab'], petase=previous['petase']) if 'annotations' in previous: benchmark_provenance['annotations'] = previous['annotations'] save('provenance.json', benchmark_provenance) print(json.dumps({'selected': cases[0]['id'], 'plddt': cases[0]['plddt'], 'metric_rows': len(metrics), 'examples': [e['id'] for e in examples]}, indent=2))