Download scripts/export_assets.py from REXWind/ouroboros: direct link, hf CLI and curl.
- Browser
- Download file 7.43 kB
-
https://huggingface.co/spaces/REXWind/ouroboros/resolve/main/scripts/export_assets.py
- Command line
-
hf download hf://spaces/REXWind/ouroboros/scripts/export_assets.py
-
curl -L -o export_assets.py https://huggingface.co/spaces/REXWind/ouroboros/resolve/main/scripts/export_assets.py
7.43 kB
| """Export traceable, precomputed examples. Does not run or modify any model.""" | |
| import csv | |
| import hashlib | |
| import json | |
| import shutil | |
| from pathlib import Path | |
| from benchmark_metrics import augment | |
| ROOT = Path(__file__).resolve().parents[1] | |
| PROJECT = Path('/mnt/shared-storage-user/dnacoding/protein-text-dual-learn') | |
| PAPER = Path('/mnt/shared-storage-user/dnacoding/sbr/Ouroboros_Thesis/prot_downstream') | |
| AA = dict(zip('ALA ARG ASN ASP CYS GLN GLU GLY HIS ILE LEU LYS MET PHE PRO SER THR TRP TYR VAL UNK'.split(), 'ARNDCQEGHILKMFPSTWYVX')) | |
| provenance = [] | |
| def read_json(path): | |
| data = path.read_bytes() | |
| provenance.append({'file': path.name, 'sha256': hashlib.sha256(data).hexdigest()}) | |
| return json.loads(data) | |
| def csv_rows(path): | |
| provenance.append({'file': path.name, 'sha256': hashlib.sha256(path.read_bytes()).hexdigest()}) | |
| with path.open() as f: | |
| return list(csv.DictReader(f)) | |
| def save(name, value): | |
| (ROOT / 'public/data' / name).write_text(json.dumps(value, ensure_ascii=False, indent=2) + '\n') | |
| folder = PROJECT / 'data/petase/petase_ft' | |
| records = read_json(folder / 'pred_seq_plddt_pae.json') | |
| ranked = sorted([(row, p) for row in records for p in row['pred_seqs']], key=lambda rp: rp[1]['plddt'], reverse=True) | |
| cases = [] | |
| for rank, (row, p) in enumerate(ranked[:3], 1): | |
| path = folder / (p['protein_id'] + '.pdb') | |
| ca = [s for s in path.read_text().splitlines() if s.startswith('ATOM ') and s[12:16].strip() == 'CA'] | |
| seq = ''.join(p['pred_seq'].split()) | |
| assert ''.join(AA.get(s[17:20], 'X') for s in ca) == seq, f'Sequence mismatch: {path}' | |
| residues = [{'index': int(s[22:26]), 'chain': s[21].strip(), 'code': AA.get(s[17:20], 'X'), 'confidence': float(s[60:66])} for s in ca] | |
| asset_name = 'petase-ft-' + p['protein_id'] | |
| shutil.copyfile(path, ROOT / 'public/structures' / (asset_name + '.pdb')) | |
| (ROOT / 'public/structures' / (asset_name + '.fasta')).write_text('>' + asset_name + '\n' + seq + '\n') | |
| cases.append({ | |
| 'id': p['protein_id'], 'assetId': asset_name, 'title': 'PET hydrolase candidate', | |
| 'cohort': 'PETase · fine-tuned', 'rank': rank, 'cohortSize': len(ranked), | |
| 'sequence': seq, 'prompt': row['real_text'], 'residues': residues, | |
| 'plddt': p['plddt'], 'pae': p['pae'], 'length': len(seq), | |
| 'pdbUrl': '/structures/' + asset_name + '.pdb', 'fastaUrl': '/structures/' + asset_name + '.fasta', | |
| 'selection': 'Selected by mean pLDDT among 439 archived fine-tuned candidates. This is not a wet-lab activity ranking.', | |
| 'structureStatus': 'Precomputed predicted structure', | |
| 'experimentalStatus': 'Quantitative wet-lab results are not included in this release.', | |
| 'pdbSha256': hashlib.sha256(path.read_bytes()).hexdigest(), | |
| 'sequenceSha256': hashlib.sha256(seq.encode()).hexdigest(), | |
| 'provenance': 'Historical petase_ft cohort; exact adapter revision was not recorded in the source metadata. Distinct from lora_petase_3epoch (9,214 sequences).', | |
| }) | |
| save('petase-archive.json', cases) | |
| if not (ROOT / 'public/data/petase.json').exists(): | |
| save('petase.json', cases) | |
| summary = csv_rows(PAPER / 'figures/text2prot/source_data/summary_statistics.csv') | |
| text_rows = csv_rows(PAPER / 'figures/prot2text/main/source_data/text_distribution_statistics.csv') | |
| metrics = [] | |
| for row in summary: | |
| if row['metric'] in ['TM-score', 'Sequence recovery', 'pLDDT'] and row['model'] not in ['Random', 'Ground Truth']: | |
| metrics.append({'task': 't2p', 'metric': row['metric'], 'split': row['split'], 'model': row['model'].replace('Ours (large)', 'Ouroboros'), 'n': int(row['n']), 'mean': float(row['mean'])}) | |
| for row in text_rows: | |
| if row['metric'] in ['bleu_4', 'rouge_l']: | |
| metrics.append({'task': 'p2t', 'metric': 'BLEU-4' if row['metric']=='bleu_4' else 'ROUGE-L', 'split': row['split'], 'model': row['label'].replace('Ours', 'Ouroboros'), 'n': int(row['n']), 'mean': float(row['mean'])}) | |
| scaling = csv_rows(PAPER / 'benchmarks/results/scaling_hard1_500/sample_main_aligned_summary.csv') | |
| metrics = augment(metrics, PAPER, ROOT.parent, provenance) | |
| save('evidence.json', {'metrics': metrics, 'scaling': [{'size': r['size'], 'parameters': int(r['parameters']), 'recovery': float(r['sequence_recovery_pct']), 'n': int(r['n'])} for r in scaling], 'date': '2026-10-07'}) | |
| predictions = read_json(PAPER / 'benchmarks/results/text2prot_generation/hard1/ours.json') | |
| sequence_scores = read_json(ROOT.parent / 'prot_downstream/text2prot/results/sequence_scores/ours_hard1.json')['rows'] | |
| # Chosen from the archived hard1 outputs after checking both generated directions | |
| # and excluding the long homopolymer loops present in the first qualifying rows. | |
| example_indices = [385, 313, 232, 490, 476, 479, 497, 496, 259, 272] | |
| example_functions = [ | |
| 'superoxide_dismutase', 'capsule_biosynthesis', 'kallikrein_protease', | |
| 'fetal_hemoglobin', 'ribosomal_protein_us4', 'mitochondrial_complex_i', | |
| 'calmodulin', 'neurophysin_oxytocin', 'is1_transposase', | |
| 'udp_glucose_pyrophosphorylase', | |
| ] | |
| assert len(predictions) == len(sequence_scores) and len(example_indices) == len(example_functions) == 10 | |
| examples = [] | |
| for i, function in zip(example_indices, example_functions): | |
| row = predictions[i] | |
| sequence = ''.join(row['pred_seq'].split()) | |
| reference = ''.join(row['real_seq'].split()) | |
| assert sequence_scores[i]['sequence_id'] == f'{i:04d}' | |
| assert sequence_scores[i]['sequence_recovery'] >= 0.75 | |
| assert len(reference) <= 511 | |
| assert sequence and len(sequence) < 512 and max(sequence.count(aa) / len(sequence) for aa in set(sequence)) < 0.2 | |
| assert max(sequence.count(sequence[j:j + 4]) for j in range(len(sequence) - 3)) <= 3 | |
| assert row['bleu_4'] >= 0.55 | |
| name = f'hard_{i:04d}_{function}' | |
| examples.append({'id': name, 'name': name, 'text': row['real_text'], 'protein': reference, 'generatedProtein': sequence, 'generatedText': row['pred_text'], 'mode': 'Precomputed · large-v5', 'note': 'Two independent directions: reference text → generated sequence; reference sequence → generated text. These are not round-trip outputs.'}) | |
| save('examples.json', examples) | |
| benchmark_provenance = {'date': '2026-10-09', 'model': 'Ouroboros original large-v5 for benchmark and live generation', 'petase': 'Historical fine-tuned PETase cohort, selected by pLDDT. Adapter revision not recorded; not the newer 3-epoch cohort.', 'sources': provenance, 'notes': ['pLDDT is predicted structural confidence, not measured enzyme activity.', 'Hard1 is a temporally held-out, cluster-diverse benchmark; low homology is not established.', 'Text comparisons select best observed valid comparator configurations on each evaluation set.', 'Scaling uses matched references; Large historical sampling settings and seed are not fully verified. No fitted scaling law or training-seed confidence intervals.']} | |
| previous_file = ROOT / 'public/data/provenance.json' | |
| if previous_file.exists(): | |
| previous = json.loads(previous_file.read_text()) | |
| if 'wetlab' in previous: | |
| benchmark_provenance.update(wetlab=previous['wetlab'], petase=previous['petase']) | |
| if 'annotations' in previous: | |
| benchmark_provenance['annotations'] = previous['annotations'] | |
| save('provenance.json', benchmark_provenance) | |
| print(json.dumps({'selected': cases[0]['id'], 'plddt': cases[0]['plddt'], 'metric_rows': len(metrics), 'examples': [e['id'] for e in examples]}, indent=2)) | |