Download examples/inference.py from jlu-wsj/scRep: direct link, hf CLI and curl.
- Browser
- Download file 2.28 kB
-
https://huggingface.co/jlu-wsj/scRep/resolve/main/examples/inference.py
- Command line
-
hf download hf://jlu-wsj/scRep/examples/inference.py
-
curl -L -o inference.py https://huggingface.co/jlu-wsj/scRep/resolve/main/examples/inference.py
2.28 kB
| #!/usr/bin/env python3 | |
| """Extract L2-normalized scRep cell embeddings from an .h5ad file.""" | |
| from __future__ import annotations | |
| import argparse | |
| import sys | |
| from pathlib import Path | |
| from types import SimpleNamespace | |
| import numpy as np | |
| RELEASE_ROOT = Path(__file__).resolve().parents[1] | |
| if str(RELEASE_ROOT) not in sys.path: | |
| sys.path.insert(0, str(RELEASE_ROOT)) | |
| from scRep_inference import encode_embeddings, load_examples, load_scRep_bundle | |
| from scRep_pretrain.vocab import gene_vocab_to_map | |
| def parse_args() -> argparse.Namespace: | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--input_h5ad", required=True, help="Input AnnData (.h5ad) file.") | |
| parser.add_argument("--output", default="embeddings.npy", help="Output .npy embedding matrix.") | |
| parser.add_argument("--checkpoint", default="checkpoints/scRep_20260625_30M") | |
| parser.add_argument("--device", default="cuda" if __import__("torch").cuda.is_available() else "cpu") | |
| parser.add_argument("--batch_size", type=int, default=128) | |
| parser.add_argument("--max_input_genes", type=int, default=2048) | |
| parser.add_argument("--n_bins", type=int, default=50) | |
| parser.add_argument("--max_cells", type=int, default=0, help="0 means all cells.") | |
| parser.add_argument("--use_raw", action="store_true", help="Use adata.raw instead of adata.X.") | |
| return parser.parse_args() | |
| def main() -> None: | |
| args = parse_args() | |
| checkpoint = Path(args.checkpoint) | |
| if not checkpoint.is_absolute(): | |
| checkpoint = RELEASE_ROOT / checkpoint | |
| model, gene_vocab, *_ = load_scRep_bundle(checkpoint, SimpleNamespace(model_dir=str(checkpoint), asset_dir="", device=args.device)) | |
| gene_name_to_id = gene_vocab_to_map(gene_vocab) | |
| examples, _ = load_examples(args.input_h5ad, gene_name_to_id, max_total_cells=args.max_cells, use_raw=args.use_raw) | |
| if not examples: | |
| raise ValueError("No input cells had genes present in the checkpoint vocabulary.") | |
| embeddings = encode_embeddings(model, examples, args) | |
| output = Path(args.output) | |
| output.parent.mkdir(parents=True, exist_ok=True) | |
| np.save(output, embeddings) | |
| print(f"Saved {embeddings.shape[0]} x {embeddings.shape[1]} embeddings to {output}") | |
| if __name__ == "__main__": | |
| main() | |