Download deployment/verify_node.py from BonanDing/worldmem-baseline-evals: direct link, hf CLI and curl.
- Browser
- Download file 882 Bytes
-
https://huggingface.co/BonanDing/worldmem-baseline-evals/resolve/main/deployment/verify_node.py
- Command line
-
hf download hf://BonanDing/worldmem-baseline-evals/deployment/verify_node.py
-
curl -L -o verify_node.py https://huggingface.co/BonanDing/worldmem-baseline-evals/resolve/main/deployment/verify_node.py
882 Bytes
| """Validate the requested H200 allocation before smoke or full evaluation.""" | |
| import argparse | |
| import json | |
| import os | |
| from pathlib import Path | |
| import torch | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument("--output", type=Path, required=True) | |
| parser.add_argument("--expected-gpus", type=int, default=8) | |
| args = parser.parse_args() | |
| devices = [torch.cuda.get_device_name(i) for i in range(torch.cuda.device_count())] | |
| if len(devices) != args.expected_gpus or any("H200" not in name for name in devices): | |
| raise RuntimeError(f"Evaluation requires {args.expected_gpus} H200 GPUs; allocated devices: {devices}") | |
| args.output.mkdir(parents=True, exist_ok=True) | |
| record = {"devices": devices, "torch": torch.__version__, "slurm_job_id": os.environ.get("SLURM_JOB_ID")} | |
| (args.output / "node.json").write_text(json.dumps(record, indent=2) + "\n") | |
| print(json.dumps(record), flush=True) | |