Download deploy_runpod.py from zotowata/pc-sho-dlm-code: direct link, hf CLI and curl.
- Browser
- Download file 7.95 kB
-
https://huggingface.co/zotowata/pc-sho-dlm-code/resolve/main/deploy_runpod.py
- Command line
-
hf download hf://zotowata/pc-sho-dlm-code/deploy_runpod.py
-
curl -L -o deploy_runpod.py https://huggingface.co/zotowata/pc-sho-dlm-code/resolve/main/deploy_runpod.py
7.95 kB
| """ | |
| Launch a PC-SHO-DLM training pod on RunPod using the current REST API. | |
| Usage: | |
| export RUNPOD_API_KEY=... | |
| python3 deploy_runpod.py --list | |
| python3 deploy_runpod.py --create --gpu "<exact RunPod GPU type ID>" --steps 10 | |
| """ | |
| import argparse | |
| import json | |
| import os | |
| import sys | |
| import urllib.error | |
| import urllib.parse | |
| import urllib.request | |
| RUNPOD_API = "https://rest.runpod.io/v1" | |
| DEFAULT_IMAGE = "runpod/pytorch:2.1.0-py3.10-cuda11.8.0-devel-ubuntu22.04" | |
| DEFAULT_VOLUME_MOUNT = "/workspace" | |
| DEFAULT_HF_CODE_REPO = "https://huggingface.co/zotowata/pc-sho-dlm-code" | |
| def get_runpod_key() -> str: | |
| key = os.environ.get("RUNPOD_API_KEY") or os.environ.get("RUNPOD_API_TOKEN") | |
| if not key: | |
| raise SystemExit( | |
| "Missing RunPod API key. Export RUNPOD_API_KEY before using this script." | |
| ) | |
| return key | |
| def request(method: str, path: str, payload: dict | None = None) -> dict | list: | |
| key = get_runpod_key() | |
| url = f"{RUNPOD_API}{path}" | |
| body = None if payload is None else json.dumps(payload).encode() | |
| req = urllib.request.Request( | |
| url, | |
| data=body, | |
| method=method, | |
| headers={ | |
| "Authorization": f"Bearer {key}", | |
| "Content-Type": "application/json", | |
| }, | |
| ) | |
| try: | |
| with urllib.request.urlopen(req, timeout=60) as resp: | |
| data = resp.read().decode() | |
| return json.loads(data) if data else {} | |
| except urllib.error.HTTPError as e: | |
| detail = e.read().decode(errors="ignore") | |
| raise RuntimeError(f"RunPod API {e.code}: {detail}") from e | |
| def list_pods() -> list[dict]: | |
| data = request("GET", "/pods") | |
| if not isinstance(data, list): | |
| raise RuntimeError(f"Unexpected pods response: {data}") | |
| return data | |
| def list_gpu_types() -> list[dict]: | |
| # Legacy GraphQL gpuTypes is not used; current pod list exposes live machine data only | |
| # via pod creation filters. This launcher accepts a preferred GPU display name and lets | |
| # RunPod place the pod on the first matching inventory. | |
| return [] | |
| def make_remote_bootstrap(args: argparse.Namespace) -> str: | |
| lines = [ | |
| "set -euo pipefail", | |
| "cd /workspace", | |
| "if ! command -v git >/dev/null 2>&1 || ! command -v git-lfs >/dev/null 2>&1; then apt-get update && apt-get install -y --no-install-recommends git git-lfs; fi", | |
| "export GIT_LFS_SKIP_SMUDGE=1", | |
| "git lfs install", | |
| ] | |
| if os.environ.get("HF_TOKEN"): | |
| lines.append('git config --global credential.helper store') | |
| lines.append('printf "https://user:%s@huggingface.co\\n" "$HF_TOKEN" > ~/.git-credentials') | |
| lines += [ | |
| f"if [ ! -d pc-sho-dlm ]; then git clone {sh_quote(args.code_repo)} pc-sho-dlm; fi", | |
| "cd pc-sho-dlm", | |
| "git fetch --all --tags || true", | |
| ] | |
| if args.code_revision: | |
| lines.append(f"git checkout {sh_quote(args.code_revision)}") | |
| lines += [ | |
| "chmod +x ./train_runpod.sh", | |
| "bash ./train_runpod.sh", | |
| ] | |
| return " && ".join(lines) | |
| def sh_quote(value: str) -> str: | |
| return "'" + value.replace("'", "'\"'\"'") + "'" | |
| def create_pod(args: argparse.Namespace) -> dict: | |
| env = { | |
| "HF_TOKEN": os.environ.get("HF_TOKEN", ""), | |
| "MODEL_PRESET": args.model_preset, | |
| "MODE": args.mode, | |
| "TOKENIZER": args.tokenizer, | |
| "DATA": args.data, | |
| "STEPS": str(args.steps), | |
| "BATCH_SIZE": str(args.batch_size), | |
| "LR": str(args.lr), | |
| "LOG_INTERVAL": str(args.log_interval), | |
| "SAVE_INTERVAL": str(args.save_interval), | |
| "TENSORBOARD_DIR": args.tensorboard_dir, | |
| "TENSORBOARD_PORT": str(args.tensorboard_port), | |
| "PRETRAIN_SHUFFLE_BUFFER": str(args.pretrain_shuffle_buffer), | |
| "PRETRAIN_SHUFFLE_SEED": str(args.pretrain_shuffle_seed), | |
| "SEQ_LEN_OVERRIDE": str(args.seq_len_override or ""), | |
| "EVAL_INTERVAL": str(args.eval_interval), | |
| "EVAL_BATCHES": str(args.eval_batches), | |
| "EVAL_BATCH_SIZE": str(args.eval_batch_size), | |
| "EVAL_RECORD_SKIP": str(args.eval_record_skip), | |
| "HF_REPO_ID": args.hf_repo_id, | |
| "CODE_REPO_URL": args.code_repo, | |
| "CODE_REPO_REVISION": args.code_revision or "", | |
| } | |
| docker_start_cmd = ( | |
| ["bash", "-lc", "sleep infinity"] | |
| if args.idle_start | |
| else ["bash", "-lc", make_remote_bootstrap(args)] | |
| ) | |
| payload = { | |
| "name": args.name, | |
| "cloudType": args.cloud_type, | |
| "computeType": "GPU", | |
| "gpuCount": args.gpu_count, | |
| "gpuTypeIds": [args.gpu], | |
| "gpuTypePriority": "availability", | |
| "containerDiskInGb": args.container_disk_gb, | |
| "volumeInGb": args.volume_gb, | |
| "volumeMountPath": DEFAULT_VOLUME_MOUNT, | |
| "imageName": args.image, | |
| "ports": ["22/tcp", "8888/http"], | |
| "supportPublicIp": True, | |
| "globalNetworking": True, | |
| "interruptible": args.interruptible, | |
| "env": env, | |
| "dockerStartCmd": docker_start_cmd, | |
| } | |
| return request("POST", "/pods", payload) | |
| def main() -> None: | |
| parser = argparse.ArgumentParser(description="RunPod launcher for PC-SHO-DLM 2B") | |
| parser.add_argument("--list", action="store_true", help="List existing pods") | |
| parser.add_argument("--create", action="store_true", help="Create a new pod") | |
| parser.add_argument("--name", default="pc-sho-dlm-2b") | |
| parser.add_argument( | |
| "--gpu", | |
| default=None, | |
| help="Exact RunPod GPU type ID, e.g. one of the values shown in the RunPod console/API", | |
| ) | |
| parser.add_argument("--gpu-count", type=int, default=1) | |
| parser.add_argument("--cloud-type", choices=["SECURE", "COMMUNITY"], default="COMMUNITY") | |
| parser.add_argument("--interruptible", action="store_true") | |
| parser.add_argument( | |
| "--idle-start", | |
| action="store_true", | |
| help="Start container with 'sleep infinity' instead of auto-running bootstrap", | |
| ) | |
| parser.add_argument("--image", default=DEFAULT_IMAGE) | |
| parser.add_argument("--volume-gb", type=int, default=150) | |
| parser.add_argument("--container-disk-gb", type=int, default=100) | |
| parser.add_argument("--code-repo", default=DEFAULT_HF_CODE_REPO) | |
| parser.add_argument("--code-revision", default=None) | |
| parser.add_argument("--model-preset", default="2b") | |
| parser.add_argument("--mode", default="unified") | |
| parser.add_argument("--tokenizer", default="HuggingFaceTB/SmolLM2-1.7B") | |
| parser.add_argument("--data", default="HuggingFaceFW/fineweb-edu") | |
| parser.add_argument("--steps", type=int, default=50000) | |
| parser.add_argument("--batch-size", type=int, default=1) | |
| parser.add_argument("--lr", type=float, default=1e-4) | |
| parser.add_argument("--log-interval", type=int, default=10) | |
| parser.add_argument("--save-interval", type=int, default=500) | |
| parser.add_argument("--tensorboard-dir", default="checkpoints/tensorboard") | |
| parser.add_argument("--tensorboard-port", type=int, default=8888) | |
| parser.add_argument("--pretrain-shuffle-buffer", type=int, default=4096) | |
| parser.add_argument("--pretrain-shuffle-seed", type=int, default=17) | |
| parser.add_argument("--seq-len-override", type=int, default=None) | |
| parser.add_argument("--eval-interval", type=int, default=500) | |
| parser.add_argument("--eval-batches", type=int, default=8) | |
| parser.add_argument("--eval-batch-size", type=int, default=1) | |
| parser.add_argument("--eval-record-skip", type=int, default=200000) | |
| parser.add_argument("--hf-repo-id", default="zotowata/pc-sho-dlm-2b-pretrain") | |
| args = parser.parse_args() | |
| if args.list: | |
| pods = list_pods() | |
| print(json.dumps(pods, indent=2)) | |
| return | |
| if args.create: | |
| if not args.gpu: | |
| raise SystemExit("--gpu is required with --create") | |
| pod = create_pod(args) | |
| print(json.dumps(pod, indent=2)) | |
| return | |
| parser.print_help() | |
| if __name__ == "__main__": | |
| main() | |