Image-Text-to-Text
PEFT
Safetensors
vision-language
multimodal
llava
lora
siglip2
n-atlas
nigerian-languages
Instructions to use Modularcomputing/AtlasVision with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use Modularcomputing/AtlasVision with PEFT:
Task type is invalid.
- Notebooks
- Google Colab
- Kaggle
File size: 1,713 Bytes
2a2540a | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 | #!/usr/bin/env python3
"""Download everything onto the node's persistent disk. Safe to re-run (skips finished files).
Needs HF_TOKEN in the environment for the gated N-ATLaS repo."""
import os
import sys
import time
from huggingface_hub import hf_hub_download, snapshot_download
HOME = os.path.expanduser("~")
DATA_DIR = f"{HOME}/data/llava_pretrain"
LLM_NAME = os.environ.get("LLM_NAME", "NCAIR1/N-ATLaS")
VISION_NAME = os.environ.get("VISION_NAME", "google/siglip2-base-patch16-224")
def log(msg):
print(f"[{time.strftime('%H:%M:%S')}] {msg}", flush=True)
if not os.environ.get("HF_TOKEN"):
sys.exit("HF_TOKEN is not set - needed for the gated N-ATLaS repo")
# 1. Fail fast if the token can't see the gated model
try:
hf_hub_download(LLM_NAME, "config.json")
log(f"gated access OK for {LLM_NAME}")
except Exception as e:
sys.exit(f"Cannot access {LLM_NAME}: {type(e).__name__}: {e}\n"
"Check that access was granted on the model page and the token can read gated repos.")
t = time.time()
snapshot_download(LLM_NAME, allow_patterns=["*.json", "*.safetensors", "tokenizer*", "*.txt", "*.model", "*.jinja"])
log(f"LLM downloaded ({time.time() - t:.0f}s)")
t = time.time()
snapshot_download(VISION_NAME)
log(f"vision encoder downloaded ({time.time() - t:.0f}s)")
t = time.time()
os.makedirs(DATA_DIR, exist_ok=True)
hf_hub_download("liuhaotian/LLaVA-Pretrain", "blip_laion_cc_sbu_558k.json", repo_type="dataset", local_dir=DATA_DIR)
hf_hub_download("liuhaotian/LLaVA-Pretrain", "images.zip", repo_type="dataset", local_dir=DATA_DIR)
size_gb = os.path.getsize(f"{DATA_DIR}/images.zip") / 1e9
log(f"dataset downloaded: images.zip {size_gb:.1f} GB ({time.time() - t:.0f}s)")
|