univer-api-training-scripts / train_sft_hf_cloud.py
yangluo's picture
Upload train_sft_hf_cloud.py with huggingface_hub
72b4aee verified
Raw History Blame Contribute Delete
13.3 kB
#!/usr/bin/env python3
# /// script
# requires-python = ">=3.10"
# dependencies = [
# "torch",
# "trl>=0.12.0",
# "peft>=0.7.0",
# "transformers>=4.36.0",
# "accelerate>=0.24.0",
# "bitsandbytes",
# "datasets",
# "trackio",
# ]
# ///
"""
SFT Training Script for Hugging Face Jobs (Cloud Training)
This script is designed to run on Hugging Face Jobs infrastructure.
It uses PEP 723 inline dependencies for automatic installation.
Supports both Dense and MoE models with 8-bit quantization.
Usage:
# Submit via hf_jobs MCP tool
hf_jobs("uv", {
"script": "https://huggingface.co/yangluo/univer-api-training-scripts/resolve/main/train_sft_hf_cloud.py",
"flavor": "a100-80gb-x1",
"timeout": "4h",
"secrets": {"HF_TOKEN": "$HF_TOKEN"},
})
# Or submit via Makefile
make train-cloud
"""
import os
import torch
import trackio
from datasets import load_dataset
from huggingface_hub import login
from peft import LoraConfig
from transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig
from trl import SFTConfig, SFTTrainer
# ═══════════════════════════════════════════════════════════════
# CONFIGURATION - Modify these values as needed
# ═══════════════════════════════════════════════════════════════
# Model - Qwen3-Coder-Next is an 80B MoE model with 3B active params
MODEL_NAME = "Qwen/Qwen3-Coder-Next"
# Dataset
DATASET_NAME = "yangluo/univer-api-sft"
# Output - Where to push the trained model
HUB_MODEL_ID = "yangluo/univer-api-qwen3-next-coder"
# Training hyperparameters (optimized for large MoE model)
NUM_EPOCHS = 2
BATCH_SIZE = 2
GRADIENT_ACCUMULATION_STEPS = 8 # Effective batch = 2 * 8 = 16
LEARNING_RATE = 5e-5 # Lower LR for larger model
# LoRA configuration (Attention-only for MoE)
LORA_R = 16
LORA_ALPHA = 32
LORA_DROPOUT = 0.05
# Only target Attention layers for MoE models
# MLP layers (gate_proj, up_proj, down_proj) are in 512 experts - too many params
LORA_TARGET_MODULES = ["q_proj", "k_proj", "v_proj", "o_proj"]
# Quantization - 8-bit for 80B model on multi-GPU (A100x4)
USE_4BIT = False
USE_8BIT = True
# Trackio monitoring
TRACKIO_PROJECT = "univer-api-training"
TRACKIO_RUN_NAME = "qwen3-coder-next-sft"
# ═══════════════════════════════════════════════════════════════
# SYSTEM PROMPT - Used to format training examples
# ═══════════════════════════════════════════════════════════════
SYSTEM_PROMPT = """You are a Univer API expert. Generate JavaScript code to perform spreadsheet operations using the Univer API.
The code should be an async arrow function that uses the univerAPI global object.
IMPORTANT PRINCIPLES:
1. ALWAYS use spreadsheet formulas (setFormula) for calculations like sum, average, max, min, count, etc.
- Use setFormula('=SUM(...)'), setFormula('=AVERAGE(...)'), etc.
- Do NOT use JavaScript calculations (reduce, Math.max, etc.) for these operations
- Formulas are preferred because they update dynamically when data changes
2. Use the correct API for the task:
- To delete rows: use deleteRows(), NOT clear()
- To insert rows: use insertRows(), NOT manual approaches
3. ONLY use documented Univer API methods. Do NOT invent or guess method names."""
# ═══════════════════════════════════════════════════════════════
# TRAINING SCRIPT
# ═══════════════════════════════════════════════════════════════
def main():
# Login to Hugging Face Hub using token from environment
hf_token = os.environ.get("HF_TOKEN")
if hf_token:
print("Logging in to Hugging Face Hub...")
try:
login(token=hf_token, add_to_git_credential=False)
print(" Login successful!")
except Exception as e:
print(f" Login failed: {e}")
print(" Continuing without explicit login (will use default credentials)...")
else:
print("WARNING: HF_TOKEN not found in environment!")
print("Model will NOT be pushed to Hub!")
print("=" * 60)
print(" UNIVER API SFT TRAINING (HF Cloud)")
print("=" * 60)
print(f"Model: {MODEL_NAME}")
print(f"Dataset: {DATASET_NAME}")
print(f"Output: {HUB_MODEL_ID}")
print(f"Epochs: {NUM_EPOCHS}")
print(f"Batch: {BATCH_SIZE} x {GRADIENT_ACCUMULATION_STEPS} = {BATCH_SIZE * GRADIENT_ACCUMULATION_STEPS}")
print(f"LR: {LEARNING_RATE}")
print(f"LoRA: r={LORA_R}, alpha={LORA_ALPHA}")
print(f"Targets: {LORA_TARGET_MODULES}")
print(f"4-bit: {USE_4BIT}")
print(f"8-bit: {USE_8BIT}")
print("=" * 60)
# ─────────────────────────────────────────────────────
# Load Dataset
# ─────────────────────────────────────────────────────
print("\n[1/5] Loading dataset...")
dataset = load_dataset(DATASET_NAME)
print(f" Train: {len(dataset['train'])} examples")
print(f" Validation: {len(dataset['validation'])} examples")
# ─────────────────────────────────────────────────────
# Load Tokenizer & Model
# ─────────────────────────────────────────────────────
print(f"\n[2/5] Loading model: {MODEL_NAME}")
tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME, trust_remote_code=True)
if tokenizer.pad_token is None:
tokenizer.pad_token = tokenizer.eos_token
# Configure quantization
if USE_4BIT:
print(" Using 4-bit QLoRA quantization...")
bnb_config = BitsAndBytesConfig(
load_in_4bit=True,
bnb_4bit_quant_type="nf4",
bnb_4bit_compute_dtype=torch.bfloat16,
bnb_4bit_use_double_quant=True,
)
model = AutoModelForCausalLM.from_pretrained(
MODEL_NAME,
quantization_config=bnb_config,
device_map="auto",
trust_remote_code=True,
)
elif USE_8BIT:
print(" Using 8-bit quantization...")
bnb_config = BitsAndBytesConfig(
load_in_8bit=True,
llm_int8_enable_fp32_cpu_offload=False, # Keep everything on GPU
)
# For multi-GPU, set max_memory per device
n_gpus = torch.cuda.device_count()
print(f" Detected {n_gpus} GPUs")
if n_gpus > 1:
# Distribute model across GPUs with balanced memory
max_memory = {i: "70GiB" for i in range(n_gpus)}
max_memory["cpu"] = "0GiB" # Don't offload to CPU
print(f" Max memory per GPU: 70GiB")
else:
max_memory = None
model = AutoModelForCausalLM.from_pretrained(
MODEL_NAME,
quantization_config=bnb_config,
device_map="auto",
max_memory=max_memory,
trust_remote_code=True,
)
else:
model = AutoModelForCausalLM.from_pretrained(
MODEL_NAME,
torch_dtype=torch.bfloat16,
device_map="auto",
trust_remote_code=True,
)
# Print GPU memory usage
if torch.cuda.is_available():
allocated = torch.cuda.memory_allocated() / 1024**3
print(f" Model loaded, GPU memory: {allocated:.1f} GB")
# ─────────────────────────────────────────────────────
# LoRA Configuration
# ─────────────────────────────────────────────────────
print("\n[3/5] Configuring LoRA...")
lora_config = LoraConfig(
r=LORA_R,
lora_alpha=LORA_ALPHA,
lora_dropout=LORA_DROPOUT,
target_modules=LORA_TARGET_MODULES,
bias="none",
task_type="CAUSAL_LM",
)
print(f" LoRA rank: {LORA_R}, alpha: {LORA_ALPHA}")
print(f" Target modules: {LORA_TARGET_MODULES}")
# ─────────────────────────────────────────────────────
# Format Data
# ─────────────────────────────────────────────────────
print("\n[4/5] Formatting dataset...")
def format_example(example):
messages = [
{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user", "content": example["instruction"]},
{"role": "assistant", "content": example["code"]},
]
text = tokenizer.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=False,
)
return {"text": text}
train_dataset = dataset["train"].map(
format_example,
remove_columns=dataset["train"].column_names,
)
eval_dataset = dataset["validation"].map(
format_example,
remove_columns=dataset["validation"].column_names,
)
print(f" Formatted {len(train_dataset)} training examples")
# ─────────────────────────────────────────────────────
# Training Configuration
# ─────────────────────────────────────────────────────
print("\n[5/5] Setting up trainer...")
# Calculate total steps for logging
total_steps = (len(train_dataset) // (BATCH_SIZE * GRADIENT_ACCUMULATION_STEPS)) * NUM_EPOCHS
print(f" Estimated total steps: {total_steps}")
training_args = SFTConfig(
# Output & Hub
output_dir="./outputs",
push_to_hub=True,
hub_model_id=HUB_MODEL_ID,
hub_strategy="every_save",
# Training
num_train_epochs=NUM_EPOCHS,
per_device_train_batch_size=BATCH_SIZE,
per_device_eval_batch_size=BATCH_SIZE,
gradient_accumulation_steps=GRADIENT_ACCUMULATION_STEPS,
learning_rate=LEARNING_RATE,
lr_scheduler_type="cosine",
warmup_ratio=0.1,
optim="adamw_torch",
weight_decay=0.01,
max_grad_norm=1.0,
bf16=True,
# Logging
logging_steps=10,
logging_first_step=True,
report_to="trackio",
project=TRACKIO_PROJECT,
run_name=TRACKIO_RUN_NAME,
# Evaluation
eval_strategy="steps",
eval_steps=100,
# Checkpointing
save_strategy="steps",
save_steps=200,
save_total_limit=2,
load_best_model_at_end=True,
metric_for_best_model="eval_loss",
greater_is_better=False,
)
trainer = SFTTrainer(
model=model,
args=training_args,
train_dataset=train_dataset,
eval_dataset=eval_dataset,
processing_class=tokenizer,
peft_config=lora_config,
)
# ─────────────────────────────────────────────────────
# Train
# ─────────────────────────────────────────────────────
print("\n" + "=" * 60)
print("Starting training...")
print("=" * 60)
trainer.train()
# ─────────────────────────────────────────────────────
# Save & Push
# ─────────────────────────────────────────────────────
print("\nSaving and pushing to Hub...")
trainer.save_model()
trainer.push_to_hub()
# Finish Trackio
trackio.finish()
print("\n" + "=" * 60)
print(" TRAINING COMPLETE!")
print("=" * 60)
print(f"Model: https://huggingface.co/{HUB_MODEL_ID}")
print(f"Monitor: https://huggingface.co/spaces/yangluo/trackio")
if __name__ == "__main__":
main()