#!/usr/bin/env python3 # /// script # requires-python = ">=3.10" # dependencies = [ # "torch", # "trl>=0.12.0", # "peft>=0.7.0", # "transformers>=4.36.0", # "accelerate>=0.24.0", # "bitsandbytes", # "datasets", # "trackio", # ] # /// """ SFT Training Script for Hugging Face Jobs (Cloud Training) This script is designed to run on Hugging Face Jobs infrastructure. It uses PEP 723 inline dependencies for automatic installation. Supports both Dense and MoE models with 8-bit quantization. Usage: # Submit via hf_jobs MCP tool hf_jobs("uv", { "script": "https://huggingface.co/yangluo/univer-api-training-scripts/resolve/main/train_sft_hf_cloud.py", "flavor": "a100-80gb-x1", "timeout": "4h", "secrets": {"HF_TOKEN": "$HF_TOKEN"}, }) # Or submit via Makefile make train-cloud """ import os import torch import trackio from datasets import load_dataset from huggingface_hub import login from peft import LoraConfig from transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig from trl import SFTConfig, SFTTrainer # ═══════════════════════════════════════════════════════════════ # CONFIGURATION - Modify these values as needed # ═══════════════════════════════════════════════════════════════ # Model - Qwen3-Coder-Next is an 80B MoE model with 3B active params MODEL_NAME = "Qwen/Qwen3-Coder-Next" # Dataset DATASET_NAME = "yangluo/univer-api-sft" # Output - Where to push the trained model HUB_MODEL_ID = "yangluo/univer-api-qwen3-next-coder" # Training hyperparameters (optimized for large MoE model) NUM_EPOCHS = 2 BATCH_SIZE = 2 GRADIENT_ACCUMULATION_STEPS = 8 # Effective batch = 2 * 8 = 16 LEARNING_RATE = 5e-5 # Lower LR for larger model # LoRA configuration (Attention-only for MoE) LORA_R = 16 LORA_ALPHA = 32 LORA_DROPOUT = 0.05 # Only target Attention layers for MoE models # MLP layers (gate_proj, up_proj, down_proj) are in 512 experts - too many params LORA_TARGET_MODULES = ["q_proj", "k_proj", "v_proj", "o_proj"] # Quantization - 8-bit for 80B model on multi-GPU (A100x4) USE_4BIT = False USE_8BIT = True # Trackio monitoring TRACKIO_PROJECT = "univer-api-training" TRACKIO_RUN_NAME = "qwen3-coder-next-sft" # ═══════════════════════════════════════════════════════════════ # SYSTEM PROMPT - Used to format training examples # ═══════════════════════════════════════════════════════════════ SYSTEM_PROMPT = """You are a Univer API expert. Generate JavaScript code to perform spreadsheet operations using the Univer API. The code should be an async arrow function that uses the univerAPI global object. IMPORTANT PRINCIPLES: 1. ALWAYS use spreadsheet formulas (setFormula) for calculations like sum, average, max, min, count, etc. - Use setFormula('=SUM(...)'), setFormula('=AVERAGE(...)'), etc. - Do NOT use JavaScript calculations (reduce, Math.max, etc.) for these operations - Formulas are preferred because they update dynamically when data changes 2. Use the correct API for the task: - To delete rows: use deleteRows(), NOT clear() - To insert rows: use insertRows(), NOT manual approaches 3. ONLY use documented Univer API methods. Do NOT invent or guess method names.""" # ═══════════════════════════════════════════════════════════════ # TRAINING SCRIPT # ═══════════════════════════════════════════════════════════════ def main(): # Login to Hugging Face Hub using token from environment hf_token = os.environ.get("HF_TOKEN") if hf_token: print("Logging in to Hugging Face Hub...") try: login(token=hf_token, add_to_git_credential=False) print(" Login successful!") except Exception as e: print(f" Login failed: {e}") print(" Continuing without explicit login (will use default credentials)...") else: print("WARNING: HF_TOKEN not found in environment!") print("Model will NOT be pushed to Hub!") print("=" * 60) print(" UNIVER API SFT TRAINING (HF Cloud)") print("=" * 60) print(f"Model: {MODEL_NAME}") print(f"Dataset: {DATASET_NAME}") print(f"Output: {HUB_MODEL_ID}") print(f"Epochs: {NUM_EPOCHS}") print(f"Batch: {BATCH_SIZE} x {GRADIENT_ACCUMULATION_STEPS} = {BATCH_SIZE * GRADIENT_ACCUMULATION_STEPS}") print(f"LR: {LEARNING_RATE}") print(f"LoRA: r={LORA_R}, alpha={LORA_ALPHA}") print(f"Targets: {LORA_TARGET_MODULES}") print(f"4-bit: {USE_4BIT}") print(f"8-bit: {USE_8BIT}") print("=" * 60) # ───────────────────────────────────────────────────── # Load Dataset # ───────────────────────────────────────────────────── print("\n[1/5] Loading dataset...") dataset = load_dataset(DATASET_NAME) print(f" Train: {len(dataset['train'])} examples") print(f" Validation: {len(dataset['validation'])} examples") # ───────────────────────────────────────────────────── # Load Tokenizer & Model # ───────────────────────────────────────────────────── print(f"\n[2/5] Loading model: {MODEL_NAME}") tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME, trust_remote_code=True) if tokenizer.pad_token is None: tokenizer.pad_token = tokenizer.eos_token # Configure quantization if USE_4BIT: print(" Using 4-bit QLoRA quantization...") bnb_config = BitsAndBytesConfig( load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16, bnb_4bit_use_double_quant=True, ) model = AutoModelForCausalLM.from_pretrained( MODEL_NAME, quantization_config=bnb_config, device_map="auto", trust_remote_code=True, ) elif USE_8BIT: print(" Using 8-bit quantization...") bnb_config = BitsAndBytesConfig( load_in_8bit=True, llm_int8_enable_fp32_cpu_offload=False, # Keep everything on GPU ) # For multi-GPU, set max_memory per device n_gpus = torch.cuda.device_count() print(f" Detected {n_gpus} GPUs") if n_gpus > 1: # Distribute model across GPUs with balanced memory max_memory = {i: "70GiB" for i in range(n_gpus)} max_memory["cpu"] = "0GiB" # Don't offload to CPU print(f" Max memory per GPU: 70GiB") else: max_memory = None model = AutoModelForCausalLM.from_pretrained( MODEL_NAME, quantization_config=bnb_config, device_map="auto", max_memory=max_memory, trust_remote_code=True, ) else: model = AutoModelForCausalLM.from_pretrained( MODEL_NAME, torch_dtype=torch.bfloat16, device_map="auto", trust_remote_code=True, ) # Print GPU memory usage if torch.cuda.is_available(): allocated = torch.cuda.memory_allocated() / 1024**3 print(f" Model loaded, GPU memory: {allocated:.1f} GB") # ───────────────────────────────────────────────────── # LoRA Configuration # ───────────────────────────────────────────────────── print("\n[3/5] Configuring LoRA...") lora_config = LoraConfig( r=LORA_R, lora_alpha=LORA_ALPHA, lora_dropout=LORA_DROPOUT, target_modules=LORA_TARGET_MODULES, bias="none", task_type="CAUSAL_LM", ) print(f" LoRA rank: {LORA_R}, alpha: {LORA_ALPHA}") print(f" Target modules: {LORA_TARGET_MODULES}") # ───────────────────────────────────────────────────── # Format Data # ───────────────────────────────────────────────────── print("\n[4/5] Formatting dataset...") def format_example(example): messages = [ {"role": "system", "content": SYSTEM_PROMPT}, {"role": "user", "content": example["instruction"]}, {"role": "assistant", "content": example["code"]}, ] text = tokenizer.apply_chat_template( messages, tokenize=False, add_generation_prompt=False, ) return {"text": text} train_dataset = dataset["train"].map( format_example, remove_columns=dataset["train"].column_names, ) eval_dataset = dataset["validation"].map( format_example, remove_columns=dataset["validation"].column_names, ) print(f" Formatted {len(train_dataset)} training examples") # ───────────────────────────────────────────────────── # Training Configuration # ───────────────────────────────────────────────────── print("\n[5/5] Setting up trainer...") # Calculate total steps for logging total_steps = (len(train_dataset) // (BATCH_SIZE * GRADIENT_ACCUMULATION_STEPS)) * NUM_EPOCHS print(f" Estimated total steps: {total_steps}") training_args = SFTConfig( # Output & Hub output_dir="./outputs", push_to_hub=True, hub_model_id=HUB_MODEL_ID, hub_strategy="every_save", # Training num_train_epochs=NUM_EPOCHS, per_device_train_batch_size=BATCH_SIZE, per_device_eval_batch_size=BATCH_SIZE, gradient_accumulation_steps=GRADIENT_ACCUMULATION_STEPS, learning_rate=LEARNING_RATE, lr_scheduler_type="cosine", warmup_ratio=0.1, optim="adamw_torch", weight_decay=0.01, max_grad_norm=1.0, bf16=True, # Logging logging_steps=10, logging_first_step=True, report_to="trackio", project=TRACKIO_PROJECT, run_name=TRACKIO_RUN_NAME, # Evaluation eval_strategy="steps", eval_steps=100, # Checkpointing save_strategy="steps", save_steps=200, save_total_limit=2, load_best_model_at_end=True, metric_for_best_model="eval_loss", greater_is_better=False, ) trainer = SFTTrainer( model=model, args=training_args, train_dataset=train_dataset, eval_dataset=eval_dataset, processing_class=tokenizer, peft_config=lora_config, ) # ───────────────────────────────────────────────────── # Train # ───────────────────────────────────────────────────── print("\n" + "=" * 60) print("Starting training...") print("=" * 60) trainer.train() # ───────────────────────────────────────────────────── # Save & Push # ───────────────────────────────────────────────────── print("\nSaving and pushing to Hub...") trainer.save_model() trainer.push_to_hub() # Finish Trackio trackio.finish() print("\n" + "=" * 60) print(" TRAINING COMPLETE!") print("=" * 60) print(f"Model: https://huggingface.co/{HUB_MODEL_ID}") print(f"Monitor: https://huggingface.co/spaces/yangluo/trackio") if __name__ == "__main__": main()