Download train_sft_hf_cloud.py from yangluo/univer-api-training-scripts: direct link, hf CLI and curl.
- Browser
- Download file 13.3 kB
-
https://huggingface.co/yangluo/univer-api-training-scripts/resolve/main/train_sft_hf_cloud.py
- Command line
-
hf download hf://yangluo/univer-api-training-scripts/train_sft_hf_cloud.py
-
curl -L -o train_sft_hf_cloud.py https://huggingface.co/yangluo/univer-api-training-scripts/resolve/main/train_sft_hf_cloud.py
13.3 kB
| #!/usr/bin/env python3 | |
| # /// script | |
| # requires-python = ">=3.10" | |
| # dependencies = [ | |
| # "torch", | |
| # "trl>=0.12.0", | |
| # "peft>=0.7.0", | |
| # "transformers>=4.36.0", | |
| # "accelerate>=0.24.0", | |
| # "bitsandbytes", | |
| # "datasets", | |
| # "trackio", | |
| # ] | |
| # /// | |
| """ | |
| SFT Training Script for Hugging Face Jobs (Cloud Training) | |
| This script is designed to run on Hugging Face Jobs infrastructure. | |
| It uses PEP 723 inline dependencies for automatic installation. | |
| Supports both Dense and MoE models with 8-bit quantization. | |
| Usage: | |
| # Submit via hf_jobs MCP tool | |
| hf_jobs("uv", { | |
| "script": "https://huggingface.co/yangluo/univer-api-training-scripts/resolve/main/train_sft_hf_cloud.py", | |
| "flavor": "a100-80gb-x1", | |
| "timeout": "4h", | |
| "secrets": {"HF_TOKEN": "$HF_TOKEN"}, | |
| }) | |
| # Or submit via Makefile | |
| make train-cloud | |
| """ | |
| import os | |
| import torch | |
| import trackio | |
| from datasets import load_dataset | |
| from huggingface_hub import login | |
| from peft import LoraConfig | |
| from transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig | |
| from trl import SFTConfig, SFTTrainer | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # CONFIGURATION - Modify these values as needed | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Model - Qwen3-Coder-Next is an 80B MoE model with 3B active params | |
| MODEL_NAME = "Qwen/Qwen3-Coder-Next" | |
| # Dataset | |
| DATASET_NAME = "yangluo/univer-api-sft" | |
| # Output - Where to push the trained model | |
| HUB_MODEL_ID = "yangluo/univer-api-qwen3-next-coder" | |
| # Training hyperparameters (optimized for large MoE model) | |
| NUM_EPOCHS = 2 | |
| BATCH_SIZE = 2 | |
| GRADIENT_ACCUMULATION_STEPS = 8 # Effective batch = 2 * 8 = 16 | |
| LEARNING_RATE = 5e-5 # Lower LR for larger model | |
| # LoRA configuration (Attention-only for MoE) | |
| LORA_R = 16 | |
| LORA_ALPHA = 32 | |
| LORA_DROPOUT = 0.05 | |
| # Only target Attention layers for MoE models | |
| # MLP layers (gate_proj, up_proj, down_proj) are in 512 experts - too many params | |
| LORA_TARGET_MODULES = ["q_proj", "k_proj", "v_proj", "o_proj"] | |
| # Quantization - 8-bit for 80B model on multi-GPU (A100x4) | |
| USE_4BIT = False | |
| USE_8BIT = True | |
| # Trackio monitoring | |
| TRACKIO_PROJECT = "univer-api-training" | |
| TRACKIO_RUN_NAME = "qwen3-coder-next-sft" | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # SYSTEM PROMPT - Used to format training examples | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| SYSTEM_PROMPT = """You are a Univer API expert. Generate JavaScript code to perform spreadsheet operations using the Univer API. | |
| The code should be an async arrow function that uses the univerAPI global object. | |
| IMPORTANT PRINCIPLES: | |
| 1. ALWAYS use spreadsheet formulas (setFormula) for calculations like sum, average, max, min, count, etc. | |
| - Use setFormula('=SUM(...)'), setFormula('=AVERAGE(...)'), etc. | |
| - Do NOT use JavaScript calculations (reduce, Math.max, etc.) for these operations | |
| - Formulas are preferred because they update dynamically when data changes | |
| 2. Use the correct API for the task: | |
| - To delete rows: use deleteRows(), NOT clear() | |
| - To insert rows: use insertRows(), NOT manual approaches | |
| 3. ONLY use documented Univer API methods. Do NOT invent or guess method names.""" | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # TRAINING SCRIPT | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def main(): | |
| # Login to Hugging Face Hub using token from environment | |
| hf_token = os.environ.get("HF_TOKEN") | |
| if hf_token: | |
| print("Logging in to Hugging Face Hub...") | |
| try: | |
| login(token=hf_token, add_to_git_credential=False) | |
| print(" Login successful!") | |
| except Exception as e: | |
| print(f" Login failed: {e}") | |
| print(" Continuing without explicit login (will use default credentials)...") | |
| else: | |
| print("WARNING: HF_TOKEN not found in environment!") | |
| print("Model will NOT be pushed to Hub!") | |
| print("=" * 60) | |
| print(" UNIVER API SFT TRAINING (HF Cloud)") | |
| print("=" * 60) | |
| print(f"Model: {MODEL_NAME}") | |
| print(f"Dataset: {DATASET_NAME}") | |
| print(f"Output: {HUB_MODEL_ID}") | |
| print(f"Epochs: {NUM_EPOCHS}") | |
| print(f"Batch: {BATCH_SIZE} x {GRADIENT_ACCUMULATION_STEPS} = {BATCH_SIZE * GRADIENT_ACCUMULATION_STEPS}") | |
| print(f"LR: {LEARNING_RATE}") | |
| print(f"LoRA: r={LORA_R}, alpha={LORA_ALPHA}") | |
| print(f"Targets: {LORA_TARGET_MODULES}") | |
| print(f"4-bit: {USE_4BIT}") | |
| print(f"8-bit: {USE_8BIT}") | |
| print("=" * 60) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Load Dataset | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| print("\n[1/5] Loading dataset...") | |
| dataset = load_dataset(DATASET_NAME) | |
| print(f" Train: {len(dataset['train'])} examples") | |
| print(f" Validation: {len(dataset['validation'])} examples") | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Load Tokenizer & Model | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| print(f"\n[2/5] Loading model: {MODEL_NAME}") | |
| tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME, trust_remote_code=True) | |
| if tokenizer.pad_token is None: | |
| tokenizer.pad_token = tokenizer.eos_token | |
| # Configure quantization | |
| if USE_4BIT: | |
| print(" Using 4-bit QLoRA quantization...") | |
| bnb_config = BitsAndBytesConfig( | |
| load_in_4bit=True, | |
| bnb_4bit_quant_type="nf4", | |
| bnb_4bit_compute_dtype=torch.bfloat16, | |
| bnb_4bit_use_double_quant=True, | |
| ) | |
| model = AutoModelForCausalLM.from_pretrained( | |
| MODEL_NAME, | |
| quantization_config=bnb_config, | |
| device_map="auto", | |
| trust_remote_code=True, | |
| ) | |
| elif USE_8BIT: | |
| print(" Using 8-bit quantization...") | |
| bnb_config = BitsAndBytesConfig( | |
| load_in_8bit=True, | |
| llm_int8_enable_fp32_cpu_offload=False, # Keep everything on GPU | |
| ) | |
| # For multi-GPU, set max_memory per device | |
| n_gpus = torch.cuda.device_count() | |
| print(f" Detected {n_gpus} GPUs") | |
| if n_gpus > 1: | |
| # Distribute model across GPUs with balanced memory | |
| max_memory = {i: "70GiB" for i in range(n_gpus)} | |
| max_memory["cpu"] = "0GiB" # Don't offload to CPU | |
| print(f" Max memory per GPU: 70GiB") | |
| else: | |
| max_memory = None | |
| model = AutoModelForCausalLM.from_pretrained( | |
| MODEL_NAME, | |
| quantization_config=bnb_config, | |
| device_map="auto", | |
| max_memory=max_memory, | |
| trust_remote_code=True, | |
| ) | |
| else: | |
| model = AutoModelForCausalLM.from_pretrained( | |
| MODEL_NAME, | |
| torch_dtype=torch.bfloat16, | |
| device_map="auto", | |
| trust_remote_code=True, | |
| ) | |
| # Print GPU memory usage | |
| if torch.cuda.is_available(): | |
| allocated = torch.cuda.memory_allocated() / 1024**3 | |
| print(f" Model loaded, GPU memory: {allocated:.1f} GB") | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # LoRA Configuration | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| print("\n[3/5] Configuring LoRA...") | |
| lora_config = LoraConfig( | |
| r=LORA_R, | |
| lora_alpha=LORA_ALPHA, | |
| lora_dropout=LORA_DROPOUT, | |
| target_modules=LORA_TARGET_MODULES, | |
| bias="none", | |
| task_type="CAUSAL_LM", | |
| ) | |
| print(f" LoRA rank: {LORA_R}, alpha: {LORA_ALPHA}") | |
| print(f" Target modules: {LORA_TARGET_MODULES}") | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Format Data | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| print("\n[4/5] Formatting dataset...") | |
| def format_example(example): | |
| messages = [ | |
| {"role": "system", "content": SYSTEM_PROMPT}, | |
| {"role": "user", "content": example["instruction"]}, | |
| {"role": "assistant", "content": example["code"]}, | |
| ] | |
| text = tokenizer.apply_chat_template( | |
| messages, | |
| tokenize=False, | |
| add_generation_prompt=False, | |
| ) | |
| return {"text": text} | |
| train_dataset = dataset["train"].map( | |
| format_example, | |
| remove_columns=dataset["train"].column_names, | |
| ) | |
| eval_dataset = dataset["validation"].map( | |
| format_example, | |
| remove_columns=dataset["validation"].column_names, | |
| ) | |
| print(f" Formatted {len(train_dataset)} training examples") | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Training Configuration | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| print("\n[5/5] Setting up trainer...") | |
| # Calculate total steps for logging | |
| total_steps = (len(train_dataset) // (BATCH_SIZE * GRADIENT_ACCUMULATION_STEPS)) * NUM_EPOCHS | |
| print(f" Estimated total steps: {total_steps}") | |
| training_args = SFTConfig( | |
| # Output & Hub | |
| output_dir="./outputs", | |
| push_to_hub=True, | |
| hub_model_id=HUB_MODEL_ID, | |
| hub_strategy="every_save", | |
| # Training | |
| num_train_epochs=NUM_EPOCHS, | |
| per_device_train_batch_size=BATCH_SIZE, | |
| per_device_eval_batch_size=BATCH_SIZE, | |
| gradient_accumulation_steps=GRADIENT_ACCUMULATION_STEPS, | |
| learning_rate=LEARNING_RATE, | |
| lr_scheduler_type="cosine", | |
| warmup_ratio=0.1, | |
| optim="adamw_torch", | |
| weight_decay=0.01, | |
| max_grad_norm=1.0, | |
| bf16=True, | |
| # Logging | |
| logging_steps=10, | |
| logging_first_step=True, | |
| report_to="trackio", | |
| project=TRACKIO_PROJECT, | |
| run_name=TRACKIO_RUN_NAME, | |
| # Evaluation | |
| eval_strategy="steps", | |
| eval_steps=100, | |
| # Checkpointing | |
| save_strategy="steps", | |
| save_steps=200, | |
| save_total_limit=2, | |
| load_best_model_at_end=True, | |
| metric_for_best_model="eval_loss", | |
| greater_is_better=False, | |
| ) | |
| trainer = SFTTrainer( | |
| model=model, | |
| args=training_args, | |
| train_dataset=train_dataset, | |
| eval_dataset=eval_dataset, | |
| processing_class=tokenizer, | |
| peft_config=lora_config, | |
| ) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Train | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| print("\n" + "=" * 60) | |
| print("Starting training...") | |
| print("=" * 60) | |
| trainer.train() | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Save & Push | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| print("\nSaving and pushing to Hub...") | |
| trainer.save_model() | |
| trainer.push_to_hub() | |
| # Finish Trackio | |
| trackio.finish() | |
| print("\n" + "=" * 60) | |
| print(" TRAINING COMPLETE!") | |
| print("=" * 60) | |
| print(f"Model: https://huggingface.co/{HUB_MODEL_ID}") | |
| print(f"Monitor: https://huggingface.co/spaces/yangluo/trackio") | |
| if __name__ == "__main__": | |
| main() | |