#!/usr/bin/env python3 -u """ FSI_FELON ยท Unified Training Pipeline Usage: python train.py # Train 4.5M model (default) python train.py --size 10m # Train 10M model python train.py --resume # Resume from best checkpoint python train.py --eval-only # Evaluate best model only python train.py --push # Train, then push best to HF """ import os, sys, time, random, gc, json, argparse import torch torch.set_num_threads(4) sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from quantum.gated_conv_engine import FelonGatedConvModel, get_tokenizer from quantum.bpe_tokenizer import BPETokenizer MODEL_CONFIGS = { "4.5m": {"hidden": 192, "n_layers": 8, "n_heads": 6, "n_kv_heads": 3, "inter": 384, "max_seq": 512, "dropout": 0.1, "batch": 8, "lr": 3e-4, "epochs": 30}, "10m": {"hidden": 256, "n_layers": 12, "n_heads": 6, "n_kv_heads": 3, "inter": 512, "max_seq": 512, "dropout": 0.1, "batch": 4, "lr": 3e-4, "epochs": 20}, } def make_model(size, vocab_size=4096): cfg = MODEL_CONFIGS[size] return FelonGatedConvModel( vocab_size=vocab_size, hidden=cfg["hidden"], n_layers=cfg["n_layers"], n_heads=cfg["n_heads"], n_kv_heads=cfg["n_kv_heads"], inter=cfg["inter"], kernel=3, max_seq=cfg["max_seq"], dropout=cfg["dropout"], ) def build_blocks(data_ids, block_size=256): stride = block_size * 2 blocks = [data_ids[i:i + block_size] for i in range(0, len(data_ids) - block_size, stride)] gc.collect() return blocks def train_epoch(model, blocks, optimizer, scheduler, loss_fn, batch_size, epoch, epochs, t_start, best_val, best_path): model.train() total, n = 0.0, 0 random.shuffle(blocks) t0 = time.time() for i in range(0, len(blocks) - batch_size, batch_size): batch = blocks[i:i + batch_size] x = torch.tensor([b[:-1] for b in batch], dtype=torch.long) y = torch.tensor([b[1:] for b in batch], dtype=torch.long) logits = model(x) loss = loss_fn(logits.view(-1, logits.size(-1)), y.view(-1)) optimizer.zero_grad() loss.backward() torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0) optimizer.step() scheduler.step() total += loss.item() n += 1 if n % 100 == 0: elapsed = time.time() - t0 print(f" Ep{epoch+1} s{n}: loss={total/n:.4f} [{elapsed:.0f}s]", flush=True) train_loss = total / max(n, 1) model.eval() with torch.no_grad(): vb = blocks[:min(batch_size, len(blocks))] vx = torch.tensor([b[:-1] for b in vb], dtype=torch.long) vy = torch.tensor([b[1:] for b in vb], dtype=torch.long) vl = loss_fn(model(vx).view(-1, model.head.out_features), vy.view(-1)).item() if vl < best_val: best_val = vl print(f" *** New best val_loss={vl:.4f} ***", flush=True) torch.save({'state': model.state_dict(), 'optimizer': optimizer.state_dict(), 'epoch': epoch + 1, 'step': n, 'best_val': best_val, 'config': {'hidden': model.hidden, 'n_layers': len(model.blocks), 'n_heads': 6, 'n_kv_heads': 3, 'inter': 512, 'max_seq': model.max_seq, 'vocab_size': model.head.out_features}}, best_path) elapsed_total = time.time() - t_start eta = (elapsed_total / (epoch + 1)) * (epochs - epoch - 1) if epoch < epochs - 1 else 0 print(f" Ep{epoch+1}/{epochs} | train={train_loss:.4f} val={vl:.4f} best={best_val:.4f} | {elapsed_total/3600:.1f}h total | ETA {eta/3600:.1f}h", flush=True) gc.collect() return best_val def train(size="4.5m", resume=False, corpus_path="felon_deep_corpus.txt", max_tokens=None): cfg = MODEL_CONFIGS[size] best_path = best_path_for(size) final_path = f"felon_{size}_codegen_final.pt" tok = BPETokenizer(vocab_size=4096) tok_path = "quantum/felon_bpe_tokenizer.json" if os.path.exists(tok_path): tok.load(tok_path) print(f" Tokenizer: {tok.vocab_size_actual} tokens", flush=True) if not os.path.exists(corpus_path): print(f" Generating corpus...", flush=True) import gen_code_corpus gen_code_corpus.main() with open(corpus_path) as f: text = f.read() all_ids = tok.encode(text) if max_tokens and len(all_ids) > max_tokens: all_ids = all_ids[:max_tokens] print(f" Corpus: {len(all_ids):,} tokens ({len(text):,} chars)", flush=True) blocks = build_blocks(all_ids) n_val = max(1, len(blocks) // 10) random.shuffle(blocks) val_blocks = blocks[:n_val] train_blocks = blocks[n_val:] print(f" Blocks: {len(train_blocks):,} train + {len(val_blocks)} val", flush=True) model = make_model(size) print(f" Params: {sum(p.numel() for p in model.parameters()):,}", flush=True) start_epoch = 0 best_val = float('inf') ckpt = None if resume and os.path.exists(best_path): ckpt = torch.load(best_path, map_location='cpu', weights_only=False) state_dict = ckpt.get('state', ckpt.get('state_dict', ckpt)) model.load_state_dict(state_dict, strict=False) start_epoch = ckpt.get('epoch', 1) if 'epoch' in ckpt else 1 best_val = ckpt.get('best_val', float('inf')) print(f" Resumed (model loaded, optimizer cold start) epoch={start_epoch} best_val={best_val:.4f}", flush=True) optimizer = torch.optim.AdamW(model.parameters(), lr=cfg["lr"], weight_decay=0.01) if ckpt and 'optimizer' in ckpt: optimizer.load_state_dict(ckpt['optimizer']) print(f" Optimizer state restored", flush=True) total_steps = cfg["epochs"] * max(1, len(train_blocks) // cfg["batch"]) scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=total_steps) loss_fn = torch.nn.CrossEntropyLoss() t_start = time.time() for epoch in range(start_epoch, cfg["epochs"]): try: best_val = train_epoch(model, train_blocks, optimizer, scheduler, loss_fn, cfg["batch"], epoch, cfg["epochs"], t_start, best_val, best_path) except Exception as e: import traceback traceback.print_exc() print(f" Epoch {epoch+1} failed: {e}", flush=True) print(f" Saving emergency checkpoint...", flush=True) torch.save({'state': model.state_dict(), 'optimizer': optimizer.state_dict(), 'epoch': epoch, 'best_val': best_val, 'config': {'hidden': model.hidden, 'n_layers': len(model.blocks), 'n_heads': 6, 'n_kv_heads': 3, 'inter': 512, 'max_seq': model.max_seq, 'vocab_size': model.head.out_features}}, best_path.replace('.pt', '_emergency.pt')) raise save_config = { 'hidden': getattr(model, 'hidden', 256), 'n_layers': len(getattr(model, 'blocks', [])), 'n_heads': 6, 'n_kv_heads': 3, 'inter': 512, 'max_seq': getattr(model, 'max_seq', 512), 'vocab_size': getattr(getattr(model, 'head', None), 'out_features', 4096) } torch.save({'state': model.state_dict(), 'optimizer': optimizer.state_dict(), 'epoch': cfg["epochs"], 'best_val': best_val, 'config': save_config}, final_path) print(f" Training complete. Best: {best_path} Final: {final_path}", flush=True) return best_path def best_path_for(size): p = f"felon_{size}_codegen_best.pt" if os.path.exists(p): return p legacy = {"4.5m": "felon_codegen_best.pt", "10m": "felon_10m_codegen_best.pt"} if size in legacy: return legacy[size] return p def eval_model(size="4.5m"): best_path = best_path_for(size) if not os.path.exists(best_path): print(f" No checkpoint at {best_path}") return cfg = MODEL_CONFIGS[size] model = make_model(size) ckpt = torch.load(best_path, map_location='cpu', weights_only=False) model.load_state_dict(ckpt['state'], strict=False) model.eval() tok = BPETokenizer(vocab_size=4096) tok.load("quantum/felon_bpe_tokenizer.json") print(f"\n{'='*55}", flush=True) print(f" EVALUATING {size.upper()} MODEL", flush=True) print(f"{'='*55}", flush=True) prompts = [ "DESCRIPTION: function to add two numbers\nCODE:\n", "DESCRIPTION: REST API with FastAPI\nCODE:\n", "DESCRIPTION: CLI tool with argparse\nCODE:\n", "DESCRIPTION: class for a binary search tree\nCODE:\n", "DESCRIPTION: respond to user\nCODE:\n", ] for prompt in prompts: ids = tok.encode(prompt) out = model.generate(ids, max_new=100, temp=0.6, top_k=30, eos_id=tok.EOS) text = tok.decode(out) print(f"\n >>> {prompt.split(chr(10))[0][:60]}", flush=True) print(f" {text[:250]}", flush=True) print(f" ---", flush=True) def push_to_hf(size="4.5m"): best_path = f"felon_{size}_codegen_best.pt" if not os.path.exists(best_path): print(f" No checkpoint at {best_path}, skipping push") return try: from huggingface_hub import HfApi api = HfApi() repo_id = "FerrellSyntheticIntelligence/FSI_FELON" api.upload_file(path_or_fileobj=best_path, path_in_repo=best_path, repo_id=repo_id, repo_type="model", commit_message=f"auto-push {best_path} from training pipeline") print(f" Pushed {best_path} to HF", flush=True) except Exception as e: print(f" Push failed: {e}", flush=True) if __name__ == "__main__": parser = argparse.ArgumentParser(description="FSI_FELON Training Pipeline") parser.add_argument("--size", choices=["4.5m", "10m"], default="4.5m") parser.add_argument("--resume", action="store_true", help="Resume from best checkpoint") parser.add_argument("--eval-only", action="store_true", help="Evaluate best model only") parser.add_argument("--push", action="store_true", help="Push best checkpoint to HF") parser.add_argument("--max-tokens", type=int, default=None, help="Max tokens to use") args = parser.parse_args() if args.eval_only: eval_model(args.size) else: try: best = train(args.size, resume=args.resume, max_tokens=args.max_tokens) except Exception as e: import traceback traceback.print_exc() print(f"\n Training failed: {e}", flush=True) print(f" You can resume with: python3 train.py --size {args.size} --resume", flush=True) sys.exit(1) eval_model(args.size) if args.push: push_to_hf(args.size)