"""upload_v6_5_7ds.py — V6.5-7ds upload to HuggingFace. User requirement (latest): "upload dos arquivos testados e os aprimorados corrigidos (remover os antigos)" This script: 1. Pre-upload safety: scrub HF_TOKEN from all scripts/*.py via regex. 2. Defines OLD_FILES_TO_REMOVE — legacy v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive reports and scripts, plus old v6_5_* files (superseded by v6_5_7ds_*). 3. Uploads the tested+improved files (current BiGRU_T_version tree). 4. Deletes OLD_FILES_TO_REMOVE from the HF repo via HfApi.delete_file. 5. Verifies critical V6.5-7ds files are present in repo. 6. Saves upload report to v6_5_7ds_upload_report.json. After successful upload, the HF_TOKEN env var MUST be deleted by the caller. """ from __future__ import annotations import json, os, sys, time, re from pathlib import Path PROJECT_ROOT = Path("/home/z/my-project") BIGRU_ROOT = PROJECT_ROOT / "BiGRU_T_version" # ────────────────────────────────────────────────────────────────────────────── # Files to REMOVE from the HF repo (legacy / superseded by V6.5-7ds) # ────────────────────────────────────────────────────────────────────────────── # User requirement: "upload dos arquivos testados e os aprimorados corrigidos # (remover os antigos)" # # Removal strategy: # - All legacy version reports (v3, v4, v5, v6, v6_1, v6_2, v6_3, v6_progressive) # - All legacy upload reports (v3, v5, v6_1, v6_2, v6_4, v6_5) # - Old V6.5 files superseded by V6.5-7ds (v6_5_*.json and v6_5_final_*.json) # - Old training scripts (train.py, train_v2.py, train_fast.py, smoke_test.py, # parse_v6_log.py, train_v6_1.py, train_v6_2.py, train_v6_3.py, # train_v6_progressive.py, train_v6_5.py, train_v6_5_final.py) # - Old upload scripts (upload_to_hf.py, upload_v6_1_resilient.py, # upload_v6_2_resilient.py, upload_v6_3_resilient.py, upload_v6_4_resilient.py, # upload_v6_5_resilient.py) # - Legacy root files: bug_hunt_v3_report.json, v32_test_report.json, # v32_upload_report.json, training_report.json, v6_progressive_train.log, # v6_progressive_report.json, v6_report.json, v6_upload_report.json, # config.json (legacy model_final/config.json stays) OLD_FILES_TO_REMOVE = [ # ── Legacy root reports (v3/v4/v5/v6/v32) ───────────────────────────────── "bug_hunt_v3_report.json", "v3_report.json", "v4_report.json", "v5_report.json", "v5_upload_report.json", "v32_test_report.json", "v32_upload_report.json", "training_report.json", "v6_report.json", "v6_upload_report.json", "v6_progressive_report.json", "v6_progressive_train.log", # ── Legacy v6.1/v6.2/v6.3 reports ───────────────────────────────────────── "v6_1_report.json", "v6_1_upload_report.json", "v6_2_report.json", "v6_2_upload_report.json", "v6_3_report.json", "v6_3_training_metrics.json", # ── Old V6.5 reports (superseded by V6.5-7ds) ───────────────────────────── # These are removed because V6.5-7ds is the new canonical version. "v6_5_report.json", "v6_5_training_metrics.json", "v6_5_module_analysis.json", "v6_5_script_activity.json", "v6_5_ewc_w8a8_benchmark.json", "v6_5_reasoning_eval.json", "v6_5_upload_report.json", # ── Old V6.5-final reports (also superseded by V6.5-7ds) ────────────────── "v6_5_final_report.json", "v6_5_final_training_metrics.json", "v6_5_final_module_analysis.json", "v6_5_final_script_activity.json", "v6_5_final_ewc_w8a8_benchmark.json", "v6_5_final_reasoning_eval.json", "v6_5_final_w8a8_compression.json", # ── Legacy training scripts ─────────────────────────────────────────────── "scripts/parse_v6_log.py", "scripts/smoke_test.py", "scripts/train.py", "scripts/train_fast.py", "scripts/train_v2.py", "scripts/train_v6_1.py", "scripts/train_v6_2.py", "scripts/train_v6_3.py", "scripts/train_v6_progressive.py", "scripts/train_v6_5.py", # superseded by train_v6_5_7ds.py "scripts/train_v6_5_final.py", # superseded by train_v6_5_7ds.py # ── Legacy upload scripts (only upload_v6_5_7ds.py stays) ───────────────── "scripts/upload_to_hf.py", "scripts/upload_v6_1_resilient.py", "scripts/upload_v6_2_resilient.py", "scripts/upload_v6_3_resilient.py", "scripts/upload_v6_4_resilient.py", "scripts/upload_v6_5_resilient.py", # superseded by upload_v6_5_7ds.py # ── Legacy tokenizer/ root config (kept model_final/ tree intact) ───────── "config.json", # legacy root config; not used by V6.5 "tokenizer/tokenizer.json", # legacy root tokenizer; model_final/ has its own # ── Legacy data_augmentation.py (was moved to training/ in V6.4) ────────── "src/bigru_t/data/data_augmentation.py", # ── Legacy xeon_runtime.py at scripts/ (canonical is utils/xeon_runtime.py) "scripts/xeon_runtime.py", ] # Critical files that MUST be present after upload CRITICAL_FILES_V65_7DS = [ # Core model "src/bigru_t/model/kohonen_learning_system.py", "src/bigru_t/model/__init__.py", "src/bigru_t/__init__.py", "src/bigru_t/model/hyp_t.py", "src/bigru_t/model/vqvae2_hierarchical.py", "src/bigru_t/model/vqvae2_hierarchical_flexnet.py", "src/bigru_t/model/token_compress.py", "src/bigru_t/model/embedding_reconfig.py", "src/bigru_t/model/attention_multimodal.py", # Training "src/bigru_t/training/mtp.py", "src/bigru_t/training/ewc.py", # Quantization "src/bigru_t/quantization/smoothquant_compressor.py", "src/bigru_t/quantization/w8a8_smoothquant.py", "src/bigru_t/quantization/quantized_linear.py", # Reasoning "src/bigru_t/reasoning/thinking.py", "src/bigru_t/reasoning/reasoning_engine.py", "src/bigru_t/reasoning/circular_orchestration.py", "src/bigru_t/reasoning/tool_agent.py", "src/bigru_t/reasoning/distributed_reasoning_system.py", "src/bigru_t/reasoning/cyclic_reasoning.py", "src/bigru_t/reasoning/consensus_sampling.py", # Data + utils "src/bigru_t/data/streaming_datasets.py", "src/bigru_t/utils/xeon_runtime.py", # V6.5-7ds reports (NEW canonical) "v6_5_7ds_report.json", "v6_5_7ds_training_metrics.json", "v6_5_7ds_module_analysis.json", "v6_5_7ds_script_activity.json", "v6_5_7ds_ewc_w8a8_benchmark.json", "v6_5_7ds_reasoning_eval.json", "v6_5_7ds_w8a8_compression.json", "v6_5_7ds_upload_report.json", # V6.4 reports (kept as the immediate predecessor baseline) "v6_4_report.json", "v6_4_training_metrics.json", "v6_4_upload_report.json", # V6.5-7ds scripts (the new canonicals) "scripts/train_v6_5_7ds.py", "scripts/upload_v6_5_7ds.py", "scripts/train_v6_4.py", # kept as predecessor reference # Misc "requirements.txt", "README.md", "docs/analysis.md", ] def main() -> int: print("=" * 72) print("V6.5-7DS — UPLOAD TO HUGGINGFACE (with removal of old files)") print("=" * 72) hf_token = os.environ.get("HF_TOKEN") if not hf_token: print("ERROR: HF_TOKEN not set in env") return 1 print(f" HF_TOKEN loaded from env (length={len(hf_token)})") # ── Step 1: Pre-upload safety — scrub any HF token from scripts ────────── token_pattern = re.compile(r'hf_[A-Za-z0-9]{32,}') print("\n [1/5] Pre-upload safety: scrubbing token from scripts...") scrubbed_count = 0 for script_path in BIGRU_ROOT.glob("scripts/*.py"): try: content = script_path.read_text() if token_pattern.search(content): scrubbed = token_pattern.sub("hf_", content) script_path.write_text(scrubbed) scrubbed_count += 1 print(f" SCRUBBED: {script_path.name}") except Exception as e: print(f" SKIP {script_path.name}: {e}") if scrubbed_count == 0: print(" (no token leakage found in scripts)") # ── Step 2: Import HF API + check repo access ──────────────────────────── print("\n [2/5] Connecting to HuggingFace repo...") try: from huggingface_hub import HfApi, upload_folder except ImportError: print("ERROR: huggingface_hub not installed") return 1 api = HfApi(token=hf_token) repo_id = "PowerMachine/BiGRU_T_version" try: info = api.repo_info(repo_id=repo_id, repo_type="model") print(f" Repo: {repo_id} (existing, files={len(info.siblings)})") except Exception as e: print(f" ERROR: cannot access repo: {e}") return 1 # Snapshot of files in repo BEFORE upload files_before = set(s.rfilename for s in info.siblings) print(f" Files in repo BEFORE upload: {len(files_before)}") # ── Step 3: Upload current BiGRU_T_version tree as single commit ───────── print(f"\n [3/5] Uploading {BIGRU_ROOT} as single commit...") t0 = time.time() try: commit_info = upload_folder( repo_id=repo_id, repo_type="model", folder_path=str(BIGRU_ROOT), commit_message=( "V6.5-7ds: 7 datasets streaming (1400 samples, 112 steps), " "864 neurons, SmoothQuant W8A8 (err=0.022), MTP K=6, " "tool_coordinator workers reactivated, 10/10 verification PASS, " "reasoning GOOD, removed legacy v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive " "files + old v6_5_* reports/scripts" ), token=hf_token, ) t1 = time.time() print(f" Upload completed in {t1 - t0:.2f}s") print(f" Commit: {commit_info.oid}") print(f" URL: https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_info.oid}") except Exception as e: print(f" ERROR: upload failed: {e}") return 1 # ── Step 4: Remove OLD files from repo (single delete commit) ──────────── print(f"\n [4/5] Removing {len(OLD_FILES_TO_REMOVE)} old files from repo...") # Refresh repo info to know what files currently exist try: info_after = api.repo_info(repo_id=repo_id, repo_type="model") files_after_upload = set(s.rfilename for s in info_after.siblings) except Exception as e: print(f" WARNING: cannot refresh repo info: {e}") files_after_upload = files_before # Determine which old files actually exist in repo (to delete) old_files_in_repo = [f for f in OLD_FILES_TO_REMOVE if f in files_after_upload] old_files_not_in_repo = [f for f in OLD_FILES_TO_REMOVE if f not in files_after_upload] print(f" Old files present in repo (will delete): {len(old_files_in_repo)}") print(f" Old files NOT in repo (skip): {len(old_files_not_in_repo)}") if old_files_not_in_repo: print(f" examples: {old_files_not_in_repo[:5]}") # Use HfApi.create_commit with CommitOperationDelete for batch removal # (more efficient than calling delete_file one-by-one) delete_errors = [] if old_files_in_repo: try: from huggingface_hub import CommitOperation operations = [ CommitOperationDelete(path_in_repo=f) for f in old_files_in_repo ] print(f" Creating batch delete commit ({len(operations)} files)...") t_del_start = time.time() commit_del = api.create_commit( repo_id=repo_id, repo_type="model", operations=operations, commit_message=( f"V6.5-7ds cleanup: remove {len(operations)} legacy/superseded files " f"(v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive + old v6_5_*/v6_5_final_* " f"reports/scripts) — replaced by V6.5-7ds canonicals" ), ) t_del_end = time.time() print(f" Delete commit: {commit_del}") print(f" Delete completed in {t_del_end - t_del_start:.2f}s") print(f" URL: https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_del}") except Exception as e: print(f" ERROR: batch delete failed: {e}") print(f" Falling back to per-file delete...") # Fallback: delete one-by-one for f in old_files_in_repo: try: api.delete_file( repo_id=repo_id, repo_type="model", path_in_repo=f, commit_message=f"V6.5-7ds cleanup: remove legacy {f}", token=hf_token, ) print(f" DELETED: {f}") except Exception as e2: delete_errors.append((f, str(e2)[:100])) print(f" FAILED: {f} — {str(e2)[:100]}") # ── Step 5: Verify critical files in repo ──────────────────────────────── print(f"\n [5/5] Verifying critical V6.5-7ds files in repo...") try: # Refresh repo info after deletes info_final = api.repo_info(repo_id=repo_id, repo_type="model") files_in_repo = set(s.rfilename for s in info_final.siblings) all_present = True for cf in CRITICAL_FILES_V65_7DS: present = cf in files_in_repo mark = "OK" if present else "MISS" print(f" [{mark}] {cf}") if not present: all_present = False print(f"\n Files in repo AFTER cleanup: {len(files_in_repo)}") if not all_present: print(" WARNING: some critical files missing!") except Exception as e: print(f" WARNING: cannot verify files: {e}") files_in_repo = set() # ── Save upload report ─────────────────────────────────────────────────── report = { "version": "V6.5-7ds", "upload_timestamp": time.strftime("%Y-%m-%dT%H:%M:%S"), "repo_id": repo_id, "upload_commit_oid": commit_info.oid, "upload_commit_url": f"https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_info.oid}", "upload_duration_s": float(t1 - t0), "delete_commit_oid": commit_del if old_files_in_repo else None, "delete_count": len(old_files_in_repo) if old_files_in_repo else 0, "delete_errors": delete_errors, "n_files_before": len(files_before), "n_files_after_upload": len(files_after_upload), "n_files_after_cleanup": len(files_in_repo) if files_in_repo else None, "old_files_removed": old_files_in_repo, "old_files_not_in_repo_skipped": old_files_not_in_repo, "critical_files": CRITICAL_FILES_V65_7DS, "all_critical_files_present": all_present if 'all_present' in dir() else None, "token_scrubbed_count": scrubbed_count, } report_path = BIGRU_ROOT / "v6_5_7ds_upload_report.json" report_path.write_text(json.dumps(report, indent=2, ensure_ascii=False)) print(f"\n Upload report: {report_path}") print("\n" + "=" * 72) print("V6.5-7DS — UPLOAD COMPLETED (with old file removal)") print("=" * 72) return 0 if __name__ == "__main__": sys.exit(main())