Download scripts/deprecated/upload_v6_5_7ds.py from PowerMachine/BiGRU_T_version: direct link, hf CLI and curl.
- Browser
- Download file 16.2 kB
-
https://huggingface.co/PowerMachine/BiGRU_T_version/resolve/main/scripts/deprecated/upload_v6_5_7ds.py
- Command line
-
hf download hf://PowerMachine/BiGRU_T_version/scripts/deprecated/upload_v6_5_7ds.py
-
curl -L -o upload_v6_5_7ds.py https://huggingface.co/PowerMachine/BiGRU_T_version/resolve/main/scripts/deprecated/upload_v6_5_7ds.py
16.2 kB
| """upload_v6_5_7ds.py β V6.5-7ds upload to HuggingFace. | |
| User requirement (latest): "upload dos arquivos testados e os aprimorados | |
| corrigidos (remover os antigos)" | |
| This script: | |
| 1. Pre-upload safety: scrub HF_TOKEN from all scripts/*.py via regex. | |
| 2. Defines OLD_FILES_TO_REMOVE β legacy v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive | |
| reports and scripts, plus old v6_5_* files (superseded by v6_5_7ds_*). | |
| 3. Uploads the tested+improved files (current BiGRU_T_version tree). | |
| 4. Deletes OLD_FILES_TO_REMOVE from the HF repo via HfApi.delete_file. | |
| 5. Verifies critical V6.5-7ds files are present in repo. | |
| 6. Saves upload report to v6_5_7ds_upload_report.json. | |
| After successful upload, the HF_TOKEN env var MUST be deleted by the caller. | |
| """ | |
| from __future__ import annotations | |
| import json, os, sys, time, re | |
| from pathlib import Path | |
| PROJECT_ROOT = Path("/home/z/my-project") | |
| BIGRU_ROOT = PROJECT_ROOT / "BiGRU_T_version" | |
| # ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Files to REMOVE from the HF repo (legacy / superseded by V6.5-7ds) | |
| # ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # User requirement: "upload dos arquivos testados e os aprimorados corrigidos | |
| # (remover os antigos)" | |
| # | |
| # Removal strategy: | |
| # - All legacy version reports (v3, v4, v5, v6, v6_1, v6_2, v6_3, v6_progressive) | |
| # - All legacy upload reports (v3, v5, v6_1, v6_2, v6_4, v6_5) | |
| # - Old V6.5 files superseded by V6.5-7ds (v6_5_*.json and v6_5_final_*.json) | |
| # - Old training scripts (train.py, train_v2.py, train_fast.py, smoke_test.py, | |
| # parse_v6_log.py, train_v6_1.py, train_v6_2.py, train_v6_3.py, | |
| # train_v6_progressive.py, train_v6_5.py, train_v6_5_final.py) | |
| # - Old upload scripts (upload_to_hf.py, upload_v6_1_resilient.py, | |
| # upload_v6_2_resilient.py, upload_v6_3_resilient.py, upload_v6_4_resilient.py, | |
| # upload_v6_5_resilient.py) | |
| # - Legacy root files: bug_hunt_v3_report.json, v32_test_report.json, | |
| # v32_upload_report.json, training_report.json, v6_progressive_train.log, | |
| # v6_progressive_report.json, v6_report.json, v6_upload_report.json, | |
| # config.json (legacy model_final/config.json stays) | |
| OLD_FILES_TO_REMOVE = [ | |
| # ββ Legacy root reports (v3/v4/v5/v6/v32) βββββββββββββββββββββββββββββββββ | |
| "bug_hunt_v3_report.json", | |
| "v3_report.json", | |
| "v4_report.json", | |
| "v5_report.json", | |
| "v5_upload_report.json", | |
| "v32_test_report.json", | |
| "v32_upload_report.json", | |
| "training_report.json", | |
| "v6_report.json", | |
| "v6_upload_report.json", | |
| "v6_progressive_report.json", | |
| "v6_progressive_train.log", | |
| # ββ Legacy v6.1/v6.2/v6.3 reports βββββββββββββββββββββββββββββββββββββββββ | |
| "v6_1_report.json", | |
| "v6_1_upload_report.json", | |
| "v6_2_report.json", | |
| "v6_2_upload_report.json", | |
| "v6_3_report.json", | |
| "v6_3_training_metrics.json", | |
| # ββ Old V6.5 reports (superseded by V6.5-7ds) βββββββββββββββββββββββββββββ | |
| # These are removed because V6.5-7ds is the new canonical version. | |
| "v6_5_report.json", | |
| "v6_5_training_metrics.json", | |
| "v6_5_module_analysis.json", | |
| "v6_5_script_activity.json", | |
| "v6_5_ewc_w8a8_benchmark.json", | |
| "v6_5_reasoning_eval.json", | |
| "v6_5_upload_report.json", | |
| # ββ Old V6.5-final reports (also superseded by V6.5-7ds) ββββββββββββββββββ | |
| "v6_5_final_report.json", | |
| "v6_5_final_training_metrics.json", | |
| "v6_5_final_module_analysis.json", | |
| "v6_5_final_script_activity.json", | |
| "v6_5_final_ewc_w8a8_benchmark.json", | |
| "v6_5_final_reasoning_eval.json", | |
| "v6_5_final_w8a8_compression.json", | |
| # ββ Legacy training scripts βββββββββββββββββββββββββββββββββββββββββββββββ | |
| "scripts/parse_v6_log.py", | |
| "scripts/smoke_test.py", | |
| "scripts/train.py", | |
| "scripts/train_fast.py", | |
| "scripts/train_v2.py", | |
| "scripts/train_v6_1.py", | |
| "scripts/train_v6_2.py", | |
| "scripts/train_v6_3.py", | |
| "scripts/train_v6_progressive.py", | |
| "scripts/train_v6_5.py", # superseded by train_v6_5_7ds.py | |
| "scripts/train_v6_5_final.py", # superseded by train_v6_5_7ds.py | |
| # ββ Legacy upload scripts (only upload_v6_5_7ds.py stays) βββββββββββββββββ | |
| "scripts/upload_to_hf.py", | |
| "scripts/upload_v6_1_resilient.py", | |
| "scripts/upload_v6_2_resilient.py", | |
| "scripts/upload_v6_3_resilient.py", | |
| "scripts/upload_v6_4_resilient.py", | |
| "scripts/upload_v6_5_resilient.py", # superseded by upload_v6_5_7ds.py | |
| # ββ Legacy tokenizer/ root config (kept model_final/ tree intact) βββββββββ | |
| "config.json", # legacy root config; not used by V6.5 | |
| "tokenizer/tokenizer.json", # legacy root tokenizer; model_final/ has its own | |
| # ββ Legacy data_augmentation.py (was moved to training/ in V6.4) ββββββββββ | |
| "src/bigru_t/data/data_augmentation.py", | |
| # ββ Legacy xeon_runtime.py at scripts/ (canonical is utils/xeon_runtime.py) | |
| "scripts/xeon_runtime.py", | |
| ] | |
| # Critical files that MUST be present after upload | |
| CRITICAL_FILES_V65_7DS = [ | |
| # Core model | |
| "src/bigru_t/model/kohonen_learning_system.py", | |
| "src/bigru_t/model/__init__.py", | |
| "src/bigru_t/__init__.py", | |
| "src/bigru_t/model/hyp_t.py", | |
| "src/bigru_t/model/vqvae2_hierarchical.py", | |
| "src/bigru_t/model/vqvae2_hierarchical_flexnet.py", | |
| "src/bigru_t/model/token_compress.py", | |
| "src/bigru_t/model/embedding_reconfig.py", | |
| "src/bigru_t/model/attention_multimodal.py", | |
| # Training | |
| "src/bigru_t/training/mtp.py", | |
| "src/bigru_t/training/ewc.py", | |
| # Quantization | |
| "src/bigru_t/quantization/smoothquant_compressor.py", | |
| "src/bigru_t/quantization/w8a8_smoothquant.py", | |
| "src/bigru_t/quantization/quantized_linear.py", | |
| # Reasoning | |
| "src/bigru_t/reasoning/thinking.py", | |
| "src/bigru_t/reasoning/reasoning_engine.py", | |
| "src/bigru_t/reasoning/circular_orchestration.py", | |
| "src/bigru_t/reasoning/tool_agent.py", | |
| "src/bigru_t/reasoning/distributed_reasoning_system.py", | |
| "src/bigru_t/reasoning/cyclic_reasoning.py", | |
| "src/bigru_t/reasoning/consensus_sampling.py", | |
| # Data + utils | |
| "src/bigru_t/data/streaming_datasets.py", | |
| "src/bigru_t/utils/xeon_runtime.py", | |
| # V6.5-7ds reports (NEW canonical) | |
| "v6_5_7ds_report.json", | |
| "v6_5_7ds_training_metrics.json", | |
| "v6_5_7ds_module_analysis.json", | |
| "v6_5_7ds_script_activity.json", | |
| "v6_5_7ds_ewc_w8a8_benchmark.json", | |
| "v6_5_7ds_reasoning_eval.json", | |
| "v6_5_7ds_w8a8_compression.json", | |
| "v6_5_7ds_upload_report.json", | |
| # V6.4 reports (kept as the immediate predecessor baseline) | |
| "v6_4_report.json", | |
| "v6_4_training_metrics.json", | |
| "v6_4_upload_report.json", | |
| # V6.5-7ds scripts (the new canonicals) | |
| "scripts/train_v6_5_7ds.py", | |
| "scripts/upload_v6_5_7ds.py", | |
| "scripts/train_v6_4.py", # kept as predecessor reference | |
| # Misc | |
| "requirements.txt", | |
| "README.md", | |
| "docs/analysis.md", | |
| ] | |
| def main() -> int: | |
| print("=" * 72) | |
| print("V6.5-7DS β UPLOAD TO HUGGINGFACE (with removal of old files)") | |
| print("=" * 72) | |
| hf_token = os.environ.get("HF_TOKEN") | |
| if not hf_token: | |
| print("ERROR: HF_TOKEN not set in env") | |
| return 1 | |
| print(f" HF_TOKEN loaded from env (length={len(hf_token)})") | |
| # ββ Step 1: Pre-upload safety β scrub any HF token from scripts ββββββββββ | |
| token_pattern = re.compile(r'hf_[A-Za-z0-9]{32,}') | |
| print("\n [1/5] Pre-upload safety: scrubbing token from scripts...") | |
| scrubbed_count = 0 | |
| for script_path in BIGRU_ROOT.glob("scripts/*.py"): | |
| try: | |
| content = script_path.read_text() | |
| if token_pattern.search(content): | |
| scrubbed = token_pattern.sub("hf_<REDACTED_TOKEN>", content) | |
| script_path.write_text(scrubbed) | |
| scrubbed_count += 1 | |
| print(f" SCRUBBED: {script_path.name}") | |
| except Exception as e: | |
| print(f" SKIP {script_path.name}: {e}") | |
| if scrubbed_count == 0: | |
| print(" (no token leakage found in scripts)") | |
| # ββ Step 2: Import HF API + check repo access ββββββββββββββββββββββββββββ | |
| print("\n [2/5] Connecting to HuggingFace repo...") | |
| try: | |
| from huggingface_hub import HfApi, upload_folder | |
| except ImportError: | |
| print("ERROR: huggingface_hub not installed") | |
| return 1 | |
| api = HfApi(token=hf_token) | |
| repo_id = "PowerMachine/BiGRU_T_version" | |
| try: | |
| info = api.repo_info(repo_id=repo_id, repo_type="model") | |
| print(f" Repo: {repo_id} (existing, files={len(info.siblings)})") | |
| except Exception as e: | |
| print(f" ERROR: cannot access repo: {e}") | |
| return 1 | |
| # Snapshot of files in repo BEFORE upload | |
| files_before = set(s.rfilename for s in info.siblings) | |
| print(f" Files in repo BEFORE upload: {len(files_before)}") | |
| # ββ Step 3: Upload current BiGRU_T_version tree as single commit βββββββββ | |
| print(f"\n [3/5] Uploading {BIGRU_ROOT} as single commit...") | |
| t0 = time.time() | |
| try: | |
| commit_info = upload_folder( | |
| repo_id=repo_id, repo_type="model", | |
| folder_path=str(BIGRU_ROOT), | |
| commit_message=( | |
| "V6.5-7ds: 7 datasets streaming (1400 samples, 112 steps), " | |
| "864 neurons, SmoothQuant W8A8 (err=0.022), MTP K=6, " | |
| "tool_coordinator workers reactivated, 10/10 verification PASS, " | |
| "reasoning GOOD, removed legacy v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive " | |
| "files + old v6_5_* reports/scripts" | |
| ), | |
| token=hf_token, | |
| ) | |
| t1 = time.time() | |
| print(f" Upload completed in {t1 - t0:.2f}s") | |
| print(f" Commit: {commit_info.oid}") | |
| print(f" URL: https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_info.oid}") | |
| except Exception as e: | |
| print(f" ERROR: upload failed: {e}") | |
| return 1 | |
| # ββ Step 4: Remove OLD files from repo (single delete commit) ββββββββββββ | |
| print(f"\n [4/5] Removing {len(OLD_FILES_TO_REMOVE)} old files from repo...") | |
| # Refresh repo info to know what files currently exist | |
| try: | |
| info_after = api.repo_info(repo_id=repo_id, repo_type="model") | |
| files_after_upload = set(s.rfilename for s in info_after.siblings) | |
| except Exception as e: | |
| print(f" WARNING: cannot refresh repo info: {e}") | |
| files_after_upload = files_before | |
| # Determine which old files actually exist in repo (to delete) | |
| old_files_in_repo = [f for f in OLD_FILES_TO_REMOVE if f in files_after_upload] | |
| old_files_not_in_repo = [f for f in OLD_FILES_TO_REMOVE if f not in files_after_upload] | |
| print(f" Old files present in repo (will delete): {len(old_files_in_repo)}") | |
| print(f" Old files NOT in repo (skip): {len(old_files_not_in_repo)}") | |
| if old_files_not_in_repo: | |
| print(f" examples: {old_files_not_in_repo[:5]}") | |
| # Use HfApi.create_commit with CommitOperationDelete for batch removal | |
| # (more efficient than calling delete_file one-by-one) | |
| delete_errors = [] | |
| if old_files_in_repo: | |
| try: | |
| from huggingface_hub import CommitOperation | |
| operations = [ | |
| CommitOperationDelete(path_in_repo=f) for f in old_files_in_repo | |
| ] | |
| print(f" Creating batch delete commit ({len(operations)} files)...") | |
| t_del_start = time.time() | |
| commit_del = api.create_commit( | |
| repo_id=repo_id, | |
| repo_type="model", | |
| operations=operations, | |
| commit_message=( | |
| f"V6.5-7ds cleanup: remove {len(operations)} legacy/superseded files " | |
| f"(v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive + old v6_5_*/v6_5_final_* " | |
| f"reports/scripts) β replaced by V6.5-7ds canonicals" | |
| ), | |
| ) | |
| t_del_end = time.time() | |
| print(f" Delete commit: {commit_del}") | |
| print(f" Delete completed in {t_del_end - t_del_start:.2f}s") | |
| print(f" URL: https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_del}") | |
| except Exception as e: | |
| print(f" ERROR: batch delete failed: {e}") | |
| print(f" Falling back to per-file delete...") | |
| # Fallback: delete one-by-one | |
| for f in old_files_in_repo: | |
| try: | |
| api.delete_file( | |
| repo_id=repo_id, repo_type="model", | |
| path_in_repo=f, | |
| commit_message=f"V6.5-7ds cleanup: remove legacy {f}", | |
| token=hf_token, | |
| ) | |
| print(f" DELETED: {f}") | |
| except Exception as e2: | |
| delete_errors.append((f, str(e2)[:100])) | |
| print(f" FAILED: {f} β {str(e2)[:100]}") | |
| # ββ Step 5: Verify critical files in repo ββββββββββββββββββββββββββββββββ | |
| print(f"\n [5/5] Verifying critical V6.5-7ds files in repo...") | |
| try: | |
| # Refresh repo info after deletes | |
| info_final = api.repo_info(repo_id=repo_id, repo_type="model") | |
| files_in_repo = set(s.rfilename for s in info_final.siblings) | |
| all_present = True | |
| for cf in CRITICAL_FILES_V65_7DS: | |
| present = cf in files_in_repo | |
| mark = "OK" if present else "MISS" | |
| print(f" [{mark}] {cf}") | |
| if not present: | |
| all_present = False | |
| print(f"\n Files in repo AFTER cleanup: {len(files_in_repo)}") | |
| if not all_present: | |
| print(" WARNING: some critical files missing!") | |
| except Exception as e: | |
| print(f" WARNING: cannot verify files: {e}") | |
| files_in_repo = set() | |
| # ββ Save upload report βββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| report = { | |
| "version": "V6.5-7ds", | |
| "upload_timestamp": time.strftime("%Y-%m-%dT%H:%M:%S"), | |
| "repo_id": repo_id, | |
| "upload_commit_oid": commit_info.oid, | |
| "upload_commit_url": f"https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_info.oid}", | |
| "upload_duration_s": float(t1 - t0), | |
| "delete_commit_oid": commit_del if old_files_in_repo else None, | |
| "delete_count": len(old_files_in_repo) if old_files_in_repo else 0, | |
| "delete_errors": delete_errors, | |
| "n_files_before": len(files_before), | |
| "n_files_after_upload": len(files_after_upload), | |
| "n_files_after_cleanup": len(files_in_repo) if files_in_repo else None, | |
| "old_files_removed": old_files_in_repo, | |
| "old_files_not_in_repo_skipped": old_files_not_in_repo, | |
| "critical_files": CRITICAL_FILES_V65_7DS, | |
| "all_critical_files_present": all_present if 'all_present' in dir() else None, | |
| "token_scrubbed_count": scrubbed_count, | |
| } | |
| report_path = BIGRU_ROOT / "v6_5_7ds_upload_report.json" | |
| report_path.write_text(json.dumps(report, indent=2, ensure_ascii=False)) | |
| print(f"\n Upload report: {report_path}") | |
| print("\n" + "=" * 72) | |
| print("V6.5-7DS β UPLOAD COMPLETED (with old file removal)") | |
| print("=" * 72) | |
| return 0 | |
| if __name__ == "__main__": | |
| sys.exit(main()) | |