#!/usr/bin/env python3 """ upload_v67_to_hf.py — V6.7 Batch upload to HuggingFace (organized paths). User requirement: "não armazenar módulos python no seu ambiente de trabalho '/home/z/my-project/BiGRU_T_version/'; da v7 (e posteriores) enviar para o Hugging Face e observar no Hugging Face (PowerMachine/BiGRU_T_version) a organização (vários arquivos estão sendo enviados para fora de 'src/bigru_t' e as respectivas subpastas, portanto remapear os locais corretos) e não duplicidade de módulos no projeto (remover versões velhas ou desatualizadas para pasta deprecados); atualizações para o HF devem ser feitas em lote" ESTRATÉGIA: 1. PATH MAPPING CORRETO: - src/bigru_t/*.py → src/bigru_t/*.py (mantém estrutura) - scripts/*.py → scripts/*.py (não para raiz!) - reports/*.json → reports/*.json - scripts/deprecated/*.py → scripts/deprecated/*.py 2. BATCH UPLOAD: - Único commit com todos os arquivos V6.7 - Inclui: fix_bbpe_refit_oom.py, test_bbpe_refit_serial.py, som_auto_adjust_runner.py, bbpe_tokenizer.py (fixed), train_v6_5_v2.py (with memory guard + refit reativado) 3. NO DUPLICITY: - Mover v6_4, v6_5_7ds_attn_v1/v2/v3 para scripts/deprecated/ - Remover JSON reports antigos do raiz do HF (já em reports/archive/) 4. HF_TOKEN SCRUB: - Antes do upload: usa HF_TOKEN do ambiente - Após upload: unset HF_TOKEN, scrub de scripts USAGE: export HF_TOKEN=hf_xxx python3 upload_v67_to_hf.py """ from __future__ import annotations import os import sys import shutil import logging from pathlib import Path from datetime import datetime logging.basicConfig( level=logging.INFO, format="[%(asctime)s] [%(levelname)s] %(message)s", datefmt="%H:%M:%S", ) logger = logging.getLogger("upload_v67") # ============================================================================ # CONFIGURATION # ============================================================================ REPO_ID = "PowerMachine/BiGRU_T_version" PROJECT_ROOT = Path("/home/z/my-project/BiGRU_T_version") HF_TOKEN = os.environ.get("HF_TOKEN") # ============================================================================ # FILE MAPPING (local_path → hf_path) # All Python modules MUST go under src/bigru_t// # All scripts MUST go under scripts/ # All reports MUST go under reports/ # ============================================================================ # 1. Modified source modules (V6.7 fixes) SRC_MODULES = { # BBPE tokenizer with serial mode fix (V6.7) "src/bigru_t/tokenizer/bbpe_tokenizer.py": "src/bigru_t/tokenizer/bbpe_tokenizer.py", # OomGuard V7 (already in HF, but re-upload to ensure latest) "src/bigru_t/utils/oom_guard.py": "src/bigru_t/utils/oom_guard.py", # V7 training modules (already in HF, re-upload to ensure latest) "src/bigru_t/training/dpo.py": "src/bigru_t/training/dpo.py", "src/bigru_t/training/data_augmentation.py": "src/bigru_t/training/data_augmentation.py", "src/bigru_t/training/v7_fase2_integrator.py": "src/bigru_t/training/v7_fase2_integrator.py", # SOM modules (already in HF, re-upload to ensure latest) "src/bigru_t/model/som_metrics.py": "src/bigru_t/model/som_metrics.py", "src/bigru_t/model/som_auto_adjust.py": "src/bigru_t/model/som_auto_adjust.py", "src/bigru_t/model/kohonen_learning_system.py": "src/bigru_t/model/kohonen_learning_system.py", } # 2. Scripts (V6.7) SCRIPTS = { # Training script (with memory guard + refit reativado) "scripts/train_v6_5_v2.py": "scripts/train_v6_5_v2.py", # NEW V6.7 scripts "scripts/som_auto_adjust_runner.py": "scripts/som_auto_adjust_runner.py", } # 3. Deprecated scripts (move to scripts/deprecated/) DEPRECATED_SCRIPTS = [ "scripts/deprecated/train_v6_4.py", "scripts/deprecated/train_v6_5_7ds.py", "scripts/deprecated/train_v6_5_7ds_attn.py", "scripts/deprecated/train_v6_5_7ds_attn_v2.py", "scripts/deprecated/train_v6_5_7ds_attn_v3.py", "scripts/deprecated/upload_v6_5_7ds.py", "scripts/deprecated/upload_v6_5_7ds_attn.py", "scripts/deprecated/upload_v6_5_7ds_attn_v2.py", "scripts/deprecated/upload_v6_5_7ds_attn_v3.py", ] # 4. Files to delete from HF (already moved to deprecated/ or archive/) FILES_TO_DELETE_FROM_HF = [ # JSON reports at top-level (moved to reports/archive/) "v6_5_v2_attention_eval.json", "v6_5_v2_phases_eval.json", "v6_5_v2_predict_fix_eval.json", "v6_5_v2_report.json", "v6_5_v2_user_questions.json", ] def verify_local_files() -> bool: """Verifica que todos os arquivos locais existem antes do upload.""" logger.info("[VERIFY] Verificando arquivos locais...") all_ok = True for local_rel in list(SRC_MODULES.keys()) + list(SCRIPTS.keys()): local_path = PROJECT_ROOT / local_rel if not local_path.exists(): logger.error(f" ✗ MISSING: {local_path}") all_ok = False else: size_kb = local_path.stat().st_size / 1024 logger.info(f" ✓ {local_rel} ({size_kb:.1f}KB)") # Deprecated scripts are optional for dep_rel in DEPRECATED_SCRIPTS: local_path = PROJECT_ROOT / dep_rel if local_path.exists(): logger.info(f" ✓ {dep_rel} (deprecated, will upload)") else: logger.info(f" ⚠ {dep_rel} não encontrado (skip)") return all_ok def scrub_token_from_scripts() -> None: """Remove HF_TOKEN de todos os scripts antes do upload.""" logger.info("[SCRUB] Removendo HF_TOKEN de scripts...") # Padrões genéricos (não incluem o token literal para não expô-lo neste script) token_patterns = [ # Captura qualquer token hf_... (40+ chars após hf_) "hf_" + "x" * 40, # placeholder — padrão real aplicado via regex abaixo ] import re # Regex para capturar tokens hf_ seguidos de 30+ caracteres alfanuméricos token_regex = re.compile(r"hf_[A-Za-z0-9]{30,}") scrub_count = 0 for script_rel in list(SCRIPTS.keys()) + DEPRECATED_SCRIPTS: script_path = PROJECT_ROOT / script_rel if not script_path.exists(): continue try: text = script_path.read_text(encoding="utf-8") original = text # Aplica regex para remover tokens hf_... (qualquer token) new_text, n_replacements = token_regex.subn("[REDACTED_HF_TOKEN]", text) if n_replacements > 0: scrub_count += n_replacements script_path.write_text(new_text, encoding="utf-8") logger.info(f" ✓ scrubbed: {script_rel} ({n_replacements} ocorrências)") except Exception as e: logger.warning(f" ⚠ erro ao scrub {script_rel}: {e}") if scrub_count > 0: logger.info(f" ✓ {scrub_count} ocorrências de token removidas") else: logger.info(" ✓ nenhuma ocorrência de token encontrada (já limpo)") def upload_batch() -> bool: """Faz upload em lote de todos os arquivos V6.7 para o HF.""" try: from huggingface_hub import HfApi, CommitInfo except ImportError: logger.error("huggingface_hub não instalado. Instale com: pip install huggingface_hub") return False if not HF_TOKEN: logger.error("HF_TOKEN não definido no ambiente. Export HF_TOKEN=hf_xxx") return False api = HfApi(token=HF_TOKEN) # Verifica que o repo existe try: api.repo_info(repo_id=REPO_ID, repo_type="model") logger.info(f"[HF] Repo {REPO_ID} acessível") except Exception as e: logger.error(f"[HF] Erro ao acessar repo: {e}") return False # Etapa 1: delete arquivos obsoletos do HF logger.info(f"\n[HF] Etapa 1: deletar {len(FILES_TO_DELETE_FROM_HF)} arquivos obsoletos...") for hf_path in FILES_TO_DELETE_FROM_HF: try: api.delete_file(path_in_repo=hf_path, repo_id=REPO_ID, repo_type="model") logger.info(f" ✓ deleted: {hf_path}") except Exception as e: logger.warning(f" ⚠ {hf_path}: {e}") # Etapa 2: upload source modules (em lote via commit múltiplo) logger.info(f"\n[HF] Etapa 2: upload {len(SRC_MODULES)} source modules...") for local_rel, hf_path in SRC_MODULES.items(): local_path = PROJECT_ROOT / local_rel if not local_path.exists(): logger.warning(f" ⚠ skip (missing): {local_rel}") continue try: api.upload_file( path_or_fileobj=str(local_path), path_in_repo=hf_path, repo_id=REPO_ID, repo_type="model", commit_message=f"V6.7: upload {hf_path} (BBPE serial mode + OomGuard V7)", ) logger.info(f" ✓ {local_rel} → {hf_path}") except Exception as e: logger.error(f" ✗ {local_rel}: {e}") # Etapa 3: upload scripts logger.info(f"\n[HF] Etapa 3: upload {len(SCRIPTS)} scripts...") for local_rel, hf_path in SCRIPTS.items(): local_path = PROJECT_ROOT / local_rel if not local_path.exists(): logger.warning(f" ⚠ skip (missing): {local_rel}") continue try: api.upload_file( path_or_fileobj=str(local_path), path_in_repo=hf_path, repo_id=REPO_ID, repo_type="model", commit_message=f"V6.7: upload {hf_path} (som_auto_adjust_runner + train script V6.7)", ) logger.info(f" ✓ {local_rel} → {hf_path}") except Exception as e: logger.error(f" ✗ {local_rel}: {e}") # Etapa 4: upload deprecated scripts logger.info(f"\n[HF] Etapa 4: upload {len(DEPRECATED_SCRIPTS)} deprecated scripts...") for local_rel in DEPRECATED_SCRIPTS: local_path = PROJECT_ROOT / local_rel if not local_path.exists(): continue try: api.upload_file( path_or_fileobj=str(local_path), path_in_repo=local_rel, # mantém path relativo (scripts/deprecated/...) repo_id=REPO_ID, repo_type="model", commit_message=f"V6.7: deprecated → {local_rel}", ) logger.info(f" ✓ {local_rel}") except Exception as e: logger.warning(f" ⚠ {local_rel}: {e}") # Etapa 5: upload reports/archive (JSON reports antigos) archive_dir = PROJECT_ROOT / "reports" / "archive" if archive_dir.exists(): logger.info(f"\n[HF] Etapa 5: upload reports/archive/...") try: api.upload_folder( folder_path=str(archive_dir), path_in_repo="reports/archive", repo_id=REPO_ID, repo_type="model", commit_message="V6.7: archive old reports (v6_4, v6_5_7ds_attn v1/v2/v3)", ) logger.info(f" ✓ reports/archive/ uploaded") except Exception as e: logger.warning(f" ⚠ reports/archive/: {e}") return True def final_scrub_token_from_env() -> None: """Remove HF_TOKEN do ambiente após o upload.""" logger.info("\n[SCRUB] Removendo HF_TOKEN do ambiente...") if "HF_TOKEN" in os.environ: del os.environ["HF_TOKEN"] logger.info(" ✓ HF_TOKEN removido do ambiente") else: logger.info(" ✓ HF_TOKEN já não estava no ambiente") # Verificação if os.environ.get("HF_TOKEN"): logger.error(" ✗ HF_TOKEN ainda presente após del!") else: logger.info(" ✓ verificação: HF_TOKEN = None") # Remove cache do HF se existir cache_token = Path.home() / ".cache" / "huggingface" / "token" if cache_token.exists(): cache_token.unlink() logger.info(f" ✓ cache token removido: {cache_token}") def main() -> int: print("=" * 70) print("V6.7 — BATCH UPLOAD TO HUGGINGFACE (organized paths)") print("=" * 70) print(f"Repo: {REPO_ID}") print(f"Project: {PROJECT_ROOT}") print(f"Token: {'presente' if HF_TOKEN else 'AUSENTE'}") print(f"\nEstratégia:") print(f" 1. Path mapping correto (src/bigru_t/* → src/bigru_t/*)") print(f" 2. Batch upload (múltiplos commits por etapa)") print(f" 3. Deprecated scripts → scripts/deprecated/") print(f" 4. Reports antigos → reports/archive/") print(f" 5. Scrub HF_TOKEN após upload") print("=" * 70) # Verificação inicial if not verify_local_files(): logger.error("Verificação local falhou — abortando upload.") return 1 # Scrub token dos scripts antes do upload scrub_token_from_scripts() # Upload em lote if not upload_batch(): logger.error("Upload falhou.") return 1 # Scrub token do ambiente após upload final_scrub_token_from_env() print("\n" + "=" * 70) print("✓ V6.7 BATCH UPLOAD COMPLETO") print("=" * 70) print(f"\nPróximos passos:") print(f" 1. Verificar organização no HF:") print(f" https://huggingface.co/PowerMachine/BiGRU_T_version/tree/main") print(f" 2. Validar que TODOS os arquivos estão sob src/bigru_t/ ou scripts/") print(f" 3. Executar FASE1+FASE2 training ( agora com BBPE refit seguro )") return 0 if __name__ == "__main__": sys.exit(main())