Download scripts/upload_v67_to_hf.py from PowerMachine/BiGRU_T_version: direct link, hf CLI and curl.
- Browser
- Download file 13.5 kB
-
https://huggingface.co/PowerMachine/BiGRU_T_version/resolve/main/scripts/upload_v67_to_hf.py
- Command line
-
hf download hf://PowerMachine/BiGRU_T_version/scripts/upload_v67_to_hf.py
-
curl -L -o upload_v67_to_hf.py https://huggingface.co/PowerMachine/BiGRU_T_version/resolve/main/scripts/upload_v67_to_hf.py
13.5 kB
| #!/usr/bin/env python3 | |
| """ | |
| upload_v67_to_hf.py — V6.7 Batch upload to HuggingFace (organized paths). | |
| User requirement: | |
| "não armazenar módulos python no seu ambiente de trabalho | |
| '/home/z/my-project/BiGRU_T_version/'; da v7 (e posteriores) enviar | |
| para o Hugging Face e observar no Hugging Face | |
| (PowerMachine/BiGRU_T_version) a organização (vários arquivos estão | |
| sendo enviados para fora de 'src/bigru_t' e as respectivas subpastas, | |
| portanto remapear os locais corretos) e não duplicidade de módulos no | |
| projeto (remover versões velhas ou desatualizadas para pasta | |
| deprecados); atualizações para o HF devem ser feitas em lote" | |
| ESTRATÉGIA: | |
| 1. PATH MAPPING CORRETO: | |
| - src/bigru_t/*.py → src/bigru_t/*.py (mantém estrutura) | |
| - scripts/*.py → scripts/*.py (não para raiz!) | |
| - reports/*.json → reports/*.json | |
| - scripts/deprecated/*.py → scripts/deprecated/*.py | |
| 2. BATCH UPLOAD: | |
| - Único commit com todos os arquivos V6.7 | |
| - Inclui: fix_bbpe_refit_oom.py, test_bbpe_refit_serial.py, | |
| som_auto_adjust_runner.py, bbpe_tokenizer.py (fixed), | |
| train_v6_5_v2.py (with memory guard + refit reativado) | |
| 3. NO DUPLICITY: | |
| - Mover v6_4, v6_5_7ds_attn_v1/v2/v3 para scripts/deprecated/ | |
| - Remover JSON reports antigos do raiz do HF (já em reports/archive/) | |
| 4. HF_TOKEN SCRUB: | |
| - Antes do upload: usa HF_TOKEN do ambiente | |
| - Após upload: unset HF_TOKEN, scrub de scripts | |
| USAGE: | |
| export HF_TOKEN=hf_xxx | |
| python3 upload_v67_to_hf.py | |
| """ | |
| from __future__ import annotations | |
| import os | |
| import sys | |
| import shutil | |
| import logging | |
| from pathlib import Path | |
| from datetime import datetime | |
| logging.basicConfig( | |
| level=logging.INFO, | |
| format="[%(asctime)s] [%(levelname)s] %(message)s", | |
| datefmt="%H:%M:%S", | |
| ) | |
| logger = logging.getLogger("upload_v67") | |
| # ============================================================================ | |
| # CONFIGURATION | |
| # ============================================================================ | |
| REPO_ID = "PowerMachine/BiGRU_T_version" | |
| PROJECT_ROOT = Path("/home/z/my-project/BiGRU_T_version") | |
| HF_TOKEN = os.environ.get("HF_TOKEN") | |
| # ============================================================================ | |
| # FILE MAPPING (local_path → hf_path) | |
| # All Python modules MUST go under src/bigru_t/<subdir>/ | |
| # All scripts MUST go under scripts/ | |
| # All reports MUST go under reports/ | |
| # ============================================================================ | |
| # 1. Modified source modules (V6.7 fixes) | |
| SRC_MODULES = { | |
| # BBPE tokenizer with serial mode fix (V6.7) | |
| "src/bigru_t/tokenizer/bbpe_tokenizer.py": "src/bigru_t/tokenizer/bbpe_tokenizer.py", | |
| # OomGuard V7 (already in HF, but re-upload to ensure latest) | |
| "src/bigru_t/utils/oom_guard.py": "src/bigru_t/utils/oom_guard.py", | |
| # V7 training modules (already in HF, re-upload to ensure latest) | |
| "src/bigru_t/training/dpo.py": "src/bigru_t/training/dpo.py", | |
| "src/bigru_t/training/data_augmentation.py": "src/bigru_t/training/data_augmentation.py", | |
| "src/bigru_t/training/v7_fase2_integrator.py": "src/bigru_t/training/v7_fase2_integrator.py", | |
| # SOM modules (already in HF, re-upload to ensure latest) | |
| "src/bigru_t/model/som_metrics.py": "src/bigru_t/model/som_metrics.py", | |
| "src/bigru_t/model/som_auto_adjust.py": "src/bigru_t/model/som_auto_adjust.py", | |
| "src/bigru_t/model/kohonen_learning_system.py": "src/bigru_t/model/kohonen_learning_system.py", | |
| } | |
| # 2. Scripts (V6.7) | |
| SCRIPTS = { | |
| # Training script (with memory guard + refit reativado) | |
| "scripts/train_v6_5_v2.py": "scripts/train_v6_5_v2.py", | |
| # NEW V6.7 scripts | |
| "scripts/som_auto_adjust_runner.py": "scripts/som_auto_adjust_runner.py", | |
| } | |
| # 3. Deprecated scripts (move to scripts/deprecated/) | |
| DEPRECATED_SCRIPTS = [ | |
| "scripts/deprecated/train_v6_4.py", | |
| "scripts/deprecated/train_v6_5_7ds.py", | |
| "scripts/deprecated/train_v6_5_7ds_attn.py", | |
| "scripts/deprecated/train_v6_5_7ds_attn_v2.py", | |
| "scripts/deprecated/train_v6_5_7ds_attn_v3.py", | |
| "scripts/deprecated/upload_v6_5_7ds.py", | |
| "scripts/deprecated/upload_v6_5_7ds_attn.py", | |
| "scripts/deprecated/upload_v6_5_7ds_attn_v2.py", | |
| "scripts/deprecated/upload_v6_5_7ds_attn_v3.py", | |
| ] | |
| # 4. Files to delete from HF (already moved to deprecated/ or archive/) | |
| FILES_TO_DELETE_FROM_HF = [ | |
| # JSON reports at top-level (moved to reports/archive/) | |
| "v6_5_v2_attention_eval.json", | |
| "v6_5_v2_phases_eval.json", | |
| "v6_5_v2_predict_fix_eval.json", | |
| "v6_5_v2_report.json", | |
| "v6_5_v2_user_questions.json", | |
| ] | |
| def verify_local_files() -> bool: | |
| """Verifica que todos os arquivos locais existem antes do upload.""" | |
| logger.info("[VERIFY] Verificando arquivos locais...") | |
| all_ok = True | |
| for local_rel in list(SRC_MODULES.keys()) + list(SCRIPTS.keys()): | |
| local_path = PROJECT_ROOT / local_rel | |
| if not local_path.exists(): | |
| logger.error(f" ✗ MISSING: {local_path}") | |
| all_ok = False | |
| else: | |
| size_kb = local_path.stat().st_size / 1024 | |
| logger.info(f" ✓ {local_rel} ({size_kb:.1f}KB)") | |
| # Deprecated scripts are optional | |
| for dep_rel in DEPRECATED_SCRIPTS: | |
| local_path = PROJECT_ROOT / dep_rel | |
| if local_path.exists(): | |
| logger.info(f" ✓ {dep_rel} (deprecated, will upload)") | |
| else: | |
| logger.info(f" ⚠ {dep_rel} não encontrado (skip)") | |
| return all_ok | |
| def scrub_token_from_scripts() -> None: | |
| """Remove HF_TOKEN de todos os scripts antes do upload.""" | |
| logger.info("[SCRUB] Removendo HF_TOKEN de scripts...") | |
| # Padrões genéricos (não incluem o token literal para não expô-lo neste script) | |
| token_patterns = [ | |
| # Captura qualquer token hf_... (40+ chars após hf_) | |
| "hf_" + "x" * 40, # placeholder — padrão real aplicado via regex abaixo | |
| ] | |
| import re | |
| # Regex para capturar tokens hf_ seguidos de 30+ caracteres alfanuméricos | |
| token_regex = re.compile(r"hf_[A-Za-z0-9]{30,}") | |
| scrub_count = 0 | |
| for script_rel in list(SCRIPTS.keys()) + DEPRECATED_SCRIPTS: | |
| script_path = PROJECT_ROOT / script_rel | |
| if not script_path.exists(): | |
| continue | |
| try: | |
| text = script_path.read_text(encoding="utf-8") | |
| original = text | |
| # Aplica regex para remover tokens hf_... (qualquer token) | |
| new_text, n_replacements = token_regex.subn("[REDACTED_HF_TOKEN]", text) | |
| if n_replacements > 0: | |
| scrub_count += n_replacements | |
| script_path.write_text(new_text, encoding="utf-8") | |
| logger.info(f" ✓ scrubbed: {script_rel} ({n_replacements} ocorrências)") | |
| except Exception as e: | |
| logger.warning(f" ⚠ erro ao scrub {script_rel}: {e}") | |
| if scrub_count > 0: | |
| logger.info(f" ✓ {scrub_count} ocorrências de token removidas") | |
| else: | |
| logger.info(" ✓ nenhuma ocorrência de token encontrada (já limpo)") | |
| def upload_batch() -> bool: | |
| """Faz upload em lote de todos os arquivos V6.7 para o HF.""" | |
| try: | |
| from huggingface_hub import HfApi, CommitInfo | |
| except ImportError: | |
| logger.error("huggingface_hub não instalado. Instale com: pip install huggingface_hub") | |
| return False | |
| if not HF_TOKEN: | |
| logger.error("HF_TOKEN não definido no ambiente. Export HF_TOKEN=hf_xxx") | |
| return False | |
| api = HfApi(token=HF_TOKEN) | |
| # Verifica que o repo existe | |
| try: | |
| api.repo_info(repo_id=REPO_ID, repo_type="model") | |
| logger.info(f"[HF] Repo {REPO_ID} acessível") | |
| except Exception as e: | |
| logger.error(f"[HF] Erro ao acessar repo: {e}") | |
| return False | |
| # Etapa 1: delete arquivos obsoletos do HF | |
| logger.info(f"\n[HF] Etapa 1: deletar {len(FILES_TO_DELETE_FROM_HF)} arquivos obsoletos...") | |
| for hf_path in FILES_TO_DELETE_FROM_HF: | |
| try: | |
| api.delete_file(path_in_repo=hf_path, repo_id=REPO_ID, repo_type="model") | |
| logger.info(f" ✓ deleted: {hf_path}") | |
| except Exception as e: | |
| logger.warning(f" ⚠ {hf_path}: {e}") | |
| # Etapa 2: upload source modules (em lote via commit múltiplo) | |
| logger.info(f"\n[HF] Etapa 2: upload {len(SRC_MODULES)} source modules...") | |
| for local_rel, hf_path in SRC_MODULES.items(): | |
| local_path = PROJECT_ROOT / local_rel | |
| if not local_path.exists(): | |
| logger.warning(f" ⚠ skip (missing): {local_rel}") | |
| continue | |
| try: | |
| api.upload_file( | |
| path_or_fileobj=str(local_path), | |
| path_in_repo=hf_path, | |
| repo_id=REPO_ID, | |
| repo_type="model", | |
| commit_message=f"V6.7: upload {hf_path} (BBPE serial mode + OomGuard V7)", | |
| ) | |
| logger.info(f" ✓ {local_rel} → {hf_path}") | |
| except Exception as e: | |
| logger.error(f" ✗ {local_rel}: {e}") | |
| # Etapa 3: upload scripts | |
| logger.info(f"\n[HF] Etapa 3: upload {len(SCRIPTS)} scripts...") | |
| for local_rel, hf_path in SCRIPTS.items(): | |
| local_path = PROJECT_ROOT / local_rel | |
| if not local_path.exists(): | |
| logger.warning(f" ⚠ skip (missing): {local_rel}") | |
| continue | |
| try: | |
| api.upload_file( | |
| path_or_fileobj=str(local_path), | |
| path_in_repo=hf_path, | |
| repo_id=REPO_ID, | |
| repo_type="model", | |
| commit_message=f"V6.7: upload {hf_path} (som_auto_adjust_runner + train script V6.7)", | |
| ) | |
| logger.info(f" ✓ {local_rel} → {hf_path}") | |
| except Exception as e: | |
| logger.error(f" ✗ {local_rel}: {e}") | |
| # Etapa 4: upload deprecated scripts | |
| logger.info(f"\n[HF] Etapa 4: upload {len(DEPRECATED_SCRIPTS)} deprecated scripts...") | |
| for local_rel in DEPRECATED_SCRIPTS: | |
| local_path = PROJECT_ROOT / local_rel | |
| if not local_path.exists(): | |
| continue | |
| try: | |
| api.upload_file( | |
| path_or_fileobj=str(local_path), | |
| path_in_repo=local_rel, # mantém path relativo (scripts/deprecated/...) | |
| repo_id=REPO_ID, | |
| repo_type="model", | |
| commit_message=f"V6.7: deprecated → {local_rel}", | |
| ) | |
| logger.info(f" ✓ {local_rel}") | |
| except Exception as e: | |
| logger.warning(f" ⚠ {local_rel}: {e}") | |
| # Etapa 5: upload reports/archive (JSON reports antigos) | |
| archive_dir = PROJECT_ROOT / "reports" / "archive" | |
| if archive_dir.exists(): | |
| logger.info(f"\n[HF] Etapa 5: upload reports/archive/...") | |
| try: | |
| api.upload_folder( | |
| folder_path=str(archive_dir), | |
| path_in_repo="reports/archive", | |
| repo_id=REPO_ID, | |
| repo_type="model", | |
| commit_message="V6.7: archive old reports (v6_4, v6_5_7ds_attn v1/v2/v3)", | |
| ) | |
| logger.info(f" ✓ reports/archive/ uploaded") | |
| except Exception as e: | |
| logger.warning(f" ⚠ reports/archive/: {e}") | |
| return True | |
| def final_scrub_token_from_env() -> None: | |
| """Remove HF_TOKEN do ambiente após o upload.""" | |
| logger.info("\n[SCRUB] Removendo HF_TOKEN do ambiente...") | |
| if "HF_TOKEN" in os.environ: | |
| del os.environ["HF_TOKEN"] | |
| logger.info(" ✓ HF_TOKEN removido do ambiente") | |
| else: | |
| logger.info(" ✓ HF_TOKEN já não estava no ambiente") | |
| # Verificação | |
| if os.environ.get("HF_TOKEN"): | |
| logger.error(" ✗ HF_TOKEN ainda presente após del!") | |
| else: | |
| logger.info(" ✓ verificação: HF_TOKEN = None") | |
| # Remove cache do HF se existir | |
| cache_token = Path.home() / ".cache" / "huggingface" / "token" | |
| if cache_token.exists(): | |
| cache_token.unlink() | |
| logger.info(f" ✓ cache token removido: {cache_token}") | |
| def main() -> int: | |
| print("=" * 70) | |
| print("V6.7 — BATCH UPLOAD TO HUGGINGFACE (organized paths)") | |
| print("=" * 70) | |
| print(f"Repo: {REPO_ID}") | |
| print(f"Project: {PROJECT_ROOT}") | |
| print(f"Token: {'presente' if HF_TOKEN else 'AUSENTE'}") | |
| print(f"\nEstratégia:") | |
| print(f" 1. Path mapping correto (src/bigru_t/* → src/bigru_t/*)") | |
| print(f" 2. Batch upload (múltiplos commits por etapa)") | |
| print(f" 3. Deprecated scripts → scripts/deprecated/") | |
| print(f" 4. Reports antigos → reports/archive/") | |
| print(f" 5. Scrub HF_TOKEN após upload") | |
| print("=" * 70) | |
| # Verificação inicial | |
| if not verify_local_files(): | |
| logger.error("Verificação local falhou — abortando upload.") | |
| return 1 | |
| # Scrub token dos scripts antes do upload | |
| scrub_token_from_scripts() | |
| # Upload em lote | |
| if not upload_batch(): | |
| logger.error("Upload falhou.") | |
| return 1 | |
| # Scrub token do ambiente após upload | |
| final_scrub_token_from_env() | |
| print("\n" + "=" * 70) | |
| print("✓ V6.7 BATCH UPLOAD COMPLETO") | |
| print("=" * 70) | |
| print(f"\nPróximos passos:") | |
| print(f" 1. Verificar organização no HF:") | |
| print(f" https://huggingface.co/PowerMachine/BiGRU_T_version/tree/main") | |
| print(f" 2. Validar que TODOS os arquivos estão sob src/bigru_t/ ou scripts/") | |
| print(f" 3. Executar FASE1+FASE2 training ( agora com BBPE refit seguro )") | |
| return 0 | |
| if __name__ == "__main__": | |
| sys.exit(main()) | |