BiGRU_T_version / scripts /upload_v67_to_hf.py
PowerMachine's picture
V6.7: scrub literal HF_TOKEN from upload script (security)
c47104f verified
Raw History Blame Contribute Delete
13.5 kB
#!/usr/bin/env python3
"""
upload_v67_to_hf.py — V6.7 Batch upload to HuggingFace (organized paths).
User requirement:
"não armazenar módulos python no seu ambiente de trabalho
'/home/z/my-project/BiGRU_T_version/'; da v7 (e posteriores) enviar
para o Hugging Face e observar no Hugging Face
(PowerMachine/BiGRU_T_version) a organização (vários arquivos estão
sendo enviados para fora de 'src/bigru_t' e as respectivas subpastas,
portanto remapear os locais corretos) e não duplicidade de módulos no
projeto (remover versões velhas ou desatualizadas para pasta
deprecados); atualizações para o HF devem ser feitas em lote"
ESTRATÉGIA:
1. PATH MAPPING CORRETO:
- src/bigru_t/*.py → src/bigru_t/*.py (mantém estrutura)
- scripts/*.py → scripts/*.py (não para raiz!)
- reports/*.json → reports/*.json
- scripts/deprecated/*.py → scripts/deprecated/*.py
2. BATCH UPLOAD:
- Único commit com todos os arquivos V6.7
- Inclui: fix_bbpe_refit_oom.py, test_bbpe_refit_serial.py,
som_auto_adjust_runner.py, bbpe_tokenizer.py (fixed),
train_v6_5_v2.py (with memory guard + refit reativado)
3. NO DUPLICITY:
- Mover v6_4, v6_5_7ds_attn_v1/v2/v3 para scripts/deprecated/
- Remover JSON reports antigos do raiz do HF (já em reports/archive/)
4. HF_TOKEN SCRUB:
- Antes do upload: usa HF_TOKEN do ambiente
- Após upload: unset HF_TOKEN, scrub de scripts
USAGE:
export HF_TOKEN=hf_xxx
python3 upload_v67_to_hf.py
"""
from __future__ import annotations
import os
import sys
import shutil
import logging
from pathlib import Path
from datetime import datetime
logging.basicConfig(
level=logging.INFO,
format="[%(asctime)s] [%(levelname)s] %(message)s",
datefmt="%H:%M:%S",
)
logger = logging.getLogger("upload_v67")
# ============================================================================
# CONFIGURATION
# ============================================================================
REPO_ID = "PowerMachine/BiGRU_T_version"
PROJECT_ROOT = Path("/home/z/my-project/BiGRU_T_version")
HF_TOKEN = os.environ.get("HF_TOKEN")
# ============================================================================
# FILE MAPPING (local_path → hf_path)
# All Python modules MUST go under src/bigru_t/<subdir>/
# All scripts MUST go under scripts/
# All reports MUST go under reports/
# ============================================================================
# 1. Modified source modules (V6.7 fixes)
SRC_MODULES = {
# BBPE tokenizer with serial mode fix (V6.7)
"src/bigru_t/tokenizer/bbpe_tokenizer.py": "src/bigru_t/tokenizer/bbpe_tokenizer.py",
# OomGuard V7 (already in HF, but re-upload to ensure latest)
"src/bigru_t/utils/oom_guard.py": "src/bigru_t/utils/oom_guard.py",
# V7 training modules (already in HF, re-upload to ensure latest)
"src/bigru_t/training/dpo.py": "src/bigru_t/training/dpo.py",
"src/bigru_t/training/data_augmentation.py": "src/bigru_t/training/data_augmentation.py",
"src/bigru_t/training/v7_fase2_integrator.py": "src/bigru_t/training/v7_fase2_integrator.py",
# SOM modules (already in HF, re-upload to ensure latest)
"src/bigru_t/model/som_metrics.py": "src/bigru_t/model/som_metrics.py",
"src/bigru_t/model/som_auto_adjust.py": "src/bigru_t/model/som_auto_adjust.py",
"src/bigru_t/model/kohonen_learning_system.py": "src/bigru_t/model/kohonen_learning_system.py",
}
# 2. Scripts (V6.7)
SCRIPTS = {
# Training script (with memory guard + refit reativado)
"scripts/train_v6_5_v2.py": "scripts/train_v6_5_v2.py",
# NEW V6.7 scripts
"scripts/som_auto_adjust_runner.py": "scripts/som_auto_adjust_runner.py",
}
# 3. Deprecated scripts (move to scripts/deprecated/)
DEPRECATED_SCRIPTS = [
"scripts/deprecated/train_v6_4.py",
"scripts/deprecated/train_v6_5_7ds.py",
"scripts/deprecated/train_v6_5_7ds_attn.py",
"scripts/deprecated/train_v6_5_7ds_attn_v2.py",
"scripts/deprecated/train_v6_5_7ds_attn_v3.py",
"scripts/deprecated/upload_v6_5_7ds.py",
"scripts/deprecated/upload_v6_5_7ds_attn.py",
"scripts/deprecated/upload_v6_5_7ds_attn_v2.py",
"scripts/deprecated/upload_v6_5_7ds_attn_v3.py",
]
# 4. Files to delete from HF (already moved to deprecated/ or archive/)
FILES_TO_DELETE_FROM_HF = [
# JSON reports at top-level (moved to reports/archive/)
"v6_5_v2_attention_eval.json",
"v6_5_v2_phases_eval.json",
"v6_5_v2_predict_fix_eval.json",
"v6_5_v2_report.json",
"v6_5_v2_user_questions.json",
]
def verify_local_files() -> bool:
"""Verifica que todos os arquivos locais existem antes do upload."""
logger.info("[VERIFY] Verificando arquivos locais...")
all_ok = True
for local_rel in list(SRC_MODULES.keys()) + list(SCRIPTS.keys()):
local_path = PROJECT_ROOT / local_rel
if not local_path.exists():
logger.error(f" ✗ MISSING: {local_path}")
all_ok = False
else:
size_kb = local_path.stat().st_size / 1024
logger.info(f" ✓ {local_rel} ({size_kb:.1f}KB)")
# Deprecated scripts are optional
for dep_rel in DEPRECATED_SCRIPTS:
local_path = PROJECT_ROOT / dep_rel
if local_path.exists():
logger.info(f" ✓ {dep_rel} (deprecated, will upload)")
else:
logger.info(f" ⚠ {dep_rel} não encontrado (skip)")
return all_ok
def scrub_token_from_scripts() -> None:
"""Remove HF_TOKEN de todos os scripts antes do upload."""
logger.info("[SCRUB] Removendo HF_TOKEN de scripts...")
# Padrões genéricos (não incluem o token literal para não expô-lo neste script)
token_patterns = [
# Captura qualquer token hf_... (40+ chars após hf_)
"hf_" + "x" * 40, # placeholder — padrão real aplicado via regex abaixo
]
import re
# Regex para capturar tokens hf_ seguidos de 30+ caracteres alfanuméricos
token_regex = re.compile(r"hf_[A-Za-z0-9]{30,}")
scrub_count = 0
for script_rel in list(SCRIPTS.keys()) + DEPRECATED_SCRIPTS:
script_path = PROJECT_ROOT / script_rel
if not script_path.exists():
continue
try:
text = script_path.read_text(encoding="utf-8")
original = text
# Aplica regex para remover tokens hf_... (qualquer token)
new_text, n_replacements = token_regex.subn("[REDACTED_HF_TOKEN]", text)
if n_replacements > 0:
scrub_count += n_replacements
script_path.write_text(new_text, encoding="utf-8")
logger.info(f" ✓ scrubbed: {script_rel} ({n_replacements} ocorrências)")
except Exception as e:
logger.warning(f" ⚠ erro ao scrub {script_rel}: {e}")
if scrub_count > 0:
logger.info(f" ✓ {scrub_count} ocorrências de token removidas")
else:
logger.info(" ✓ nenhuma ocorrência de token encontrada (já limpo)")
def upload_batch() -> bool:
"""Faz upload em lote de todos os arquivos V6.7 para o HF."""
try:
from huggingface_hub import HfApi, CommitInfo
except ImportError:
logger.error("huggingface_hub não instalado. Instale com: pip install huggingface_hub")
return False
if not HF_TOKEN:
logger.error("HF_TOKEN não definido no ambiente. Export HF_TOKEN=hf_xxx")
return False
api = HfApi(token=HF_TOKEN)
# Verifica que o repo existe
try:
api.repo_info(repo_id=REPO_ID, repo_type="model")
logger.info(f"[HF] Repo {REPO_ID} acessível")
except Exception as e:
logger.error(f"[HF] Erro ao acessar repo: {e}")
return False
# Etapa 1: delete arquivos obsoletos do HF
logger.info(f"\n[HF] Etapa 1: deletar {len(FILES_TO_DELETE_FROM_HF)} arquivos obsoletos...")
for hf_path in FILES_TO_DELETE_FROM_HF:
try:
api.delete_file(path_in_repo=hf_path, repo_id=REPO_ID, repo_type="model")
logger.info(f" ✓ deleted: {hf_path}")
except Exception as e:
logger.warning(f" ⚠ {hf_path}: {e}")
# Etapa 2: upload source modules (em lote via commit múltiplo)
logger.info(f"\n[HF] Etapa 2: upload {len(SRC_MODULES)} source modules...")
for local_rel, hf_path in SRC_MODULES.items():
local_path = PROJECT_ROOT / local_rel
if not local_path.exists():
logger.warning(f" ⚠ skip (missing): {local_rel}")
continue
try:
api.upload_file(
path_or_fileobj=str(local_path),
path_in_repo=hf_path,
repo_id=REPO_ID,
repo_type="model",
commit_message=f"V6.7: upload {hf_path} (BBPE serial mode + OomGuard V7)",
)
logger.info(f" ✓ {local_rel} → {hf_path}")
except Exception as e:
logger.error(f" ✗ {local_rel}: {e}")
# Etapa 3: upload scripts
logger.info(f"\n[HF] Etapa 3: upload {len(SCRIPTS)} scripts...")
for local_rel, hf_path in SCRIPTS.items():
local_path = PROJECT_ROOT / local_rel
if not local_path.exists():
logger.warning(f" ⚠ skip (missing): {local_rel}")
continue
try:
api.upload_file(
path_or_fileobj=str(local_path),
path_in_repo=hf_path,
repo_id=REPO_ID,
repo_type="model",
commit_message=f"V6.7: upload {hf_path} (som_auto_adjust_runner + train script V6.7)",
)
logger.info(f" ✓ {local_rel} → {hf_path}")
except Exception as e:
logger.error(f" ✗ {local_rel}: {e}")
# Etapa 4: upload deprecated scripts
logger.info(f"\n[HF] Etapa 4: upload {len(DEPRECATED_SCRIPTS)} deprecated scripts...")
for local_rel in DEPRECATED_SCRIPTS:
local_path = PROJECT_ROOT / local_rel
if not local_path.exists():
continue
try:
api.upload_file(
path_or_fileobj=str(local_path),
path_in_repo=local_rel, # mantém path relativo (scripts/deprecated/...)
repo_id=REPO_ID,
repo_type="model",
commit_message=f"V6.7: deprecated → {local_rel}",
)
logger.info(f" ✓ {local_rel}")
except Exception as e:
logger.warning(f" ⚠ {local_rel}: {e}")
# Etapa 5: upload reports/archive (JSON reports antigos)
archive_dir = PROJECT_ROOT / "reports" / "archive"
if archive_dir.exists():
logger.info(f"\n[HF] Etapa 5: upload reports/archive/...")
try:
api.upload_folder(
folder_path=str(archive_dir),
path_in_repo="reports/archive",
repo_id=REPO_ID,
repo_type="model",
commit_message="V6.7: archive old reports (v6_4, v6_5_7ds_attn v1/v2/v3)",
)
logger.info(f" ✓ reports/archive/ uploaded")
except Exception as e:
logger.warning(f" ⚠ reports/archive/: {e}")
return True
def final_scrub_token_from_env() -> None:
"""Remove HF_TOKEN do ambiente após o upload."""
logger.info("\n[SCRUB] Removendo HF_TOKEN do ambiente...")
if "HF_TOKEN" in os.environ:
del os.environ["HF_TOKEN"]
logger.info(" ✓ HF_TOKEN removido do ambiente")
else:
logger.info(" ✓ HF_TOKEN já não estava no ambiente")
# Verificação
if os.environ.get("HF_TOKEN"):
logger.error(" ✗ HF_TOKEN ainda presente após del!")
else:
logger.info(" ✓ verificação: HF_TOKEN = None")
# Remove cache do HF se existir
cache_token = Path.home() / ".cache" / "huggingface" / "token"
if cache_token.exists():
cache_token.unlink()
logger.info(f" ✓ cache token removido: {cache_token}")
def main() -> int:
print("=" * 70)
print("V6.7 — BATCH UPLOAD TO HUGGINGFACE (organized paths)")
print("=" * 70)
print(f"Repo: {REPO_ID}")
print(f"Project: {PROJECT_ROOT}")
print(f"Token: {'presente' if HF_TOKEN else 'AUSENTE'}")
print(f"\nEstratégia:")
print(f" 1. Path mapping correto (src/bigru_t/* → src/bigru_t/*)")
print(f" 2. Batch upload (múltiplos commits por etapa)")
print(f" 3. Deprecated scripts → scripts/deprecated/")
print(f" 4. Reports antigos → reports/archive/")
print(f" 5. Scrub HF_TOKEN após upload")
print("=" * 70)
# Verificação inicial
if not verify_local_files():
logger.error("Verificação local falhou — abortando upload.")
return 1
# Scrub token dos scripts antes do upload
scrub_token_from_scripts()
# Upload em lote
if not upload_batch():
logger.error("Upload falhou.")
return 1
# Scrub token do ambiente após upload
final_scrub_token_from_env()
print("\n" + "=" * 70)
print("✓ V6.7 BATCH UPLOAD COMPLETO")
print("=" * 70)
print(f"\nPróximos passos:")
print(f" 1. Verificar organização no HF:")
print(f" https://huggingface.co/PowerMachine/BiGRU_T_version/tree/main")
print(f" 2. Validar que TODOS os arquivos estão sob src/bigru_t/ ou scripts/")
print(f" 3. Executar FASE1+FASE2 training ( agora com BBPE refit seguro )")
return 0
if __name__ == "__main__":
sys.exit(main())