File size: 16,244 Bytes
ae70aec | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 | """upload_v6_5_7ds.py β V6.5-7ds upload to HuggingFace.
User requirement (latest): "upload dos arquivos testados e os aprimorados
corrigidos (remover os antigos)"
This script:
1. Pre-upload safety: scrub HF_TOKEN from all scripts/*.py via regex.
2. Defines OLD_FILES_TO_REMOVE β legacy v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive
reports and scripts, plus old v6_5_* files (superseded by v6_5_7ds_*).
3. Uploads the tested+improved files (current BiGRU_T_version tree).
4. Deletes OLD_FILES_TO_REMOVE from the HF repo via HfApi.delete_file.
5. Verifies critical V6.5-7ds files are present in repo.
6. Saves upload report to v6_5_7ds_upload_report.json.
After successful upload, the HF_TOKEN env var MUST be deleted by the caller.
"""
from __future__ import annotations
import json, os, sys, time, re
from pathlib import Path
PROJECT_ROOT = Path("/home/z/my-project")
BIGRU_ROOT = PROJECT_ROOT / "BiGRU_T_version"
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# Files to REMOVE from the HF repo (legacy / superseded by V6.5-7ds)
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# User requirement: "upload dos arquivos testados e os aprimorados corrigidos
# (remover os antigos)"
#
# Removal strategy:
# - All legacy version reports (v3, v4, v5, v6, v6_1, v6_2, v6_3, v6_progressive)
# - All legacy upload reports (v3, v5, v6_1, v6_2, v6_4, v6_5)
# - Old V6.5 files superseded by V6.5-7ds (v6_5_*.json and v6_5_final_*.json)
# - Old training scripts (train.py, train_v2.py, train_fast.py, smoke_test.py,
# parse_v6_log.py, train_v6_1.py, train_v6_2.py, train_v6_3.py,
# train_v6_progressive.py, train_v6_5.py, train_v6_5_final.py)
# - Old upload scripts (upload_to_hf.py, upload_v6_1_resilient.py,
# upload_v6_2_resilient.py, upload_v6_3_resilient.py, upload_v6_4_resilient.py,
# upload_v6_5_resilient.py)
# - Legacy root files: bug_hunt_v3_report.json, v32_test_report.json,
# v32_upload_report.json, training_report.json, v6_progressive_train.log,
# v6_progressive_report.json, v6_report.json, v6_upload_report.json,
# config.json (legacy model_final/config.json stays)
OLD_FILES_TO_REMOVE = [
# ββ Legacy root reports (v3/v4/v5/v6/v32) βββββββββββββββββββββββββββββββββ
"bug_hunt_v3_report.json",
"v3_report.json",
"v4_report.json",
"v5_report.json",
"v5_upload_report.json",
"v32_test_report.json",
"v32_upload_report.json",
"training_report.json",
"v6_report.json",
"v6_upload_report.json",
"v6_progressive_report.json",
"v6_progressive_train.log",
# ββ Legacy v6.1/v6.2/v6.3 reports βββββββββββββββββββββββββββββββββββββββββ
"v6_1_report.json",
"v6_1_upload_report.json",
"v6_2_report.json",
"v6_2_upload_report.json",
"v6_3_report.json",
"v6_3_training_metrics.json",
# ββ Old V6.5 reports (superseded by V6.5-7ds) βββββββββββββββββββββββββββββ
# These are removed because V6.5-7ds is the new canonical version.
"v6_5_report.json",
"v6_5_training_metrics.json",
"v6_5_module_analysis.json",
"v6_5_script_activity.json",
"v6_5_ewc_w8a8_benchmark.json",
"v6_5_reasoning_eval.json",
"v6_5_upload_report.json",
# ββ Old V6.5-final reports (also superseded by V6.5-7ds) ββββββββββββββββββ
"v6_5_final_report.json",
"v6_5_final_training_metrics.json",
"v6_5_final_module_analysis.json",
"v6_5_final_script_activity.json",
"v6_5_final_ewc_w8a8_benchmark.json",
"v6_5_final_reasoning_eval.json",
"v6_5_final_w8a8_compression.json",
# ββ Legacy training scripts βββββββββββββββββββββββββββββββββββββββββββββββ
"scripts/parse_v6_log.py",
"scripts/smoke_test.py",
"scripts/train.py",
"scripts/train_fast.py",
"scripts/train_v2.py",
"scripts/train_v6_1.py",
"scripts/train_v6_2.py",
"scripts/train_v6_3.py",
"scripts/train_v6_progressive.py",
"scripts/train_v6_5.py", # superseded by train_v6_5_7ds.py
"scripts/train_v6_5_final.py", # superseded by train_v6_5_7ds.py
# ββ Legacy upload scripts (only upload_v6_5_7ds.py stays) βββββββββββββββββ
"scripts/upload_to_hf.py",
"scripts/upload_v6_1_resilient.py",
"scripts/upload_v6_2_resilient.py",
"scripts/upload_v6_3_resilient.py",
"scripts/upload_v6_4_resilient.py",
"scripts/upload_v6_5_resilient.py", # superseded by upload_v6_5_7ds.py
# ββ Legacy tokenizer/ root config (kept model_final/ tree intact) βββββββββ
"config.json", # legacy root config; not used by V6.5
"tokenizer/tokenizer.json", # legacy root tokenizer; model_final/ has its own
# ββ Legacy data_augmentation.py (was moved to training/ in V6.4) ββββββββββ
"src/bigru_t/data/data_augmentation.py",
# ββ Legacy xeon_runtime.py at scripts/ (canonical is utils/xeon_runtime.py)
"scripts/xeon_runtime.py",
]
# Critical files that MUST be present after upload
CRITICAL_FILES_V65_7DS = [
# Core model
"src/bigru_t/model/kohonen_learning_system.py",
"src/bigru_t/model/__init__.py",
"src/bigru_t/__init__.py",
"src/bigru_t/model/hyp_t.py",
"src/bigru_t/model/vqvae2_hierarchical.py",
"src/bigru_t/model/vqvae2_hierarchical_flexnet.py",
"src/bigru_t/model/token_compress.py",
"src/bigru_t/model/embedding_reconfig.py",
"src/bigru_t/model/attention_multimodal.py",
# Training
"src/bigru_t/training/mtp.py",
"src/bigru_t/training/ewc.py",
# Quantization
"src/bigru_t/quantization/smoothquant_compressor.py",
"src/bigru_t/quantization/w8a8_smoothquant.py",
"src/bigru_t/quantization/quantized_linear.py",
# Reasoning
"src/bigru_t/reasoning/thinking.py",
"src/bigru_t/reasoning/reasoning_engine.py",
"src/bigru_t/reasoning/circular_orchestration.py",
"src/bigru_t/reasoning/tool_agent.py",
"src/bigru_t/reasoning/distributed_reasoning_system.py",
"src/bigru_t/reasoning/cyclic_reasoning.py",
"src/bigru_t/reasoning/consensus_sampling.py",
# Data + utils
"src/bigru_t/data/streaming_datasets.py",
"src/bigru_t/utils/xeon_runtime.py",
# V6.5-7ds reports (NEW canonical)
"v6_5_7ds_report.json",
"v6_5_7ds_training_metrics.json",
"v6_5_7ds_module_analysis.json",
"v6_5_7ds_script_activity.json",
"v6_5_7ds_ewc_w8a8_benchmark.json",
"v6_5_7ds_reasoning_eval.json",
"v6_5_7ds_w8a8_compression.json",
"v6_5_7ds_upload_report.json",
# V6.4 reports (kept as the immediate predecessor baseline)
"v6_4_report.json",
"v6_4_training_metrics.json",
"v6_4_upload_report.json",
# V6.5-7ds scripts (the new canonicals)
"scripts/train_v6_5_7ds.py",
"scripts/upload_v6_5_7ds.py",
"scripts/train_v6_4.py", # kept as predecessor reference
# Misc
"requirements.txt",
"README.md",
"docs/analysis.md",
]
def main() -> int:
print("=" * 72)
print("V6.5-7DS β UPLOAD TO HUGGINGFACE (with removal of old files)")
print("=" * 72)
hf_token = os.environ.get("HF_TOKEN")
if not hf_token:
print("ERROR: HF_TOKEN not set in env")
return 1
print(f" HF_TOKEN loaded from env (length={len(hf_token)})")
# ββ Step 1: Pre-upload safety β scrub any HF token from scripts ββββββββββ
token_pattern = re.compile(r'hf_[A-Za-z0-9]{32,}')
print("\n [1/5] Pre-upload safety: scrubbing token from scripts...")
scrubbed_count = 0
for script_path in BIGRU_ROOT.glob("scripts/*.py"):
try:
content = script_path.read_text()
if token_pattern.search(content):
scrubbed = token_pattern.sub("hf_<REDACTED_TOKEN>", content)
script_path.write_text(scrubbed)
scrubbed_count += 1
print(f" SCRUBBED: {script_path.name}")
except Exception as e:
print(f" SKIP {script_path.name}: {e}")
if scrubbed_count == 0:
print(" (no token leakage found in scripts)")
# ββ Step 2: Import HF API + check repo access ββββββββββββββββββββββββββββ
print("\n [2/5] Connecting to HuggingFace repo...")
try:
from huggingface_hub import HfApi, upload_folder
except ImportError:
print("ERROR: huggingface_hub not installed")
return 1
api = HfApi(token=hf_token)
repo_id = "PowerMachine/BiGRU_T_version"
try:
info = api.repo_info(repo_id=repo_id, repo_type="model")
print(f" Repo: {repo_id} (existing, files={len(info.siblings)})")
except Exception as e:
print(f" ERROR: cannot access repo: {e}")
return 1
# Snapshot of files in repo BEFORE upload
files_before = set(s.rfilename for s in info.siblings)
print(f" Files in repo BEFORE upload: {len(files_before)}")
# ββ Step 3: Upload current BiGRU_T_version tree as single commit βββββββββ
print(f"\n [3/5] Uploading {BIGRU_ROOT} as single commit...")
t0 = time.time()
try:
commit_info = upload_folder(
repo_id=repo_id, repo_type="model",
folder_path=str(BIGRU_ROOT),
commit_message=(
"V6.5-7ds: 7 datasets streaming (1400 samples, 112 steps), "
"864 neurons, SmoothQuant W8A8 (err=0.022), MTP K=6, "
"tool_coordinator workers reactivated, 10/10 verification PASS, "
"reasoning GOOD, removed legacy v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive "
"files + old v6_5_* reports/scripts"
),
token=hf_token,
)
t1 = time.time()
print(f" Upload completed in {t1 - t0:.2f}s")
print(f" Commit: {commit_info.oid}")
print(f" URL: https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_info.oid}")
except Exception as e:
print(f" ERROR: upload failed: {e}")
return 1
# ββ Step 4: Remove OLD files from repo (single delete commit) ββββββββββββ
print(f"\n [4/5] Removing {len(OLD_FILES_TO_REMOVE)} old files from repo...")
# Refresh repo info to know what files currently exist
try:
info_after = api.repo_info(repo_id=repo_id, repo_type="model")
files_after_upload = set(s.rfilename for s in info_after.siblings)
except Exception as e:
print(f" WARNING: cannot refresh repo info: {e}")
files_after_upload = files_before
# Determine which old files actually exist in repo (to delete)
old_files_in_repo = [f for f in OLD_FILES_TO_REMOVE if f in files_after_upload]
old_files_not_in_repo = [f for f in OLD_FILES_TO_REMOVE if f not in files_after_upload]
print(f" Old files present in repo (will delete): {len(old_files_in_repo)}")
print(f" Old files NOT in repo (skip): {len(old_files_not_in_repo)}")
if old_files_not_in_repo:
print(f" examples: {old_files_not_in_repo[:5]}")
# Use HfApi.create_commit with CommitOperationDelete for batch removal
# (more efficient than calling delete_file one-by-one)
delete_errors = []
if old_files_in_repo:
try:
from huggingface_hub import CommitOperation
operations = [
CommitOperationDelete(path_in_repo=f) for f in old_files_in_repo
]
print(f" Creating batch delete commit ({len(operations)} files)...")
t_del_start = time.time()
commit_del = api.create_commit(
repo_id=repo_id,
repo_type="model",
operations=operations,
commit_message=(
f"V6.5-7ds cleanup: remove {len(operations)} legacy/superseded files "
f"(v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive + old v6_5_*/v6_5_final_* "
f"reports/scripts) β replaced by V6.5-7ds canonicals"
),
)
t_del_end = time.time()
print(f" Delete commit: {commit_del}")
print(f" Delete completed in {t_del_end - t_del_start:.2f}s")
print(f" URL: https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_del}")
except Exception as e:
print(f" ERROR: batch delete failed: {e}")
print(f" Falling back to per-file delete...")
# Fallback: delete one-by-one
for f in old_files_in_repo:
try:
api.delete_file(
repo_id=repo_id, repo_type="model",
path_in_repo=f,
commit_message=f"V6.5-7ds cleanup: remove legacy {f}",
token=hf_token,
)
print(f" DELETED: {f}")
except Exception as e2:
delete_errors.append((f, str(e2)[:100]))
print(f" FAILED: {f} β {str(e2)[:100]}")
# ββ Step 5: Verify critical files in repo ββββββββββββββββββββββββββββββββ
print(f"\n [5/5] Verifying critical V6.5-7ds files in repo...")
try:
# Refresh repo info after deletes
info_final = api.repo_info(repo_id=repo_id, repo_type="model")
files_in_repo = set(s.rfilename for s in info_final.siblings)
all_present = True
for cf in CRITICAL_FILES_V65_7DS:
present = cf in files_in_repo
mark = "OK" if present else "MISS"
print(f" [{mark}] {cf}")
if not present:
all_present = False
print(f"\n Files in repo AFTER cleanup: {len(files_in_repo)}")
if not all_present:
print(" WARNING: some critical files missing!")
except Exception as e:
print(f" WARNING: cannot verify files: {e}")
files_in_repo = set()
# ββ Save upload report βββββββββββββββββββββββββββββββββββββββββββββββββββ
report = {
"version": "V6.5-7ds",
"upload_timestamp": time.strftime("%Y-%m-%dT%H:%M:%S"),
"repo_id": repo_id,
"upload_commit_oid": commit_info.oid,
"upload_commit_url": f"https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_info.oid}",
"upload_duration_s": float(t1 - t0),
"delete_commit_oid": commit_del if old_files_in_repo else None,
"delete_count": len(old_files_in_repo) if old_files_in_repo else 0,
"delete_errors": delete_errors,
"n_files_before": len(files_before),
"n_files_after_upload": len(files_after_upload),
"n_files_after_cleanup": len(files_in_repo) if files_in_repo else None,
"old_files_removed": old_files_in_repo,
"old_files_not_in_repo_skipped": old_files_not_in_repo,
"critical_files": CRITICAL_FILES_V65_7DS,
"all_critical_files_present": all_present if 'all_present' in dir() else None,
"token_scrubbed_count": scrubbed_count,
}
report_path = BIGRU_ROOT / "v6_5_7ds_upload_report.json"
report_path.write_text(json.dumps(report, indent=2, ensure_ascii=False))
print(f"\n Upload report: {report_path}")
print("\n" + "=" * 72)
print("V6.5-7DS β UPLOAD COMPLETED (with old file removal)")
print("=" * 72)
return 0
if __name__ == "__main__":
sys.exit(main())
|