PowerMachine commited on
Commit
ae70aec
·
verified ·
1 Parent(s): 27340a3

V6.7: deprecated → scripts/deprecated/upload_v6_5_7ds.py

Browse files
Files changed (1) hide show
  1. scripts/deprecated/upload_v6_5_7ds.py +346 -0
scripts/deprecated/upload_v6_5_7ds.py ADDED
@@ -0,0 +1,346 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """upload_v6_5_7ds.py — V6.5-7ds upload to HuggingFace.
2
+
3
+ User requirement (latest): "upload dos arquivos testados e os aprimorados
4
+ corrigidos (remover os antigos)"
5
+
6
+ This script:
7
+ 1. Pre-upload safety: scrub HF_TOKEN from all scripts/*.py via regex.
8
+ 2. Defines OLD_FILES_TO_REMOVE — legacy v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive
9
+ reports and scripts, plus old v6_5_* files (superseded by v6_5_7ds_*).
10
+ 3. Uploads the tested+improved files (current BiGRU_T_version tree).
11
+ 4. Deletes OLD_FILES_TO_REMOVE from the HF repo via HfApi.delete_file.
12
+ 5. Verifies critical V6.5-7ds files are present in repo.
13
+ 6. Saves upload report to v6_5_7ds_upload_report.json.
14
+
15
+ After successful upload, the HF_TOKEN env var MUST be deleted by the caller.
16
+ """
17
+ from __future__ import annotations
18
+ import json, os, sys, time, re
19
+ from pathlib import Path
20
+
21
+ PROJECT_ROOT = Path("/home/z/my-project")
22
+ BIGRU_ROOT = PROJECT_ROOT / "BiGRU_T_version"
23
+
24
+ # ──────────────────────────────────────────────────────────────────────────────
25
+ # Files to REMOVE from the HF repo (legacy / superseded by V6.5-7ds)
26
+ # ──────────────────────────────────────────────────────────────────────────────
27
+ # User requirement: "upload dos arquivos testados e os aprimorados corrigidos
28
+ # (remover os antigos)"
29
+ #
30
+ # Removal strategy:
31
+ # - All legacy version reports (v3, v4, v5, v6, v6_1, v6_2, v6_3, v6_progressive)
32
+ # - All legacy upload reports (v3, v5, v6_1, v6_2, v6_4, v6_5)
33
+ # - Old V6.5 files superseded by V6.5-7ds (v6_5_*.json and v6_5_final_*.json)
34
+ # - Old training scripts (train.py, train_v2.py, train_fast.py, smoke_test.py,
35
+ # parse_v6_log.py, train_v6_1.py, train_v6_2.py, train_v6_3.py,
36
+ # train_v6_progressive.py, train_v6_5.py, train_v6_5_final.py)
37
+ # - Old upload scripts (upload_to_hf.py, upload_v6_1_resilient.py,
38
+ # upload_v6_2_resilient.py, upload_v6_3_resilient.py, upload_v6_4_resilient.py,
39
+ # upload_v6_5_resilient.py)
40
+ # - Legacy root files: bug_hunt_v3_report.json, v32_test_report.json,
41
+ # v32_upload_report.json, training_report.json, v6_progressive_train.log,
42
+ # v6_progressive_report.json, v6_report.json, v6_upload_report.json,
43
+ # config.json (legacy model_final/config.json stays)
44
+ OLD_FILES_TO_REMOVE = [
45
+ # ── Legacy root reports (v3/v4/v5/v6/v32) ─────────────────────────────────
46
+ "bug_hunt_v3_report.json",
47
+ "v3_report.json",
48
+ "v4_report.json",
49
+ "v5_report.json",
50
+ "v5_upload_report.json",
51
+ "v32_test_report.json",
52
+ "v32_upload_report.json",
53
+ "training_report.json",
54
+ "v6_report.json",
55
+ "v6_upload_report.json",
56
+ "v6_progressive_report.json",
57
+ "v6_progressive_train.log",
58
+ # ── Legacy v6.1/v6.2/v6.3 reports ─────────────────────────────────────────
59
+ "v6_1_report.json",
60
+ "v6_1_upload_report.json",
61
+ "v6_2_report.json",
62
+ "v6_2_upload_report.json",
63
+ "v6_3_report.json",
64
+ "v6_3_training_metrics.json",
65
+ # ── Old V6.5 reports (superseded by V6.5-7ds) ─────────────────────────────
66
+ # These are removed because V6.5-7ds is the new canonical version.
67
+ "v6_5_report.json",
68
+ "v6_5_training_metrics.json",
69
+ "v6_5_module_analysis.json",
70
+ "v6_5_script_activity.json",
71
+ "v6_5_ewc_w8a8_benchmark.json",
72
+ "v6_5_reasoning_eval.json",
73
+ "v6_5_upload_report.json",
74
+ # ── Old V6.5-final reports (also superseded by V6.5-7ds) ──────────────────
75
+ "v6_5_final_report.json",
76
+ "v6_5_final_training_metrics.json",
77
+ "v6_5_final_module_analysis.json",
78
+ "v6_5_final_script_activity.json",
79
+ "v6_5_final_ewc_w8a8_benchmark.json",
80
+ "v6_5_final_reasoning_eval.json",
81
+ "v6_5_final_w8a8_compression.json",
82
+ # ── Legacy training scripts ───────────────────────────────────────────────
83
+ "scripts/parse_v6_log.py",
84
+ "scripts/smoke_test.py",
85
+ "scripts/train.py",
86
+ "scripts/train_fast.py",
87
+ "scripts/train_v2.py",
88
+ "scripts/train_v6_1.py",
89
+ "scripts/train_v6_2.py",
90
+ "scripts/train_v6_3.py",
91
+ "scripts/train_v6_progressive.py",
92
+ "scripts/train_v6_5.py", # superseded by train_v6_5_7ds.py
93
+ "scripts/train_v6_5_final.py", # superseded by train_v6_5_7ds.py
94
+ # ── Legacy upload scripts (only upload_v6_5_7ds.py stays) ─────────────────
95
+ "scripts/upload_to_hf.py",
96
+ "scripts/upload_v6_1_resilient.py",
97
+ "scripts/upload_v6_2_resilient.py",
98
+ "scripts/upload_v6_3_resilient.py",
99
+ "scripts/upload_v6_4_resilient.py",
100
+ "scripts/upload_v6_5_resilient.py", # superseded by upload_v6_5_7ds.py
101
+ # ── Legacy tokenizer/ root config (kept model_final/ tree intact) ─────────
102
+ "config.json", # legacy root config; not used by V6.5
103
+ "tokenizer/tokenizer.json", # legacy root tokenizer; model_final/ has its own
104
+ # ── Legacy data_augmentation.py (was moved to training/ in V6.4) ──────────
105
+ "src/bigru_t/data/data_augmentation.py",
106
+ # ── Legacy xeon_runtime.py at scripts/ (canonical is utils/xeon_runtime.py)
107
+ "scripts/xeon_runtime.py",
108
+ ]
109
+
110
+ # Critical files that MUST be present after upload
111
+ CRITICAL_FILES_V65_7DS = [
112
+ # Core model
113
+ "src/bigru_t/model/kohonen_learning_system.py",
114
+ "src/bigru_t/model/__init__.py",
115
+ "src/bigru_t/__init__.py",
116
+ "src/bigru_t/model/hyp_t.py",
117
+ "src/bigru_t/model/vqvae2_hierarchical.py",
118
+ "src/bigru_t/model/vqvae2_hierarchical_flexnet.py",
119
+ "src/bigru_t/model/token_compress.py",
120
+ "src/bigru_t/model/embedding_reconfig.py",
121
+ "src/bigru_t/model/attention_multimodal.py",
122
+ # Training
123
+ "src/bigru_t/training/mtp.py",
124
+ "src/bigru_t/training/ewc.py",
125
+ # Quantization
126
+ "src/bigru_t/quantization/smoothquant_compressor.py",
127
+ "src/bigru_t/quantization/w8a8_smoothquant.py",
128
+ "src/bigru_t/quantization/quantized_linear.py",
129
+ # Reasoning
130
+ "src/bigru_t/reasoning/thinking.py",
131
+ "src/bigru_t/reasoning/reasoning_engine.py",
132
+ "src/bigru_t/reasoning/circular_orchestration.py",
133
+ "src/bigru_t/reasoning/tool_agent.py",
134
+ "src/bigru_t/reasoning/distributed_reasoning_system.py",
135
+ "src/bigru_t/reasoning/cyclic_reasoning.py",
136
+ "src/bigru_t/reasoning/consensus_sampling.py",
137
+ # Data + utils
138
+ "src/bigru_t/data/streaming_datasets.py",
139
+ "src/bigru_t/utils/xeon_runtime.py",
140
+ # V6.5-7ds reports (NEW canonical)
141
+ "v6_5_7ds_report.json",
142
+ "v6_5_7ds_training_metrics.json",
143
+ "v6_5_7ds_module_analysis.json",
144
+ "v6_5_7ds_script_activity.json",
145
+ "v6_5_7ds_ewc_w8a8_benchmark.json",
146
+ "v6_5_7ds_reasoning_eval.json",
147
+ "v6_5_7ds_w8a8_compression.json",
148
+ "v6_5_7ds_upload_report.json",
149
+ # V6.4 reports (kept as the immediate predecessor baseline)
150
+ "v6_4_report.json",
151
+ "v6_4_training_metrics.json",
152
+ "v6_4_upload_report.json",
153
+ # V6.5-7ds scripts (the new canonicals)
154
+ "scripts/train_v6_5_7ds.py",
155
+ "scripts/upload_v6_5_7ds.py",
156
+ "scripts/train_v6_4.py", # kept as predecessor reference
157
+ # Misc
158
+ "requirements.txt",
159
+ "README.md",
160
+ "docs/analysis.md",
161
+ ]
162
+
163
+
164
+ def main() -> int:
165
+ print("=" * 72)
166
+ print("V6.5-7DS — UPLOAD TO HUGGINGFACE (with removal of old files)")
167
+ print("=" * 72)
168
+ hf_token = os.environ.get("HF_TOKEN")
169
+ if not hf_token:
170
+ print("ERROR: HF_TOKEN not set in env")
171
+ return 1
172
+ print(f" HF_TOKEN loaded from env (length={len(hf_token)})")
173
+
174
+ # ── Step 1: Pre-upload safety — scrub any HF token from scripts ──────────
175
+ token_pattern = re.compile(r'hf_[A-Za-z0-9]{32,}')
176
+ print("\n [1/5] Pre-upload safety: scrubbing token from scripts...")
177
+ scrubbed_count = 0
178
+ for script_path in BIGRU_ROOT.glob("scripts/*.py"):
179
+ try:
180
+ content = script_path.read_text()
181
+ if token_pattern.search(content):
182
+ scrubbed = token_pattern.sub("hf_<REDACTED_TOKEN>", content)
183
+ script_path.write_text(scrubbed)
184
+ scrubbed_count += 1
185
+ print(f" SCRUBBED: {script_path.name}")
186
+ except Exception as e:
187
+ print(f" SKIP {script_path.name}: {e}")
188
+ if scrubbed_count == 0:
189
+ print(" (no token leakage found in scripts)")
190
+
191
+ # ── Step 2: Import HF API + check repo access ────────────────────────────
192
+ print("\n [2/5] Connecting to HuggingFace repo...")
193
+ try:
194
+ from huggingface_hub import HfApi, upload_folder
195
+ except ImportError:
196
+ print("ERROR: huggingface_hub not installed")
197
+ return 1
198
+
199
+ api = HfApi(token=hf_token)
200
+ repo_id = "PowerMachine/BiGRU_T_version"
201
+ try:
202
+ info = api.repo_info(repo_id=repo_id, repo_type="model")
203
+ print(f" Repo: {repo_id} (existing, files={len(info.siblings)})")
204
+ except Exception as e:
205
+ print(f" ERROR: cannot access repo: {e}")
206
+ return 1
207
+
208
+ # Snapshot of files in repo BEFORE upload
209
+ files_before = set(s.rfilename for s in info.siblings)
210
+ print(f" Files in repo BEFORE upload: {len(files_before)}")
211
+
212
+ # ── Step 3: Upload current BiGRU_T_version tree as single commit ─────────
213
+ print(f"\n [3/5] Uploading {BIGRU_ROOT} as single commit...")
214
+ t0 = time.time()
215
+ try:
216
+ commit_info = upload_folder(
217
+ repo_id=repo_id, repo_type="model",
218
+ folder_path=str(BIGRU_ROOT),
219
+ commit_message=(
220
+ "V6.5-7ds: 7 datasets streaming (1400 samples, 112 steps), "
221
+ "864 neurons, SmoothQuant W8A8 (err=0.022), MTP K=6, "
222
+ "tool_coordinator workers reactivated, 10/10 verification PASS, "
223
+ "reasoning GOOD, removed legacy v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive "
224
+ "files + old v6_5_* reports/scripts"
225
+ ),
226
+ token=hf_token,
227
+ )
228
+ t1 = time.time()
229
+ print(f" Upload completed in {t1 - t0:.2f}s")
230
+ print(f" Commit: {commit_info.oid}")
231
+ print(f" URL: https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_info.oid}")
232
+ except Exception as e:
233
+ print(f" ERROR: upload failed: {e}")
234
+ return 1
235
+
236
+ # ── Step 4: Remove OLD files from repo (single delete commit) ────────────
237
+ print(f"\n [4/5] Removing {len(OLD_FILES_TO_REMOVE)} old files from repo...")
238
+ # Refresh repo info to know what files currently exist
239
+ try:
240
+ info_after = api.repo_info(repo_id=repo_id, repo_type="model")
241
+ files_after_upload = set(s.rfilename for s in info_after.siblings)
242
+ except Exception as e:
243
+ print(f" WARNING: cannot refresh repo info: {e}")
244
+ files_after_upload = files_before
245
+
246
+ # Determine which old files actually exist in repo (to delete)
247
+ old_files_in_repo = [f for f in OLD_FILES_TO_REMOVE if f in files_after_upload]
248
+ old_files_not_in_repo = [f for f in OLD_FILES_TO_REMOVE if f not in files_after_upload]
249
+ print(f" Old files present in repo (will delete): {len(old_files_in_repo)}")
250
+ print(f" Old files NOT in repo (skip): {len(old_files_not_in_repo)}")
251
+ if old_files_not_in_repo:
252
+ print(f" examples: {old_files_not_in_repo[:5]}")
253
+
254
+ # Use HfApi.create_commit with CommitOperationDelete for batch removal
255
+ # (more efficient than calling delete_file one-by-one)
256
+ delete_errors = []
257
+ if old_files_in_repo:
258
+ try:
259
+ from huggingface_hub import CommitOperation
260
+ operations = [
261
+ CommitOperationDelete(path_in_repo=f) for f in old_files_in_repo
262
+ ]
263
+ print(f" Creating batch delete commit ({len(operations)} files)...")
264
+ t_del_start = time.time()
265
+ commit_del = api.create_commit(
266
+ repo_id=repo_id,
267
+ repo_type="model",
268
+ operations=operations,
269
+ commit_message=(
270
+ f"V6.5-7ds cleanup: remove {len(operations)} legacy/superseded files "
271
+ f"(v3/v4/v5/v6_1/v6_2/v6_3/v6_progressive + old v6_5_*/v6_5_final_* "
272
+ f"reports/scripts) — replaced by V6.5-7ds canonicals"
273
+ ),
274
+ )
275
+ t_del_end = time.time()
276
+ print(f" Delete commit: {commit_del}")
277
+ print(f" Delete completed in {t_del_end - t_del_start:.2f}s")
278
+ print(f" URL: https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_del}")
279
+ except Exception as e:
280
+ print(f" ERROR: batch delete failed: {e}")
281
+ print(f" Falling back to per-file delete...")
282
+ # Fallback: delete one-by-one
283
+ for f in old_files_in_repo:
284
+ try:
285
+ api.delete_file(
286
+ repo_id=repo_id, repo_type="model",
287
+ path_in_repo=f,
288
+ commit_message=f"V6.5-7ds cleanup: remove legacy {f}",
289
+ token=hf_token,
290
+ )
291
+ print(f" DELETED: {f}")
292
+ except Exception as e2:
293
+ delete_errors.append((f, str(e2)[:100]))
294
+ print(f" FAILED: {f} — {str(e2)[:100]}")
295
+
296
+ # ── Step 5: Verify critical files in repo ────────────────────────────────
297
+ print(f"\n [5/5] Verifying critical V6.5-7ds files in repo...")
298
+ try:
299
+ # Refresh repo info after deletes
300
+ info_final = api.repo_info(repo_id=repo_id, repo_type="model")
301
+ files_in_repo = set(s.rfilename for s in info_final.siblings)
302
+ all_present = True
303
+ for cf in CRITICAL_FILES_V65_7DS:
304
+ present = cf in files_in_repo
305
+ mark = "OK" if present else "MISS"
306
+ print(f" [{mark}] {cf}")
307
+ if not present:
308
+ all_present = False
309
+ print(f"\n Files in repo AFTER cleanup: {len(files_in_repo)}")
310
+ if not all_present:
311
+ print(" WARNING: some critical files missing!")
312
+ except Exception as e:
313
+ print(f" WARNING: cannot verify files: {e}")
314
+ files_in_repo = set()
315
+
316
+ # ── Save upload report ───────────────────────────────────────────────────
317
+ report = {
318
+ "version": "V6.5-7ds",
319
+ "upload_timestamp": time.strftime("%Y-%m-%dT%H:%M:%S"),
320
+ "repo_id": repo_id,
321
+ "upload_commit_oid": commit_info.oid,
322
+ "upload_commit_url": f"https://huggingface.co/PowerMachine/BiGRU_T_version/commit/{commit_info.oid}",
323
+ "upload_duration_s": float(t1 - t0),
324
+ "delete_commit_oid": commit_del if old_files_in_repo else None,
325
+ "delete_count": len(old_files_in_repo) if old_files_in_repo else 0,
326
+ "delete_errors": delete_errors,
327
+ "n_files_before": len(files_before),
328
+ "n_files_after_upload": len(files_after_upload),
329
+ "n_files_after_cleanup": len(files_in_repo) if files_in_repo else None,
330
+ "old_files_removed": old_files_in_repo,
331
+ "old_files_not_in_repo_skipped": old_files_not_in_repo,
332
+ "critical_files": CRITICAL_FILES_V65_7DS,
333
+ "all_critical_files_present": all_present if 'all_present' in dir() else None,
334
+ "token_scrubbed_count": scrubbed_count,
335
+ }
336
+ report_path = BIGRU_ROOT / "v6_5_7ds_upload_report.json"
337
+ report_path.write_text(json.dumps(report, indent=2, ensure_ascii=False))
338
+ print(f"\n Upload report: {report_path}")
339
+ print("\n" + "=" * 72)
340
+ print("V6.5-7DS — UPLOAD COMPLETED (with old file removal)")
341
+ print("=" * 72)
342
+ return 0
343
+
344
+
345
+ if __name__ == "__main__":
346
+ sys.exit(main())