PowerMachine commited on
Commit
acb0bc3
·
verified ·
1 Parent(s): 8793991

V6.7: upload scripts/train_v6_5_v2.py (som_auto_adjust_runner + train script V6.7)

Browse files
Files changed (1) hide show
  1. scripts/train_v6_5_v2.py +79 -3
scripts/train_v6_5_v2.py CHANGED
@@ -230,6 +230,17 @@ ALPHA0 = 0.5
230
  SIGMA0 = 2.0
231
  DIM_CHOICE = "y"
232
 
 
 
 
 
 
 
 
 
 
 
 
233
  # V2-dynamic — HypothesisEnsemble parameters (valores canônicos restaurados)
234
  N_HYPOTHESES = 16 # V6.5-V2-metrics-FIX-3: restored from 8
235
  MAX_N_HYPOTHESES = 32 # V6.5-V2-metrics-FIX-3: restored from 16
@@ -990,7 +1001,10 @@ def stream_dataset_in_chunks(
990
  continue
991
 
992
  if sample.raw_text and len(sample.raw_text.strip()) > 0:
993
- chunk.append(sample.raw_text.strip()[:200])
 
 
 
994
  if len(chunk) >= chunk_size:
995
  yield chunk
996
  yielded_total += len(chunk)
@@ -1217,7 +1231,13 @@ def run_fase_conhecimento(
1217
  # fixo (16384 rows), então qualquer crescimento de vocab é acomodado.
1218
  # O SOM adapta-se automaticamente às novas projeções 4D do embedding.
1219
  tokenizer_corpus_buffer: List[str] = []
1220
- TOKENIZER_REFIT_INTERVAL_SAMPLES = 1000 # refit a cada 1000 amostras
 
 
 
 
 
 
1221
  TOKENIZER_CORPUS_CAP = 2000 # cap para evitar OOM (memória: ~2MB)
1222
  last_tokenizer_refit_sample_count = 0
1223
  tokenizer_growth_log: List[Dict[str, Any]] = []
@@ -1511,10 +1531,66 @@ def run_fase_conhecimento(
1511
  samples_since_refit = total_so_far_now - last_tokenizer_refit_sample_count
1512
  if (samples_since_refit >= TOKENIZER_REFIT_INTERVAL_SAMPLES
1513
  and len(tokenizer_corpus_buffer) >= 200):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1514
  # Refit tokenizer com corpus acumulado
1515
  vocab_before = int(getattr(kls.tokenizer, "vocab_size", 0))
1516
  try:
1517
- kls.tokenizer.fit(list(tokenizer_corpus_buffer))
1518
  vocab_after = int(getattr(kls.tokenizer, "vocab_size", 0))
1519
  growth = vocab_after - vocab_before
1520
  tokenizer_growth_log.append({
 
230
  SIGMA0 = 2.0
231
  DIM_CHOICE = "y"
232
 
233
+ # V6.6 — User requirement: "aumentar o truncamento de textos (line 993 do
234
+ # train script) de 200 para 1000+ chars, permitindo mais diversidade de
235
+ # pares byte-level no BBPE".
236
+ # Justificativa matemática: com 200 chars, o BBPE byte-level produz ~200
237
+ # tokens (1 token/byte), limitando a diversidade de pares observados pelo
238
+ # algoritmo de merges. Com 1000+ chars, a diversidade de pares byte-level
239
+ # aumenta ~5×, permitindo que o BBPE aprenda merges mais representativos
240
+ # e efetivamente utilize o vocabulário alvo de 16384 tokens (em vez de
241
+ # ficar limitado a ~293 palavras únicas como na versão word-level).
242
+ TEXT_TRUNCATION_CHARS = 1000
243
+
244
  # V2-dynamic — HypothesisEnsemble parameters (valores canônicos restaurados)
245
  N_HYPOTHESES = 16 # V6.5-V2-metrics-FIX-3: restored from 8
246
  MAX_N_HYPOTHESES = 32 # V6.5-V2-metrics-FIX-3: restored from 16
 
1001
  continue
1002
 
1003
  if sample.raw_text and len(sample.raw_text.strip()) > 0:
1004
+ # V6.6 — User requirement: "aumentar o truncamento de textos
1005
+ # (line 993 do train script) de 200 para 1000+ chars, permitindo
1006
+ # mais diversidade de pares byte-level no BBPE".
1007
+ chunk.append(sample.raw_text.strip()[:TEXT_TRUNCATION_CHARS])
1008
  if len(chunk) >= chunk_size:
1009
  yield chunk
1010
  yielded_total += len(chunk)
 
1231
  # fixo (16384 rows), então qualquer crescimento de vocab é acomodado.
1232
  # O SOM adapta-se automaticamente às novas projeções 4D do embedding.
1233
  tokenizer_corpus_buffer: List[str] = []
1234
+ # V6.7: REATIVADO refit com serial mode + memory guard (fix BBPE OOM)
1235
+ # User requirement: "processar aprimoramento (matemático e lógico) para
1236
+ # resolver: tokenizer-growth refit (was causing crashes during BBPE
1237
+ # parallel training at 1000-sample mark)".
1238
+ # Prova 14: agora serial mode (no fork) + memory guard (skip se RSS>85%)
1239
+ # tornam o refit seguro. Era 10**9 (desativado); agora 1000 (reativado).
1240
+ TOKENIZER_REFIT_INTERVAL_SAMPLES = 1000
1241
  TOKENIZER_CORPUS_CAP = 2000 # cap para evitar OOM (memória: ~2MB)
1242
  last_tokenizer_refit_sample_count = 0
1243
  tokenizer_growth_log: List[Dict[str, Any]] = []
 
1531
  samples_since_refit = total_so_far_now - last_tokenizer_refit_sample_count
1532
  if (samples_since_refit >= TOKENIZER_REFIT_INTERVAL_SAMPLES
1533
  and len(tokenizer_corpus_buffer) >= 200):
1534
+ # V6.7 — MEMORY GUARD before tokenizer refit
1535
+ # User requirement: "tokenizer-growth refit (was causing
1536
+ # crashes during BBPE parallel training at 1000-sample mark)".
1537
+ # Prova 14: ProcessPoolExecutor fork duplica RSS do processo
1538
+ # pai (modelo + tensores + buffers). Se RSS > 85% do cgroup
1539
+ # limit, SKIP refit (non-fatal) para evitar OOM-killer.
1540
+ import os as _os_mod_v67
1541
+ try:
1542
+ with open("/proc/self/status") as _f_v67:
1543
+ _rss_line_v67 = [l for l in _f_v67 if l.startswith("VmRSS:")]
1544
+ _rss_kb_v67 = int(_rss_line_v67[0].split()[1]) if _rss_line_v67 else 0
1545
+ _rss_mb_v67 = _rss_kb_v67 / 1024.0
1546
+ # cgroup limit
1547
+ _cg_limit_mb_v67 = 4096.0 # default fallback
1548
+ try:
1549
+ with open("/sys/fs/cgroup/memory.max") as _f_cg_v67:
1550
+ _cg_val_v67 = _f_cg_v67.read().strip()
1551
+ if _cg_val_v67 and _cg_val_v67 != "max":
1552
+ _cg_limit_mb_v67 = int(_cg_val_v67) / 1024 / 1024
1553
+ except Exception:
1554
+ pass # fallback default
1555
+ _rss_pct_v67 = _rss_mb_v67 / _cg_limit_mb_v67 if _cg_limit_mb_v67 > 0 else 0
1556
+ logger.info(
1557
+ f"[V6.7-memory-guard] RSS={_rss_mb_v67:.0f}MB "
1558
+ f"({100*_rss_pct_v67:.1f}% of cgroup "
1559
+ f"{_cg_limit_mb_v67:.0f}MB)"
1560
+ )
1561
+ if _rss_pct_v67 > 0.85:
1562
+ logger.warning(
1563
+ f"[V6.7-memory-guard] SKIP refit: RSS "
1564
+ f"{100*_rss_pct_v67:.1f}% > 85% of cgroup "
1565
+ f"(would trigger OOM via fork). "
1566
+ f"gc.collect() e prosseguir sem refit."
1567
+ )
1568
+ gc.collect()
1569
+ if hasattr(torch, "cpu") and hasattr(torch.cpu, "empty_cache"):
1570
+ try: torch.cpu.empty_cache()
1571
+ except Exception: pass
1572
+ last_tokenizer_refit_sample_count = total_so_far_now
1573
+ tokenizer_growth_log.append({
1574
+ "step": step,
1575
+ "chunk_idx": chunk_idx_global,
1576
+ "total_samples": total_so_far_now,
1577
+ "skipped_reason": "rss_exceeded_85pct",
1578
+ "rss_mb": _rss_mb_v67,
1579
+ "cgroup_limit_mb": _cg_limit_mb_v67,
1580
+ })
1581
+ # SKIP refit — continue para próximo chunk
1582
+ raise _SkipRefitV67()
1583
+ except _SkipRefitV67:
1584
+ pass # já tratado acima
1585
+ except Exception as _guard_err_v67:
1586
+ logger.warning(
1587
+ f"[V6.7-memory-guard] guard failed (non-fatal): {_guard_err_v67}"
1588
+ )
1589
+
1590
  # Refit tokenizer com corpus acumulado
1591
  vocab_before = int(getattr(kls.tokenizer, "vocab_size", 0))
1592
  try:
1593
+ kls.tokenizer.fit(list(tokenizer_corpus_buffer), min_frequency=2)
1594
  vocab_after = int(getattr(kls.tokenizer, "vocab_size", 0))
1595
  growth = vocab_after - vocab_before
1596
  tokenizer_growth_log.append({