V6.7: upload scripts/train_v6_5_v2.py (som_auto_adjust_runner + train script V6.7)
Browse files- scripts/train_v6_5_v2.py +79 -3
scripts/train_v6_5_v2.py
CHANGED
|
@@ -230,6 +230,17 @@ ALPHA0 = 0.5
|
|
| 230 |
SIGMA0 = 2.0
|
| 231 |
DIM_CHOICE = "y"
|
| 232 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 233 |
# V2-dynamic — HypothesisEnsemble parameters (valores canônicos restaurados)
|
| 234 |
N_HYPOTHESES = 16 # V6.5-V2-metrics-FIX-3: restored from 8
|
| 235 |
MAX_N_HYPOTHESES = 32 # V6.5-V2-metrics-FIX-3: restored from 16
|
|
@@ -990,7 +1001,10 @@ def stream_dataset_in_chunks(
|
|
| 990 |
continue
|
| 991 |
|
| 992 |
if sample.raw_text and len(sample.raw_text.strip()) > 0:
|
| 993 |
-
|
|
|
|
|
|
|
|
|
|
| 994 |
if len(chunk) >= chunk_size:
|
| 995 |
yield chunk
|
| 996 |
yielded_total += len(chunk)
|
|
@@ -1217,7 +1231,13 @@ def run_fase_conhecimento(
|
|
| 1217 |
# fixo (16384 rows), então qualquer crescimento de vocab é acomodado.
|
| 1218 |
# O SOM adapta-se automaticamente às novas projeções 4D do embedding.
|
| 1219 |
tokenizer_corpus_buffer: List[str] = []
|
| 1220 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1221 |
TOKENIZER_CORPUS_CAP = 2000 # cap para evitar OOM (memória: ~2MB)
|
| 1222 |
last_tokenizer_refit_sample_count = 0
|
| 1223 |
tokenizer_growth_log: List[Dict[str, Any]] = []
|
|
@@ -1511,10 +1531,66 @@ def run_fase_conhecimento(
|
|
| 1511 |
samples_since_refit = total_so_far_now - last_tokenizer_refit_sample_count
|
| 1512 |
if (samples_since_refit >= TOKENIZER_REFIT_INTERVAL_SAMPLES
|
| 1513 |
and len(tokenizer_corpus_buffer) >= 200):
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1514 |
# Refit tokenizer com corpus acumulado
|
| 1515 |
vocab_before = int(getattr(kls.tokenizer, "vocab_size", 0))
|
| 1516 |
try:
|
| 1517 |
-
kls.tokenizer.fit(list(tokenizer_corpus_buffer))
|
| 1518 |
vocab_after = int(getattr(kls.tokenizer, "vocab_size", 0))
|
| 1519 |
growth = vocab_after - vocab_before
|
| 1520 |
tokenizer_growth_log.append({
|
|
|
|
| 230 |
SIGMA0 = 2.0
|
| 231 |
DIM_CHOICE = "y"
|
| 232 |
|
| 233 |
+
# V6.6 — User requirement: "aumentar o truncamento de textos (line 993 do
|
| 234 |
+
# train script) de 200 para 1000+ chars, permitindo mais diversidade de
|
| 235 |
+
# pares byte-level no BBPE".
|
| 236 |
+
# Justificativa matemática: com 200 chars, o BBPE byte-level produz ~200
|
| 237 |
+
# tokens (1 token/byte), limitando a diversidade de pares observados pelo
|
| 238 |
+
# algoritmo de merges. Com 1000+ chars, a diversidade de pares byte-level
|
| 239 |
+
# aumenta ~5×, permitindo que o BBPE aprenda merges mais representativos
|
| 240 |
+
# e efetivamente utilize o vocabulário alvo de 16384 tokens (em vez de
|
| 241 |
+
# ficar limitado a ~293 palavras únicas como na versão word-level).
|
| 242 |
+
TEXT_TRUNCATION_CHARS = 1000
|
| 243 |
+
|
| 244 |
# V2-dynamic — HypothesisEnsemble parameters (valores canônicos restaurados)
|
| 245 |
N_HYPOTHESES = 16 # V6.5-V2-metrics-FIX-3: restored from 8
|
| 246 |
MAX_N_HYPOTHESES = 32 # V6.5-V2-metrics-FIX-3: restored from 16
|
|
|
|
| 1001 |
continue
|
| 1002 |
|
| 1003 |
if sample.raw_text and len(sample.raw_text.strip()) > 0:
|
| 1004 |
+
# V6.6 — User requirement: "aumentar o truncamento de textos
|
| 1005 |
+
# (line 993 do train script) de 200 para 1000+ chars, permitindo
|
| 1006 |
+
# mais diversidade de pares byte-level no BBPE".
|
| 1007 |
+
chunk.append(sample.raw_text.strip()[:TEXT_TRUNCATION_CHARS])
|
| 1008 |
if len(chunk) >= chunk_size:
|
| 1009 |
yield chunk
|
| 1010 |
yielded_total += len(chunk)
|
|
|
|
| 1231 |
# fixo (16384 rows), então qualquer crescimento de vocab é acomodado.
|
| 1232 |
# O SOM adapta-se automaticamente às novas projeções 4D do embedding.
|
| 1233 |
tokenizer_corpus_buffer: List[str] = []
|
| 1234 |
+
# V6.7: REATIVADO refit com serial mode + memory guard (fix BBPE OOM)
|
| 1235 |
+
# User requirement: "processar aprimoramento (matemático e lógico) para
|
| 1236 |
+
# resolver: tokenizer-growth refit (was causing crashes during BBPE
|
| 1237 |
+
# parallel training at 1000-sample mark)".
|
| 1238 |
+
# Prova 14: agora serial mode (no fork) + memory guard (skip se RSS>85%)
|
| 1239 |
+
# tornam o refit seguro. Era 10**9 (desativado); agora 1000 (reativado).
|
| 1240 |
+
TOKENIZER_REFIT_INTERVAL_SAMPLES = 1000
|
| 1241 |
TOKENIZER_CORPUS_CAP = 2000 # cap para evitar OOM (memória: ~2MB)
|
| 1242 |
last_tokenizer_refit_sample_count = 0
|
| 1243 |
tokenizer_growth_log: List[Dict[str, Any]] = []
|
|
|
|
| 1531 |
samples_since_refit = total_so_far_now - last_tokenizer_refit_sample_count
|
| 1532 |
if (samples_since_refit >= TOKENIZER_REFIT_INTERVAL_SAMPLES
|
| 1533 |
and len(tokenizer_corpus_buffer) >= 200):
|
| 1534 |
+
# V6.7 — MEMORY GUARD before tokenizer refit
|
| 1535 |
+
# User requirement: "tokenizer-growth refit (was causing
|
| 1536 |
+
# crashes during BBPE parallel training at 1000-sample mark)".
|
| 1537 |
+
# Prova 14: ProcessPoolExecutor fork duplica RSS do processo
|
| 1538 |
+
# pai (modelo + tensores + buffers). Se RSS > 85% do cgroup
|
| 1539 |
+
# limit, SKIP refit (non-fatal) para evitar OOM-killer.
|
| 1540 |
+
import os as _os_mod_v67
|
| 1541 |
+
try:
|
| 1542 |
+
with open("/proc/self/status") as _f_v67:
|
| 1543 |
+
_rss_line_v67 = [l for l in _f_v67 if l.startswith("VmRSS:")]
|
| 1544 |
+
_rss_kb_v67 = int(_rss_line_v67[0].split()[1]) if _rss_line_v67 else 0
|
| 1545 |
+
_rss_mb_v67 = _rss_kb_v67 / 1024.0
|
| 1546 |
+
# cgroup limit
|
| 1547 |
+
_cg_limit_mb_v67 = 4096.0 # default fallback
|
| 1548 |
+
try:
|
| 1549 |
+
with open("/sys/fs/cgroup/memory.max") as _f_cg_v67:
|
| 1550 |
+
_cg_val_v67 = _f_cg_v67.read().strip()
|
| 1551 |
+
if _cg_val_v67 and _cg_val_v67 != "max":
|
| 1552 |
+
_cg_limit_mb_v67 = int(_cg_val_v67) / 1024 / 1024
|
| 1553 |
+
except Exception:
|
| 1554 |
+
pass # fallback default
|
| 1555 |
+
_rss_pct_v67 = _rss_mb_v67 / _cg_limit_mb_v67 if _cg_limit_mb_v67 > 0 else 0
|
| 1556 |
+
logger.info(
|
| 1557 |
+
f"[V6.7-memory-guard] RSS={_rss_mb_v67:.0f}MB "
|
| 1558 |
+
f"({100*_rss_pct_v67:.1f}% of cgroup "
|
| 1559 |
+
f"{_cg_limit_mb_v67:.0f}MB)"
|
| 1560 |
+
)
|
| 1561 |
+
if _rss_pct_v67 > 0.85:
|
| 1562 |
+
logger.warning(
|
| 1563 |
+
f"[V6.7-memory-guard] SKIP refit: RSS "
|
| 1564 |
+
f"{100*_rss_pct_v67:.1f}% > 85% of cgroup "
|
| 1565 |
+
f"(would trigger OOM via fork). "
|
| 1566 |
+
f"gc.collect() e prosseguir sem refit."
|
| 1567 |
+
)
|
| 1568 |
+
gc.collect()
|
| 1569 |
+
if hasattr(torch, "cpu") and hasattr(torch.cpu, "empty_cache"):
|
| 1570 |
+
try: torch.cpu.empty_cache()
|
| 1571 |
+
except Exception: pass
|
| 1572 |
+
last_tokenizer_refit_sample_count = total_so_far_now
|
| 1573 |
+
tokenizer_growth_log.append({
|
| 1574 |
+
"step": step,
|
| 1575 |
+
"chunk_idx": chunk_idx_global,
|
| 1576 |
+
"total_samples": total_so_far_now,
|
| 1577 |
+
"skipped_reason": "rss_exceeded_85pct",
|
| 1578 |
+
"rss_mb": _rss_mb_v67,
|
| 1579 |
+
"cgroup_limit_mb": _cg_limit_mb_v67,
|
| 1580 |
+
})
|
| 1581 |
+
# SKIP refit — continue para próximo chunk
|
| 1582 |
+
raise _SkipRefitV67()
|
| 1583 |
+
except _SkipRefitV67:
|
| 1584 |
+
pass # já tratado acima
|
| 1585 |
+
except Exception as _guard_err_v67:
|
| 1586 |
+
logger.warning(
|
| 1587 |
+
f"[V6.7-memory-guard] guard failed (non-fatal): {_guard_err_v67}"
|
| 1588 |
+
)
|
| 1589 |
+
|
| 1590 |
# Refit tokenizer com corpus acumulado
|
| 1591 |
vocab_before = int(getattr(kls.tokenizer, "vocab_size", 0))
|
| 1592 |
try:
|
| 1593 |
+
kls.tokenizer.fit(list(tokenizer_corpus_buffer), min_frequency=2)
|
| 1594 |
vocab_after = int(getattr(kls.tokenizer, "vocab_size", 0))
|
| 1595 |
growth = vocab_after - vocab_before
|
| 1596 |
tokenizer_growth_log.append({
|