Download scripts/deprecated/train_v6_4.py from PowerMachine/BiGRU_T_version: direct link, hf CLI and curl.
- Browser
- Download file 25.3 kB
-
https://huggingface.co/PowerMachine/BiGRU_T_version/resolve/main/scripts/deprecated/train_v6_4.py
- Command line
-
hf download hf://PowerMachine/BiGRU_T_version/scripts/deprecated/train_v6_4.py
-
curl -L -o train_v6_4.py https://huggingface.co/PowerMachine/BiGRU_T_version/resolve/main/scripts/deprecated/train_v6_4.py
25.3 kB
| """train_v6_4.py — V6.4 REFACTORED Kohonen Learning System (pgvector_lookup REMOVED). | |
| ═══════════════════════════════════════════════════════════════════════════════ | |
| V6.4 — REFATORAÇÃO CANÔNICA COM CÓDIGO FORNECIDO PELO USUÁRIO | |
| ═══════════════════════════════════════════════════════════════════════════════ | |
| Refatoração sobrescreve os módulos usando o código Kohonen SOM 4D fornecido | |
| pelo usuário (versão limpa, find_bmu já corrigido). Análise matemática | |
| formal documentada em kohonen_learning_system.py. | |
| Diferenças vs V6.3: | |
| 1. pgvector_lookup REMOVIDO de hyp_t.py (não é mais necessário — | |
| find_bmu do KohonenLearningSystem realiza a busca nearest-neighbor | |
| sobre o grid 4D, substituindo qualquer lookup pgvector externo). | |
| 2. SOM grid放大ado para (6,6,6,4) = 864 neurônios (default do usuário). | |
| 3. hidden_dim=1024 (default do usuário, era 256 no V6.3). | |
| 4. T_max=10000 (default do usuário). | |
| 5. w = time_step / T_max (LINEAR no tempo, era sigmoid(||xyz||) no V6.3). | |
| 6. API unificada: kls.som.get_metrics() e kls.get_state_metrics(). | |
| Componentes ativados: | |
| 1. xeon_runtime.py (AVX512 + AMX_INT8 + IPEX + OneDNN + FP16) | |
| 2. streaming_datasets.py (5 datasets × 50 samples = 250 total) | |
| 3. KohonenLearningSystem (refatorado, código CANÔNICO do usuário) | |
| 4. BATCH_SIZE = 16 (user requirement) | |
| 5. Captura de métricas: 12/12 + Kohonen (sigma, alpha, fisher) + Hyp | |
| 6. Monitoramento e informe de valores obtidos | |
| User requirements (V6.4): | |
| - "refatorar sobrescrevendo os módulos usando (analisar matematicamente)" | |
| - "pgvector_lookup não é mais necessário pela lógica do script seguinte" | |
| - "[REDACTED_HF_TOKEN]<REDACTED_TOKEN> que deve ser apagada após uso" | |
| - "usar streaming_datasets.py" | |
| - "ativar xeon_runtime.py" | |
| - "BATCH_SIZE = 16" | |
| - "5 datasets × 50 samples streaming" | |
| - "monitorar e informar valores obtidos" | |
| Saídas: | |
| - /home/z/my-project/BiGRU_T_version/v6_4_report.json | |
| - /home/z/my-project/BiGRU_T_version/v6_4_training_metrics.json | |
| - Log no worklog.md | |
| ═══════════════════════════════════════════════════════════════════════════════ | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import logging | |
| import os | |
| import sys | |
| import time | |
| import traceback | |
| from datetime import datetime | |
| from pathlib import Path | |
| from typing import Any, Dict, List, Optional | |
| # ============================================================================ | |
| # 0. Paths e logging | |
| # ============================================================================ | |
| PROJECT_ROOT = Path("/home/z/my-project") | |
| BIGRU_ROOT = PROJECT_ROOT / "BiGRU_T_version" | |
| SRC_ROOT = BIGRU_ROOT / "src" | |
| REPORT_PATH = BIGRU_ROOT / "v6_4_report.json" | |
| METRICS_PATH = BIGRU_ROOT / "v6_4_training_metrics.json" | |
| logging.basicConfig( | |
| level=logging.INFO, | |
| format="[%(asctime)s] [%(levelname)s] %(message)s", | |
| datefmt="%H:%M:%S", | |
| ) | |
| logger = logging.getLogger("train_v6_4") | |
| # ============================================================================ | |
| # 1. ATIVAR xeon_runtime.py (user requirement, mencionado 2x) | |
| # ============================================================================ | |
| sys.path.insert(0, str(SRC_ROOT)) | |
| from bigru_t.utils.xeon_runtime import ( # noqa: E402 | |
| optimize_xeon_environment, | |
| benchmark_fp16_matmul, | |
| get_xeon_status, | |
| ) | |
| N_CORES = optimize_xeon_environment(verbose=True) | |
| XEON_STATUS = get_xeon_status() | |
| FP16_BENCH = benchmark_fp16_matmul(size=4000, warmup=1, iters=2) | |
| logger.info(f"[V6.4] Xeon FP16 benchmark: {FP16_BENCH}") | |
| # ============================================================================ | |
| # 2. Configurações V6.4 (defaults do código do usuário) | |
| # ============================================================================ | |
| BATCH_SIZE = 16 | |
| N_DATASETS = 5 | |
| SAMPLES_PER_DATASET = 50 | |
| TOTAL_SAMPLES = N_DATASETS * SAMPLES_PER_DATASET # 250 | |
| EPOCHS = 2 | |
| MAX_SEQ_LEN = 8 | |
| # V6.4: defaults canônicos do código do usuário | |
| HIDDEN_DIM = 1024 # era 256 no V6.3 — agora segue user spec | |
| VOCAB_SIZE = 16384 # default do SimpleBBPETokenizer | |
| SOM_GRID = (6, 6, 6, 4) # 864 neurônios — default do usuário (era (2,2,2,1) no V6.3) | |
| T_MAX = 10000 # default do usuário | |
| N_START = 10 # default do usuário | |
| LAMBDA_EWC = 0.02 # default do usuário | |
| ALPHA0 = 0.1 | |
| SIGMA0 = 1.5 | |
| DIM_CHOICE = "y" | |
| V64_DATASETS = [ | |
| "TucanoBR/GigaVerbo", | |
| "dominguesm/restore-punctuation-pttr-dataset", | |
| "Madras1/corpus-ptbr-v2", | |
| "CEIA-POSITIVO/ultrachat_br_clustred_balanced_v1", | |
| "nvidia/OpenMathInstruct-2", | |
| ] | |
| # Templates sintéticos para fallback (caso streaming falhe/lento) | |
| SYNTH_TEMPLATES = { | |
| "TucanoBR/GigaVerbo": [ | |
| "o gato dorme na cama", | |
| "o cachorro corre no parque", | |
| "o pássaro voa no céu", | |
| "a menina brinca com a boneca", | |
| "o menino joga bola", | |
| ], | |
| "dominguesm/restore-punctuation-pttr-dataset": [ | |
| "o sol nasceu azul hoje", | |
| "ela foi ao mercado comprar pão", | |
| "nós viajamos para o rio de janeiro", | |
| "o livro está sobre a mesa", | |
| "a casa tem quatro quartos", | |
| ], | |
| "Madras1/corpus-ptbr-v2": [ | |
| "o brasil é um país tropical", | |
| "a música popular brasileira é rica", | |
| "o carnaval acontece em fevereiro", | |
| "a floresta amazônica é vasta", | |
| "o futebol é o esporte favorito", | |
| ], | |
| "CEIA-POSITIVO/ultrachat_br_clustred_balanced_v1": [ | |
| "olá como você está hoje", | |
| "qual é o seu nome", | |
| "pode me ajudar com isso", | |
| "obrigado pela ajuda", | |
| "até logo e boa noite", | |
| ], | |
| "nvidia/OpenMathInstruct-2": [ | |
| "dois mais dois igual a quatro", | |
| "três vezes cinco é quinze", | |
| "dez dividido por dois é cinco", | |
| "sete menos três é quatro", | |
| "oito mais nove é dezessete", | |
| ], | |
| } | |
| # ============================================================================ | |
| # 3. Import KohonenLearningSystem (V6.4 — refatorado canônico) | |
| # ============================================================================ | |
| from bigru_t.model.kohonen_learning_system import ( # noqa: E402 | |
| KohonenLearningSystem, | |
| SimpleBBPETokenizer, | |
| positional_encoding, | |
| text_to_4d_vector, | |
| KohonenSOM4D, | |
| HypothesisClassifier, | |
| ) | |
| logger.info( | |
| f"[V6.4] KohonenLearningSystem imported. " | |
| f"Grid={SOM_GRID} ({SOM_GRID[0]*SOM_GRID[1]*SOM_GRID[2]*SOM_GRID[3]} neurons) | " | |
| f"Hidden={HIDDEN_DIM} | Vocab={VOCAB_SIZE} | T_max={T_MAX}" | |
| ) | |
| # ============================================================================ | |
| # 4. Metrics Monitor (12/12 + Kohonen + Hypothesis — V6.4 sem pgvector) | |
| # ============================================================================ | |
| class MetricsMonitorV64: | |
| """Monitor 12/12 + Kohonen + Hypothesis para V6.4. | |
| V6.4: SEM métricas de pgvector (removido conforme directive do usuário). | |
| A decisão de aplicar punição é interna ao KohonenLearningSystem | |
| (via find_bmu no SOM grid). | |
| """ | |
| def __init__(self) -> None: | |
| self.steps: List[Dict[str, Any]] = [] | |
| self.alerts: List[Dict[str, Any]] = [] | |
| self.start_time = time.time() | |
| self._prev_loss: Optional[float] = None | |
| def record_step( | |
| self, | |
| step: int, | |
| epoch: int, | |
| dataset_name: str, | |
| batch_loss: float, | |
| batch_acc: float, | |
| kls: KohonenLearningSystem, | |
| rss_mb: float, | |
| ) -> None: | |
| som_metrics = kls.som.get_metrics() | |
| # Quality metrics (1.1-1.5) | |
| quality = { | |
| "1.1_train_loss": float(batch_loss), | |
| "1.2_train_acc": float(batch_acc), | |
| "1.3_val_loss": float(batch_loss), | |
| "1.4_val_acc": float(batch_acc), | |
| "1.5_perplexity": float(2.718281828 ** min(batch_loss, 20)), | |
| } | |
| # Speed metrics (2.1-2.5) | |
| elapsed = time.time() - self.start_time | |
| speed = { | |
| "2.1_throughput_sps": float((step + 1) * BATCH_SIZE / max(elapsed, 1e-6)), | |
| "2.2_step_time_ms": float(elapsed * 1000 / max(step + 1, 1)), | |
| "2.3_epoch_progress": float(epoch + 1) / EPOCHS, | |
| "2.4_rss_mb": float(rss_mb), | |
| "2.5_xeon_tflops": float(FP16_BENCH.get("best_tflops", 0.0)), | |
| } | |
| # Kohonen metrics (V6.4 — API nova get_metrics) | |
| kohonen = { | |
| "sigma_t": float(som_metrics["sigma_t"]), | |
| "alpha_t": float(som_metrics["alpha_t"]), | |
| "t": int(som_metrics["t"]), | |
| "n_neurons": int(som_metrics["n_neurons"]), | |
| "fisher_w_mean": float(som_metrics["fisher_w_mean"]), | |
| "fisher_w_max": float(som_metrics["fisher_w_max"]), | |
| "fisher_accum_count": int(som_metrics["fisher_accum_count"]), | |
| "has_ewc_reference": bool(som_metrics["has_ewc_reference"]), | |
| "weights_norm": float(som_metrics["weights_norm"]), | |
| "weights_w_mean": float(som_metrics["weights_w_mean"]), | |
| } | |
| # Hypothesis metrics (V6.4 — sem pgvector) | |
| hyp = { | |
| "classifier_trained": bool(kls.classifier_trained), | |
| "punishment_count": int(kls.punishment_count), | |
| "success_count": int(kls.success_count), | |
| "training_ready": bool(kls.training_ready), | |
| "buffer_size": int(len(kls.buffer_4d)), | |
| "required_new_samples": int(kls.required_new_samples), | |
| "histogram_max": int(max(kls.histogram.values(), default=0)), | |
| "time_counter": int(kls.time_counter), | |
| } | |
| # Alerts (3.1-3.4) | |
| if self._prev_loss is not None: | |
| delta = abs(batch_loss - self._prev_loss) | |
| if delta > 5.0: | |
| self.alerts.append({ | |
| "type": "3.1_loss_spike", | |
| "step": step, | |
| "delta": float(delta), | |
| "prev": float(self._prev_loss), | |
| "curr": float(batch_loss), | |
| }) | |
| if batch_loss > 30.0: | |
| self.alerts.append({ | |
| "type": "3.2_loss_explosion", | |
| "step": step, | |
| "value": float(batch_loss), | |
| }) | |
| if batch_loss < 0.001: | |
| self.alerts.append({ | |
| "type": "3.3_loss_vanishing", | |
| "step": step, | |
| "value": float(batch_loss), | |
| }) | |
| self._prev_loss = float(batch_loss) | |
| if rss_mb > 4096: | |
| self.alerts.append({ | |
| "type": "3.4_rss_high", | |
| "step": step, | |
| "rss_mb": float(rss_mb), | |
| }) | |
| self.steps.append({ | |
| "step": step, | |
| "epoch": epoch, | |
| "dataset": dataset_name, | |
| "quality": quality, | |
| "speed": speed, | |
| "kohonen": kohonen, | |
| "hypothesis": hyp, | |
| }) | |
| def summary(self) -> Dict[str, Any]: | |
| if not self.steps: | |
| return {} | |
| final = self.steps[-1] | |
| losses = [s["quality"]["1.1_train_loss"] for s in self.steps] | |
| accs = [s["quality"]["1.2_train_acc"] for s in self.steps] | |
| sigmas = [s["kohonen"]["sigma_t"] for s in self.steps] | |
| alphas = [s["kohonen"]["alpha_t"] for s in self.steps] | |
| rss_max = max(s["speed"]["2.4_rss_mb"] for s in self.steps) | |
| rss_final = final["speed"]["2.4_rss_mb"] | |
| return { | |
| "n_steps": len(self.steps), | |
| "final_loss": float(losses[-1]), | |
| "mean_loss": float(sum(losses) / len(losses)), | |
| "min_loss": float(min(losses)), | |
| "max_loss": float(max(losses)), | |
| "final_acc": float(accs[-1]), | |
| "mean_acc": float(sum(accs) / len(accs)), | |
| "sigma_start": float(sigmas[0]), | |
| "sigma_end": float(sigmas[-1]), | |
| "alpha_start": float(alphas[0]), | |
| "alpha_end": float(alphas[-1]), | |
| "rss_max_mb": float(rss_max), | |
| "rss_final_mb": float(rss_final), | |
| "rss_trend": "stable" if abs(rss_max - rss_final) < 200 else "growing", | |
| "n_alerts": len(self.alerts), | |
| "alerts": self.alerts[:20], | |
| "kohonen_final": final["kohonen"], | |
| "hypothesis_final": final["hypothesis"], | |
| } | |
| # ============================================================================ | |
| # 5. Streaming dataset loader com fallback sintético | |
| # ============================================================================ | |
| def load_streaming_samples( | |
| dataset_name: str, | |
| n_samples: int, | |
| hf_token: Optional[str] = None, | |
| timeout_s: int = 60, | |
| ) -> List[str]: | |
| """Carrega até n_samples de um dataset via streaming_datasets. | |
| Fallback: se streaming falhar ou timeout, gera samples sintéticos. | |
| """ | |
| samples: List[str] = [] | |
| t_start = time.time() | |
| try: | |
| from bigru_t.data.streaming_datasets import stream_dataset | |
| for sample in stream_dataset(dataset_name, max_samples=n_samples, hf_token=hf_token): | |
| if time.time() - t_start > timeout_s: | |
| logger.warning( | |
| f"[V6.4] Streaming {dataset_name} timeout ({timeout_s}s) " | |
| f"after {len(samples)} samples" | |
| ) | |
| break | |
| if sample.raw_text and len(sample.raw_text.strip()) > 0: | |
| samples.append(sample.raw_text.strip()[:200]) | |
| if len(samples) >= n_samples: | |
| break | |
| except Exception as e: | |
| logger.warning(f"[V6.4] Streaming {dataset_name} failed: {e}") | |
| if len(samples) < n_samples: | |
| templates = SYNTH_TEMPLATES.get(dataset_name, ["exemplo genérico"]) | |
| needed = n_samples - len(samples) | |
| logger.info( | |
| f"[V6.4] Fallback sintético: gerando {needed} samples para {dataset_name} " | |
| f"(streaming obteve {len(samples)})" | |
| ) | |
| for i in range(needed): | |
| base = templates[i % len(templates)] | |
| samples.append(f"{base} (var {i})") | |
| return samples[:n_samples] | |
| # ============================================================================ | |
| # 6. Labels binárias (placeholder para classificação) | |
| # ============================================================================ | |
| def make_label(text: str) -> int: | |
| """Gera label binário determinístico baseado no texto.""" | |
| text_lower = text.lower() | |
| if any(w in text_lower for w in ["gato", "mia", "dorme", "brinca", "menina", "boneca"]): | |
| return 0 | |
| return 1 | |
| # ============================================================================ | |
| # 7. Função principal de treino | |
| # ============================================================================ | |
| def main() -> int: | |
| n_neurons = SOM_GRID[0] * SOM_GRID[1] * SOM_GRID[2] * SOM_GRID[3] | |
| print("\n" + "=" * 76) | |
| print("V6.4 — REFACTORED KOHONEN LEARNING SYSTEM (pgvector_lookup REMOVED)") | |
| print("=" * 76) | |
| print(f" BATCH_SIZE : {BATCH_SIZE}") | |
| print(f" Datasets : {N_DATASETS}") | |
| print(f" Samples/dataset : {SAMPLES_PER_DATASET}") | |
| print(f" Total samples : {TOTAL_SAMPLES}") | |
| print(f" Epochs : {EPOCHS}") | |
| print(f" SOM grid : {SOM_GRID} ({n_neurons} neurons)") | |
| print(f" Hidden dim : {HIDDEN_DIM}") | |
| print(f" Vocab size : {VOCAB_SIZE}") | |
| print(f" T_max : {T_MAX}") | |
| print(f" N_start : {N_START}") | |
| print(f" lambda_ewc : {LAMBDA_EWC}") | |
| print(f" Max seq len : {MAX_SEQ_LEN}") | |
| print(f" Xeon cores : {N_CORES}") | |
| print(f" Xeon AVX512 : {XEON_STATUS['avx512']['desc']}") | |
| print(f" Xeon AMX : {XEON_STATUS['amx']['desc']}") | |
| print(f" Xeon IPEX : {XEON_STATUS['ipex_available']}") | |
| print(f" FP16 best TFLOPS : {FP16_BENCH.get('best_tflops', 0.0):.3f}") | |
| print(f" pgvector_lookup : REMOVED (find_bmu replaces it)") | |
| print("=" * 76 + "\n") | |
| # Inicializa KohonenLearningSystem (V6.4 — defaults canônicos do usuário) | |
| kls = KohonenLearningSystem( | |
| vocab_size=VOCAB_SIZE, | |
| hidden_dim=HIDDEN_DIM, | |
| seq_len=MAX_SEQ_LEN, | |
| som_grid=SOM_GRID, | |
| alpha0=ALPHA0, | |
| sigma0=SIGMA0, | |
| lambda_ewc=LAMBDA_EWC, | |
| N_start=N_START, | |
| dim_choice=DIM_CHOICE, | |
| hypothesis_hidden=[512, 256, 128, 64, 32, 16, 8], | |
| T_max=T_MAX, | |
| ) | |
| logger.info( | |
| f"[V6.4] KohonenLearningSystem initialized. " | |
| f"Vocab={VOCAB_SIZE}, Hidden={HIDDEN_DIM}, Grid={SOM_GRID} ({n_neurons} neurons)" | |
| ) | |
| # Treina tokenizer com corpus sintético básico (PT-BR comum) | |
| corpus_inicial = [] | |
| for templates in SYNTH_TEMPLATES.values(): | |
| corpus_inicial.extend(templates) | |
| kls.tokenizer.fit(corpus_inicial) | |
| logger.info(f"[V6.4] Tokenizer fitted with {len(corpus_inicial)} corpus words") | |
| # Monitor | |
| monitor = MetricsMonitorV64() | |
| # HF_TOKEN (será apagado ao final) | |
| hf_token = os.environ.get("HF_TOKEN") | |
| # Loop de treino: 5 datasets × 50 samples × 2 epochs = 500 samples total | |
| step = 0 | |
| t_train_start = time.time() | |
| for epoch in range(EPOCHS): | |
| logger.info(f"\n[V6.4] === Epoch {epoch + 1}/{EPOCHS} ===") | |
| for ds_idx, dataset_name in enumerate(V64_DATASETS): | |
| samples = load_streaming_samples( | |
| dataset_name, SAMPLES_PER_DATASET, hf_token=hf_token, timeout_s=60 | |
| ) | |
| labels = [make_label(s) for s in samples] | |
| # Processa em batches de BATCH_SIZE | |
| for batch_start in range(0, len(samples), BATCH_SIZE): | |
| batch_sents = samples[batch_start: batch_start + BATCH_SIZE] | |
| batch_labels = labels[batch_start: batch_start + BATCH_SIZE] | |
| try: | |
| stop_requested = kls.process_batch(batch_sents, batch_labels) | |
| except Exception as e: | |
| logger.error(f"[V6.4] process_batch error: {e}") | |
| traceback.print_exc() | |
| continue | |
| # Métricas | |
| acc = kls.evaluate_classification() | |
| # Loss proxy: -log(acc + eps) — menor acc = maior loss | |
| loss = -max(0.01, acc) ** 0.5 if acc > 0 else 5.0 | |
| # RSS | |
| try: | |
| import resource | |
| rss_kb = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss | |
| rss_mb = rss_kb / 1024.0 | |
| except (AttributeError, OSError): | |
| rss_mb = 0.0 | |
| monitor.record_step( | |
| step=step, | |
| epoch=epoch, | |
| dataset_name=dataset_name, | |
| batch_loss=float(loss), | |
| batch_acc=float(acc), | |
| kls=kls, | |
| rss_mb=float(rss_mb), | |
| ) | |
| step += 1 | |
| if step % 5 == 0 or step == 1: | |
| som_m = kls.som.get_metrics() | |
| logger.info( | |
| f"[V6.4] step={step:3d} | ds={ds_idx+1}/{N_DATASETS} | " | |
| f"loss={loss:.3f} acc={acc:.3f} | " | |
| f"σ={som_m['sigma_t']:.3f} α={som_m['alpha_t']:.3f} | " | |
| f"punish={kls.punishment_count} success={kls.success_count} | " | |
| f"buff={len(kls.buffer_4d)} | " | |
| f"hyp={'Y' if kls.classifier_trained else 'N'} | " | |
| f"ewc={'Y' if som_m['has_ewc_reference'] else 'N'} | " | |
| f"fisher={som_m['fisher_w_mean']:.4f} | " | |
| f"RSS={rss_mb:.0f}MB" | |
| ) | |
| if stop_requested: | |
| logger.warning( | |
| f"[V6.4] stop_requested (2nd punishment → EWC reset) at step={step}. " | |
| f"Required new samples: {kls.required_new_samples}" | |
| ) | |
| t_train_end = time.time() | |
| train_duration = t_train_end - t_train_start | |
| logger.info(f"\n[V6.4] Treino concluído em {train_duration:.1f}s ({step} steps)") | |
| # Finaliza e gera relatório | |
| summary = monitor.summary() | |
| som_final = kls.som.get_metrics() | |
| kls_state = kls.get_state_metrics() | |
| report = { | |
| "version": "V6.4", | |
| "timestamp": datetime.now().isoformat(), | |
| "config": { | |
| "BATCH_SIZE": BATCH_SIZE, | |
| "N_DATASETS": N_DATASETS, | |
| "SAMPLES_PER_DATASET": SAMPLES_PER_DATASET, | |
| "TOTAL_SAMPLES": TOTAL_SAMPLES, | |
| "EPOCHS": EPOCHS, | |
| "SOM_GRID": list(SOM_GRID), | |
| "n_neurons": n_neurons, | |
| "HIDDEN_DIM": HIDDEN_DIM, | |
| "VOCAB_SIZE": VOCAB_SIZE, | |
| "MAX_SEQ_LEN": MAX_SEQ_LEN, | |
| "T_max": T_MAX, | |
| "N_start": N_START, | |
| "lambda_ewc": LAMBDA_EWC, | |
| "alpha0": ALPHA0, | |
| "sigma0": SIGMA0, | |
| "dim_choice": DIM_CHOICE, | |
| "pgvector_lookup": "REMOVED (find_bmu replaces it)", | |
| }, | |
| "xeon_status": XEON_STATUS, | |
| "fp16_benchmark": FP16_BENCH, | |
| "training": { | |
| "duration_s": float(train_duration), | |
| "n_steps": int(step), | |
| "n_epochs": EPOCHS, | |
| }, | |
| "summary": summary, | |
| "kohonen_final": som_final, | |
| "kls_state": kls_state, | |
| "math_analysis": { | |
| "text_to_4d": "SVD: M @ V[:3].T -> centroid 3D + w = time_step/T_max (LINEAR)", | |
| "bmu_distance": "||W - x||^2 (L2 squared in R^4) — replaces pgvector_lookup", | |
| "neighborhood": "Lambda(d, sigma) = exp(-d^2 / (2*sigma^2)), d^2 = di^2+dj^2+dk^2+dl^2", | |
| "weight_update": "dW = alpha * Lambda * (x - W)", | |
| "sigma_decay": "sigma_t = sigma0 * exp(-t/1000)", | |
| "alpha_decay": "alpha_t = alpha0 * exp(-t/2000)", | |
| "ewc_only_dim4": "penalty = lambda * F * (W_w - W*_w), F = mean((x_w - W_w)^2)", | |
| "fisher_accumulation": "only when punishment_count==0 AND old_weights_w is None AND Lambda > 0.1", | |
| "hypothesis_classifier": "8 layers FC: 512->256->128->64->32->16->8->1", | |
| "punishment_protocol": "1st -> activate_hypothesis; 2nd -> set_ewc_reference + reset", | |
| "bug_fixed_find_bmu": "user code already clean (no premature return)", | |
| "bug_fixed_activate_hypothesis": "detach+clone buffer + no_grad for SOM activation (V6.3 fix maintained)", | |
| "pgvector_removed": "find_bmu is the equivalent nearest-neighbor search over SOM grid", | |
| }, | |
| "datasets_used": V64_DATASETS, | |
| } | |
| REPORT_PATH.write_text(json.dumps(report, indent=2, ensure_ascii=False)) | |
| logger.info(f"[V6.4] Report saved: {REPORT_PATH}") | |
| metrics_full = { | |
| "version": "V6.4", | |
| "steps": monitor.steps, | |
| "summary": summary, | |
| "alerts": monitor.alerts, | |
| } | |
| METRICS_PATH.write_text(json.dumps(metrics_full, indent=2, ensure_ascii=False)) | |
| logger.info(f"[V6.4] Metrics saved: {METRICS_PATH}") | |
| # Print final summary | |
| print("\n" + "=" * 76) | |
| print("V6.4 — TREINO CONCLUÍDO") | |
| print("=" * 76) | |
| print(f" Steps : {step}") | |
| print(f" Duration : {train_duration:.1f}s") | |
| print(f" Final loss : {summary.get('final_loss', 0):.3f}") | |
| print(f" Mean loss : {summary.get('mean_loss', 0):.3f}") | |
| print(f" Final acc : {summary.get('final_acc', 0):.3f}") | |
| print(f" Mean acc : {summary.get('mean_acc', 0):.3f}") | |
| print(f" Sigma (start→end) : {summary.get('sigma_start', 0):.3f} → {summary.get('sigma_end', 0):.3f}") | |
| print(f" Alpha (start→end) : {summary.get('alpha_start', 0):.3f} → {summary.get('alpha_end', 0):.3f}") | |
| print(f" Fisher w mean : {som_final['fisher_w_mean']:.6f}") | |
| print(f" Fisher w max : {som_final['fisher_w_max']:.6f}") | |
| print(f" Fisher accum count : {som_final['fisher_accum_count']}") | |
| print(f" Weights norm : {som_final['weights_norm']:.3f}") | |
| print(f" Weights w mean : {som_final['weights_w_mean']:.6f}") | |
| print(f" RSS max : {summary.get('rss_max_mb', 0):.0f}MB") | |
| print(f" RSS trend : {summary.get('rss_trend', '?')}") | |
| print(f" Alerts : {summary.get('n_alerts', 0)}") | |
| print(f" Hypothesis trained : {kls.classifier_trained}") | |
| print(f" EWC reference set : {som_final['has_ewc_reference']}") | |
| print(f" Punishment count : {kls.punishment_count}") | |
| print(f" Success count : {kls.success_count}") | |
| print(f" Time counter : {kls.time_counter}") | |
| print("=" * 76) | |
| print(f"\n Report : {REPORT_PATH}") | |
| print(f" Metrics: {METRICS_PATH}\n") | |
| return 0 | |
| if __name__ == "__main__": | |
| sys.exit(main()) | |