PowerMachine commited on
Commit
02df069
·
verified ·
1 Parent(s): 81ff26f

V6.5-V2-dynamic: upload batch (scripts + state + model) [79 files]

Browse files
scripts/train_v6_5_v2.py CHANGED
@@ -580,7 +580,12 @@ def save_model_states_for_evaluation(
580
  "step": int(step),
581
  "total_samples": int(total_samples),
582
  "timestamp": datetime.now().isoformat(),
583
- "version": "V6.5-V2-metrics-FIX-2",
 
 
 
 
 
584
  "som_grid": list(kls.som_grid),
585
  "n_neurons": int(kls.som_neuron_count),
586
  "hidden_dim": int(kls.hidden_dim),
@@ -1199,6 +1204,24 @@ def run_fase_conhecimento(
1199
  traceback.print_exc()
1200
  continue
1201
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1202
  # Métricas
1203
  acc = kls.evaluate_classification()
1204
  accuracies_log.append({
@@ -2751,13 +2774,18 @@ def main() -> int:
2751
  f"atingida: {conhecimento_result['meta_atingida']})"
2752
  )
2753
 
2754
- # Salva estados do modelo após CONHECIMENTO
 
 
 
 
 
2755
  save_model_states_for_evaluation(
2756
  kls=kls,
2757
  reason="end_of_fase_conhecimento",
2758
  step=conhecimento_result["n_steps"],
2759
  total_samples=conhecimento_result["total_samples"],
2760
- output_path=BIGRU_ROOT / "v6_5_v2_model_states_after_conhecimento.pt",
2761
  )
2762
 
2763
  # Aggressive cleanup entre fases
@@ -2797,7 +2825,10 @@ def main() -> int:
2797
  # (b) permitir que um processo separado (apenas FASE2) seja iniciado
2798
  # do zero apontando para o estado salvo da FASE1.
2799
  logger.info("\n[V6.5-V2-metrics] Carregando estado do modelo da FASE1 (FASE2 uses FASE1 completed state)...")
2800
- fase1_state_path = BIGRU_ROOT / "v6_5_v2_model_states_after_conhecimento.pt"
 
 
 
2801
  fase1_load_result = load_fase1_state_into_kls(kls, fase1_state_path)
2802
 
2803
  logger.info("\n[V6.5-V2] Iniciando FASE 2 — TREINAMENTO COM PUNIÇÃO...")
@@ -2916,9 +2947,9 @@ def main() -> int:
2916
 
2917
  # Copia arquivos relevantes para download/
2918
  try:
 
2919
  for src in [REPORT_PATH, USER_QUESTIONS_PATH, MODEL_STATES_PATH,
2920
- V2_PHASES_EVAL_PATH, PREDICT_FIX_EVAL_PATH, ATTENTION_EVAL_PATH,
2921
- BIGRU_ROOT / "v6_5_v2_model_states_after_conhecimento.pt"]:
2922
  if src.exists():
2923
  dst = DOWNLOAD_DIR / src.name
2924
  shutil.copy2(src, dst)
@@ -2975,9 +3006,9 @@ def main() -> int:
2975
  # Inclui: v6_5_v2_model_states.pt, v6_5_v2_model_states_after_conhecimento.pt
2976
  # Verifica tamanho ≤ 500MB (HF LFS free tier sem autenticar LFS).
2977
  _MAX_HF_LFS_SIZE_BYTES = 500 * 1024 * 1024 # 500MB
 
2978
  for pt_candidate in [
2979
  MODEL_STATES_PATH,
2980
- BIGRU_ROOT / "v6_5_v2_model_states_after_conhecimento.pt",
2981
  ]:
2982
  if pt_candidate.exists():
2983
  pt_size = pt_candidate.stat().st_size
@@ -3117,6 +3148,8 @@ Updated: {datetime.now().isoformat()}
3117
  f"({hf_upload_result.get('n_files', 0)} files)")
3118
  print(f" HF_TOKEN cleaned : True (env + scripts)")
3119
  print(f" .pt files removed : {len(pt_files_removed)} (ALL local .pt cleaned)")
 
 
3120
  print(f" Storage cleanup : {final_storage_cleanup['n_files_removed']} files, "
3121
  f"{final_storage_cleanup['mb_freed']:.1f}MB freed")
3122
  print(f" Final n_hypotheses : {kls.n_hypotheses} (active={kls.hypothesis_ensemble.active_count})")
 
580
  "step": int(step),
581
  "total_samples": int(total_samples),
582
  "timestamp": datetime.now().isoformat(),
583
+ "version": "V6.5-V2-unified-state",
584
+ # V6.5-V2-unified — fase que gerou este estado. Permite que
585
+ # um processo separado (ex: retomar treino) saiba se o estado
586
+ # é pós-FASE1 (pronto para FASE2) ou pós-FASE2 (treino completo).
587
+ "phase": str(reason), # "end_of_fase_conhecimento" | "end_of_training_v2"
588
+ "kmeans_pp_applied": bool(getattr(kls.som, "_kmeans_pp_initialized", False)),
589
  "som_grid": list(kls.som_grid),
590
  "n_neurons": int(kls.som_neuron_count),
591
  "hidden_dim": int(kls.hidden_dim),
 
1204
  traceback.print_exc()
1205
  continue
1206
 
1207
+ # V6.5-V2-kmeans-pp - Apos 1o chunk, inicializa pesos
1208
+ # do SOM via k-means++ (deferred init).
1209
+ # User requirement: "inicializacao com k-means++ sobre
1210
+ # embeddings em vez de grid coords+ruido para ambas as
1211
+ # FASE1 e FASE2".
1212
+ try:
1213
+ _km = kls.init_weights_kmeans_pp_if_ready()
1214
+ if isinstance(_km, dict) and _km.get("initialized"):
1215
+ logger.info(
1216
+ f"[V6.5-V2-kmeans-pp] SOM weights initialized via k-means++ "
1217
+ f"(samples={_km.get('n_samples_used')}, "
1218
+ f"inertia={_km.get('inertia'):.4f})"
1219
+ )
1220
+ except Exception as _km_err:
1221
+ logger.warning(
1222
+ f"[V6.5-V2-kmeans-pp] init failed (non-fatal): {_km_err}"
1223
+ )
1224
+
1225
  # Métricas
1226
  acc = kls.evaluate_classification()
1227
  accuracies_log.append({
 
2774
  f"atingida: {conhecimento_result['meta_atingida']})"
2775
  )
2776
 
2777
+ # V6.5-V2-unified — Salva estado após CONHECIMENTO no MESMO arquivo
2778
+ # unificado (v6_5_v2_model_states.pt). User requirement: "unir os
2779
+ # estados do modelo de ambas as FASE1 e FASE2 para um estado contínuo
2780
+ # em um único arquivo de treinamento".
2781
+ # O campo _meta.phase = "end_of_fase_conhecimento" permite identificar
2782
+ # que este estado é pós-FASE1 (pronto para iniciar FASE2).
2783
  save_model_states_for_evaluation(
2784
  kls=kls,
2785
  reason="end_of_fase_conhecimento",
2786
  step=conhecimento_result["n_steps"],
2787
  total_samples=conhecimento_result["total_samples"],
2788
+ output_path=MODEL_STATES_PATH, # ARQUIVO UNIFICADO
2789
  )
2790
 
2791
  # Aggressive cleanup entre fases
 
2825
  # (b) permitir que um processo separado (apenas FASE2) seja iniciado
2826
  # do zero apontando para o estado salvo da FASE1.
2827
  logger.info("\n[V6.5-V2-metrics] Carregando estado do modelo da FASE1 (FASE2 uses FASE1 completed state)...")
2828
+ # V6.5-V2-unified — Carrega estado unificado (FASE1+FASE2 no mesmo arquivo).
2829
+ # User requirement: "LEMBRANDO que agora FASE1 e FASE2 estão treinadas
2830
+ # no mesmo estado do modelo".
2831
+ fase1_state_path = MODEL_STATES_PATH
2832
  fase1_load_result = load_fase1_state_into_kls(kls, fase1_state_path)
2833
 
2834
  logger.info("\n[V6.5-V2] Iniciando FASE 2 — TREINAMENTO COM PUNIÇÃO...")
 
2947
 
2948
  # Copia arquivos relevantes para download/
2949
  try:
2950
+ # V6.5-V2-unified — Apenas MODEL_STATES_PATH (arquivo unificado)
2951
  for src in [REPORT_PATH, USER_QUESTIONS_PATH, MODEL_STATES_PATH,
2952
+ V2_PHASES_EVAL_PATH, PREDICT_FIX_EVAL_PATH, ATTENTION_EVAL_PATH]:
 
2953
  if src.exists():
2954
  dst = DOWNLOAD_DIR / src.name
2955
  shutil.copy2(src, dst)
 
3006
  # Inclui: v6_5_v2_model_states.pt, v6_5_v2_model_states_after_conhecimento.pt
3007
  # Verifica tamanho ≤ 500MB (HF LFS free tier sem autenticar LFS).
3008
  _MAX_HF_LFS_SIZE_BYTES = 500 * 1024 * 1024 # 500MB
3009
+ # V6.5-V2-unified — Apenas UM arquivo de estado unificado.
3010
  for pt_candidate in [
3011
  MODEL_STATES_PATH,
 
3012
  ]:
3013
  if pt_candidate.exists():
3014
  pt_size = pt_candidate.stat().st_size
 
3148
  f"({hf_upload_result.get('n_files', 0)} files)")
3149
  print(f" HF_TOKEN cleaned : True (env + scripts)")
3150
  print(f" .pt files removed : {len(pt_files_removed)} (ALL local .pt cleaned)")
3151
+ print(f" Unified state file : {MODEL_STATES_PATH.name} (FASE1+FASE2 continuous)")
3152
+ print(f" k-means++ applied : {getattr(kls.som, '_kmeans_pp_initialized', False)}")
3153
  print(f" Storage cleanup : {final_storage_cleanup['n_files_removed']} files, "
3154
  f"{final_storage_cleanup['mb_freed']:.1f}MB freed")
3155
  print(f" Final n_hypotheses : {kls.n_hypotheses} (active={kls.hypothesis_ensemble.active_count})")
src/bigru_t/model/kohonen_learning_system.py CHANGED
@@ -507,6 +507,138 @@ class KohonenSOM4D:
507
  # colapso de neurônios mais cedo e ajustar γ a tempo.
508
  self._auto_adjust_interval = 10 # ajusta γ a cada 10 updates
509
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
510
  def _neighborhood(self, bmu_idx):
511
  """Vizinhança Gaussiana 4D: d² = Δi² + Δj² + Δk² + Δl²."""
512
  i, j, k, l = bmu_idx
@@ -671,20 +803,55 @@ class KohonenSOM4D:
671
  except Exception:
672
  buffer_tensor = None
673
 
674
- # Itera sobre neurônios mortos e reinicializa
 
 
 
 
 
 
 
 
675
  dead_indices = dead_mask.nonzero(as_tuple=False)
 
 
 
 
 
 
 
 
 
676
  for idx_tensor in dead_indices:
677
  i, j, k, l = idx_tensor.tolist()
678
  if buffer_tensor is not None and buffer_tensor.shape[0] > 0:
679
- # Amostra aleatória do buffer
680
- sample_idx = torch.randint(0, buffer_tensor.shape[0], (1,)).item()
681
- new_w = buffer_tensor[sample_idx].clone()
682
- # Pequeno ruído para evitar duplicação exata
683
- new_w = new_w + 0.05 * torch.randn(4)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
684
  else:
685
  # Reinicialização gaussiana pequena
686
  new_w = 0.1 * torch.randn(4)
687
- # Clamp para segurança
688
  new_w = torch.clamp(new_w, -10.0, 10.0)
689
  self.weights[i, j, k, l] = new_w
690
  # Reset counters
@@ -2931,8 +3098,30 @@ class KohonenLearningSystemV2(KohonenLearningSystem):
2931
  return {"active": False, "reason": "empty_buffer"}
2932
 
2933
  import time as _time
 
2934
  t0 = _time.time()
2935
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2936
  # Prepara dados
2937
  data = torch.stack(buffer_4d).detach() # (N, 4)
2938
  labels = torch.tensor(buffer_labels, dtype=torch.float, device=data.device)
@@ -2948,102 +3137,117 @@ class KohonenLearningSystemV2(KohonenLearningSystem):
2948
 
2949
  self.hypothesis_ensemble.train()
2950
  losses = []
 
 
2951
  # V6.5-V2-metrics-FIX: pré-computa data_norm_sq uma única vez (N, 1)
2952
  # para reuso em todos os steps — evita recomputação redundante.
2953
  data_norm_sq = (data * data).sum(dim=-1, keepdim=True).t() # (1, N)
2954
  labels_expanded = labels # (N,)
2955
 
2956
  for step in range(self.hyp_train_steps):
2957
- self.hyp_optimizer.zero_grad()
2958
-
2959
- # Ativação média como representação do estado do SOM
2960
- x_mean = som_activations.mean(dim=0, keepdim=True) # (1, P_som)
2961
-
2962
- # Gera deltas: (1, n_hypotheses, P)
2963
- deltas_stack = self.hypothesis_ensemble.forward_stacked(x_mean)
2964
- deltas_stack = self.delta_scale * deltas_stack # escala
2965
-
2966
- # V6.5-V2-metrics-FIX: Avaliação VETORIZADA de todas as hipóteses
2967
- # em paralelo (substitui loop que materializava 16 cópias do SOM).
2968
- # Memória: (n_hyp, P, N) em vez de 16 * (P + 2*P*N).
2969
- if self.classifier is not None and self.classifier_trained:
2970
- # deltas_stack: (1, n_hyp, P) → (n_hyp, P, 4) reshape
2971
- n_hyp = self.n_hypotheses
2972
- P_neurons = self.som_neuron_count
2973
- # som_weights_flat: (P, 4) detached
2974
- W_base = som_weights_flat.reshape(P_neurons, 4) # (P, 4)
2975
- # delta_h: (n_hyp, P) → reshape para (n_hyp, P, 4)
2976
- deltas_3d = deltas_stack[0].reshape(n_hyp, P_neurons, 4) # (n_hyp, P, 4)
2977
- # W_new[h, p, d] = W_base[p, d] + deltas_3d[h, p, d]
2978
- W_new = W_base.unsqueeze(0) + deltas_3d # (n_hyp, P, 4) — broadcast
2979
- # dist²[h, p, n] = ||W_new[h, p] - data[n]||²
2980
- # = ||W_new[h, p]||² + ||data[n]||² - 2*W_new[h, p]·data[n]
2981
- W_new_norm_sq = (W_new * W_new).sum(dim=-1) # (n_hyp, P)
2982
- # cross[h, p, n] = W_new[h, p] · data[n]
2983
- cross = torch.matmul(W_new, data.t()) # (n_hyp, P, N)
2984
- dist_sq = (
2985
- W_new_norm_sq.unsqueeze(-1) # (n_hyp, P, 1)
2986
- + data_norm_sq # (1, N) → broadcast (n_hyp, P, N)
2987
- - 2.0 * cross
2988
- ) # (n_hyp, P, N)
2989
- dist_sq = torch.clamp(dist_sq, min=0.0)
2990
- # activations[h, n, p] = dist_sq[h, p, n]
2991
- activations = dist_sq.transpose(1, 2) # (n_hyp, N, P)
2992
- # Classifier forward (congelado)
2993
- classifier_params_were_grad = [
2994
- p.requires_grad for p in self.classifier.parameters()
2995
- ]
2996
- for p in self.classifier.parameters():
2997
- p.requires_grad_(False)
2998
- try:
2999
- logits = self.classifier(
3000
- activations.reshape(n_hyp * len(data), P_neurons)
3001
- ).reshape(n_hyp, len(data)) # (n_hyp, N)
3002
- # BCE por hipótese, depois média
3003
- loss_per_hyp = F.binary_cross_entropy_with_logits(
3004
- logits,
3005
- labels_expanded.unsqueeze(0).expand(n_hyp, -1),
3006
- reduction='none',
3007
- ).mean(dim=1) # (n_hyp,)
3008
- loss_total = loss_per_hyp.mean()
3009
- # Regularização L2 sobre os deltas
3010
- reg_loss = 0.01 * deltas_stack.norm()
3011
- loss_total = loss_total + reg_loss
3012
- finally:
3013
- for p, was_grad in zip(self.classifier.parameters(),
3014
- classifier_params_were_grad):
3015
- p.requires_grad_(was_grad)
3016
- else:
3017
- # Fallback: minimizar norma do delta (regularização pura)
3018
- loss_total = deltas_stack.norm()
3019
-
3020
- loss_total.backward()
3021
- # V6.5-V2-metrics-FIX-3 — gradient clipping no treino de hipóteses
3022
- # previne explosão de gradientes em batches degenerados (ex: todos
3023
- # os labels iguais → BCE produz gradientes grandes). max_norm=1.0
3024
- # é o valor canônico recomendado pela literatura para BCE heads.
3025
- torch.nn.utils.clip_grad_norm_(
3026
- self.hypothesis_ensemble.parameters(), max_norm=1.0
3027
- )
3028
- # V6.5-V2-metrics-FIX-3 — zera grad do delta_scale (que não está
3029
- # no optimizer mas aparece no grafo de forward, aculumando .grad
3030
- # silenciosamente a cada step).
3031
- if self.delta_scale.grad is not None:
3032
- self.delta_scale.grad = None
3033
- self.hyp_optimizer.step()
3034
- losses.append(float(loss_total.item()))
3035
-
3036
- # V6.5-V2-metrics-FIX: libera tensores intermediários explicitamente
3037
- # para reduzir pico de memória entre steps.
3038
- del loss_total, deltas_stack
3039
- if 'W_new' in dir():
3040
- del W_new, cross, dist_sq, activations
3041
-
3042
- # Atualiza a escala de delta (decai suavemente)
3043
- with torch.no_grad():
3044
- self.delta_scale.data = torch.clamp(
3045
- self.delta_scale.data * 0.99, 0.001, 0.1
3046
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3047
 
3048
  self.hypothesis_ensemble.eval()
3049
  # V6.5-V2-metrics-FIX: libera som_activations e som_weights_flat
@@ -3063,6 +3267,9 @@ class KohonenLearningSystemV2(KohonenLearningSystem):
3063
  result = {
3064
  "active": True,
3065
  "n_steps": self.hyp_train_steps,
 
 
 
3066
  "n_hypotheses": self.n_hypotheses,
3067
  "loss_initial": float(losses[0]) if losses else 0.0,
3068
  "loss_final": loss_final_val,
@@ -3827,6 +4034,40 @@ class KohonenLearningSystemV2(KohonenLearningSystem):
3827
  # ou para N(0, 0.1) se buffer vazio. Isto acelera a diversificação
3828
  # quando o conscience mechanism sozinho não basta.
3829
  # ------------------------------------------------------------------
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3830
  def revive_dead_neurons(
3831
  self,
3832
  dead_threshold: int = 0,
 
507
  # colapso de neurônios mais cedo e ajustar γ a tempo.
508
  self._auto_adjust_interval = 10 # ajusta γ a cada 10 updates
509
 
510
+ # V6.5-V2-kmeans-pp — flag para inicialização k-means++ deferida.
511
+ # User requirement: "inicialização com k-means++ sobre embeddings em
512
+ # vez de grid coords+ruído dos embeddings que saem do tokenizador
513
+ # para ambas as FASE1 e FASE2". Como no momento do __init__ ainda
514
+ # não temos dados, a init k-means++ é DEFERIDA: ela acontece na
515
+ # primeira chamada de init_weights_kmeans_pp(data_buffer) feita
516
+ # pelo KLS quando o buffer_4d tem >= 64 amostras.
517
+ self._kmeans_pp_initialized = False
518
+ self._kmeans_pp_min_samples = 64
519
+
520
+ # ======================================================================
521
+ # V6.5-V2-kmeans-pp — Inicialização k-means++ sobre embeddings
522
+ # User requirement: "inicialização com k-means++ sobre embeddings em
523
+ # vez de grid coords+ruído dos embeddings que saem do tokenizador
524
+ # para ambas as FASE1 e FASE2".
525
+ # Math (Arthur & Vassilvitskii 2007):
526
+ # 1. c_1 = data[índice aleatório]
527
+ # 2. Para c_2..c_N: escolhe c_i = data[j] com prob ∝ D(j)^2
528
+ # onde D(j) = min_k ||x_j - c_k||^2
529
+ # 3. Após seeding, roda n_iter iterações de Lloyd para refinar centros.
530
+ # Vantagem: distribui pesos nas regiões de ALTA densidade dos dados,
531
+ # reduzindo dead neuron rate inicial.
532
+ # ======================================================================
533
+ def init_weights_kmeans_pp(
534
+ self,
535
+ data_buffer,
536
+ n_iter: int = 5,
537
+ random_seed=None,
538
+ ):
539
+ """Inicializa pesos do SOM via k-means++ (Arthur & Vassilvitskii 2007).
540
+
541
+ Args:
542
+ data_buffer: lista de tensores [4] — amostras do buffer_4d do KLS.
543
+ n_iter: número de iterações de Lloyd (default 5).
544
+ random_seed: seed para reprodutibilidade.
545
+
546
+ Returns:
547
+ Dict com: initialized, n_samples_used, n_neurons_assigned,
548
+ inertia (soma das dist^2 ao centro mais próximo).
549
+ """
550
+ if not data_buffer or len(data_buffer) < 8:
551
+ return {
552
+ "initialized": False,
553
+ "reason": f"insufficient_data ({len(data_buffer) if data_buffer else 0} < 8)",
554
+ }
555
+
556
+ if random_seed is not None:
557
+ g = torch.Generator().manual_seed(int(random_seed))
558
+ else:
559
+ g = None
560
+
561
+ with torch.no_grad():
562
+ try:
563
+ data = torch.stack(
564
+ [v.detach().clone().float() if isinstance(v, torch.Tensor)
565
+ else torch.tensor(v, dtype=torch.float)
566
+ for v in data_buffer]
567
+ ).float()
568
+ except Exception:
569
+ return {"initialized": False, "reason": "stack_failed"}
570
+
571
+ data = torch.nan_to_num(data, nan=0.0, posinf=1.0, neginf=-1.0)
572
+ N = data.shape[0]
573
+ P = self.n_neurons
574
+
575
+ if N < P:
576
+ repeats = (P // N) + 1
577
+ data = data.repeat(repeats, 1)[:P * 4]
578
+ N = data.shape[0]
579
+
580
+ # ---- k-means++ seeding ----
581
+ if g is not None:
582
+ first_idx = int(torch.randint(0, N, (1,), generator=g).item())
583
+ else:
584
+ first_idx = int(torch.randint(0, N, (1,)).item())
585
+ centers = [data[first_idx].clone()]
586
+
587
+ for k_idx in range(1, P):
588
+ C = torch.stack(centers)
589
+ data_sq = (data * data).sum(dim=-1, keepdim=True)
590
+ C_sq = (C * C).sum(dim=-1, keepdim=True).t()
591
+ cross = data @ C.t()
592
+ dist_sq = data_sq + C_sq - 2.0 * cross
593
+ dist_sq = torch.clamp(dist_sq, min=0.0)
594
+ D_sq_min = dist_sq.min(dim=-1).values
595
+ D_sum = D_sq_min.sum().clamp(min=1e-12)
596
+ probs = D_sq_min / D_sum
597
+ if g is not None:
598
+ new_idx = int(torch.multinomial(probs, 1, generator=g).item())
599
+ else:
600
+ new_idx = int(torch.multinomial(probs, 1).item())
601
+ centers.append(data[new_idx].clone())
602
+
603
+ centers_tensor = torch.stack(centers)
604
+ centers_tensor = centers_tensor + 0.02 * torch.randn(P, 4)
605
+
606
+ # ---- Lloyd iterations (refinamento) ----
607
+ for it in range(n_iter):
608
+ data_sq = (data * data).sum(dim=-1, keepdim=True)
609
+ C_sq = (centers_tensor * centers_tensor).sum(dim=-1, keepdim=True).t()
610
+ cross = data @ centers_tensor.t()
611
+ dist_sq = data_sq + C_sq - 2.0 * cross
612
+ dist_sq = torch.clamp(dist_sq, min=0.0)
613
+ nearest = dist_sq.argmin(dim=-1)
614
+ for p in range(P):
615
+ mask = (nearest == p)
616
+ if mask.any():
617
+ centers_tensor[p] = data[mask].mean(dim=0)
618
+
619
+ data_sq = (data * data).sum(dim=-1, keepdim=True)
620
+ C_sq = (centers_tensor * centers_tensor).sum(dim=-1, keepdim=True).t()
621
+ cross = data @ centers_tensor.t()
622
+ dist_sq = data_sq + C_sq - 2.0 * cross
623
+ dist_sq = torch.clamp(dist_sq, min=0.0)
624
+ inertia = float(dist_sq.min(dim=-1).values.sum().item())
625
+
626
+ perm = torch.randperm(P, generator=g)
627
+ centers_shuffled = centers_tensor[perm]
628
+ new_weights = centers_shuffled.view(self.I, self.J, self.K, self.L, 4).clone()
629
+ new_weights = torch.clamp(new_weights, -10.0, 10.0)
630
+ self.weights = new_weights
631
+ self._kmeans_pp_initialized = True
632
+
633
+ return {
634
+ "initialized": True,
635
+ "n_samples_used": int(N),
636
+ "n_neurons_assigned": int(P),
637
+ "n_lloyd_iters": int(n_iter),
638
+ "inertia": float(inertia),
639
+ "method": "kmeans++ (Arthur & Vassilvitskii 2007)",
640
+ }
641
+
642
  def _neighborhood(self, bmu_idx):
643
  """Vizinhança Gaussiana 4D: d² = Δi² + Δj² + Δk² + Δl²."""
644
  i, j, k, l = bmu_idx
 
803
  except Exception:
804
  buffer_tensor = None
805
 
806
+ # V6.5-V2-kmeans-pp-revive — Reinicializa neurônios mortos usando
807
+ # estratégia k-means++ (max-min selection): em vez de amostrar um
808
+ # ponto aleatório do buffer, escolhe o ponto do buffer que está
809
+ # MAIS DISTANTE de todos os neurônios ativos atuais. Isto garante
810
+ # que os pesos revividos cubram regiões do espaço de entrada que
811
+ # não estavam sendo representadas, maximizando a diversidade
812
+ # topológica do SOM.
813
+ # User requirement: "reforçar auto_revive para reinicializar pesos
814
+ # dos neurônios mortos (não só boostar γ)".
815
  dead_indices = dead_mask.nonzero(as_tuple=False)
816
+ # Identifica neurônios ATIVOS (win_count > 0) para cálculo de distância
817
+ active_mask = self.bmu_win_count > 0
818
+ active_indices = active_mask.nonzero(as_tuple=False)
819
+ # Stack pesos ativos (centros atuais do SOM)
820
+ if active_indices.shape[0] > 0:
821
+ active_weights = self.weights[active_mask] # (n_active, 4)
822
+ else:
823
+ active_weights = None
824
+
825
  for idx_tensor in dead_indices:
826
  i, j, k, l = idx_tensor.tolist()
827
  if buffer_tensor is not None and buffer_tensor.shape[0] > 0:
828
+ if active_weights is not None and active_weights.shape[0] > 0:
829
+ # k-means++ selection: ponto mais distante dos ativos
830
+ # dist²[j, k] = ||buffer[j] - active[k]||²
831
+ buf_sq = (buffer_tensor * buffer_tensor).sum(dim=-1, keepdim=True) # (B, 1)
832
+ act_sq = (active_weights * active_weights).sum(dim=-1, keepdim=True).t() # (1, A)
833
+ cross = buffer_tensor @ active_weights.t() # (B, A)
834
+ dist_sq = buf_sq + act_sq - 2.0 * cross
835
+ dist_sq = torch.clamp(dist_sq, min=0.0)
836
+ # D²(j) = min_k ||buf[j] - active[k]||²
837
+ D_sq_min = dist_sq.min(dim=-1).values # (B,)
838
+ # Escolhe o ponto com maior D² (mais distante)
839
+ sample_idx = int(torch.argmax(D_sq_min).item())
840
+ new_w = buffer_tensor[sample_idx].clone()
841
+ # Pequeno ruído para evitar duplicação exata
842
+ new_w = new_w + 0.05 * torch.randn(4)
843
+ # Adiciona o novo peso à lista de ativos para próxima iteração
844
+ active_weights = torch.cat([active_weights, new_w.unsqueeze(0)], dim=0)
845
+ else:
846
+ # Sem ativos: amostra aleatória simples
847
+ sample_idx = torch.randint(0, buffer_tensor.shape[0], (1,)).item()
848
+ new_w = buffer_tensor[sample_idx].clone()
849
+ new_w = new_w + 0.05 * torch.randn(4)
850
+ # Inicia lista de ativos
851
+ active_weights = new_w.unsqueeze(0).clone()
852
  else:
853
  # Reinicialização gaussiana pequena
854
  new_w = 0.1 * torch.randn(4)
 
855
  new_w = torch.clamp(new_w, -10.0, 10.0)
856
  self.weights[i, j, k, l] = new_w
857
  # Reset counters
 
3098
  return {"active": False, "reason": "empty_buffer"}
3099
 
3100
  import time as _time
3101
+ import gc as _gc
3102
  t0 = _time.time()
3103
 
3104
+ # V6.5-V2-oom-guard — Pré-check de memória antes de alocar tensores
3105
+ # grandes (n_hyp * P * N * 4 bytes). Se memória livre < 2x estimado,
3106
+ # trunca o buffer para evitar OOM-killer.
3107
+ # User requirement: "o processo vem sendo morto OOM-kiler".
3108
+ try:
3109
+ import os as _os
3110
+ # Estima uso: n_hyp * P * N * 4 bytes (float32) + overhead 50%
3111
+ _est_bytes = int(self.n_hypotheses * self.som_neuron_count *
3112
+ len(buffer_4d) * 4 * 1.5)
3113
+ _est_mb = _est_bytes / 1e6
3114
+ # Se estimativa > 800MB, trunca buffer para últimos 64 samples
3115
+ if _est_mb > 800 and len(buffer_4d) > 64:
3116
+ _orig_n = len(buffer_4d)
3117
+ buffer_4d = buffer_4d[-64:]
3118
+ buffer_labels = buffer_labels[-64:]
3119
+ # Log silently (não podemos usar logger aqui para evitar circular import)
3120
+ print(f"[V6.5-V2-oom-guard] train_hypotheses: buffer truncated "
3121
+ f"{_orig_n}→64 (est. {_est_mb:.0f}MB > 800MB)")
3122
+ except Exception:
3123
+ pass # Nunca deixa o pré-check quebrar o treino
3124
+
3125
  # Prepara dados
3126
  data = torch.stack(buffer_4d).detach() # (N, 4)
3127
  labels = torch.tensor(buffer_labels, dtype=torch.float, device=data.device)
 
3137
 
3138
  self.hypothesis_ensemble.train()
3139
  losses = []
3140
+ _oom_truncated = False # V6.5-V2-oom-guard: flag setada em caso de OOM
3141
+ _oom_step = -1 # V6.5-V2-oom-guard: step onde OOM ocorreu
3142
  # V6.5-V2-metrics-FIX: pré-computa data_norm_sq uma única vez (N, 1)
3143
  # para reuso em todos os steps — evita recomputação redundante.
3144
  data_norm_sq = (data * data).sum(dim=-1, keepdim=True).t() # (1, N)
3145
  labels_expanded = labels # (N,)
3146
 
3147
  for step in range(self.hyp_train_steps):
3148
+ # V6.5-V2-oom-guard — captura OOM/RuntimeError por step.
3149
+ # Em caso de OOM, interrompe o treino de hipóteses graciosamente.
3150
+ try:
3151
+ self.hyp_optimizer.zero_grad()
3152
+
3153
+ # Ativação média como representação do estado do SOM
3154
+ x_mean = som_activations.mean(dim=0, keepdim=True) # (1, P_som)
3155
+
3156
+ # Gera deltas: (1, n_hypotheses, P)
3157
+ deltas_stack = self.hypothesis_ensemble.forward_stacked(x_mean)
3158
+ deltas_stack = self.delta_scale * deltas_stack # escala
3159
+
3160
+ # V6.5-V2-metrics-FIX: Avaliação VETORIZADA de todas as hipóteses
3161
+ # em paralelo (substitui loop que materializava 16 cópias do SOM).
3162
+ # Memória: (n_hyp, P, N) em vez de 16 * (P + 2*P*N).
3163
+ if self.classifier is not None and self.classifier_trained:
3164
+ # deltas_stack: (1, n_hyp, P) → (n_hyp, P, 4) reshape
3165
+ n_hyp = self.n_hypotheses
3166
+ P_neurons = self.som_neuron_count
3167
+ # som_weights_flat: (P, 4) detached
3168
+ W_base = som_weights_flat.reshape(P_neurons, 4) # (P, 4)
3169
+ # delta_h: (n_hyp, P) → reshape para (n_hyp, P, 4)
3170
+ deltas_3d = deltas_stack[0].reshape(n_hyp, P_neurons, 4) # (n_hyp, P, 4)
3171
+ # W_new[h, p, d] = W_base[p, d] + deltas_3d[h, p, d]
3172
+ W_new = W_base.unsqueeze(0) + deltas_3d # (n_hyp, P, 4) — broadcast
3173
+ # dist²[h, p, n] = ||W_new[h, p] - data[n]||²
3174
+ # = ||W_new[h, p]||² + ||data[n]||² - 2*W_new[h, p]·data[n]
3175
+ W_new_norm_sq = (W_new * W_new).sum(dim=-1) # (n_hyp, P)
3176
+ # cross[h, p, n] = W_new[h, p] · data[n]
3177
+ cross = torch.matmul(W_new, data.t()) # (n_hyp, P, N)
3178
+ dist_sq = (
3179
+ W_new_norm_sq.unsqueeze(-1) # (n_hyp, P, 1)
3180
+ + data_norm_sq # (1, N) → broadcast (n_hyp, P, N)
3181
+ - 2.0 * cross
3182
+ ) # (n_hyp, P, N)
3183
+ dist_sq = torch.clamp(dist_sq, min=0.0)
3184
+ # activations[h, n, p] = dist_sq[h, p, n]
3185
+ activations = dist_sq.transpose(1, 2) # (n_hyp, N, P)
3186
+ # Classifier forward (congelado)
3187
+ classifier_params_were_grad = [
3188
+ p.requires_grad for p in self.classifier.parameters()
3189
+ ]
3190
+ for p in self.classifier.parameters():
3191
+ p.requires_grad_(False)
3192
+ try:
3193
+ logits = self.classifier(
3194
+ activations.reshape(n_hyp * len(data), P_neurons)
3195
+ ).reshape(n_hyp, len(data)) # (n_hyp, N)
3196
+ # BCE por hipótese, depois média
3197
+ loss_per_hyp = F.binary_cross_entropy_with_logits(
3198
+ logits,
3199
+ labels_expanded.unsqueeze(0).expand(n_hyp, -1),
3200
+ reduction='none',
3201
+ ).mean(dim=1) # (n_hyp,)
3202
+ loss_total = loss_per_hyp.mean()
3203
+ # Regularização L2 sobre os deltas
3204
+ reg_loss = 0.01 * deltas_stack.norm()
3205
+ loss_total = loss_total + reg_loss
3206
+ finally:
3207
+ for p, was_grad in zip(self.classifier.parameters(),
3208
+ classifier_params_were_grad):
3209
+ p.requires_grad_(was_grad)
3210
+ else:
3211
+ # Fallback: minimizar norma do delta (regularização pura)
3212
+ loss_total = deltas_stack.norm()
3213
+
3214
+ loss_total.backward()
3215
+ # V6.5-V2-metrics-FIX-3 — gradient clipping no treino de hipóteses
3216
+ # previne explosão de gradientes em batches degenerados (ex: todos
3217
+ # os labels iguais → BCE produz gradientes grandes). max_norm=1.0
3218
+ # é o valor canônico recomendado pela literatura para BCE heads.
3219
+ torch.nn.utils.clip_grad_norm_(
3220
+ self.hypothesis_ensemble.parameters(), max_norm=1.0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3221
  )
3222
+ # V6.5-V2-metrics-FIX-3 — zera grad do delta_scale (que não está
3223
+ # no optimizer mas aparece no grafo de forward, aculumando .grad
3224
+ # silenciosamente a cada step).
3225
+ if self.delta_scale.grad is not None:
3226
+ self.delta_scale.grad = None
3227
+ self.hyp_optimizer.step()
3228
+ losses.append(float(loss_total.item()))
3229
+
3230
+ # V6.5-V2-metrics-FIX: libera tensores intermediários explicitamente
3231
+ # para reduzir pico de memória entre steps.
3232
+ del loss_total, deltas_stack
3233
+ if 'W_new' in dir():
3234
+ del W_new, cross, dist_sq, activations
3235
+
3236
+ # Atualiza a escala de delta (decai suavemente)
3237
+ with torch.no_grad():
3238
+ self.delta_scale.data = torch.clamp(
3239
+ self.delta_scale.data * 0.99, 0.001, 0.1
3240
+ )
3241
+ except (RuntimeError, MemoryError) as _oom_err:
3242
+ # V6.5-V2-oom-guard — OOM detectado durante o step.
3243
+ # Interrompe o treino de hipóteses e retorna partial results.
3244
+ _oom_truncated = True
3245
+ _oom_step = step
3246
+ try:
3247
+ _gc.collect()
3248
+ except Exception:
3249
+ pass
3250
+ break
3251
 
3252
  self.hypothesis_ensemble.eval()
3253
  # V6.5-V2-metrics-FIX: libera som_activations e som_weights_flat
 
3267
  result = {
3268
  "active": True,
3269
  "n_steps": self.hyp_train_steps,
3270
+ "n_steps_executed": len(losses),
3271
+ "oom_truncated": bool(_oom_truncated),
3272
+ "oom_step": int(_oom_step),
3273
  "n_hypotheses": self.n_hypotheses,
3274
  "loss_initial": float(losses[0]) if losses else 0.0,
3275
  "loss_final": loss_final_val,
 
4034
  # ou para N(0, 0.1) se buffer vazio. Isto acelera a diversificação
4035
  # quando o conscience mechanism sozinho não basta.
4036
  # ------------------------------------------------------------------
4037
+ def init_weights_kmeans_pp_if_ready(self):
4038
+ """V6.5-V2-kmeans-pp - Chama init_weights_kmeans_pp no SOM se buffer >= 64.
4039
+
4040
+ Tenta uma unica vez durante o treino (idempotente via flag
4041
+ _kmeans_pp_init_attempted). Se falhar (buffer pequeno, erro), retorna
4042
+ silently e o SOM mantem a inicializacao grid coords+ruido original.
4043
+
4044
+ Returns:
4045
+ Dict com status da inicializacao.
4046
+ """
4047
+ if getattr(self, "_kmeans_pp_init_attempted", False):
4048
+ return {"initialized": False, "reason": "already_attempted"}
4049
+ self._kmeans_pp_init_attempted = True
4050
+
4051
+ if len(self.buffer_4d) < self.som._kmeans_pp_min_samples:
4052
+ return {
4053
+ "initialized": False,
4054
+ "reason": f"buffer_too_small ({len(self.buffer_4d)} < {self.som._kmeans_pp_min_samples})",
4055
+ }
4056
+ try:
4057
+ result = self.som.init_weights_kmeans_pp(
4058
+ data_buffer=self.buffer_4d,
4059
+ n_iter=5,
4060
+ random_seed=42,
4061
+ )
4062
+ if result.get("initialized"):
4063
+ self.som.bmu_win_count.zero_()
4064
+ target_p = 1.0 / self.som.n_neurons
4065
+ self.som.win_frequency.fill_(target_p)
4066
+ self.som._recent_bmu_flat.clear()
4067
+ return result
4068
+ except Exception as e:
4069
+ return {"initialized": False, "reason": f"error: {e}"}
4070
+
4071
  def revive_dead_neurons(
4072
  self,
4073
  dead_threshold: int = 0,
v6_5_v2_attention_eval.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "evaluation": "attention_v65_v2",
3
+ "user_requirement": "verificar se o mecanismo de atenção está ativo e acessado logicamente funcional",
4
+ "metrics": {
5
+ "active": true,
6
+ "n_calls": 10000,
7
+ "n_errors": 0,
8
+ "last_norm_in": 112.07373046875,
9
+ "last_norm_out": 117.50370788574219,
10
+ "last_attn_activated": true,
11
+ "last_attn_diff_norm": 116.66998291015625,
12
+ "n_heads": 8,
13
+ "logic_functional": true
14
+ },
15
+ "active": true,
16
+ "logic_functional": true,
17
+ "n_calls": 10000,
18
+ "n_errors": 0,
19
+ "n_heads": 8,
20
+ "assessment": "PASS"
21
+ }
v6_5_v2_model_states.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:49b0636b59f766efc18ed05732a7f663a4c7b3588c485dd65d22136f7b4b9d83
3
+ size 220157236
v6_5_v2_phases_eval.json ADDED
The diff for this file is too large to render. See raw diff
 
v6_5_v2_predict_fix_eval.json ADDED
@@ -0,0 +1,147 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "evaluation": "predict_fix_v65_v2",
3
+ "user_requirement": "os strings 'gato' e 'cachorro' são fixos quando deveriam ser extrações variáveis e flexíveis de rótulos proveniente de dados dos datasets anteriormente treinados",
4
+ "n_test_queries": 6,
5
+ "results": [
6
+ {
7
+ "query": "o gato dorme na cama",
8
+ "prediction": "short_text",
9
+ "probability": 0.42794111371040344,
10
+ "is_gato_hardcoded": false,
11
+ "is_cachorro_hardcoded": false,
12
+ "is_registry_label": true,
13
+ "is_default_label": false
14
+ },
15
+ {
16
+ "query": "calcule dois mais dois",
17
+ "prediction": "short_text",
18
+ "probability": 0.20115551352500916,
19
+ "is_gato_hardcoded": false,
20
+ "is_cachorro_hardcoded": false,
21
+ "is_registry_label": true,
22
+ "is_default_label": false
23
+ },
24
+ {
25
+ "query": "qual é a capital do brasil",
26
+ "prediction": "short_text",
27
+ "probability": 0.2258892059326172,
28
+ "is_gato_hardcoded": false,
29
+ "is_cachorro_hardcoded": false,
30
+ "is_registry_label": true,
31
+ "is_default_label": false
32
+ },
33
+ {
34
+ "query": "explique o que é uma rede neural",
35
+ "prediction": "short_text",
36
+ "probability": 0.47971081733703613,
37
+ "is_gato_hardcoded": false,
38
+ "is_cachorro_hardcoded": false,
39
+ "is_registry_label": true,
40
+ "is_default_label": false
41
+ },
42
+ {
43
+ "query": "olá como você está",
44
+ "prediction": "short_text",
45
+ "probability": 0.447274774312973,
46
+ "is_gato_hardcoded": false,
47
+ "is_cachorro_hardcoded": false,
48
+ "is_registry_label": true,
49
+ "is_default_label": false
50
+ },
51
+ {
52
+ "query": "traduza hello para portugues",
53
+ "prediction": "short_text",
54
+ "probability": 0.4153057634830475,
55
+ "is_gato_hardcoded": false,
56
+ "is_cachorro_hardcoded": false,
57
+ "is_registry_label": true,
58
+ "is_default_label": false
59
+ }
60
+ ],
61
+ "per_dataset_results": [
62
+ {
63
+ "dataset": "dominguesm/restore-punctuation-ptbr-dataset",
64
+ "expected_labels": [
65
+ "unpunctuated",
66
+ "punctuated"
67
+ ],
68
+ "prediction": "punctuated",
69
+ "valid_for_dataset": true
70
+ },
71
+ {
72
+ "dataset": "carolina-c4ai/corpus-carolina",
73
+ "expected_labels": [
74
+ "raw_corpus",
75
+ "normalized_text"
76
+ ],
77
+ "prediction": "normalized_text",
78
+ "valid_for_dataset": true
79
+ },
80
+ {
81
+ "dataset": "CEIA-POSITIVO/ultrachat_br_clustred_balanced_v1",
82
+ "expected_labels": [
83
+ "user_turn",
84
+ "assistant_turn"
85
+ ],
86
+ "prediction": "assistant_turn",
87
+ "valid_for_dataset": true
88
+ },
89
+ {
90
+ "dataset": "dominguesm/Canarim-Instruct-PTBR-Dataset",
91
+ "expected_labels": [
92
+ "instruction",
93
+ "response"
94
+ ],
95
+ "prediction": "response",
96
+ "valid_for_dataset": true
97
+ },
98
+ {
99
+ "dataset": "adalbertojunior/punctuation-ptbr",
100
+ "expected_labels": [
101
+ "unpunctuated",
102
+ "punctuated"
103
+ ],
104
+ "prediction": "punctuated",
105
+ "valid_for_dataset": true
106
+ },
107
+ {
108
+ "dataset": "iara-project/news-articles-ptbr-dataset",
109
+ "expected_labels": [
110
+ "headline",
111
+ "body"
112
+ ],
113
+ "prediction": "body",
114
+ "valid_for_dataset": true
115
+ },
116
+ {
117
+ "dataset": "manoela/noticias_ptbr",
118
+ "expected_labels": [
119
+ "headline",
120
+ "body"
121
+ ],
122
+ "prediction": "body",
123
+ "valid_for_dataset": true
124
+ },
125
+ {
126
+ "dataset": "BrunoN-Dev/corpus-ptbr-v1",
127
+ "expected_labels": [
128
+ "short_text",
129
+ "long_text"
130
+ ],
131
+ "prediction": "long_text",
132
+ "valid_for_dataset": true
133
+ }
134
+ ],
135
+ "summary": {
136
+ "n_returns_gato": 0,
137
+ "n_returns_cachorro": 0,
138
+ "n_returns_registry_label": 6,
139
+ "n_returns_default_label": 0,
140
+ "hardcoded_bug_present": false,
141
+ "predict_fix_verified": true
142
+ },
143
+ "quality_assessment": {
144
+ "fix_status": "PASS",
145
+ "note": "predict() não retorna mais 'gato'/'cachorro' fixos — usa label_registry dinâmico."
146
+ }
147
+ }
v6_5_v2_report.json ADDED
@@ -0,0 +1,309 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "version": "V6.5-V2",
3
+ "timestamp": "2026-08-09T14:25:12.703041",
4
+ "config": {
5
+ "som_grid": [
6
+ 6,
7
+ 6,
8
+ 6,
9
+ 4
10
+ ],
11
+ "n_neurons": 864,
12
+ "hidden_dim": 1024,
13
+ "vocab_size": 16384,
14
+ "n_hypotheses": 16,
15
+ "n_trials": 3,
16
+ "hyp_train_steps": 30,
17
+ "stream_batch_size": 100,
18
+ "meta_conhecimento": 8000,
19
+ "meta_punicão": 2000,
20
+ "conhecimento_datasets": [
21
+ "dominguesm/restore-punctuation-ptbr-dataset",
22
+ "carolina-c4ai/corpus-carolina",
23
+ "CEIA-POSITIVO/ultrachat_br_clustred_balanced_v1",
24
+ "dominguesm/Canarim-Instruct-PTBR-Dataset",
25
+ "adalbertojunior/punctuation-ptbr",
26
+ "iara-project/news-articles-ptbr-dataset",
27
+ "manoela/noticias_ptbr",
28
+ "BrunoN-Dev/corpus-ptbr-v1"
29
+ ],
30
+ "punicão_dataset": "BrunoN-Dev/corpus-ptbr-v1"
31
+ },
32
+ "xeon_status": {
33
+ "version": "V6",
34
+ "physical_cores": 2,
35
+ "env": {
36
+ "MKL_ENABLE_INSTRUCTIONS": "AVX512",
37
+ "MKL_NUM_THREADS": "2",
38
+ "OMP_NUM_THREADS": "2",
39
+ "MKL_DYNAMIC": "FALSE",
40
+ "DNNL_PRIMITIVE_CACHE_CAPACITY": "1024",
41
+ "ONEDNN_MAX_CPU_ISA": "AMX_INT8",
42
+ "KMP_AFFINITY": "granularity=fine,compact,1,0",
43
+ "KMP_BLOCKTIME": "1"
44
+ },
45
+ "avx512": {
46
+ "supported": true,
47
+ "desc": "AVX512_VNNI (full INT8 acceleration)"
48
+ },
49
+ "amx": {
50
+ "supported": true,
51
+ "desc": "AMX (tile + int8 + bf16) — full AMX acceleration"
52
+ },
53
+ "ipex_available": false,
54
+ "init_done": true
55
+ },
56
+ "fp16_benchmark": {
57
+ "best_time_ms": 92.64165200011121,
58
+ "avg_time_ms": 92.91077149998728,
59
+ "best_tflops": 1.3816679348490715,
60
+ "avg_tflops": 1.3776658823677677,
61
+ "matrix_size": 4000.0
62
+ },
63
+ "v2_verification": {
64
+ "checks": {
65
+ "v2_instantiation": {
66
+ "status": "PASS",
67
+ "details": "n_hyp=4/8 (active=4), n_trials=2, delta_scale=0.0100"
68
+ },
69
+ "predict_dynamic_labels": {
70
+ "status": "PASS",
71
+ "details": "predict returned: negative_class (dynamic, not hardcoded)"
72
+ },
73
+ "process_batch_v2_conhecimento": {
74
+ "status": "PASS",
75
+ "details": "phase=CONHECIMENTO, action=none"
76
+ },
77
+ "v2_methods_exist": {
78
+ "status": "PASS",
79
+ "details": "train_hypotheses, select_best_delta, apply_best_delta_and_consolidate, process_batch_v2, get_v2_metrics"
80
+ }
81
+ },
82
+ "n_pass": 4,
83
+ "n_fail": 0,
84
+ "all_pass": true
85
+ },
86
+ "fase_1_conhecimento_summary": {
87
+ "total_samples": 8000,
88
+ "meta_atingida": true,
89
+ "elapsed_s": 529.4711573123932,
90
+ "storage_critical_stopped": false
91
+ },
92
+ "fase_2_punicão_summary": {
93
+ "total_samples": 2000,
94
+ "meta_atingida": true,
95
+ "elapsed_s": 571.2624096870422,
96
+ "punishment_events": 76,
97
+ "hypotheses_trainings": 38,
98
+ "delta_applications": 38,
99
+ "skipped": false
100
+ },
101
+ "attention_eval_summary": {
102
+ "active": true,
103
+ "logic_functional": true,
104
+ "n_calls": 10000,
105
+ "assessment": "PASS"
106
+ },
107
+ "predict_fix_summary": {
108
+ "n_returns_gato": 0,
109
+ "n_returns_cachorro": 0,
110
+ "n_returns_registry_label": 6,
111
+ "n_returns_default_label": 0,
112
+ "hardcoded_bug_present": false,
113
+ "predict_fix_verified": true
114
+ },
115
+ "user_questions_summary": {
116
+ "n_with_answer": 3,
117
+ "n_with_think": 0,
118
+ "answer_rate": 1.0,
119
+ "think_rate": 0.0,
120
+ "avg_latency_ms": 4.942417144775391,
121
+ "avg_reasoning_length": 69.0
122
+ },
123
+ "model_states_saved_to": "/home/z/my-project/BiGRU_T_version/v6_5_v2_model_states.pt",
124
+ "save_info": {
125
+ "saved": true,
126
+ "path": "/home/z/my-project/BiGRU_T_version/v6_5_v2_model_states.pt",
127
+ "size_mb": 220.157236,
128
+ "size_gb": 0.2050374038517475,
129
+ "size_status": "OK",
130
+ "size_within_1gb_limit": true,
131
+ "reason": "end_of_training_v2",
132
+ "step": 700,
133
+ "total_samples": 10000,
134
+ "n_tensors": 7,
135
+ "n_buffer_tail": 64
136
+ },
137
+ "final_v2_metrics": {
138
+ "version": "V2-dynamic",
139
+ "n_hypotheses": 4,
140
+ "n_hypotheses_active": 4,
141
+ "max_n_hypotheses": 32,
142
+ "n_trials": 6,
143
+ "hyp_train_steps": 80,
144
+ "hyp_lr": 0.0001,
145
+ "delta_scale": 0.0010000000474974513,
146
+ "n_generators": 32,
147
+ "punishment_count": 0,
148
+ "success_count": 0,
149
+ "training_ready": false,
150
+ "classifier_trained": true,
151
+ "ewc_reference_set": true,
152
+ "buffer_size": 128,
153
+ "total_hyp_steps_executed": 2670,
154
+ "n_train_hyp_calls": 38,
155
+ "dynamic_adaptation": {
156
+ "loss_history_len": 8,
157
+ "loss_stats": {
158
+ "slope": 0.020730994996570405,
159
+ "volatility": 0.09847807385506234,
160
+ "mean": 0.6678446382284164,
161
+ "std": 0.06576805360716538,
162
+ "n": 8
163
+ },
164
+ "punishment_rate": 1.0,
165
+ "punishment_window_size": 12,
166
+ "n_adaptations": 20,
167
+ "limits": {
168
+ "min_n_hypotheses": 4,
169
+ "max_n_hypotheses": 32,
170
+ "min_n_trials": 1,
171
+ "max_n_trials": 6,
172
+ "min_hyp_train_steps": 10,
173
+ "max_hyp_train_steps": 80
174
+ },
175
+ "last_5_adaptations": [
176
+ {
177
+ "trigger": "punishment",
178
+ "step": 71,
179
+ "before": {
180
+ "n_hypotheses": 4,
181
+ "n_trials": 6,
182
+ "hyp_train_steps": 55
183
+ },
184
+ "after": {
185
+ "n_hypotheses": 4,
186
+ "n_trials": 6,
187
+ "hyp_train_steps": 60
188
+ },
189
+ "adapted": true,
190
+ "rules_fired": [
191
+ "steps+5 (slope=0.01182 → 60)"
192
+ ],
193
+ "loss_stats": {
194
+ "slope": 0.011820448296410697,
195
+ "volatility": 0.10509967848483093,
196
+ "mean": 0.6714299917221069,
197
+ "std": 0.07056707625506613,
198
+ "n": 8
199
+ },
200
+ "punishment_rate": 1.0
201
+ },
202
+ {
203
+ "trigger": "auto",
204
+ "step": 72,
205
+ "before": {
206
+ "n_hypotheses": 4,
207
+ "n_trials": 6,
208
+ "hyp_train_steps": 60
209
+ },
210
+ "after": {
211
+ "n_hypotheses": 4,
212
+ "n_trials": 6,
213
+ "hyp_train_steps": 65
214
+ },
215
+ "adapted": true,
216
+ "rules_fired": [
217
+ "steps+5 (slope=0.01214 → 65)"
218
+ ],
219
+ "loss_stats": {
220
+ "slope": 0.01214205367224557,
221
+ "volatility": 0.10464130233226861,
222
+ "mean": 0.6752422899007797,
223
+ "std": 0.07065823260504085,
224
+ "n": 8
225
+ },
226
+ "punishment_rate": 1.0
227
+ },
228
+ {
229
+ "trigger": "punishment",
230
+ "step": 73,
231
+ "before": {
232
+ "n_hypotheses": 4,
233
+ "n_trials": 6,
234
+ "hyp_train_steps": 65
235
+ },
236
+ "after": {
237
+ "n_hypotheses": 4,
238
+ "n_trials": 6,
239
+ "hyp_train_steps": 70
240
+ },
241
+ "adapted": true,
242
+ "rules_fired": [
243
+ "steps+5 (slope=0.01214 → 70)"
244
+ ],
245
+ "loss_stats": {
246
+ "slope": 0.01214205367224557,
247
+ "volatility": 0.10464130233226861,
248
+ "mean": 0.6752422899007797,
249
+ "std": 0.07065823260504085,
250
+ "n": 8
251
+ },
252
+ "punishment_rate": 1.0
253
+ },
254
+ {
255
+ "trigger": "auto",
256
+ "step": 74,
257
+ "before": {
258
+ "n_hypotheses": 4,
259
+ "n_trials": 6,
260
+ "hyp_train_steps": 70
261
+ },
262
+ "after": {
263
+ "n_hypotheses": 4,
264
+ "n_trials": 6,
265
+ "hyp_train_steps": 75
266
+ },
267
+ "adapted": true,
268
+ "rules_fired": [
269
+ "steps+5 (slope=0.02073 → 75)"
270
+ ],
271
+ "loss_stats": {
272
+ "slope": 0.020730994996570405,
273
+ "volatility": 0.09847807385506234,
274
+ "mean": 0.6678446382284164,
275
+ "std": 0.06576805360716538,
276
+ "n": 8
277
+ },
278
+ "punishment_rate": 1.0
279
+ },
280
+ {
281
+ "trigger": "punishment",
282
+ "step": 75,
283
+ "before": {
284
+ "n_hypotheses": 4,
285
+ "n_trials": 6,
286
+ "hyp_train_steps": 75
287
+ },
288
+ "after": {
289
+ "n_hypotheses": 4,
290
+ "n_trials": 6,
291
+ "hyp_train_steps": 80
292
+ },
293
+ "adapted": true,
294
+ "rules_fired": [
295
+ "steps+5 (slope=0.02073 → 80)"
296
+ ],
297
+ "loss_stats": {
298
+ "slope": 0.020730994996570405,
299
+ "volatility": 0.09847807385506234,
300
+ "mean": 0.6678446382284164,
301
+ "std": 0.06576805360716538,
302
+ "n": 8
303
+ },
304
+ "punishment_rate": 1.0
305
+ }
306
+ ]
307
+ }
308
+ }
309
+ }
v6_5_v2_user_questions.json ADDED
@@ -0,0 +1,82 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "evaluation": "user_questions_without_help_v65_v2",
3
+ "user_requirement": "não ajudar o modelo em respostas e lançar perguntas",
4
+ "questions_sent_verbatim": true,
5
+ "no_context_added": true,
6
+ "no_system_prompt": true,
7
+ "no_few_shot": true,
8
+ "n_questions": 3,
9
+ "questions": [
10
+ "Luva de Pedreiro Távila",
11
+ "Lula reserva valor",
12
+ "Amazonas força-tarefa vítimas"
13
+ ],
14
+ "results": [
15
+ {
16
+ "query": "Luva de Pedreiro Távila",
17
+ "query_was_modified": false,
18
+ "context_provided": false,
19
+ "system_prompt_used": false,
20
+ "few_shot_examples": false,
21
+ "som_prediction": "short_text",
22
+ "reasoning_length": 69,
23
+ "has_think": false,
24
+ "has_plan": false,
25
+ "has_answer": true,
26
+ "has_decompose": false,
27
+ "think_preview": "",
28
+ "answer_preview": "ReasoningEngine disabled. SOM prediction: short_text",
29
+ "raw_response_preview": "<answer>ReasoningEngine disabled. SOM prediction: short_text</answer>",
30
+ "n_tags": 1,
31
+ "latency_ms": 4.969596862792969
32
+ },
33
+ {
34
+ "query": "Lula reserva valor",
35
+ "query_was_modified": false,
36
+ "context_provided": false,
37
+ "system_prompt_used": false,
38
+ "few_shot_examples": false,
39
+ "som_prediction": "short_text",
40
+ "reasoning_length": 69,
41
+ "has_think": false,
42
+ "has_plan": false,
43
+ "has_answer": true,
44
+ "has_decompose": false,
45
+ "think_preview": "",
46
+ "answer_preview": "ReasoningEngine disabled. SOM prediction: short_text",
47
+ "raw_response_preview": "<answer>ReasoningEngine disabled. SOM prediction: short_text</answer>",
48
+ "n_tags": 1,
49
+ "latency_ms": 4.921436309814453
50
+ },
51
+ {
52
+ "query": "Amazonas força-tarefa vítimas",
53
+ "query_was_modified": false,
54
+ "context_provided": false,
55
+ "system_prompt_used": false,
56
+ "few_shot_examples": false,
57
+ "som_prediction": "short_text",
58
+ "reasoning_length": 69,
59
+ "has_think": false,
60
+ "has_plan": false,
61
+ "has_answer": true,
62
+ "has_decompose": false,
63
+ "think_preview": "",
64
+ "answer_preview": "ReasoningEngine disabled. SOM prediction: short_text",
65
+ "raw_response_preview": "<answer>ReasoningEngine disabled. SOM prediction: short_text</answer>",
66
+ "n_tags": 1,
67
+ "latency_ms": 4.93621826171875
68
+ }
69
+ ],
70
+ "summary": {
71
+ "n_with_answer": 3,
72
+ "n_with_think": 0,
73
+ "answer_rate": 1.0,
74
+ "think_rate": 0.0,
75
+ "avg_latency_ms": 4.942417144775391,
76
+ "avg_reasoning_length": 69.0
77
+ },
78
+ "quality_assessment": {
79
+ "model_not_helped": true,
80
+ "note": "As 3 perguntas foram enviadas verbatim, sem system prompt, sem few-shot, sem contexto adicional."
81
+ }
82
+ }