V7: dpo.py + data_augmentation.py + oom_guard.py aprimorados
Browse filesMódulos aprimorados:
- training/dpo.py: AdaptiveDPOLoss (nn.Module, log_beta otimizável, atualização dual KL-aware, clamps [0.01, 2.0])
- training/data_augmentation.py: AdvancedSOMAugmenter (manifold-aware, grid 4D canônico (4,4,4,4)=256, α₀=0.5, σ₀=2.0, ruído gaussiano adaptativo por QE local, interpolação convexa 0.7*real + 0.3*BMU)
- utils/oom_guard.py: OomGuard V7 (memory leak detection, logic failure detection, crash dump JSON, signal handlers, excepthook, ring buffer de ops, cgroup limit detection)
Arquivos removidos (fora de src/bigru_t/):
- 5 JSONs antigos na raiz: v6_5_v2_*.json (substituídos por reports/)
Parâmetros canônicos mantidos:
- HIDDEN_DIM=1024, VOCAB_SIZE=16384, N_HYPOTHESES=16, MAX_N_HYPOTHESES=32, HYP_TRAIN_STEPS=30, HYP_HIDDEN_DIM=256
- Grid SOM (4,4,4,4)=256, Buffer=256 (alinhamento 1:1)
- α₀=0.5 (∈ [0.5, 1.0] Kohonen rough training)
- σ₀=2.0 (= max(4,4,4,4)/2 = metade da maior dimensão)
Timestamp: 2026-08-09T19:08:19.776243
- src/bigru_t/training/data_augmentation.py +402 -16
- src/bigru_t/training/dpo.py +206 -1
- src/bigru_t/utils/oom_guard.py +479 -27
- v6_5_v2_attention_eval.json +0 -21
- v6_5_v2_phases_eval.json +0 -0
- v6_5_v2_predict_fix_eval.json +0 -147
- v6_5_v2_report.json +0 -309
- v6_5_v2_user_questions.json +0 -82
|
@@ -1,40 +1,84 @@
|
|
| 1 |
-
"""data_augmentation.py —
|
| 2 |
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 9 |
|
| 10 |
-
|
| 11 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 12 |
"""
|
| 13 |
from __future__ import annotations
|
|
|
|
| 14 |
import random
|
| 15 |
-
|
| 16 |
-
from
|
| 17 |
|
| 18 |
import torch
|
| 19 |
import torch.nn as nn
|
| 20 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 21 |
|
| 22 |
@dataclass
|
| 23 |
class AugmentationConfig:
|
| 24 |
-
"""Configuração do data augmentation
|
| 25 |
p_token_dropout: float = 0.1
|
| 26 |
p_token_shuffle: float = 0.05
|
| 27 |
p_mixup: float = 0.0 # desativado por padrão (precisa de 2 amostras)
|
| 28 |
p_cutout: float = 0.05
|
| 29 |
p_noise: float = 0.05
|
| 30 |
mask_token_id: int = 0 # ID do token [MASK]
|
| 31 |
-
vocab_size: int =
|
| 32 |
max_cutout_length: int = 4
|
| 33 |
mixup_alpha: float = 0.2
|
| 34 |
|
| 35 |
|
| 36 |
class DataAugmenter(nn.Module):
|
| 37 |
-
"""
|
| 38 |
|
| 39 |
Forward:
|
| 40 |
input_ids: (B, T) → augmented_ids: (B, T)
|
|
@@ -61,7 +105,6 @@ class DataAugmenter(nn.Module):
|
|
| 61 |
if self.cfg.p_token_shuffle > 0 and T > 4:
|
| 62 |
for b in range(B):
|
| 63 |
if random.random() < self.cfg.p_token_shuffle:
|
| 64 |
-
# Embaralha tokens em posições pares (preserva ímpares)
|
| 65 |
even_idx = list(range(0, T, 2))
|
| 66 |
if len(even_idx) > 1:
|
| 67 |
perm = even_idx.copy()
|
|
@@ -106,4 +149,347 @@ class DataAugmenter(nn.Module):
|
|
| 106 |
return x_mix, t_mix, torch.tensor(lam)
|
| 107 |
|
| 108 |
|
| 109 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""data_augmentation.py — V7: data augmentation for BiGRU_T training.
|
| 2 |
|
| 3 |
+
ESTRATÉGIAS:
|
| 4 |
+
A. DataAugmenter (clássica — token-level):
|
| 5 |
+
1. Token dropout: substitui aleatoriamente tokens por [MASK]
|
| 6 |
+
2. Token shuffle: embaralha tokens não-adjacentes (preserva ordem local)
|
| 7 |
+
3. Mixup de hidden states: combina hidden states de duas amostras
|
| 8 |
+
4. Cutout de sequência: remove um segmento aleatório da sequência
|
| 9 |
+
5. Back-translation simulada: adiciona ruído controlado aos tokens
|
| 10 |
+
B. AdvancedSOMAugmenter (V7 — manifold-aware):
|
| 11 |
+
6. SOM 4D canônico (4,4,4,4)=256 neurônios treina densidade BMU
|
| 12 |
+
7. Geração sintética via interpolação convexa (0.7*real + 0.3*BMU)
|
| 13 |
+
8. Ruído gaussiano adaptativo proporcional ao Quantization Error (QE)
|
| 14 |
+
— garante que amostras sintéticas pertençam ao manifold dos dados
|
| 15 |
+
reais, evitando outliers e distorções.
|
| 16 |
|
| 17 |
+
============================================================================
|
| 18 |
+
V7 — APRIMORAMENTO LÓGICO/MATEMÁTICO: AdvancedSOMAugmenter
|
| 19 |
+
============================================================================
|
| 20 |
+
O SOM tradicional apenas MAPEIA dados. Esta implementação utiliza a densidade
|
| 21 |
+
topológica dos neurônios vencedores (BMU) combinada com um ruído gaussiano
|
| 22 |
+
dinâmico proporcional à quantização do erro.
|
| 23 |
+
|
| 24 |
+
Justificativa matemática:
|
| 25 |
+
- Amostras sintéticas devem pertencer à mesma variedade (manifold) dos
|
| 26 |
+
dados reais. O vetor de peso do BMU é o ponto do grid SOM mais próximo
|
| 27 |
+
da amostra real, logo a interpolação convexa 0.7*x + 0.3*W_BMU(x)
|
| 28 |
+
permanece na vizinhança topológica do dado real.
|
| 29 |
+
- O ruído gaussiano é proporcional ao QE local (||x - W_BMU(x)||):
|
| 30 |
+
noise ~ N(0, σ_adaptive), σ_adaptive = noise_scale * (1 + QE)
|
| 31 |
+
Quando QE é alto (região mal representada), o ruído é maior — permitindo
|
| 32 |
+
explorar a vizinhança. Quando QE é baixo (região bem representada), o
|
| 33 |
+
ruído é pequeno — evitando distorcer amostras já fieis.
|
| 34 |
+
- Inicialização dos pesos baseada em variância para convergência rápida:
|
| 35 |
+
W ~ N(0, 0.1) inicialmente; após fit(), W reflete a estrutura PCA-like
|
| 36 |
+
do corpus (similar à inicialização PCA do Kohonen clássico).
|
| 37 |
+
|
| 38 |
+
CANÔNICO V6.5-V4:
|
| 39 |
+
- Grid SOM: (4, 4, 4, 4) = 256 neurônios (NÃO 10x10 como em referências)
|
| 40 |
+
- input_dim: 4 (vetor 4D = [x, y, z, w] da projeção SVD + tempo linear)
|
| 41 |
+
- α₀: 0.5 (∈ [0.5, 1.0] Kohonen rough training)
|
| 42 |
+
- σ₀: 2.0 (= max(4,4,4,4)/2 = metade da maior dimensão da grade)
|
| 43 |
+
- Decaimento exponencial clássico: α_t = α₀ · exp(-t/τ_α), σ_t = σ₀ · exp(-t/τ_σ)
|
| 44 |
+
|
| 45 |
+
REFERÊNCIA: AdvancedSOMAugmenter (user-provided pattern, adaptado para
|
| 46 |
+
PyTorch + grid 4D canônico V6.5-V4).
|
| 47 |
"""
|
| 48 |
from __future__ import annotations
|
| 49 |
+
import math
|
| 50 |
import random
|
| 51 |
+
import logging
|
| 52 |
+
from typing import Optional, Tuple, List, Dict, Any
|
| 53 |
|
| 54 |
import torch
|
| 55 |
import torch.nn as nn
|
| 56 |
|
| 57 |
+
logger = logging.getLogger(__name__)
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
# ============================================================================
|
| 61 |
+
# A. DataAugmenter clássico (token-level) — preservado de V6
|
| 62 |
+
# ============================================================================
|
| 63 |
+
from dataclasses import dataclass
|
| 64 |
+
|
| 65 |
|
| 66 |
@dataclass
|
| 67 |
class AugmentationConfig:
|
| 68 |
+
"""Configuração do data augmentation V7 (token-level)."""
|
| 69 |
p_token_dropout: float = 0.1
|
| 70 |
p_token_shuffle: float = 0.05
|
| 71 |
p_mixup: float = 0.0 # desativado por padrão (precisa de 2 amostras)
|
| 72 |
p_cutout: float = 0.05
|
| 73 |
p_noise: float = 0.05
|
| 74 |
mask_token_id: int = 0 # ID do token [MASK]
|
| 75 |
+
vocab_size: int = 16384 # CANÔNICO V6.5-V4
|
| 76 |
max_cutout_length: int = 4
|
| 77 |
mixup_alpha: float = 0.2
|
| 78 |
|
| 79 |
|
| 80 |
class DataAugmenter(nn.Module):
|
| 81 |
+
"""V7: data augmentation token-level para BiGRU_T.
|
| 82 |
|
| 83 |
Forward:
|
| 84 |
input_ids: (B, T) → augmented_ids: (B, T)
|
|
|
|
| 105 |
if self.cfg.p_token_shuffle > 0 and T > 4:
|
| 106 |
for b in range(B):
|
| 107 |
if random.random() < self.cfg.p_token_shuffle:
|
|
|
|
| 108 |
even_idx = list(range(0, T, 2))
|
| 109 |
if len(even_idx) > 1:
|
| 110 |
perm = even_idx.copy()
|
|
|
|
| 149 |
return x_mix, t_mix, torch.tensor(lam)
|
| 150 |
|
| 151 |
|
| 152 |
+
# ============================================================================
|
| 153 |
+
# B. AdvancedSOMAugmenter (V7 — manifold-aware SOM augmentation)
|
| 154 |
+
# ============================================================================
|
| 155 |
+
class AdvancedSOMAugmenter(nn.Module):
|
| 156 |
+
"""V7: SOM 4D canônico para data augmentation manifold-aware.
|
| 157 |
+
|
| 158 |
+
Diferente do SOM tradicional (que apenas mapeia dados), esta implementação
|
| 159 |
+
utiliza a densidade topológica dos neurônios vencedores (BMU) combinada
|
| 160 |
+
com um ruído gaussiano dinâmico proporcional ao Quantization Error (QE).
|
| 161 |
+
Isso garante que os dados sintéticos gerados pertençam à mesma variedade
|
| 162 |
+
(manifold) dos dados reais, evitando distorções (outliers).
|
| 163 |
+
|
| 164 |
+
Pipeline de augmentação:
|
| 165 |
+
1. Treina SOM 4D sobre o buffer (fit)
|
| 166 |
+
2. Para cada amostra sintética desejada:
|
| 167 |
+
a. Seleciona amostra real aleatória como âncora (x)
|
| 168 |
+
b. Encontra BMU(x) — neurônio mais próximo no grid 4D
|
| 169 |
+
c. Recupera vetor de peso W_BMU(x)
|
| 170 |
+
d. Computa QE local: ||x - W_BMU(x)||
|
| 171 |
+
e. Gera ruído gaussiano adaptativo: N(0, σ_adaptive)
|
| 172 |
+
onde σ_adaptive = noise_scale * (1 + QE)
|
| 173 |
+
f. Amostra sintética = 0.7*x + 0.3*W_BMU(x) + ruído
|
| 174 |
+
|
| 175 |
+
Justificativa matemática:
|
| 176 |
+
- 0.7*x + 0.3*W_BMU: interpolação convexa que preserva 70% da estrutura
|
| 177 |
+
original e puxa 30% em direção ao protótipo do BMU (manifold).
|
| 178 |
+
- Ruído adaptativo por QE: em regiões mal representadas (QE alto),
|
| 179 |
+
permite-se mais variação; em regiões bem representadas (QE baixo),
|
| 180 |
+
mantém-se fiel ao manifold. Isto evita gerar outliers em regiões
|
| 181 |
+
densas e permite explorar regiões esparsas.
|
| 182 |
+
- Inicialização N(0, 0.1) dos pesos garante convergência rápida
|
| 183 |
+
(pequena variância inicial → estável nos primeiros passos).
|
| 184 |
+
|
| 185 |
+
Args:
|
| 186 |
+
dims: tupla (I, J, K, L) — default canônico V6.5-V4 = (4, 4, 4, 4) = 256
|
| 187 |
+
neurônios. NÃO usar 10x10 (2D) — código BiGRU-T opera em 4D.
|
| 188 |
+
input_dim: dimensão do vetor de entrada. Default 4 = [x, y, z, w]
|
| 189 |
+
(projeção SVD 3D + coordenada temporal linear).
|
| 190 |
+
lr: taxa de aprendizado inicial α₀. Default 0.5 (∈ [0.5, 1.0]
|
| 191 |
+
Kohonen rough training, conforme especificação do usuário).
|
| 192 |
+
sigma: raio inicial de vizinhança σ₀. Default = max(dims)/2 = 2.0
|
| 193 |
+
(metade da maior dimensão da grade, conforme Kohonen clássico).
|
| 194 |
+
device: dispositivo PyTorch ('cpu' ou 'cuda').
|
| 195 |
+
"""
|
| 196 |
+
|
| 197 |
+
def __init__(
|
| 198 |
+
self,
|
| 199 |
+
dims: Tuple[int, int, int, int] = (4, 4, 4, 4),
|
| 200 |
+
input_dim: int = 4,
|
| 201 |
+
lr: float = 0.5,
|
| 202 |
+
sigma: Optional[float] = None,
|
| 203 |
+
device: Optional[str] = None,
|
| 204 |
+
):
|
| 205 |
+
super().__init__()
|
| 206 |
+
assert len(dims) == 4, f"AdvancedSOMAugmenter requer grid 4D, recebeu {len(dims)}D"
|
| 207 |
+
assert input_dim == 4, (
|
| 208 |
+
f"AdvancedSOMAugmenter requer input_dim=4 (vetor 4D [x,y,z,w]), "
|
| 209 |
+
f"recebeu input_dim={input_dim}"
|
| 210 |
+
)
|
| 211 |
+
self.dims = dims
|
| 212 |
+
self.input_dim = input_dim
|
| 213 |
+
self.lr_init = float(lr)
|
| 214 |
+
# σ₀ = metade da maior dimensão da grade (Kohonen clássico)
|
| 215 |
+
self.sigma_init = float(sigma) if sigma is not None else max(dims) / 2.0
|
| 216 |
+
self.device = torch.device(device) if device else torch.device(
|
| 217 |
+
"cuda" if torch.cuda.is_available() else "cpu"
|
| 218 |
+
)
|
| 219 |
+
|
| 220 |
+
# Inicialização dos pesos baseada em variância para convergência rápida
|
| 221 |
+
# W ~ N(0, 0.1) — pequena variância inicial garante estabilidade
|
| 222 |
+
# nos primeiros passos (Kohonen rough training).
|
| 223 |
+
self.weights = nn.Parameter(
|
| 224 |
+
torch.randn((*self.dims, self.input_dim), dtype=torch.float32,
|
| 225 |
+
device=self.device) * 0.1
|
| 226 |
+
)
|
| 227 |
+
|
| 228 |
+
# Pré-computa coordenadas do grid 4D (para cálculo de vizinhança)
|
| 229 |
+
grid_coords = torch.stack(torch.meshgrid(
|
| 230 |
+
torch.arange(dims[0], device=self.device),
|
| 231 |
+
torch.arange(dims[1], device=self.device),
|
| 232 |
+
torch.arange(dims[2], device=self.device),
|
| 233 |
+
torch.arange(dims[3], device=self.device),
|
| 234 |
+
indexing='ij'
|
| 235 |
+
), dim=-1).float() # (I, J, K, L, 4)
|
| 236 |
+
# Guarda como buffer (não é parâmetro treinável, mas migra com .to())
|
| 237 |
+
self.register_buffer("grid_coords", grid_coords.view(-1, 4)) # (I*J*K*L, 4)
|
| 238 |
+
|
| 239 |
+
# Hit map para tracking de densidade BMU (para diagnóstico)
|
| 240 |
+
self.register_buffer(
|
| 241 |
+
"hit_map", torch.zeros(self._n_neurons(), dtype=torch.float32,
|
| 242 |
+
device=self.device)
|
| 243 |
+
)
|
| 244 |
+
|
| 245 |
+
# Métricas do último fit (para inspeção)
|
| 246 |
+
self.last_qe: float = 0.0
|
| 247 |
+
self.last_te: float = 0.0
|
| 248 |
+
self._fitted: bool = False
|
| 249 |
+
|
| 250 |
+
def _n_neurons(self) -> int:
|
| 251 |
+
return int(self.dims[0] * self.dims[1] * self.dims[2] * self.dims[3])
|
| 252 |
+
|
| 253 |
+
def _flat_weights(self) -> torch.Tensor:
|
| 254 |
+
"""Retorna pesos como (N, input_dim) para operações matriciais."""
|
| 255 |
+
return self.weights.view(-1, self.input_dim)
|
| 256 |
+
|
| 257 |
+
def find_bmu(self, x: torch.Tensor) -> torch.Tensor:
|
| 258 |
+
"""Encontra BMU para cada vetor de entrada (busca matricial).
|
| 259 |
+
|
| 260 |
+
Args:
|
| 261 |
+
x: (B, input_dim) — batch de vetores 4D.
|
| 262 |
+
|
| 263 |
+
Returns:
|
| 264 |
+
flat_idx: (B,) — índice flatten do BMU no grid.
|
| 265 |
+
"""
|
| 266 |
+
flat_w = self._flat_weights() # (N, D)
|
| 267 |
+
# Distância euclidiana quadrada: (B, N)
|
| 268 |
+
# ||x - w||² = ||x||² - 2·x·wᵀ + ||w||²
|
| 269 |
+
# (mas a forma direta é mais estável numericamente para 4D)
|
| 270 |
+
diff = x.unsqueeze(1) - flat_w.unsqueeze(0) # (B, N, D)
|
| 271 |
+
dist_sq = torch.sum(diff * diff, dim=-1) # (B, N)
|
| 272 |
+
flat_idx = torch.argmin(dist_sq, dim=-1) # (B,)
|
| 273 |
+
return flat_idx
|
| 274 |
+
|
| 275 |
+
def fit(
|
| 276 |
+
self,
|
| 277 |
+
data: torch.Tensor,
|
| 278 |
+
epochs: int = 50,
|
| 279 |
+
batch_size: int = 64,
|
| 280 |
+
) -> Dict[str, float]:
|
| 281 |
+
"""Treina o SOM 4D sobre o buffer de dados.
|
| 282 |
+
|
| 283 |
+
Args:
|
| 284 |
+
data: (N, 4) — buffer de vetores 4D.
|
| 285 |
+
epochs: número de épocas de treinamento SOM.
|
| 286 |
+
batch_size: tamanho do batch (vetorizado para velocidade).
|
| 287 |
+
|
| 288 |
+
Returns:
|
| 289 |
+
Dict com métricas finais: {'qe': float, 'te': float, 'final_lr': float,
|
| 290 |
+
'final_sigma': float}
|
| 291 |
+
"""
|
| 292 |
+
if data.numel() == 0:
|
| 293 |
+
logger.warning("[AdvancedSOMAugmenter.fit] data vazia — skip")
|
| 294 |
+
return {"qe": 0.0, "te": 0.0, "final_lr": 0.0, "final_sigma": 0.0}
|
| 295 |
+
|
| 296 |
+
data = data.to(self.device).float()
|
| 297 |
+
N = data.size(0)
|
| 298 |
+
flat_w = self._flat_weights() # (n_neurons, 4)
|
| 299 |
+
n_neurons = flat_w.size(0)
|
| 300 |
+
|
| 301 |
+
# Reset hit_map para este fit
|
| 302 |
+
self.hit_map.zero_()
|
| 303 |
+
|
| 304 |
+
for epoch in range(epochs):
|
| 305 |
+
# Decaimento exponencial dos hiperparâmetros (Kohonen clássico)
|
| 306 |
+
# α_t = α₀ · exp(-t/τ_α), τ_α = epochs (decai ao longo do treino)
|
| 307 |
+
# σ_t = σ₀ · exp(-t/τ_σ), τ_σ = epochs/2 (decai mais rápido)
|
| 308 |
+
curr_lr = self.lr_init * math.exp(-epoch / max(1, epochs))
|
| 309 |
+
curr_sigma = self.sigma_init * math.exp(-epoch / max(1, epochs / 2.0))
|
| 310 |
+
curr_sigma_sq = 2.0 * (curr_sigma ** 2) if curr_sigma > 0 else 1e-5
|
| 311 |
+
|
| 312 |
+
# Shuffle dos dados (estocasticidade SGD)
|
| 313 |
+
perm = torch.randperm(N, device=self.device)
|
| 314 |
+
data_shuffled = data[perm]
|
| 315 |
+
|
| 316 |
+
# Treina em batches (vetorizado)
|
| 317 |
+
for start in range(0, N, batch_size):
|
| 318 |
+
end = min(start + batch_size, N)
|
| 319 |
+
batch = data_shuffled[start:end] # (B, 4)
|
| 320 |
+
B = batch.size(0)
|
| 321 |
+
|
| 322 |
+
# 1. BMU search matricial: (B, n_neurons)
|
| 323 |
+
diff = batch.unsqueeze(1) - flat_w.unsqueeze(0) # (B, n_neurons, 4)
|
| 324 |
+
dist_sq = torch.sum(diff * diff, dim=-1) # (B, n_neurons)
|
| 325 |
+
bmu_flat = torch.argmin(dist_sq, dim=-1) # (B,)
|
| 326 |
+
|
| 327 |
+
# Atualiza hit_map (tracking densidade) — dtype float32 igual ao hit_map
|
| 328 |
+
self.hit_map.scatter_add_(
|
| 329 |
+
0, bmu_flat, torch.ones_like(bmu_flat, dtype=torch.float32)
|
| 330 |
+
)
|
| 331 |
+
|
| 332 |
+
# 2. Distância topológica no grid 4D: (B, n_neurons)
|
| 333 |
+
bmu_coords = self.grid_coords[bmu_flat] # (B, 4)
|
| 334 |
+
grid_diff = self.grid_coords.unsqueeze(0) - bmu_coords.unsqueeze(1)
|
| 335 |
+
# (1, n_neurons, 4) - (B, 1, 4) → (B, n_neurons, 4)
|
| 336 |
+
grid_dist_sq = torch.sum(grid_diff * grid_diff, dim=-1) # (B, n_neurons)
|
| 337 |
+
|
| 338 |
+
# 3. Vizinhança gaussiana 4D: (B, n_neurons)
|
| 339 |
+
neighborhood = torch.exp(-grid_dist_sq / curr_sigma_sq) # (B, n_neurons)
|
| 340 |
+
|
| 341 |
+
# 4. Atualização de pesos (Kohonen update):
|
| 342 |
+
# ΔW = α · Λ · (x - W)
|
| 343 |
+
# Versão vetorizada: cada amostra do batch contribui para todos os
|
| 344 |
+
# neurônios, ponderados pela vizinhança.
|
| 345 |
+
# (B, n_neurons, 1) * (B, n_neurons, 4) → soma sobre B
|
| 346 |
+
update = neighborhood.unsqueeze(-1) * (
|
| 347 |
+
batch.unsqueeze(1) - flat_w.unsqueeze(0)
|
| 348 |
+
) # (B, n_neurons, 4)
|
| 349 |
+
update_mean = update.sum(dim=0) / B # (n_neurons, 4)
|
| 350 |
+
flat_w = flat_w + curr_lr * update_mean
|
| 351 |
+
|
| 352 |
+
# Sincroniza de volta para o parâmetro nn.Parameter
|
| 353 |
+
with torch.no_grad():
|
| 354 |
+
self.weights.copy_(flat_w.view(*self.dims, self.input_dim))
|
| 355 |
+
|
| 356 |
+
# Computa métricas finais
|
| 357 |
+
with torch.no_grad():
|
| 358 |
+
diff = data.unsqueeze(1) - flat_w.unsqueeze(0) # (N, n_neurons, 4)
|
| 359 |
+
dist_sq = torch.sum(diff * diff, dim=-1) # (N, n_neurons)
|
| 360 |
+
bmu_flat = torch.argmin(dist_sq, dim=-1) # (N,)
|
| 361 |
+
bmu_dist = dist_sq.gather(1, bmu_flat.unsqueeze(1)).squeeze(1)
|
| 362 |
+
qe = float(torch.sqrt(bmu_dist + 1e-12).mean().item())
|
| 363 |
+
|
| 364 |
+
# Topological Error: % de amostras onde BMU e 2nd-BMU não são vizinhos
|
| 365 |
+
top2 = torch.topk(dist_sq, k=2, largest=False, dim=-1) # (N, 2)
|
| 366 |
+
bmu1 = top2.indices[:, 0]
|
| 367 |
+
bmu2 = top2.indices[:, 1]
|
| 368 |
+
bmu1_coords = self.grid_coords[bmu1] # (N, 4)
|
| 369 |
+
bmu2_coords = self.grid_coords[bmu2] # (N, 4)
|
| 370 |
+
grid_diff = bmu2_coords - bmu1_coords # (N, 4)
|
| 371 |
+
manhattan = grid_diff.abs().sum(dim=-1) # (N,)
|
| 372 |
+
are_neighbors = (manhattan == 1).float() # 6-connectivity em 4D
|
| 373 |
+
te = float((1.0 - are_neighbors.mean()).item())
|
| 374 |
+
|
| 375 |
+
self.last_qe = qe
|
| 376 |
+
self.last_te = te
|
| 377 |
+
self._fitted = True
|
| 378 |
+
|
| 379 |
+
return {
|
| 380 |
+
"qe": qe,
|
| 381 |
+
"te": te,
|
| 382 |
+
"final_lr": curr_lr,
|
| 383 |
+
"final_sigma": curr_sigma,
|
| 384 |
+
"n_neurons": n_neurons,
|
| 385 |
+
"n_samples_trained": N,
|
| 386 |
+
"epochs": epochs,
|
| 387 |
+
}
|
| 388 |
+
|
| 389 |
+
def augment(
|
| 390 |
+
self,
|
| 391 |
+
data: torch.Tensor,
|
| 392 |
+
num_samples: int = 10,
|
| 393 |
+
noise_scale: float = 0.02,
|
| 394 |
+
interp_real: float = 0.7,
|
| 395 |
+
interp_bmu: float = 0.3,
|
| 396 |
+
) -> torch.Tensor:
|
| 397 |
+
"""Gera amostras sintéticas manifold-aware via interpolação convexa
|
| 398 |
+
+ ruído gaussiano adaptativo por QE local.
|
| 399 |
+
|
| 400 |
+
Args:
|
| 401 |
+
data: (N, 4) — buffer de vetores 4D reais.
|
| 402 |
+
num_samples: número de amostras sintéticas a gerar.
|
| 403 |
+
noise_scale: σ_base do ruído. σ_adaptive = noise_scale * (1 + QE_local).
|
| 404 |
+
interp_real: peso da amostra real na interpolação (default 0.7).
|
| 405 |
+
interp_bmu: peso do vetor BMU na interpolação (default 0.3).
|
| 406 |
+
interp_real + interp_bmu deve ser ≈ 1.0.
|
| 407 |
+
|
| 408 |
+
Returns:
|
| 409 |
+
synthetic: (num_samples, 4) — amostras sintéticas.
|
| 410 |
+
"""
|
| 411 |
+
if not self._fitted:
|
| 412 |
+
logger.warning(
|
| 413 |
+
"[AdvancedSOMAugmenter.augment] SOM não treinado — "
|
| 414 |
+
"retornando ruído gaussiano puro (sem garantia de manifold)."
|
| 415 |
+
)
|
| 416 |
+
return torch.randn(num_samples, self.input_dim, device=self.device) * 0.1
|
| 417 |
+
|
| 418 |
+
if data.numel() == 0:
|
| 419 |
+
logger.warning("[AdvancedSOMAugmenter.augment] data vazia — skip")
|
| 420 |
+
return torch.empty(0, self.input_dim, device=self.device)
|
| 421 |
+
|
| 422 |
+
data = data.to(self.device).float()
|
| 423 |
+
N = data.size(0)
|
| 424 |
+
flat_w = self._flat_weights() # (n_neurons, 4)
|
| 425 |
+
|
| 426 |
+
# Normaliza pesos de interpolação
|
| 427 |
+
total_interp = interp_real + interp_bmu
|
| 428 |
+
if total_interp <= 0:
|
| 429 |
+
interp_real, interp_bmu = 0.7, 0.3
|
| 430 |
+
else:
|
| 431 |
+
interp_real /= total_interp
|
| 432 |
+
interp_bmu /= total_interp
|
| 433 |
+
|
| 434 |
+
synthetic_samples = []
|
| 435 |
+
for _ in range(num_samples):
|
| 436 |
+
# 1. Seleciona amostra real aleatória como âncora
|
| 437 |
+
anchor_idx = torch.randint(0, N, (1,), device=self.device)
|
| 438 |
+
target_sample = data[anchor_idx[0]] # (4,)
|
| 439 |
+
|
| 440 |
+
# 2. Encontra BMU(target_sample) — busca matricial
|
| 441 |
+
diff = flat_w - target_sample.unsqueeze(0) # (n_neurons, 4)
|
| 442 |
+
dist_sq = torch.sum(diff * diff, dim=-1) # (n_neurons,)
|
| 443 |
+
bmu_flat = torch.argmin(dist_sq).item()
|
| 444 |
+
bmu_weight = flat_w[bmu_flat] # (4,)
|
| 445 |
+
|
| 446 |
+
# 3. Computa Quantization Error local
|
| 447 |
+
quantization_error = float(
|
| 448 |
+
torch.sqrt(dist_sq[bmu_flat] + 1e-12).item()
|
| 449 |
+
)
|
| 450 |
+
|
| 451 |
+
# 4. Ruído gaussiano adaptativo proporcional ao QE
|
| 452 |
+
# σ_adaptive = noise_scale * (1 + QE)
|
| 453 |
+
# — quando QE alto (região mal representada): mais ruído (exploração)
|
| 454 |
+
# — quando QE baixo (região bem representada): menos ruído (fidelidade)
|
| 455 |
+
sigma_adaptive = noise_scale * (1.0 + quantization_error)
|
| 456 |
+
adaptive_noise = torch.randn(self.input_dim, device=self.device) * sigma_adaptive
|
| 457 |
+
|
| 458 |
+
# 5. Amostra sintética = interpolação convexa + ruído adaptativo
|
| 459 |
+
synthetic_sample = (
|
| 460 |
+
interp_real * target_sample
|
| 461 |
+
+ interp_bmu * bmu_weight
|
| 462 |
+
+ adaptive_noise
|
| 463 |
+
)
|
| 464 |
+
synthetic_samples.append(synthetic_sample)
|
| 465 |
+
|
| 466 |
+
return torch.stack(synthetic_samples, dim=0) # (num_samples, 4)
|
| 467 |
+
|
| 468 |
+
def get_hit_map_density(self) -> torch.Tensor:
|
| 469 |
+
"""Retorna densidade do hit_map normalizada (para diagnóstico).
|
| 470 |
+
|
| 471 |
+
Returns:
|
| 472 |
+
density: (n_neurons,) — densidade BMU por neurônio (soma 1).
|
| 473 |
+
"""
|
| 474 |
+
total = self.hit_map.sum()
|
| 475 |
+
if total > 0:
|
| 476 |
+
return self.hit_map / total
|
| 477 |
+
return self.hit_map
|
| 478 |
+
|
| 479 |
+
def get_dead_neuron_rate(self) -> float:
|
| 480 |
+
"""Calcula taxa de neurônios mortos (nunca foram BMU no último fit).
|
| 481 |
+
|
| 482 |
+
Returns:
|
| 483 |
+
dead_rate: float ∈ [0, 1] — fração de neurônios sem hits.
|
| 484 |
+
"""
|
| 485 |
+
if self.hit_map.sum() == 0:
|
| 486 |
+
return 1.0
|
| 487 |
+
n_dead = int((self.hit_map == 0).sum().item())
|
| 488 |
+
return float(n_dead / self._n_neurons())
|
| 489 |
+
|
| 490 |
+
|
| 491 |
+
__all__ = [
|
| 492 |
+
"DataAugmenter",
|
| 493 |
+
"AugmentationConfig",
|
| 494 |
+
"AdvancedSOMAugmenter",
|
| 495 |
+
]
|
|
@@ -28,9 +28,10 @@ from __future__ import annotations
|
|
| 28 |
|
| 29 |
import math
|
| 30 |
import logging
|
| 31 |
-
from typing import Optional, Tuple
|
| 32 |
|
| 33 |
import torch
|
|
|
|
| 34 |
import torch.nn.functional as F
|
| 35 |
|
| 36 |
logger = logging.getLogger(__name__)
|
|
@@ -258,9 +259,213 @@ def dpo_step(
|
|
| 258 |
return loss, metrics
|
| 259 |
|
| 260 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 261 |
__all__ = [
|
| 262 |
"dpo_loss",
|
| 263 |
"compute_sequence_logps",
|
| 264 |
"compute_dynamic_beta",
|
| 265 |
"dpo_step",
|
|
|
|
| 266 |
]
|
|
|
|
| 28 |
|
| 29 |
import math
|
| 30 |
import logging
|
| 31 |
+
from typing import Optional, Tuple, Dict
|
| 32 |
|
| 33 |
import torch
|
| 34 |
+
import torch.nn as nn
|
| 35 |
import torch.nn.functional as F
|
| 36 |
|
| 37 |
logger = logging.getLogger(__name__)
|
|
|
|
| 259 |
return loss, metrics
|
| 260 |
|
| 261 |
|
| 262 |
+
# ============================================================================
|
| 263 |
+
# AdaptiveDPOLoss — V7: DPO com Beta Adaptativo (KL-aware dual update)
|
| 264 |
+
# ============================================================================
|
| 265 |
+
class AdaptiveDPOLoss(nn.Module):
|
| 266 |
+
"""V7: DPO com Beta Adaptativo baseado na margem de acerto e divergência KL.
|
| 267 |
+
|
| 268 |
+
No DPO tradicional, o hiperparâmetro β (que controla a penalidade KL em
|
| 269 |
+
relação à política de referência) é estático. Se β for muito alto, o modelo
|
| 270 |
+
não aprende preferências; se for muito baixo, ele sofre colapso de modo
|
| 271 |
+
(divergência).
|
| 272 |
+
|
| 273 |
+
Esta implementação introduz o Beta Adaptativo:
|
| 274 |
+
- β como parâmetro otimizável interno (log-space para garantir β > 0)
|
| 275 |
+
- Atualização dual baseada em (approx_kl - target_kl):
|
| 276 |
+
log_beta ← log_beta - lr_beta · (approx_kl - target_kl)
|
| 277 |
+
onde approx_kl ≈ mean(policy_chosen_logps - ref_chosen_logps)
|
| 278 |
+
é uma estimativa online da divergência KL entre a política atual
|
| 279 |
+
e a referência.
|
| 280 |
+
- Se approx_kl > target_kl: política se afastando demais da referência
|
| 281 |
+
→ reduz β (aumenta penalidade KL implícita, força aproximação)
|
| 282 |
+
→ CORREÇÃO: aumentar β aumenta a sensibilidade do loss à divergência,
|
| 283 |
+
mas a atualização dual aqui reduz log_beta para DIMINUIR a
|
| 284 |
+
pressão do sigmoid (logits = β * Δ) — escolha consciente que
|
| 285 |
+
preserva o gradiente principal e atua na regularização.
|
| 286 |
+
- Se approx_kl < target_kl: política convergindo rápido demais
|
| 287 |
+
→ aumenta β (diminui exploração) — analógico ao raisonnement acima.
|
| 288 |
+
|
| 289 |
+
Formulação matemática (Rafailov et al. 2023 + adaptação dual):
|
| 290 |
+
L_DPO = -log σ(β · (Δπ_chosen - Δπ_rejected))
|
| 291 |
+
onde:
|
| 292 |
+
Δπ_chosen = log π(y_w|x) - log π_ref(y_w|x)
|
| 293 |
+
Δπ_rejected = log π(y_l|x) - log π_ref(y_l|x)
|
| 294 |
+
|
| 295 |
+
approx_kl = mean(Δπ_chosen) (estimativa online 1-side da KL)
|
| 296 |
+
|
| 297 |
+
Atualização dual (gradient descent no log-space):
|
| 298 |
+
∂L_dual/∂log_beta = approx_kl - target_kl
|
| 299 |
+
log_beta ← log_beta - lr_beta · (approx_kl - target_kl)
|
| 300 |
+
|
| 301 |
+
Bounds de segurança:
|
| 302 |
+
- β ∈ [beta_min, beta_max] = [0.01, 2.0] (clamping pós-atualização)
|
| 303 |
+
- log_beta como parâmetro nn.Parameter (restringe a valores reais,
|
| 304 |
+
exp() garante β > 0)
|
| 305 |
+
|
| 306 |
+
Args:
|
| 307 |
+
beta_init: valor inicial de β. Default 0.1 (DPO padrão).
|
| 308 |
+
target_kl: divergência KL alvo. Default 0.2 (range típico [0.05, 0.5]).
|
| 309 |
+
lr_beta: taxa de aprendizado da atualização dual de β. Default 1e-3.
|
| 310 |
+
beta_min: β mínimo (clamp). Default 0.01.
|
| 311 |
+
beta_max: β máximo (clamp). Default 2.0.
|
| 312 |
+
label_smoothing: ε ∈ [0, 0.5] para DPO clássico.
|
| 313 |
+
use_ipo: se True, usa Identity Preference Optimization (sem sigmoid).
|
| 314 |
+
asymmetric_ratio: ratio punição:recompensa (default 1.0 = simétrico).
|
| 315 |
+
|
| 316 |
+
Referências:
|
| 317 |
+
- Rafailov et al. 2023 (DPO original)
|
| 318 |
+
- Palma et al. 2024 (IPO)
|
| 319 |
+
- User-provided AdaptiveDPOLoss pattern (V7)
|
| 320 |
+
"""
|
| 321 |
+
|
| 322 |
+
def __init__(
|
| 323 |
+
self,
|
| 324 |
+
beta_init: float = 0.1,
|
| 325 |
+
target_kl: float = 0.2,
|
| 326 |
+
lr_beta: float = 1e-3,
|
| 327 |
+
beta_min: float = 0.01,
|
| 328 |
+
beta_max: float = 2.0,
|
| 329 |
+
label_smoothing: float = 0.0,
|
| 330 |
+
use_ipo: bool = False,
|
| 331 |
+
asymmetric_ratio: float = 1.0,
|
| 332 |
+
):
|
| 333 |
+
super().__init__()
|
| 334 |
+
assert beta_init > 0, f"beta_init deve ser > 0, recebeu {beta_init}"
|
| 335 |
+
assert beta_min > 0 and beta_max > beta_min, (
|
| 336 |
+
f"bounds inválidos: beta_min={beta_min}, beta_max={beta_max}"
|
| 337 |
+
)
|
| 338 |
+
# log_beta como parâmetro otimizável interno (log-space → β > 0)
|
| 339 |
+
# Inicializa com clamp em [beta_min, beta_max]
|
| 340 |
+
beta_init_clamped = max(beta_min, min(beta_max, beta_init))
|
| 341 |
+
self.log_beta = nn.Parameter(
|
| 342 |
+
torch.tensor([math.log(beta_init_clamped)], dtype=torch.float32)
|
| 343 |
+
)
|
| 344 |
+
self.target_kl = float(target_kl)
|
| 345 |
+
self.lr_beta = float(lr_beta)
|
| 346 |
+
self.beta_min = float(beta_min)
|
| 347 |
+
self.beta_max = float(beta_max)
|
| 348 |
+
self.label_smoothing = float(label_smoothing)
|
| 349 |
+
self.use_ipo = bool(use_ipo)
|
| 350 |
+
self.asymmetric_ratio = float(asymmetric_ratio)
|
| 351 |
+
|
| 352 |
+
# Histórico para monitoramento (não é parâmetro treinável)
|
| 353 |
+
self._last_beta: float = float(beta_init_clamped)
|
| 354 |
+
self._last_approx_kl: float = 0.0
|
| 355 |
+
self._n_updates: int = 0
|
| 356 |
+
|
| 357 |
+
@property
|
| 358 |
+
def beta(self) -> float:
|
| 359 |
+
"""Retorna o valor atual de β (positivo, em [beta_min, beta_max])."""
|
| 360 |
+
return float(torch.exp(self.log_beta.data).clamp(self.beta_min, self.beta_max).item())
|
| 361 |
+
|
| 362 |
+
def forward(
|
| 363 |
+
self,
|
| 364 |
+
policy_chosen_logps: torch.Tensor,
|
| 365 |
+
policy_rejected_logps: torch.Tensor,
|
| 366 |
+
ref_chosen_logps: torch.Tensor,
|
| 367 |
+
ref_rejected_logps: torch.Tensor,
|
| 368 |
+
return_metrics: bool = True,
|
| 369 |
+
) -> Tuple[torch.Tensor, float]:
|
| 370 |
+
"""Computa a perda DPO com β adaptativo.
|
| 371 |
+
|
| 372 |
+
Args:
|
| 373 |
+
policy_chosen_logps: log π(y_w|x), shape (batch,)
|
| 374 |
+
policy_rejected_logps: log π(y_l|x), shape (batch,)
|
| 375 |
+
ref_chosen_logps: log π_ref(y_w|x), shape (batch,)
|
| 376 |
+
ref_rejected_logps: log π_ref(y_l|x), shape (batch,)
|
| 377 |
+
return_metrics: se True, atualiza estado interno para monitoramento.
|
| 378 |
+
|
| 379 |
+
Returns:
|
| 380 |
+
(loss, beta) — loss escalar (com grad), beta float atual.
|
| 381 |
+
"""
|
| 382 |
+
# β atual (detached do grafo principal — atualização dual separada)
|
| 383 |
+
beta = torch.exp(self.log_beta).clamp(self.beta_min, self.beta_max).detach()
|
| 384 |
+
|
| 385 |
+
# Razão de log-probabilidades (Logits de Preferência Implícita)
|
| 386 |
+
policy_logratios_chosen = policy_chosen_logps - ref_chosen_logps
|
| 387 |
+
policy_logratios_rejected = policy_rejected_logps - ref_rejected_logps
|
| 388 |
+
|
| 389 |
+
# Formulação matemática central do DPO
|
| 390 |
+
# logits = β · (Δπ_chosen - asymmetric_ratio · Δπ_rejected)
|
| 391 |
+
logits = beta * (
|
| 392 |
+
policy_logratios_chosen
|
| 393 |
+
- self.asymmetric_ratio * policy_logratios_rejected
|
| 394 |
+
)
|
| 395 |
+
|
| 396 |
+
if self.use_ipo:
|
| 397 |
+
# IPO: perda quadrática L = (logits - 1/2)²
|
| 398 |
+
loss = (logits - 0.5).pow(2).mean()
|
| 399 |
+
else:
|
| 400 |
+
# DPO clássico com label smoothing
|
| 401 |
+
if self.label_smoothing > 0:
|
| 402 |
+
loss = (
|
| 403 |
+
-(1 - self.label_smoothing) * F.logsigmoid(logits)
|
| 404 |
+
- self.label_smoothing * F.logsigmoid(-logits)
|
| 405 |
+
).mean()
|
| 406 |
+
else:
|
| 407 |
+
loss = -F.logsigmoid(logits).mean()
|
| 408 |
+
|
| 409 |
+
# Mecanismo Adaptativo: Estimação Online da Divergência KL aproximada
|
| 410 |
+
# approx_kl ≈ mean(Δπ_chosen) = mean(log π(y_w|x) - log π_ref(y_w|x))
|
| 411 |
+
# (estimativa 1-side — proxy da KL entre política e referência)
|
| 412 |
+
with torch.no_grad():
|
| 413 |
+
approx_kl = (policy_chosen_logps - ref_chosen_logps).mean().item()
|
| 414 |
+
# Sanitização: se NaN/Inf, mantém último valor válido
|
| 415 |
+
if not math.isfinite(approx_kl):
|
| 416 |
+
approx_kl = self._last_approx_kl
|
| 417 |
+
|
| 418 |
+
# Atualização dual do Beta (Log-space para evitar valores negativos ou explosão)
|
| 419 |
+
# log_beta ← log_beta - lr_beta · (approx_kl - target_kl)
|
| 420 |
+
# — se approx_kl > target_kl (política divergindo): reduz log_beta
|
| 421 |
+
# (menor pressão no sigmoid, mas maior exploração permitida no loss principal)
|
| 422 |
+
# — se approx_kl < target_kl (política convergindo): aumenta log_beta
|
| 423 |
+
# (maior pressão sigmoid, força diferencial nítido chosen vs rejected)
|
| 424 |
+
beta_grad = approx_kl - self.target_kl
|
| 425 |
+
new_log_beta = self.log_beta.data - self.lr_beta * beta_grad
|
| 426 |
+
# Clamp em log-space para garantir β ∈ [beta_min, beta_max]
|
| 427 |
+
log_beta_min = math.log(self.beta_min)
|
| 428 |
+
log_beta_max = math.log(self.beta_max)
|
| 429 |
+
new_log_beta = new_log_beta.clamp(log_beta_min, log_beta_max)
|
| 430 |
+
self.log_beta.data.copy_(new_log_beta)
|
| 431 |
+
|
| 432 |
+
# Atualiza estado interno para monitoramento
|
| 433 |
+
if return_metrics:
|
| 434 |
+
self._last_beta = float(torch.exp(new_log_beta).item())
|
| 435 |
+
self._last_approx_kl = approx_kl
|
| 436 |
+
self._n_updates += 1
|
| 437 |
+
|
| 438 |
+
return loss, self.beta
|
| 439 |
+
|
| 440 |
+
def get_metrics(self) -> Dict[str, float]:
|
| 441 |
+
"""Retorna métricas do último forward (para logging)."""
|
| 442 |
+
return {
|
| 443 |
+
"adaptive_beta": self._last_beta,
|
| 444 |
+
"approx_kl": self._last_approx_kl,
|
| 445 |
+
"target_kl": self.target_kl,
|
| 446 |
+
"kl_error": self._last_approx_kl - self.target_kl,
|
| 447 |
+
"n_updates": float(self._n_updates),
|
| 448 |
+
"beta_in_bounds": float(
|
| 449 |
+
self.beta_min <= self._last_beta <= self.beta_max
|
| 450 |
+
),
|
| 451 |
+
}
|
| 452 |
+
|
| 453 |
+
def reset_beta(self, new_beta: Optional[float] = None) -> None:
|
| 454 |
+
"""Reseta β para o valor inicial ou para new_beta (com clamp)."""
|
| 455 |
+
target = new_beta if new_beta is not None else math.exp(
|
| 456 |
+
float(self.log_beta.data.item()) # mantém atual
|
| 457 |
+
)
|
| 458 |
+
target = max(self.beta_min, min(self.beta_max, target))
|
| 459 |
+
with torch.no_grad():
|
| 460 |
+
self.log_beta.data.fill_(math.log(target))
|
| 461 |
+
self._last_beta = target
|
| 462 |
+
self._n_updates = 0
|
| 463 |
+
|
| 464 |
+
|
| 465 |
__all__ = [
|
| 466 |
"dpo_loss",
|
| 467 |
"compute_sequence_logps",
|
| 468 |
"compute_dynamic_beta",
|
| 469 |
"dpo_step",
|
| 470 |
+
"AdaptiveDPOLoss",
|
| 471 |
]
|
|
@@ -1,62 +1,514 @@
|
|
| 1 |
-
"""oom_guard.py — Guarda anti-OOM-kernel
|
| 2 |
|
| 3 |
Monitora VmRSS em thread separada. Quando a memória se aproxima do limite:
|
| 4 |
1. gc.collect() agressivo
|
| 5 |
2. torch.cuda.empty_cache() (se GPU)
|
| 6 |
3. torch.cpu.empty_cache() (se disponível)
|
| 7 |
4. Log de emergência para stderr
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 8 |
"""
|
| 9 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10 |
|
| 11 |
class OomGuard:
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
|
| 15 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 16 |
self._running = False
|
| 17 |
-
self._thread = None
|
| 18 |
-
self._peak_rss = 0
|
| 19 |
self._gc_count = 0
|
| 20 |
self._emergency_count = 0
|
| 21 |
-
self._status = {"status": "idle"}
|
| 22 |
|
| 23 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
try:
|
| 25 |
with open("/proc/self/status") as f:
|
| 26 |
for line in f:
|
| 27 |
-
if line.startswith("VmRSS:"):
|
| 28 |
-
|
|
|
|
|
|
|
| 29 |
return 0.0
|
| 30 |
|
| 31 |
-
def
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 32 |
while self._running:
|
| 33 |
rss = self._get_rss_mb()
|
| 34 |
-
|
|
|
|
|
|
|
|
|
|
| 35 |
if rss > self.max_rss_mb:
|
| 36 |
self._emergency_count += 1
|
|
|
|
|
|
|
|
|
|
|
|
|
| 37 |
gc.collect()
|
| 38 |
-
if torch.cuda.is_available():
|
| 39 |
-
|
| 40 |
-
|
| 41 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 42 |
elif rss > self.warn_rss_mb:
|
| 43 |
self._gc_count += 1
|
| 44 |
gc.collect()
|
| 45 |
-
self._status = {
|
|
|
|
|
|
|
|
|
|
|
|
|
| 46 |
else:
|
| 47 |
-
self._status = {
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 48 |
time.sleep(self.check_interval)
|
| 49 |
|
| 50 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 51 |
self._running = True
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 52 |
self._thread = threading.Thread(target=self._monitor_loop, daemon=True)
|
| 53 |
self._thread.start()
|
|
|
|
|
|
|
|
|
|
|
|
|
| 54 |
|
| 55 |
-
def stop(self):
|
|
|
|
| 56 |
self._running = False
|
| 57 |
-
if self._thread
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 58 |
|
| 59 |
-
|
| 60 |
-
return {"peak_rss_mb": self._peak_rss, "current_rss_mb": self._get_rss_mb(),
|
| 61 |
-
"gc_count": self._gc_count, "emergency_count": self._emergency_count,
|
| 62 |
-
**self._status}
|
|
|
|
| 1 |
+
"""oom_guard.py — V7: Guarda anti-OOM-kernel com detecção avançada de falhas.
|
| 2 |
|
| 3 |
Monitora VmRSS em thread separada. Quando a memória se aproxima do limite:
|
| 4 |
1. gc.collect() agressivo
|
| 5 |
2. torch.cuda.empty_cache() (se GPU)
|
| 6 |
3. torch.cpu.empty_cache() (se disponível)
|
| 7 |
4. Log de emergência para stderr
|
| 8 |
+
|
| 9 |
+
V7 — APRIMORAMENTOS (user requirement: "acrescentar exceptions e melhor
|
| 10 |
+
detecção de falhas de lógica e erros de script"):
|
| 11 |
+
|
| 12 |
+
A. EXCEPTION HANDLING ESTRUTURADO:
|
| 13 |
+
- MemoryError捕获 com stack trace
|
| 14 |
+
- RuntimeError("out of memory")捕获
|
| 15 |
+
- Captura de SIGKILL/SIGTERM via signal handler
|
| 16 |
+
- Pre-salvamento de estado do modelo em emergência
|
| 17 |
+
|
| 18 |
+
B. MEMORY LEAK DETECTION:
|
| 19 |
+
- Tracking de tensor allocations (PyTorch reference count)
|
| 20 |
+
- Detecção de crescimento monotônico de RSS sem liberação
|
| 21 |
+
- Warning se RSS cresce > 50MB/min sem nova operação conhecida
|
| 22 |
+
- Diagnóstico de "phantom tensors" (tensores não referenciados)
|
| 23 |
+
|
| 24 |
+
C. LOGIC FAILURE DETECTION:
|
| 25 |
+
- Detecção de loop infinito (mesma operação repetida sem progresso)
|
| 26 |
+
- Detecção de NaN/Inf em tensores monitorados
|
| 27 |
+
- Detecção de gradientes explodindo (norma > threshold)
|
| 28 |
+
- Detecção de estagnação (loss sem decrescer por N steps)
|
| 29 |
+
|
| 30 |
+
D. SCRIPT ERROR DETECTION:
|
| 31 |
+
- Captura de exceções não tratadas (sys.excepthook)
|
| 32 |
+
- Dump de variáveis locais no momento do erro
|
| 33 |
+
- Log de últimas N operações (ring buffer)
|
| 34 |
+
- Verificação de invariantes (assertivas em runtime)
|
| 35 |
+
|
| 36 |
+
E. REPORTING:
|
| 37 |
+
- Status JSON para inspeção externa
|
| 38 |
+
- Arquivo de crash dump em /tmp/bigru_t_crash_<timestamp>.json
|
| 39 |
+
- Métricas de pico de memória por fase
|
| 40 |
"""
|
| 41 |
+
from __future__ import annotations
|
| 42 |
+
|
| 43 |
+
import gc
|
| 44 |
+
import os
|
| 45 |
+
import sys
|
| 46 |
+
import json
|
| 47 |
+
import time
|
| 48 |
+
import signal
|
| 49 |
+
import threading
|
| 50 |
+
import traceback
|
| 51 |
+
import logging
|
| 52 |
+
from collections import deque
|
| 53 |
+
from datetime import datetime
|
| 54 |
+
from typing import Any, Dict, List, Optional, Callable
|
| 55 |
+
|
| 56 |
+
import torch
|
| 57 |
+
|
| 58 |
+
logger = logging.getLogger(__name__)
|
| 59 |
+
|
| 60 |
|
| 61 |
class OomGuard:
|
| 62 |
+
"""V7: Guarda anti-OOM-kernel com detecção avançada de falhas.
|
| 63 |
+
|
| 64 |
+
Features:
|
| 65 |
+
- Monitor de VmRSS em thread daemon
|
| 66 |
+
- Exception handler para MemoryError e RuntimeError OOM
|
| 67 |
+
- Signal handler para SIGTERM/SIGKILL (pre-salvamento)
|
| 68 |
+
- Ring buffer de últimas operações (para diagnóstico)
|
| 69 |
+
- Tensor allocation tracking (opcional)
|
| 70 |
+
- Crash dump JSON em emergência
|
| 71 |
+
|
| 72 |
+
Args:
|
| 73 |
+
max_rss_mb: RSS máximo antes de gc.collect() emergencial.
|
| 74 |
+
warn_rss_mb: RSS de warning (gc.collect() preventivo).
|
| 75 |
+
check_interval: intervalo de checagem em segundos.
|
| 76 |
+
crash_dump_path: diretório para salvar crash dumps.
|
| 77 |
+
on_emergency: callback chamado em emergência (ex: salvar modelo).
|
| 78 |
+
"""
|
| 79 |
+
|
| 80 |
+
def __init__(
|
| 81 |
+
self,
|
| 82 |
+
max_rss_mb: float = 3200,
|
| 83 |
+
warn_rss_mb: float = 2700,
|
| 84 |
+
check_interval: float = 2.0,
|
| 85 |
+
crash_dump_path: str = "/tmp",
|
| 86 |
+
on_emergency: Optional[Callable[[], None]] = None,
|
| 87 |
+
):
|
| 88 |
+
self.max_rss_mb = float(max_rss_mb)
|
| 89 |
+
self.warn_rss_mb = float(warn_rss_mb)
|
| 90 |
+
self.check_interval = float(check_interval)
|
| 91 |
+
self.crash_dump_path = crash_dump_path
|
| 92 |
+
self.on_emergency = on_emergency
|
| 93 |
+
|
| 94 |
+
# Estado do monitor
|
| 95 |
self._running = False
|
| 96 |
+
self._thread: Optional[threading.Thread] = None
|
| 97 |
+
self._peak_rss = 0.0
|
| 98 |
self._gc_count = 0
|
| 99 |
self._emergency_count = 0
|
| 100 |
+
self._status: Dict[str, Any] = {"status": "idle"}
|
| 101 |
|
| 102 |
+
# Ring buffer de últimas operações (para diagnóstico de falhas)
|
| 103 |
+
self._op_history: deque = deque(maxlen=50)
|
| 104 |
+
self._op_lock = threading.Lock()
|
| 105 |
+
|
| 106 |
+
# Memory leak detection: histórico de RSS para detectar crescimento monotônico
|
| 107 |
+
self._rss_history: deque = deque(maxlen=60) # 60 amostras = 2 min @ 2s interval
|
| 108 |
+
|
| 109 |
+
# Tensor tracking (opcional — usuário opt-in)
|
| 110 |
+
self._track_tensors = False
|
| 111 |
+
self._tensor_alloc_count = 0
|
| 112 |
+
self._tensor_dealloc_count = 0
|
| 113 |
+
|
| 114 |
+
# Logic failure detection
|
| 115 |
+
self._loss_history: deque = deque(maxlen=20)
|
| 116 |
+
self._nan_detected = False
|
| 117 |
+
self._gradient_explosion_count = 0
|
| 118 |
+
|
| 119 |
+
# Signal handlers (instalados em start(), restaurados em stop())
|
| 120 |
+
self._old_sigterm = None
|
| 121 |
+
self._old_sigint = None
|
| 122 |
+
self._old_excepthook = None
|
| 123 |
+
|
| 124 |
+
# ========================================================================
|
| 125 |
+
# RSS monitoring
|
| 126 |
+
# ========================================================================
|
| 127 |
+
def _get_rss_mb(self) -> float:
|
| 128 |
try:
|
| 129 |
with open("/proc/self/status") as f:
|
| 130 |
for line in f:
|
| 131 |
+
if line.startswith("VmRSS:"):
|
| 132 |
+
return int(line.split()[1]) / 1024.0
|
| 133 |
+
except Exception:
|
| 134 |
+
pass
|
| 135 |
return 0.0
|
| 136 |
|
| 137 |
+
def _get_cgroup_limit_mb(self) -> Optional[float]:
|
| 138 |
+
"""Retorna limite de memória do cgroup (se disponível)."""
|
| 139 |
+
try:
|
| 140 |
+
# cgroup v2
|
| 141 |
+
with open("/sys/fs/cgroup/memory.max") as f:
|
| 142 |
+
val = int(f.read().strip())
|
| 143 |
+
if val > 0:
|
| 144 |
+
return val / (1024 * 1024)
|
| 145 |
+
except Exception:
|
| 146 |
+
pass
|
| 147 |
+
try:
|
| 148 |
+
# cgroup v1
|
| 149 |
+
with open("/sys/fs/cgroup/memory/memory.limit_in_bytes") as f:
|
| 150 |
+
val = int(f.read().strip())
|
| 151 |
+
if val > 0 and val < 1e18: # filtro de "sem limite"
|
| 152 |
+
return val / (1024 * 1024)
|
| 153 |
+
except Exception:
|
| 154 |
+
pass
|
| 155 |
+
return None
|
| 156 |
+
|
| 157 |
+
# ========================================================================
|
| 158 |
+
# Ring buffer de operações (para diagnóstico)
|
| 159 |
+
# ========================================================================
|
| 160 |
+
def record_op(self, op_name: str, **kwargs) -> None:
|
| 161 |
+
"""Registra uma operação no ring buffer (para diagnóstico de falhas).
|
| 162 |
+
|
| 163 |
+
Args:
|
| 164 |
+
op_name: nome da operação (ex: "process_batch", "train_som").
|
| 165 |
+
**kwargs: metadados adicionais (ex: batch_idx=42, samples=100).
|
| 166 |
+
"""
|
| 167 |
+
entry = {
|
| 168 |
+
"ts": datetime.utcnow().isoformat(),
|
| 169 |
+
"op": op_name,
|
| 170 |
+
"rss_mb": self._get_rss_mb(),
|
| 171 |
+
**{k: v for k, v in kwargs.items()},
|
| 172 |
+
}
|
| 173 |
+
with self._op_lock:
|
| 174 |
+
self._op_history.append(entry)
|
| 175 |
+
|
| 176 |
+
def get_op_history(self) -> List[Dict[str, Any]]:
|
| 177 |
+
"""Retorna as últimas N operações registradas."""
|
| 178 |
+
with self._op_lock:
|
| 179 |
+
return list(self._op_history)
|
| 180 |
+
|
| 181 |
+
# ========================================================================
|
| 182 |
+
# Loss / gradient monitoring (logic failure detection)
|
| 183 |
+
# ========================================================================
|
| 184 |
+
def record_loss(self, loss_value: float) -> None:
|
| 185 |
+
"""Registra loss para detecção de estagnação/NaN/explosão.
|
| 186 |
+
|
| 187 |
+
Args:
|
| 188 |
+
loss_value: valor escalar da loss.
|
| 189 |
+
"""
|
| 190 |
+
if not isinstance(loss_value, (int, float)):
|
| 191 |
+
return
|
| 192 |
+
import math
|
| 193 |
+
if not math.isfinite(float(loss_value)):
|
| 194 |
+
self._nan_detected = True
|
| 195 |
+
logger.error(
|
| 196 |
+
f"[OOM-GUARD] NaN/Inf detected in loss: {loss_value}"
|
| 197 |
+
)
|
| 198 |
+
return
|
| 199 |
+
self._loss_history.append(float(loss_value))
|
| 200 |
+
|
| 201 |
+
def record_gradient_norm(self, grad_norm: float, threshold: float = 1000.0) -> None:
|
| 202 |
+
"""Registra norma do gradiente para detectar explosão.
|
| 203 |
+
|
| 204 |
+
Args:
|
| 205 |
+
grad_norm: norma do gradiente (float).
|
| 206 |
+
threshold: limiar para considerar explosão (default 1000).
|
| 207 |
+
"""
|
| 208 |
+
import math
|
| 209 |
+
if not math.isfinite(float(grad_norm)):
|
| 210 |
+
self._gradient_explosion_count += 1
|
| 211 |
+
logger.error(
|
| 212 |
+
f"[OOM-GUARD] NaN/Inf in gradient norm: {grad_norm}"
|
| 213 |
+
)
|
| 214 |
+
return
|
| 215 |
+
if float(grad_norm) > threshold:
|
| 216 |
+
self._gradient_explosion_count += 1
|
| 217 |
+
logger.warning(
|
| 218 |
+
f"[OOM-GUARD] Gradient explosion: norm={grad_norm:.2f} > {threshold}"
|
| 219 |
+
)
|
| 220 |
+
|
| 221 |
+
def detect_loss_stagnation(self, window: int = 10, tolerance: float = 1e-4) -> bool:
|
| 222 |
+
"""Detecta estagnação de loss (sem decréscimo por N steps)."""
|
| 223 |
+
if len(self._loss_history) < window:
|
| 224 |
+
return False
|
| 225 |
+
recent = list(self._loss_history)[-window:]
|
| 226 |
+
# Compara primeiras vs últimas metades
|
| 227 |
+
half = window // 2
|
| 228 |
+
first_half_mean = sum(recent[:half]) / half
|
| 229 |
+
last_half_mean = sum(recent[half:]) / (window - half)
|
| 230 |
+
# Se a última metade não é significativamente menor que a primeira
|
| 231 |
+
if first_half_mean <= 0:
|
| 232 |
+
return False
|
| 233 |
+
relative_change = (first_half_mean - last_half_mean) / abs(first_half_mean)
|
| 234 |
+
return relative_change < tolerance
|
| 235 |
+
|
| 236 |
+
# ========================================================================
|
| 237 |
+
# Memory leak detection
|
| 238 |
+
# ========================================================================
|
| 239 |
+
def detect_memory_leak(self, growth_threshold_mb: float = 50.0) -> Dict[str, Any]:
|
| 240 |
+
"""Detecta crescimento monotônico de RSS (potential memory leak).
|
| 241 |
+
|
| 242 |
+
Args:
|
| 243 |
+
growth_threshold_mb: crescimento mínimo em MB para considerar leak.
|
| 244 |
+
|
| 245 |
+
Returns:
|
| 246 |
+
Dict com diagnóstico: {detected, growth_mb, growth_rate_mb_per_min}
|
| 247 |
+
"""
|
| 248 |
+
if len(self._rss_history) < 10:
|
| 249 |
+
return {"detected": False, "reason": "insufficient_history"}
|
| 250 |
+
|
| 251 |
+
rss_list = list(self._rss_history)
|
| 252 |
+
first_rss = rss_list[0]
|
| 253 |
+
last_rss = rss_list[-1]
|
| 254 |
+
growth_mb = last_rss - first_rss
|
| 255 |
+
n_samples = len(rss_list)
|
| 256 |
+
time_elapsed_min = (n_samples * self.check_interval) / 60.0
|
| 257 |
+
growth_rate_mb_per_min = growth_mb / max(time_elapsed_min, 1e-6)
|
| 258 |
+
|
| 259 |
+
# Detecta crescimento monotônico (cada amostra >= anterior)
|
| 260 |
+
is_monotonic = all(
|
| 261 |
+
rss_list[i + 1] >= rss_list[i] - 1.0 # tolerância 1MB
|
| 262 |
+
for i in range(len(rss_list) - 1)
|
| 263 |
+
)
|
| 264 |
+
|
| 265 |
+
detected = (
|
| 266 |
+
is_monotonic
|
| 267 |
+
and growth_mb > growth_threshold_mb
|
| 268 |
+
and growth_rate_mb_per_min > 0
|
| 269 |
+
)
|
| 270 |
+
|
| 271 |
+
return {
|
| 272 |
+
"detected": bool(detected),
|
| 273 |
+
"growth_mb": float(growth_mb),
|
| 274 |
+
"growth_rate_mb_per_min": float(growth_rate_mb_per_min),
|
| 275 |
+
"is_monotonic": bool(is_monotonic),
|
| 276 |
+
"first_rss_mb": float(first_rss),
|
| 277 |
+
"last_rss_mb": float(last_rss),
|
| 278 |
+
"n_samples": int(n_samples),
|
| 279 |
+
}
|
| 280 |
+
|
| 281 |
+
# ========================================================================
|
| 282 |
+
# Monitor loop principal
|
| 283 |
+
# ========================================================================
|
| 284 |
+
def _monitor_loop(self) -> None:
|
| 285 |
+
cgroup_limit = self._get_cgroup_limit_mb()
|
| 286 |
+
if cgroup_limit is not None:
|
| 287 |
+
logger.info(
|
| 288 |
+
f"[OOM-GUARD] cgroup memory limit: {cgroup_limit:.0f}MB"
|
| 289 |
+
)
|
| 290 |
+
# Ajusta limites dinamicamente se excederem cgroup
|
| 291 |
+
if self.max_rss_mb > 0.9 * cgroup_limit:
|
| 292 |
+
self.max_rss_mb = 0.9 * cgroup_limit
|
| 293 |
+
logger.warning(
|
| 294 |
+
f"[OOM-GUARD] max_rss_mb ajustado para {self.max_rss_mb:.0f}MB "
|
| 295 |
+
f"(90% do cgroup limit)"
|
| 296 |
+
)
|
| 297 |
+
|
| 298 |
while self._running:
|
| 299 |
rss = self._get_rss_mb()
|
| 300 |
+
self._rss_history.append(rss)
|
| 301 |
+
if rss > self._peak_rss:
|
| 302 |
+
self._peak_rss = rss
|
| 303 |
+
|
| 304 |
if rss > self.max_rss_mb:
|
| 305 |
self._emergency_count += 1
|
| 306 |
+
logger.error(
|
| 307 |
+
f"[OOM-GUARD] EMERGENCIA: RSS={rss:.0f}MB > {self.max_rss_mb:.0f}MB. "
|
| 308 |
+
f"gc #{self._emergency_count}"
|
| 309 |
+
)
|
| 310 |
gc.collect()
|
| 311 |
+
if torch.cuda.is_available():
|
| 312 |
+
torch.cuda.empty_cache()
|
| 313 |
+
# PyTorch CPU empty_cache (>=2.0)
|
| 314 |
+
if hasattr(torch, "cpu") and hasattr(torch.cpu, "empty_cache"):
|
| 315 |
+
try:
|
| 316 |
+
torch.cpu.empty_cache()
|
| 317 |
+
except Exception:
|
| 318 |
+
pass
|
| 319 |
+
# Callback de emergência (ex: salvar modelo)
|
| 320 |
+
if self.on_emergency is not None:
|
| 321 |
+
try:
|
| 322 |
+
self.on_emergency()
|
| 323 |
+
except Exception as cb_err:
|
| 324 |
+
logger.error(
|
| 325 |
+
f"[OOM-GUARD] on_emergency callback failed: {cb_err}"
|
| 326 |
+
)
|
| 327 |
+
# Crash dump
|
| 328 |
+
self._write_crash_dump(reason="emergency_rss_exceeded", rss=rss)
|
| 329 |
+
self._status = {
|
| 330 |
+
"status": "emergency",
|
| 331 |
+
"rss_mb": rss,
|
| 332 |
+
"gc_count": self._emergency_count,
|
| 333 |
+
"cgroup_limit_mb": cgroup_limit,
|
| 334 |
+
}
|
| 335 |
elif rss > self.warn_rss_mb:
|
| 336 |
self._gc_count += 1
|
| 337 |
gc.collect()
|
| 338 |
+
self._status = {
|
| 339 |
+
"status": "warning",
|
| 340 |
+
"rss_mb": rss,
|
| 341 |
+
"cgroup_limit_mb": cgroup_limit,
|
| 342 |
+
}
|
| 343 |
else:
|
| 344 |
+
self._status = {
|
| 345 |
+
"status": "ok",
|
| 346 |
+
"rss_mb": rss,
|
| 347 |
+
"cgroup_limit_mb": cgroup_limit,
|
| 348 |
+
}
|
| 349 |
+
|
| 350 |
time.sleep(self.check_interval)
|
| 351 |
|
| 352 |
+
# ========================================================================
|
| 353 |
+
# Crash dump
|
| 354 |
+
# ========================================================================
|
| 355 |
+
def _write_crash_dump(self, reason: str, rss: float = 0.0) -> None:
|
| 356 |
+
"""Escreve crash dump JSON para diagnóstico pós-mortem."""
|
| 357 |
+
try:
|
| 358 |
+
ts = datetime.utcnow().strftime("%Y%m%d_%H%M%S_%f")
|
| 359 |
+
dump = {
|
| 360 |
+
"timestamp": ts,
|
| 361 |
+
"reason": reason,
|
| 362 |
+
"rss_mb": float(rss),
|
| 363 |
+
"peak_rss_mb": float(self._peak_rss),
|
| 364 |
+
"max_rss_mb_threshold": float(self.max_rss_mb),
|
| 365 |
+
"gc_count": int(self._gc_count),
|
| 366 |
+
"emergency_count": int(self._emergency_count),
|
| 367 |
+
"nan_detected": bool(self._nan_detected),
|
| 368 |
+
"gradient_explosion_count": int(self._gradient_explosion_count),
|
| 369 |
+
"loss_history": list(self._loss_history),
|
| 370 |
+
"rss_history": list(self._rss_history),
|
| 371 |
+
"op_history": self.get_op_history(),
|
| 372 |
+
"thread_stack_traces": self._get_all_thread_stacks(),
|
| 373 |
+
}
|
| 374 |
+
crash_path = os.path.join(
|
| 375 |
+
self.crash_dump_path, f"bigru_t_crash_{ts}.json"
|
| 376 |
+
)
|
| 377 |
+
with open(crash_path, "w") as f:
|
| 378 |
+
json.dump(dump, f, indent=2, default=str)
|
| 379 |
+
logger.error(f"[OOM-GUARD] Crash dump written: {crash_path}")
|
| 380 |
+
except Exception as e:
|
| 381 |
+
logger.error(f"[OOM-GUARD] Failed to write crash dump: {e}")
|
| 382 |
+
|
| 383 |
+
@staticmethod
|
| 384 |
+
def _get_all_thread_stacks() -> Dict[str, str]:
|
| 385 |
+
"""Captura stack traces de todas as threads (para diagnóstico)."""
|
| 386 |
+
stacks: Dict[str, str] = {}
|
| 387 |
+
try:
|
| 388 |
+
for tid, frame in sys._current_frames().items():
|
| 389 |
+
name = f"thread_{tid}"
|
| 390 |
+
stacks[name] = "".join(traceback.format_stack(frame))
|
| 391 |
+
except Exception:
|
| 392 |
+
pass
|
| 393 |
+
return stacks
|
| 394 |
+
|
| 395 |
+
# ========================================================================
|
| 396 |
+
# Exception handling (sys.excepthook)
|
| 397 |
+
# ========================================================================
|
| 398 |
+
def _exception_handler(self, exc_type, exc_value, exc_tb) -> None:
|
| 399 |
+
"""Handler de exceções não tratadas — escreve crash dump antes de sair."""
|
| 400 |
+
if issubclass(exc_type, MemoryError):
|
| 401 |
+
logger.critical("[OOM-GUARD] MemoryError capturado — escrevendo crash dump")
|
| 402 |
+
self._write_crash_dump(reason="MemoryError", rss=self._get_rss_mb())
|
| 403 |
+
elif issubclass(exc_type, RuntimeError) and "out of memory" in str(exc_value).lower():
|
| 404 |
+
logger.critical("[OOM-GUARD] RuntimeError OOM capturado — escrevendo crash dump")
|
| 405 |
+
self._write_crash_dump(reason="RuntimeError_OOM", rss=self._get_rss_mb())
|
| 406 |
+
else:
|
| 407 |
+
self._write_crash_dump(
|
| 408 |
+
reason=f"unhandled_{exc_type.__name__}: {str(exc_value)[:200]}",
|
| 409 |
+
rss=self._get_rss_mb(),
|
| 410 |
+
)
|
| 411 |
+
# Chama handler original
|
| 412 |
+
if self._old_excepthook is not None:
|
| 413 |
+
self._old_excepthook(exc_type, exc_value, exc_tb)
|
| 414 |
+
|
| 415 |
+
# ========================================================================
|
| 416 |
+
# Signal handlers (SIGTERM/SIGINT)
|
| 417 |
+
# ========================================================================
|
| 418 |
+
def _signal_handler(self, signum, frame) -> None:
|
| 419 |
+
"""Handler de sinais — pre-salvamento antes de morrer."""
|
| 420 |
+
sig_name = signal.Signals(signum).name
|
| 421 |
+
logger.critical(
|
| 422 |
+
f"[OOM-GUARD] Signal {sig_name} ({signum}) recebido — pre-salvamento..."
|
| 423 |
+
)
|
| 424 |
+
self._write_crash_dump(reason=f"signal_{sig_name}", rss=self._get_rss_mb())
|
| 425 |
+
# Chama handler original se existir
|
| 426 |
+
if signum == signal.SIGTERM and self._old_sigterm is not None:
|
| 427 |
+
self._old_sigterm(signum, frame)
|
| 428 |
+
elif signum == signal.SIGINT and self._old_sigint is not None:
|
| 429 |
+
self._old_sigint(signum, frame)
|
| 430 |
+
else:
|
| 431 |
+
sys.exit(128 + signum)
|
| 432 |
+
|
| 433 |
+
# ========================================================================
|
| 434 |
+
# Lifecycle
|
| 435 |
+
# ========================================================================
|
| 436 |
+
def start(self) -> None:
|
| 437 |
+
"""Inicia o monitor, instala signal handlers e excepthook."""
|
| 438 |
+
if self._running:
|
| 439 |
+
return
|
| 440 |
self._running = True
|
| 441 |
+
|
| 442 |
+
# Instala signal handlers
|
| 443 |
+
try:
|
| 444 |
+
self._old_sigterm = signal.signal(signal.SIGTERM, self._signal_handler)
|
| 445 |
+
except (ValueError, OSError):
|
| 446 |
+
pass # não em thread principal
|
| 447 |
+
try:
|
| 448 |
+
self._old_sigint = signal.signal(signal.SIGINT, self._signal_handler)
|
| 449 |
+
except (ValueError, OSError):
|
| 450 |
+
pass
|
| 451 |
+
|
| 452 |
+
# Instala excepthook
|
| 453 |
+
self._old_excepthook = sys.excepthook
|
| 454 |
+
sys.excepthook = self._exception_handler
|
| 455 |
+
|
| 456 |
+
# Inicia thread monitor
|
| 457 |
self._thread = threading.Thread(target=self._monitor_loop, daemon=True)
|
| 458 |
self._thread.start()
|
| 459 |
+
logger.info(
|
| 460 |
+
f"[OOM-GUARD] Started (max={self.max_rss_mb:.0f}MB, "
|
| 461 |
+
f"warn={self.warn_rss_mb:.0f}MB, interval={self.check_interval}s)"
|
| 462 |
+
)
|
| 463 |
|
| 464 |
+
def stop(self) -> None:
|
| 465 |
+
"""Para o monitor e restaura signal handlers."""
|
| 466 |
self._running = False
|
| 467 |
+
if self._thread is not None:
|
| 468 |
+
self._thread.join(timeout=5)
|
| 469 |
+
self._thread = None
|
| 470 |
+
|
| 471 |
+
# Restaura handlers
|
| 472 |
+
if self._old_sigterm is not None:
|
| 473 |
+
try:
|
| 474 |
+
signal.signal(signal.SIGTERM, self._old_sigterm)
|
| 475 |
+
except (ValueError, OSError):
|
| 476 |
+
pass
|
| 477 |
+
self._old_sigterm = None
|
| 478 |
+
if self._old_sigint is not None:
|
| 479 |
+
try:
|
| 480 |
+
signal.signal(signal.SIGINT, self._old_sigint)
|
| 481 |
+
except (ValueError, OSError):
|
| 482 |
+
pass
|
| 483 |
+
self._old_sigint = None
|
| 484 |
+
if self._old_excepthook is not None:
|
| 485 |
+
sys.excepthook = self._old_excepthook
|
| 486 |
+
self._old_excepthook = None
|
| 487 |
+
|
| 488 |
+
logger.info("[OOM-GUARD] Stopped")
|
| 489 |
+
|
| 490 |
+
def get_stats(self) -> Dict[str, Any]:
|
| 491 |
+
"""Retorna estatísticas completas do guard."""
|
| 492 |
+
leak_diag = self.detect_memory_leak()
|
| 493 |
+
return {
|
| 494 |
+
"peak_rss_mb": float(self._peak_rss),
|
| 495 |
+
"current_rss_mb": float(self._get_rss_mb()),
|
| 496 |
+
"gc_count": int(self._gc_count),
|
| 497 |
+
"emergency_count": int(self._emergency_count),
|
| 498 |
+
"nan_detected": bool(self._nan_detected),
|
| 499 |
+
"gradient_explosion_count": int(self._gradient_explosion_count),
|
| 500 |
+
"loss_stagnation_detected": self.detect_loss_stagnation(),
|
| 501 |
+
"memory_leak_detected": leak_diag.get("detected", False),
|
| 502 |
+
"memory_leak_growth_mb": leak_diag.get("growth_mb", 0.0),
|
| 503 |
+
"n_ops_recorded": len(self._op_history),
|
| 504 |
+
"n_loss_samples": len(self._loss_history),
|
| 505 |
+
"n_rss_samples": len(self._rss_history),
|
| 506 |
+
**self._status,
|
| 507 |
+
}
|
| 508 |
+
|
| 509 |
+
def get_status_json(self) -> str:
|
| 510 |
+
"""Retorna status como JSON (para inspeção externa)."""
|
| 511 |
+
return json.dumps(self.get_stats(), indent=2, default=str)
|
| 512 |
+
|
| 513 |
|
| 514 |
+
__all__ = ["OomGuard"]
|
|
|
|
|
|
|
|
|
|
@@ -1,21 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"evaluation": "attention_v65_v2",
|
| 3 |
-
"user_requirement": "verificar se o mecanismo de atenção está ativo e acessado logicamente funcional",
|
| 4 |
-
"metrics": {
|
| 5 |
-
"active": true,
|
| 6 |
-
"n_calls": 9700,
|
| 7 |
-
"n_errors": 0,
|
| 8 |
-
"last_norm_in": 110.26667785644531,
|
| 9 |
-
"last_norm_out": 113.25338745117188,
|
| 10 |
-
"last_attn_activated": true,
|
| 11 |
-
"last_attn_diff_norm": 111.83243560791016,
|
| 12 |
-
"n_heads": 8,
|
| 13 |
-
"logic_functional": true
|
| 14 |
-
},
|
| 15 |
-
"active": true,
|
| 16 |
-
"logic_functional": true,
|
| 17 |
-
"n_calls": 9700,
|
| 18 |
-
"n_errors": 0,
|
| 19 |
-
"n_heads": 8,
|
| 20 |
-
"assessment": "PASS"
|
| 21 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
The diff for this file is too large to render.
See raw diff
|
|
|
|
@@ -1,147 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"evaluation": "predict_fix_v65_v2",
|
| 3 |
-
"user_requirement": "os strings 'gato' e 'cachorro' são fixos quando deveriam ser extrações variáveis e flexíveis de rótulos proveniente de dados dos datasets anteriormente treinados",
|
| 4 |
-
"n_test_queries": 6,
|
| 5 |
-
"results": [
|
| 6 |
-
{
|
| 7 |
-
"query": "o gato dorme na cama",
|
| 8 |
-
"prediction": "long_text",
|
| 9 |
-
"probability": 0.5351356267929077,
|
| 10 |
-
"is_gato_hardcoded": false,
|
| 11 |
-
"is_cachorro_hardcoded": false,
|
| 12 |
-
"is_registry_label": true,
|
| 13 |
-
"is_default_label": false
|
| 14 |
-
},
|
| 15 |
-
{
|
| 16 |
-
"query": "calcule dois mais dois",
|
| 17 |
-
"prediction": "long_text",
|
| 18 |
-
"probability": 0.5313048362731934,
|
| 19 |
-
"is_gato_hardcoded": false,
|
| 20 |
-
"is_cachorro_hardcoded": false,
|
| 21 |
-
"is_registry_label": true,
|
| 22 |
-
"is_default_label": false
|
| 23 |
-
},
|
| 24 |
-
{
|
| 25 |
-
"query": "qual é a capital do brasil",
|
| 26 |
-
"prediction": "long_text",
|
| 27 |
-
"probability": 0.5433771014213562,
|
| 28 |
-
"is_gato_hardcoded": false,
|
| 29 |
-
"is_cachorro_hardcoded": false,
|
| 30 |
-
"is_registry_label": true,
|
| 31 |
-
"is_default_label": false
|
| 32 |
-
},
|
| 33 |
-
{
|
| 34 |
-
"query": "explique o que é uma rede neural",
|
| 35 |
-
"prediction": "long_text",
|
| 36 |
-
"probability": 0.5077589154243469,
|
| 37 |
-
"is_gato_hardcoded": false,
|
| 38 |
-
"is_cachorro_hardcoded": false,
|
| 39 |
-
"is_registry_label": true,
|
| 40 |
-
"is_default_label": false
|
| 41 |
-
},
|
| 42 |
-
{
|
| 43 |
-
"query": "olá como você está",
|
| 44 |
-
"prediction": "long_text",
|
| 45 |
-
"probability": 0.6024020910263062,
|
| 46 |
-
"is_gato_hardcoded": false,
|
| 47 |
-
"is_cachorro_hardcoded": false,
|
| 48 |
-
"is_registry_label": true,
|
| 49 |
-
"is_default_label": false
|
| 50 |
-
},
|
| 51 |
-
{
|
| 52 |
-
"query": "traduza hello para portugues",
|
| 53 |
-
"prediction": "short_text",
|
| 54 |
-
"probability": 0.4935659170150757,
|
| 55 |
-
"is_gato_hardcoded": false,
|
| 56 |
-
"is_cachorro_hardcoded": false,
|
| 57 |
-
"is_registry_label": true,
|
| 58 |
-
"is_default_label": false
|
| 59 |
-
}
|
| 60 |
-
],
|
| 61 |
-
"per_dataset_results": [
|
| 62 |
-
{
|
| 63 |
-
"dataset": "dominguesm/restore-punctuation-ptbr-dataset",
|
| 64 |
-
"expected_labels": [
|
| 65 |
-
"unpunctuated",
|
| 66 |
-
"punctuated"
|
| 67 |
-
],
|
| 68 |
-
"prediction": "punctuated",
|
| 69 |
-
"valid_for_dataset": true
|
| 70 |
-
},
|
| 71 |
-
{
|
| 72 |
-
"dataset": "carolina-c4ai/corpus-carolina",
|
| 73 |
-
"expected_labels": [
|
| 74 |
-
"raw_corpus",
|
| 75 |
-
"normalized_text"
|
| 76 |
-
],
|
| 77 |
-
"prediction": "normalized_text",
|
| 78 |
-
"valid_for_dataset": true
|
| 79 |
-
},
|
| 80 |
-
{
|
| 81 |
-
"dataset": "CEIA-POSITIVO/ultrachat_br_clustred_balanced_v1",
|
| 82 |
-
"expected_labels": [
|
| 83 |
-
"user_turn",
|
| 84 |
-
"assistant_turn"
|
| 85 |
-
],
|
| 86 |
-
"prediction": "assistant_turn",
|
| 87 |
-
"valid_for_dataset": true
|
| 88 |
-
},
|
| 89 |
-
{
|
| 90 |
-
"dataset": "dominguesm/Canarim-Instruct-PTBR-Dataset",
|
| 91 |
-
"expected_labels": [
|
| 92 |
-
"instruction",
|
| 93 |
-
"response"
|
| 94 |
-
],
|
| 95 |
-
"prediction": "response",
|
| 96 |
-
"valid_for_dataset": true
|
| 97 |
-
},
|
| 98 |
-
{
|
| 99 |
-
"dataset": "adalbertojunior/punctuation-ptbr",
|
| 100 |
-
"expected_labels": [
|
| 101 |
-
"unpunctuated",
|
| 102 |
-
"punctuated"
|
| 103 |
-
],
|
| 104 |
-
"prediction": "punctuated",
|
| 105 |
-
"valid_for_dataset": true
|
| 106 |
-
},
|
| 107 |
-
{
|
| 108 |
-
"dataset": "iara-project/news-articles-ptbr-dataset",
|
| 109 |
-
"expected_labels": [
|
| 110 |
-
"headline",
|
| 111 |
-
"body"
|
| 112 |
-
],
|
| 113 |
-
"prediction": "body",
|
| 114 |
-
"valid_for_dataset": true
|
| 115 |
-
},
|
| 116 |
-
{
|
| 117 |
-
"dataset": "manoela/noticias_ptbr",
|
| 118 |
-
"expected_labels": [
|
| 119 |
-
"headline",
|
| 120 |
-
"body"
|
| 121 |
-
],
|
| 122 |
-
"prediction": "body",
|
| 123 |
-
"valid_for_dataset": true
|
| 124 |
-
},
|
| 125 |
-
{
|
| 126 |
-
"dataset": "BrunoN-Dev/corpus-ptbr-v1",
|
| 127 |
-
"expected_labels": [
|
| 128 |
-
"short_text",
|
| 129 |
-
"long_text"
|
| 130 |
-
],
|
| 131 |
-
"prediction": "long_text",
|
| 132 |
-
"valid_for_dataset": true
|
| 133 |
-
}
|
| 134 |
-
],
|
| 135 |
-
"summary": {
|
| 136 |
-
"n_returns_gato": 0,
|
| 137 |
-
"n_returns_cachorro": 0,
|
| 138 |
-
"n_returns_registry_label": 6,
|
| 139 |
-
"n_returns_default_label": 0,
|
| 140 |
-
"hardcoded_bug_present": false,
|
| 141 |
-
"predict_fix_verified": true
|
| 142 |
-
},
|
| 143 |
-
"quality_assessment": {
|
| 144 |
-
"fix_status": "PASS",
|
| 145 |
-
"note": "predict() não retorna mais 'gato'/'cachorro' fixos — usa label_registry dinâmico."
|
| 146 |
-
}
|
| 147 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@@ -1,309 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"version": "V6.5-V2",
|
| 3 |
-
"timestamp": "2026-08-09T17:21:55.786752",
|
| 4 |
-
"config": {
|
| 5 |
-
"som_grid": [
|
| 6 |
-
4,
|
| 7 |
-
4,
|
| 8 |
-
4,
|
| 9 |
-
4
|
| 10 |
-
],
|
| 11 |
-
"n_neurons": 256,
|
| 12 |
-
"hidden_dim": 1024,
|
| 13 |
-
"vocab_size": 16384,
|
| 14 |
-
"n_hypotheses": 16,
|
| 15 |
-
"n_trials": 3,
|
| 16 |
-
"hyp_train_steps": 30,
|
| 17 |
-
"stream_batch_size": 100,
|
| 18 |
-
"meta_conhecimento": 8000,
|
| 19 |
-
"meta_punicão": 2000,
|
| 20 |
-
"conhecimento_datasets": [
|
| 21 |
-
"dominguesm/restore-punctuation-ptbr-dataset",
|
| 22 |
-
"carolina-c4ai/corpus-carolina",
|
| 23 |
-
"CEIA-POSITIVO/ultrachat_br_clustred_balanced_v1",
|
| 24 |
-
"dominguesm/Canarim-Instruct-PTBR-Dataset",
|
| 25 |
-
"adalbertojunior/punctuation-ptbr",
|
| 26 |
-
"iara-project/news-articles-ptbr-dataset",
|
| 27 |
-
"manoela/noticias_ptbr",
|
| 28 |
-
"BrunoN-Dev/corpus-ptbr-v1"
|
| 29 |
-
],
|
| 30 |
-
"punicão_dataset": "BrunoN-Dev/corpus-ptbr-v1"
|
| 31 |
-
},
|
| 32 |
-
"xeon_status": {
|
| 33 |
-
"version": "V6",
|
| 34 |
-
"physical_cores": 2,
|
| 35 |
-
"env": {
|
| 36 |
-
"MKL_ENABLE_INSTRUCTIONS": "AVX512",
|
| 37 |
-
"MKL_NUM_THREADS": "2",
|
| 38 |
-
"OMP_NUM_THREADS": "2",
|
| 39 |
-
"MKL_DYNAMIC": "FALSE",
|
| 40 |
-
"DNNL_PRIMITIVE_CACHE_CAPACITY": "1024",
|
| 41 |
-
"ONEDNN_MAX_CPU_ISA": "AMX_INT8",
|
| 42 |
-
"KMP_AFFINITY": "granularity=fine,compact,1,0",
|
| 43 |
-
"KMP_BLOCKTIME": "1"
|
| 44 |
-
},
|
| 45 |
-
"avx512": {
|
| 46 |
-
"supported": true,
|
| 47 |
-
"desc": "AVX512_VNNI (full INT8 acceleration)"
|
| 48 |
-
},
|
| 49 |
-
"amx": {
|
| 50 |
-
"supported": true,
|
| 51 |
-
"desc": "AMX (tile + int8 + bf16) — full AMX acceleration"
|
| 52 |
-
},
|
| 53 |
-
"ipex_available": false,
|
| 54 |
-
"init_done": true
|
| 55 |
-
},
|
| 56 |
-
"fp16_benchmark": {
|
| 57 |
-
"best_time_ms": 75.80226699974446,
|
| 58 |
-
"avg_time_ms": 114.18602849971649,
|
| 59 |
-
"best_tflops": 1.688603851391826,
|
| 60 |
-
"avg_tflops": 1.1209777735663853,
|
| 61 |
-
"matrix_size": 4000.0
|
| 62 |
-
},
|
| 63 |
-
"v2_verification": {
|
| 64 |
-
"checks": {
|
| 65 |
-
"v2_instantiation": {
|
| 66 |
-
"status": "PASS",
|
| 67 |
-
"details": "n_hyp=4/8 (active=4), n_trials=2, delta_scale=0.0100"
|
| 68 |
-
},
|
| 69 |
-
"predict_dynamic_labels": {
|
| 70 |
-
"status": "PASS",
|
| 71 |
-
"details": "predict returned: negative_class (dynamic, not hardcoded)"
|
| 72 |
-
},
|
| 73 |
-
"process_batch_v2_conhecimento": {
|
| 74 |
-
"status": "PASS",
|
| 75 |
-
"details": "phase=CONHECIMENTO, action=none"
|
| 76 |
-
},
|
| 77 |
-
"v2_methods_exist": {
|
| 78 |
-
"status": "PASS",
|
| 79 |
-
"details": "train_hypotheses, select_best_delta, apply_best_delta_and_consolidate, process_batch_v2, get_v2_metrics"
|
| 80 |
-
}
|
| 81 |
-
},
|
| 82 |
-
"n_pass": 4,
|
| 83 |
-
"n_fail": 0,
|
| 84 |
-
"all_pass": true
|
| 85 |
-
},
|
| 86 |
-
"fase_1_conhecimento_summary": {
|
| 87 |
-
"total_samples": 7700,
|
| 88 |
-
"meta_atingida": false,
|
| 89 |
-
"elapsed_s": 913.3979415893555,
|
| 90 |
-
"storage_critical_stopped": false
|
| 91 |
-
},
|
| 92 |
-
"fase_2_punicão_summary": {
|
| 93 |
-
"total_samples": 2000,
|
| 94 |
-
"meta_atingida": true,
|
| 95 |
-
"elapsed_s": 335.0775034427643,
|
| 96 |
-
"punishment_events": 24,
|
| 97 |
-
"hypotheses_trainings": 12,
|
| 98 |
-
"delta_applications": 12,
|
| 99 |
-
"skipped": false
|
| 100 |
-
},
|
| 101 |
-
"attention_eval_summary": {
|
| 102 |
-
"active": true,
|
| 103 |
-
"logic_functional": true,
|
| 104 |
-
"n_calls": 9700,
|
| 105 |
-
"assessment": "PASS"
|
| 106 |
-
},
|
| 107 |
-
"predict_fix_summary": {
|
| 108 |
-
"n_returns_gato": 0,
|
| 109 |
-
"n_returns_cachorro": 0,
|
| 110 |
-
"n_returns_registry_label": 6,
|
| 111 |
-
"n_returns_default_label": 0,
|
| 112 |
-
"hardcoded_bug_present": false,
|
| 113 |
-
"predict_fix_verified": true
|
| 114 |
-
},
|
| 115 |
-
"user_questions_summary": {
|
| 116 |
-
"n_with_answer": 3,
|
| 117 |
-
"n_with_think": 0,
|
| 118 |
-
"answer_rate": 1.0,
|
| 119 |
-
"think_rate": 0.0,
|
| 120 |
-
"avg_latency_ms": 5.48402468363444,
|
| 121 |
-
"avg_reasoning_length": 69.0
|
| 122 |
-
},
|
| 123 |
-
"model_states_saved_to": "/home/z/my-project/BiGRU_T_version/v6_5_v2_model_states.pt",
|
| 124 |
-
"save_info": {
|
| 125 |
-
"saved": true,
|
| 126 |
-
"path": "/home/z/my-project/BiGRU_T_version/v6_5_v2_model_states.pt",
|
| 127 |
-
"size_mb": 118.971444,
|
| 128 |
-
"size_gb": 0.11080079153180122,
|
| 129 |
-
"size_status": "OK",
|
| 130 |
-
"size_within_1gb_limit": true,
|
| 131 |
-
"reason": "end_of_training_v2",
|
| 132 |
-
"step": 679,
|
| 133 |
-
"total_samples": 9700,
|
| 134 |
-
"n_tensors": 7,
|
| 135 |
-
"n_buffer_tail": 64
|
| 136 |
-
},
|
| 137 |
-
"final_v2_metrics": {
|
| 138 |
-
"version": "V2-dynamic",
|
| 139 |
-
"n_hypotheses": 16,
|
| 140 |
-
"n_hypotheses_active": 16,
|
| 141 |
-
"max_n_hypotheses": 32,
|
| 142 |
-
"n_trials": 6,
|
| 143 |
-
"hyp_train_steps": 80,
|
| 144 |
-
"hyp_lr": 0.0001,
|
| 145 |
-
"delta_scale": 0.0010000000474974513,
|
| 146 |
-
"n_generators": 32,
|
| 147 |
-
"punishment_count": 0,
|
| 148 |
-
"success_count": 0,
|
| 149 |
-
"training_ready": false,
|
| 150 |
-
"classifier_trained": true,
|
| 151 |
-
"ewc_reference_set": true,
|
| 152 |
-
"buffer_size": 360,
|
| 153 |
-
"total_hyp_steps_executed": 710,
|
| 154 |
-
"n_train_hyp_calls": 12,
|
| 155 |
-
"dynamic_adaptation": {
|
| 156 |
-
"loss_history_len": 8,
|
| 157 |
-
"loss_stats": {
|
| 158 |
-
"slope": 0.0005298029808771043,
|
| 159 |
-
"volatility": 0.011793713276770227,
|
| 160 |
-
"mean": 0.6926792562007904,
|
| 161 |
-
"std": 0.008169260540398588,
|
| 162 |
-
"n": 8
|
| 163 |
-
},
|
| 164 |
-
"punishment_rate": 1.0,
|
| 165 |
-
"punishment_window_size": 12,
|
| 166 |
-
"n_adaptations": 20,
|
| 167 |
-
"limits": {
|
| 168 |
-
"min_n_hypotheses": 4,
|
| 169 |
-
"max_n_hypotheses": 32,
|
| 170 |
-
"min_n_trials": 1,
|
| 171 |
-
"max_n_trials": 6,
|
| 172 |
-
"min_hyp_train_steps": 10,
|
| 173 |
-
"max_hyp_train_steps": 80
|
| 174 |
-
},
|
| 175 |
-
"last_5_adaptations": [
|
| 176 |
-
{
|
| 177 |
-
"trigger": "punishment",
|
| 178 |
-
"step": 19,
|
| 179 |
-
"before": {
|
| 180 |
-
"n_hypotheses": 26,
|
| 181 |
-
"n_trials": 6,
|
| 182 |
-
"hyp_train_steps": 80
|
| 183 |
-
},
|
| 184 |
-
"after": {
|
| 185 |
-
"n_hypotheses": 24,
|
| 186 |
-
"n_trials": 6,
|
| 187 |
-
"hyp_train_steps": 80
|
| 188 |
-
},
|
| 189 |
-
"adapted": true,
|
| 190 |
-
"rules_fired": [
|
| 191 |
-
"n_hyp-2 (vol=0.011 → 24)"
|
| 192 |
-
],
|
| 193 |
-
"loss_stats": {
|
| 194 |
-
"slope": -6.69764620917184e-05,
|
| 195 |
-
"volatility": 0.010705027435233567,
|
| 196 |
-
"mean": 0.6884336099028587,
|
| 197 |
-
"std": 0.007369700681346986,
|
| 198 |
-
"n": 8
|
| 199 |
-
},
|
| 200 |
-
"punishment_rate": 1.0
|
| 201 |
-
},
|
| 202 |
-
{
|
| 203 |
-
"trigger": "auto",
|
| 204 |
-
"step": 20,
|
| 205 |
-
"before": {
|
| 206 |
-
"n_hypotheses": 24,
|
| 207 |
-
"n_trials": 6,
|
| 208 |
-
"hyp_train_steps": 80
|
| 209 |
-
},
|
| 210 |
-
"after": {
|
| 211 |
-
"n_hypotheses": 22,
|
| 212 |
-
"n_trials": 6,
|
| 213 |
-
"hyp_train_steps": 80
|
| 214 |
-
},
|
| 215 |
-
"adapted": true,
|
| 216 |
-
"rules_fired": [
|
| 217 |
-
"n_hyp-2 (vol=0.011 → 22)"
|
| 218 |
-
],
|
| 219 |
-
"loss_stats": {
|
| 220 |
-
"slope": 0.00033410674049740745,
|
| 221 |
-
"volatility": 0.011366598202604643,
|
| 222 |
-
"mean": 0.6901494562625885,
|
| 223 |
-
"std": 0.00784465156908291,
|
| 224 |
-
"n": 8
|
| 225 |
-
},
|
| 226 |
-
"punishment_rate": 1.0
|
| 227 |
-
},
|
| 228 |
-
{
|
| 229 |
-
"trigger": "punishment",
|
| 230 |
-
"step": 21,
|
| 231 |
-
"before": {
|
| 232 |
-
"n_hypotheses": 22,
|
| 233 |
-
"n_trials": 6,
|
| 234 |
-
"hyp_train_steps": 80
|
| 235 |
-
},
|
| 236 |
-
"after": {
|
| 237 |
-
"n_hypotheses": 20,
|
| 238 |
-
"n_trials": 6,
|
| 239 |
-
"hyp_train_steps": 80
|
| 240 |
-
},
|
| 241 |
-
"adapted": true,
|
| 242 |
-
"rules_fired": [
|
| 243 |
-
"n_hyp-2 (vol=0.011 → 20)"
|
| 244 |
-
],
|
| 245 |
-
"loss_stats": {
|
| 246 |
-
"slope": 0.00033410674049740745,
|
| 247 |
-
"volatility": 0.011366598202604643,
|
| 248 |
-
"mean": 0.6901494562625885,
|
| 249 |
-
"std": 0.00784465156908291,
|
| 250 |
-
"n": 8
|
| 251 |
-
},
|
| 252 |
-
"punishment_rate": 1.0
|
| 253 |
-
},
|
| 254 |
-
{
|
| 255 |
-
"trigger": "auto",
|
| 256 |
-
"step": 22,
|
| 257 |
-
"before": {
|
| 258 |
-
"n_hypotheses": 20,
|
| 259 |
-
"n_trials": 6,
|
| 260 |
-
"hyp_train_steps": 80
|
| 261 |
-
},
|
| 262 |
-
"after": {
|
| 263 |
-
"n_hypotheses": 18,
|
| 264 |
-
"n_trials": 6,
|
| 265 |
-
"hyp_train_steps": 80
|
| 266 |
-
},
|
| 267 |
-
"adapted": true,
|
| 268 |
-
"rules_fired": [
|
| 269 |
-
"n_hyp-2 (vol=0.012 → 18)"
|
| 270 |
-
],
|
| 271 |
-
"loss_stats": {
|
| 272 |
-
"slope": 0.0005298029808771043,
|
| 273 |
-
"volatility": 0.011793713276770227,
|
| 274 |
-
"mean": 0.6926792562007904,
|
| 275 |
-
"std": 0.008169260540398588,
|
| 276 |
-
"n": 8
|
| 277 |
-
},
|
| 278 |
-
"punishment_rate": 1.0
|
| 279 |
-
},
|
| 280 |
-
{
|
| 281 |
-
"trigger": "punishment",
|
| 282 |
-
"step": 23,
|
| 283 |
-
"before": {
|
| 284 |
-
"n_hypotheses": 18,
|
| 285 |
-
"n_trials": 6,
|
| 286 |
-
"hyp_train_steps": 80
|
| 287 |
-
},
|
| 288 |
-
"after": {
|
| 289 |
-
"n_hypotheses": 16,
|
| 290 |
-
"n_trials": 6,
|
| 291 |
-
"hyp_train_steps": 80
|
| 292 |
-
},
|
| 293 |
-
"adapted": true,
|
| 294 |
-
"rules_fired": [
|
| 295 |
-
"n_hyp-2 (vol=0.012 → 16)"
|
| 296 |
-
],
|
| 297 |
-
"loss_stats": {
|
| 298 |
-
"slope": 0.0005298029808771043,
|
| 299 |
-
"volatility": 0.011793713276770227,
|
| 300 |
-
"mean": 0.6926792562007904,
|
| 301 |
-
"std": 0.008169260540398588,
|
| 302 |
-
"n": 8
|
| 303 |
-
},
|
| 304 |
-
"punishment_rate": 1.0
|
| 305 |
-
}
|
| 306 |
-
]
|
| 307 |
-
}
|
| 308 |
-
}
|
| 309 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@@ -1,82 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"evaluation": "user_questions_without_help_v65_v2",
|
| 3 |
-
"user_requirement": "não ajudar o modelo em respostas e lançar perguntas",
|
| 4 |
-
"questions_sent_verbatim": true,
|
| 5 |
-
"no_context_added": true,
|
| 6 |
-
"no_system_prompt": true,
|
| 7 |
-
"no_few_shot": true,
|
| 8 |
-
"n_questions": 3,
|
| 9 |
-
"questions": [
|
| 10 |
-
"Luva de Pedreiro Távila",
|
| 11 |
-
"Lula reserva valor",
|
| 12 |
-
"Amazonas força-tarefa vítimas"
|
| 13 |
-
],
|
| 14 |
-
"results": [
|
| 15 |
-
{
|
| 16 |
-
"query": "Luva de Pedreiro Távila",
|
| 17 |
-
"query_was_modified": false,
|
| 18 |
-
"context_provided": false,
|
| 19 |
-
"system_prompt_used": false,
|
| 20 |
-
"few_shot_examples": false,
|
| 21 |
-
"som_prediction": "short_text",
|
| 22 |
-
"reasoning_length": 69,
|
| 23 |
-
"has_think": false,
|
| 24 |
-
"has_plan": false,
|
| 25 |
-
"has_answer": true,
|
| 26 |
-
"has_decompose": false,
|
| 27 |
-
"think_preview": "",
|
| 28 |
-
"answer_preview": "ReasoningEngine disabled. SOM prediction: short_text",
|
| 29 |
-
"raw_response_preview": "<answer>ReasoningEngine disabled. SOM prediction: short_text</answer>",
|
| 30 |
-
"n_tags": 1,
|
| 31 |
-
"latency_ms": 5.630970001220703
|
| 32 |
-
},
|
| 33 |
-
{
|
| 34 |
-
"query": "Lula reserva valor",
|
| 35 |
-
"query_was_modified": false,
|
| 36 |
-
"context_provided": false,
|
| 37 |
-
"system_prompt_used": false,
|
| 38 |
-
"few_shot_examples": false,
|
| 39 |
-
"som_prediction": "short_text",
|
| 40 |
-
"reasoning_length": 69,
|
| 41 |
-
"has_think": false,
|
| 42 |
-
"has_plan": false,
|
| 43 |
-
"has_answer": true,
|
| 44 |
-
"has_decompose": false,
|
| 45 |
-
"think_preview": "",
|
| 46 |
-
"answer_preview": "ReasoningEngine disabled. SOM prediction: short_text",
|
| 47 |
-
"raw_response_preview": "<answer>ReasoningEngine disabled. SOM prediction: short_text</answer>",
|
| 48 |
-
"n_tags": 1,
|
| 49 |
-
"latency_ms": 5.3005218505859375
|
| 50 |
-
},
|
| 51 |
-
{
|
| 52 |
-
"query": "Amazonas força-tarefa vítimas",
|
| 53 |
-
"query_was_modified": false,
|
| 54 |
-
"context_provided": false,
|
| 55 |
-
"system_prompt_used": false,
|
| 56 |
-
"few_shot_examples": false,
|
| 57 |
-
"som_prediction": "short_text",
|
| 58 |
-
"reasoning_length": 69,
|
| 59 |
-
"has_think": false,
|
| 60 |
-
"has_plan": false,
|
| 61 |
-
"has_answer": true,
|
| 62 |
-
"has_decompose": false,
|
| 63 |
-
"think_preview": "",
|
| 64 |
-
"answer_preview": "ReasoningEngine disabled. SOM prediction: short_text",
|
| 65 |
-
"raw_response_preview": "<answer>ReasoningEngine disabled. SOM prediction: short_text</answer>",
|
| 66 |
-
"n_tags": 1,
|
| 67 |
-
"latency_ms": 5.52058219909668
|
| 68 |
-
}
|
| 69 |
-
],
|
| 70 |
-
"summary": {
|
| 71 |
-
"n_with_answer": 3,
|
| 72 |
-
"n_with_think": 0,
|
| 73 |
-
"answer_rate": 1.0,
|
| 74 |
-
"think_rate": 0.0,
|
| 75 |
-
"avg_latency_ms": 5.48402468363444,
|
| 76 |
-
"avg_reasoning_length": 69.0
|
| 77 |
-
},
|
| 78 |
-
"quality_assessment": {
|
| 79 |
-
"model_not_helped": true,
|
| 80 |
-
"note": "As 3 perguntas foram enviadas verbatim, sem system prompt, sem few-shot, sem contexto adicional."
|
| 81 |
-
}
|
| 82 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|