V6.5-V2-dynamic: upload batch (scripts + state + model) [13 files]
Browse files- kohonen_learning_system.py +437 -3
- train_v6_5_v2.py +399 -17
- v6_5_v2_attention_eval.json +5 -5
- v6_5_v2_phases_eval.json +0 -0
- v6_5_v2_predict_fix_eval.json +6 -6
- v6_5_v2_report.json +39 -39
- v6_5_v2_user_questions.json +31 -31
kohonen_learning_system.py
CHANGED
|
@@ -375,6 +375,8 @@ class KohonenSOM4D:
|
|
| 375 |
alpha0: float = 0.5,
|
| 376 |
sigma0: float = 3.0,
|
| 377 |
lambda_ewc: float = 0.01,
|
|
|
|
|
|
|
| 378 |
):
|
| 379 |
# V6.5-V2-metrics-FIX-3 — α₀=0.5 e σ₀=3.0 conforme especificação
|
| 380 |
# canônica para SOM 4D (Kohonen classic):
|
|
@@ -384,6 +386,17 @@ class KohonenSOM4D:
|
|
| 384 |
# que α_t e σ_t nunca decaiam abaixo de 0.001 e 0.1 respectivamente
|
| 385 |
# (sem isso, após ~7000 updates σ→0 e o SOM degenera em k-means puro,
|
| 386 |
# perdendo preservação topológica).
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 387 |
self.I, self.J, self.K, self.L = grid_shape
|
| 388 |
self.alpha0 = float(alpha0)
|
| 389 |
self.sigma0 = float(sigma0)
|
|
@@ -396,6 +409,23 @@ class KohonenSOM4D:
|
|
| 396 |
self.fisher_accum = torch.zeros(self.I, self.J, self.K, self.L)
|
| 397 |
self.fisher_count = torch.zeros(self.I, self.J, self.K, self.L)
|
| 398 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 399 |
def _neighborhood(self, bmu_idx):
|
| 400 |
"""Vizinhança Gaussiana 4D: d² = Δi² + Δj² + Δk² + Δl²."""
|
| 401 |
i, j, k, l = bmu_idx
|
|
@@ -409,13 +439,39 @@ class KohonenSOM4D:
|
|
| 409 |
dist_sq = (II - i) ** 2 + (JJ - j) ** 2 + (KK - k) ** 2 + (LL - l) ** 2
|
| 410 |
return dist_sq
|
| 411 |
|
| 412 |
-
def find_bmu(self, x: torch.Tensor) -> Tuple[int, int, int, int]:
|
| 413 |
-
"""Best Matching Unit: argmin ||W - x||² em ℝ⁴.
|
| 414 |
|
| 415 |
Substitui pgvector_lookup — busca nearest-neighbor flat sobre o grid.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 416 |
"""
|
|
|
|
|
|
|
|
|
|
| 417 |
dist = torch.sum((self.weights - x.view(1, 1, 1, 1, 4)) ** 2, dim=-1)
|
| 418 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 419 |
i = flat_idx // (self.J * self.K * self.L)
|
| 420 |
rest = flat_idx % (self.J * self.K * self.L)
|
| 421 |
j = rest // (self.K * self.L)
|
|
@@ -424,9 +480,232 @@ class KohonenSOM4D:
|
|
| 424 |
l = rest % self.L
|
| 425 |
return (i, j, k, l)
|
| 426 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 427 |
def update_weights(self, x: torch.Tensor, bmu_idx, accumulate_fisher=False):
|
| 428 |
"""Update Kohonen: ΔW = α·Λ·(x - W) + penalidade EWC em w.
|
| 429 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 430 |
V6.5-V2-metrics-FIX-3 — Correções matemáticas:
|
| 431 |
1. Floors explícitos em α_t e σ_t (previnem colapso topológico
|
| 432 |
após muitas épocas, quando σ_t→0 degenera o SOM em k-means).
|
|
@@ -479,6 +758,16 @@ class KohonenSOM4D:
|
|
| 479 |
|
| 480 |
self.t += 1
|
| 481 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 482 |
def finalize_fisher(self):
|
| 483 |
"""Fisher = mean((x_w - W_w)²) sobre samples acumuladas."""
|
| 484 |
cnt = self.fisher_count.clamp(min=1e-8)
|
|
@@ -498,10 +787,20 @@ class KohonenSOM4D:
|
|
| 498 |
"""Retorna métricas atuais do SOM para monitoramento."""
|
| 499 |
sigma_t = self.sigma0 * math.exp(-self.t / 1000)
|
| 500 |
alpha_t = self.alpha0 * math.exp(-self.t / 2000)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 501 |
return {
|
| 502 |
"t": int(self.t),
|
| 503 |
"sigma_t": float(sigma_t),
|
| 504 |
"alpha_t": float(alpha_t),
|
|
|
|
|
|
|
| 505 |
"sigma0": float(self.sigma0),
|
| 506 |
"alpha0": float(self.alpha0),
|
| 507 |
"lambda_ewc": float(self.lambda_ewc),
|
|
@@ -521,6 +820,17 @@ class KohonenSOM4D:
|
|
| 521 |
"fisher_accum_count": int(self.fisher_count.sum().item()),
|
| 522 |
"weights_norm": float(self.weights.norm().item()),
|
| 523 |
"weights_w_mean": float(self.weights[..., 3].mean().item()),
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 524 |
}
|
| 525 |
|
| 526 |
|
|
@@ -3246,6 +3556,130 @@ class KohonenLearningSystemV2(KohonenLearningSystem):
|
|
| 3246 |
"max_hyp_train_steps": int(self.max_hyp_train_steps),
|
| 3247 |
}
|
| 3248 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3249 |
# ==================================================================
|
| 3250 |
# V6.5-V2-metrics — Integração das métricas SOM canônicas
|
| 3251 |
# (QE, TE, Kaski-Lagus, Variância Explicada, Dead Neurons,
|
|
|
|
| 375 |
alpha0: float = 0.5,
|
| 376 |
sigma0: float = 3.0,
|
| 377 |
lambda_ewc: float = 0.01,
|
| 378 |
+
conscience_gamma: float = 0.1,
|
| 379 |
+
conscience_beta: float = 0.1,
|
| 380 |
):
|
| 381 |
# V6.5-V2-metrics-FIX-3 — α₀=0.5 e σ₀=3.0 conforme especificação
|
| 382 |
# canônica para SOM 4D (Kohonen classic):
|
|
|
|
| 386 |
# que α_t e σ_t nunca decaiam abaixo de 0.001 e 0.1 respectivamente
|
| 387 |
# (sem isso, após ~7000 updates σ→0 e o SOM degenera em k-means puro,
|
| 388 |
# perdendo preservação topológica).
|
| 389 |
+
#
|
| 390 |
+
# V6.5-V2-metrics-FIX-4 (Conscience Mechanism — DeSieno 1988):
|
| 391 |
+
# Corrige o problema neurons_active=2/864 relatado pelo usuário.
|
| 392 |
+
# Cada neurônio i mantém uma frequência de vitória p_i ∈ [0,1].
|
| 393 |
+
# BMU selection: bmu = argmin_i ( ||W_i - x||² - b_i )
|
| 394 |
+
# onde b_i = γ · (1/N - p_i) é o bias de consciência
|
| 395 |
+
# γ = 0.1 (default), N = total de neurônios (864)
|
| 396 |
+
# Update da frequência (EMA): p_i ← (1-β)·p_i + β·𝟙[i==bmu], β=0.1
|
| 397 |
+
# Quando um neurônio ganha demais (p_i > 1/N), b_i fica negativo
|
| 398 |
+
# (penaliza), e quando nunca ganha (p_i ≈ 0), b_i fica positivo
|
| 399 |
+
# (empurra para ganhar). Isto força distribuição uniforme de BMU.
|
| 400 |
self.I, self.J, self.K, self.L = grid_shape
|
| 401 |
self.alpha0 = float(alpha0)
|
| 402 |
self.sigma0 = float(sigma0)
|
|
|
|
| 409 |
self.fisher_accum = torch.zeros(self.I, self.J, self.K, self.L)
|
| 410 |
self.fisher_count = torch.zeros(self.I, self.J, self.K, self.L)
|
| 411 |
|
| 412 |
+
# V6.5-V2-metrics-FIX-4 — Conscience mechanism (DeSieno 1988)
|
| 413 |
+
# Inicializa p_i = 1/N (uniforme) — sem viés inicial.
|
| 414 |
+
# b_i = γ · (1/N - p_i) começa em 0 (sem bias).
|
| 415 |
+
self.n_neurons = self.I * self.J * self.K * self.L
|
| 416 |
+
self.conscience_gamma = float(conscience_gamma)
|
| 417 |
+
self.conscience_beta = float(conscience_beta)
|
| 418 |
+
target_p = 1.0 / self.n_neurons
|
| 419 |
+
self.win_frequency = torch.full(
|
| 420 |
+
(self.I, self.J, self.K, self.L), target_p, dtype=torch.float
|
| 421 |
+
)
|
| 422 |
+
# Contador absoluto de vitórias por neurônio (para auditoria)
|
| 423 |
+
self.bmu_win_count = torch.zeros(
|
| 424 |
+
self.I, self.J, self.K, self.L, dtype=torch.long
|
| 425 |
+
)
|
| 426 |
+
# Histórico dos últimos K BMUs (para revive_dead_neurons)
|
| 427 |
+
self._recent_bmu_flat: List[int] = []
|
| 428 |
+
|
| 429 |
def _neighborhood(self, bmu_idx):
|
| 430 |
"""Vizinhança Gaussiana 4D: d² = Δi² + Δj² + Δk² + Δl²."""
|
| 431 |
i, j, k, l = bmu_idx
|
|
|
|
| 439 |
dist_sq = (II - i) ** 2 + (JJ - j) ** 2 + (KK - k) ** 2 + (LL - l) ** 2
|
| 440 |
return dist_sq
|
| 441 |
|
| 442 |
+
def find_bmu(self, x: torch.Tensor, use_conscience: bool = True) -> Tuple[int, int, int, int]:
|
| 443 |
+
"""Best Matching Unit: argmin ||W - x||² em ℝ⁴, com conscience bias opcional.
|
| 444 |
|
| 445 |
Substitui pgvector_lookup — busca nearest-neighbor flat sobre o grid.
|
| 446 |
+
|
| 447 |
+
V6.5-V2-metrics-FIX-4 (Conscience Mechanism — DeSieno 1988):
|
| 448 |
+
Quando use_conscience=True, adiciona bias b_i = γ·(1/N - p_i) à
|
| 449 |
+
distância, forçando distribuição uniforme de BMU entre os 864
|
| 450 |
+
neurônios. Isto corrige o problema neurons_active=2/864:
|
| 451 |
+
|
| 452 |
+
bmu = argmin_i ( ||W_i - x||² - b_i )
|
| 453 |
+
|
| 454 |
+
Sem conscience, neurônios próximos a poucas amostras ficam mortos.
|
| 455 |
+
Com conscience, neurônios que ganham demais (p_i > 1/N) são
|
| 456 |
+
penalizados (b_i < 0 aumenta distância efetiva), e neurônios que
|
| 457 |
+
nunca ganham (p_i ≈ 0) são favorecidos (b_i > 0 reduz distância).
|
| 458 |
+
|
| 459 |
+
Args:
|
| 460 |
+
x: tensor [4] — vetor de entrada.
|
| 461 |
+
use_conscience: se True, aplica bias de consciência (default True).
|
| 462 |
"""
|
| 463 |
+
# Sanitiza x antes de computar distância
|
| 464 |
+
if torch.isnan(x).any() or torch.isinf(x).any():
|
| 465 |
+
x = torch.nan_to_num(x, nan=0.0, posinf=1e4, neginf=-1e4)
|
| 466 |
dist = torch.sum((self.weights - x.view(1, 1, 1, 1, 4)) ** 2, dim=-1)
|
| 467 |
+
if use_conscience:
|
| 468 |
+
# b_i = γ · (1/N - p_i) → subtraído da distância (menor distância efetiva vence)
|
| 469 |
+
target_p = 1.0 / self.n_neurons
|
| 470 |
+
bias = self.conscience_gamma * (target_p - self.win_frequency)
|
| 471 |
+
dist_effective = dist - bias
|
| 472 |
+
flat_idx = int(torch.argmin(dist_effective).item())
|
| 473 |
+
else:
|
| 474 |
+
flat_idx = int(torch.argmin(dist).item())
|
| 475 |
i = flat_idx // (self.J * self.K * self.L)
|
| 476 |
rest = flat_idx % (self.J * self.K * self.L)
|
| 477 |
j = rest // (self.K * self.L)
|
|
|
|
| 480 |
l = rest % self.L
|
| 481 |
return (i, j, k, l)
|
| 482 |
|
| 483 |
+
def update_win_frequency(self, bmu_idx: Tuple[int, int, int, int]) -> None:
|
| 484 |
+
"""V6.5-V2-metrics-FIX-4 — Atualiza frequência de vitória (EMA).
|
| 485 |
+
|
| 486 |
+
p_i ← (1-β)·p_i + β·𝟙[i==bmu], β = conscience_beta (default 0.1)
|
| 487 |
+
|
| 488 |
+
Também mantém contador absoluto bmu_win_count para auditoria.
|
| 489 |
+
Deve ser chamado APÓS find_bmu e APÓS update_weights.
|
| 490 |
+
|
| 491 |
+
Args:
|
| 492 |
+
bmu_idx: (i, j, k, l) — índice do BMU selecionado.
|
| 493 |
+
"""
|
| 494 |
+
with torch.no_grad():
|
| 495 |
+
# EMA update
|
| 496 |
+
self.win_frequency = (
|
| 497 |
+
(1.0 - self.conscience_beta) * self.win_frequency
|
| 498 |
+
)
|
| 499 |
+
self.win_frequency[bmu_idx] += self.conscience_beta
|
| 500 |
+
# Contador absoluto
|
| 501 |
+
self.bmu_win_count[bmu_idx] += 1
|
| 502 |
+
# Histórico recente (para revive_dead_neurons)
|
| 503 |
+
flat = (
|
| 504 |
+
bmu_idx[0] * (self.J * self.K * self.L)
|
| 505 |
+
+ bmu_idx[1] * (self.K * self.L)
|
| 506 |
+
+ bmu_idx[2] * self.L
|
| 507 |
+
+ bmu_idx[3]
|
| 508 |
+
)
|
| 509 |
+
self._recent_bmu_flat.append(flat)
|
| 510 |
+
# Mantém últimos 200 BMUs
|
| 511 |
+
if len(self._recent_bmu_flat) > 200:
|
| 512 |
+
self._recent_bmu_flat = self._recent_bmu_flat[-200:]
|
| 513 |
+
|
| 514 |
+
def revive_dead_neurons(
|
| 515 |
+
self,
|
| 516 |
+
data_buffer: Optional[List[torch.Tensor]] = None,
|
| 517 |
+
dead_threshold: int = 0,
|
| 518 |
+
) -> Dict[str, Any]:
|
| 519 |
+
"""V6.5-V2-metrics-FIX-4 — Revive neurônios mortos reinicializando pesos.
|
| 520 |
+
|
| 521 |
+
User requirement: "ANALISAR matematicamente e logicamente a ativação e
|
| 522 |
+
uso e acesso dos neurônios (apenas dois estão sendo ativados:
|
| 523 |
+
neurons_active=2/864) distribuindo o processamento paralelamente".
|
| 524 |
+
|
| 525 |
+
Um neurônio é considerado "morto" se bmu_win_count <= dead_threshold
|
| 526 |
+
(nunca ou raramente foi BMU). Para cada neurônio morto:
|
| 527 |
+
|
| 528 |
+
1. Se data_buffer fornecido: amostra um vetor aleatório do buffer e
|
| 529 |
+
atribui aos pesos do neurônio (reinicialização data-driven).
|
| 530 |
+
2. Se buffer vazio: reinicializa com ruído gaussiano pequeno (N(0, 0.1)).
|
| 531 |
+
3. Reseta win_frequency para 1/N (sem bias) e bmu_win_count para 0.
|
| 532 |
+
|
| 533 |
+
Isto garante que TODOS os 864 neurônios sejam utilizados, distribuindo
|
| 534 |
+
o processamento paralelo do SOM conforme solicitado.
|
| 535 |
+
|
| 536 |
+
Args:
|
| 537 |
+
data_buffer: lista de tensores [4] — amostras do buffer_4d do KLS.
|
| 538 |
+
dead_threshold: neurônios com win_count <= threshold são revividos.
|
| 539 |
+
|
| 540 |
+
Returns:
|
| 541 |
+
Dict com: n_revived, n_total, revived_indices, dead_rate_before, dead_rate_after.
|
| 542 |
+
"""
|
| 543 |
+
with torch.no_grad():
|
| 544 |
+
dead_mask = self.bmu_win_count <= dead_threshold
|
| 545 |
+
n_dead = int(dead_mask.sum().item())
|
| 546 |
+
n_total = self.n_neurons
|
| 547 |
+
dead_rate_before = float(n_dead / n_total)
|
| 548 |
+
|
| 549 |
+
revived_indices: List[Tuple[int, int, int, int]] = []
|
| 550 |
+
if n_dead == 0:
|
| 551 |
+
return {
|
| 552 |
+
"n_revived": 0,
|
| 553 |
+
"n_total": n_total,
|
| 554 |
+
"revived_indices": [],
|
| 555 |
+
"dead_rate_before": dead_rate_before,
|
| 556 |
+
"dead_rate_after": dead_rate_before,
|
| 557 |
+
"action": "none",
|
| 558 |
+
}
|
| 559 |
+
|
| 560 |
+
# Amostra pontos do buffer se disponível
|
| 561 |
+
buffer_tensor = None
|
| 562 |
+
if data_buffer and len(data_buffer) > 0:
|
| 563 |
+
try:
|
| 564 |
+
buffer_tensor = torch.stack(
|
| 565 |
+
[v.detach().clone() if isinstance(v, torch.Tensor)
|
| 566 |
+
else torch.tensor(v, dtype=torch.float)
|
| 567 |
+
for v in data_buffer]
|
| 568 |
+
).float()
|
| 569 |
+
except Exception:
|
| 570 |
+
buffer_tensor = None
|
| 571 |
+
|
| 572 |
+
# Itera sobre neurônios mortos e reinicializa
|
| 573 |
+
dead_indices = dead_mask.nonzero(as_tuple=False)
|
| 574 |
+
for idx_tensor in dead_indices:
|
| 575 |
+
i, j, k, l = idx_tensor.tolist()
|
| 576 |
+
if buffer_tensor is not None and buffer_tensor.shape[0] > 0:
|
| 577 |
+
# Amostra aleatória do buffer
|
| 578 |
+
sample_idx = torch.randint(0, buffer_tensor.shape[0], (1,)).item()
|
| 579 |
+
new_w = buffer_tensor[sample_idx].clone()
|
| 580 |
+
# Pequeno ruído para evitar duplicação exata
|
| 581 |
+
new_w = new_w + 0.05 * torch.randn(4)
|
| 582 |
+
else:
|
| 583 |
+
# Reinicialização gaussiana pequena
|
| 584 |
+
new_w = 0.1 * torch.randn(4)
|
| 585 |
+
# Clamp para segurança
|
| 586 |
+
new_w = torch.clamp(new_w, -10.0, 10.0)
|
| 587 |
+
self.weights[i, j, k, l] = new_w
|
| 588 |
+
# Reset counters
|
| 589 |
+
self.win_frequency[i, j, k, l] = 1.0 / n_total
|
| 590 |
+
self.bmu_win_count[i, j, k, l] = 0
|
| 591 |
+
revived_indices.append((i, j, k, l))
|
| 592 |
+
|
| 593 |
+
# Recalcula dead rate após revival
|
| 594 |
+
new_dead_mask = self.bmu_win_count <= dead_threshold
|
| 595 |
+
n_dead_after = int(new_dead_mask.sum().item())
|
| 596 |
+
dead_rate_after = float(n_dead_after / n_total)
|
| 597 |
+
|
| 598 |
+
return {
|
| 599 |
+
"n_revived": len(revived_indices),
|
| 600 |
+
"n_total": n_total,
|
| 601 |
+
"revived_indices": revived_indices[:50], # top 50 para log
|
| 602 |
+
"n_dead_before": n_dead,
|
| 603 |
+
"n_dead_after": n_dead_after,
|
| 604 |
+
"dead_rate_before": dead_rate_before,
|
| 605 |
+
"dead_rate_after": dead_rate_after,
|
| 606 |
+
"action": "revived" if revived_indices else "none",
|
| 607 |
+
"used_buffer": buffer_tensor is not None,
|
| 608 |
+
}
|
| 609 |
+
|
| 610 |
+
def parallel_neuron_activation_report(self) -> Dict[str, Any]:
|
| 611 |
+
"""V6.5-V2-metrics-FIX-4 — Relatório estruturado de ativação dos 864 neurônios.
|
| 612 |
+
|
| 613 |
+
User requirement: "distribuindo o processamento paralelamente" + o
|
| 614 |
+
exemplo de código fornecido mostra estruturação vetorizada para
|
| 615 |
+
análise estatística e auditoria do modelo.
|
| 616 |
+
|
| 617 |
+
Este método produz um relatório análogo ao DataFrame do exemplo,
|
| 618 |
+
mas otimizado para o grid 4D (6,6,6,4) com 864 neurônios:
|
| 619 |
+
|
| 620 |
+
Returns:
|
| 621 |
+
Dict com:
|
| 622 |
+
- n_total_neurons: int (864)
|
| 623 |
+
- n_active_neurons: int (vitória em ≥1 amostra histórica)
|
| 624 |
+
- n_dead_neurons: int (nunca foi BMU)
|
| 625 |
+
- neuron_activation_rate: float
|
| 626 |
+
- max_win_count: int (neurônio mais ativo)
|
| 627 |
+
- min_win_count: int (neurônio menos ativo)
|
| 628 |
+
- mean_win_count: float
|
| 629 |
+
- std_win_count: float
|
| 630 |
+
- max_win_frequency: float
|
| 631 |
+
- min_win_frequency: float
|
| 632 |
+
- bmu_distribution_top20: dict {flat_idx: count}
|
| 633 |
+
- bmu_distribution_bottom20: dict {flat_idx: count} (mortos)
|
| 634 |
+
- conscience_bias_mean: float (deve tender a 0 se uniforme)
|
| 635 |
+
- conscience_bias_std: float (deve tender a 0 se uniforme)
|
| 636 |
+
- uniformity_score: float (1 - CV da win_frequency, ∈ [0,1])
|
| 637 |
+
"""
|
| 638 |
+
with torch.no_grad():
|
| 639 |
+
win_counts_flat = self.bmu_win_count.flatten().float()
|
| 640 |
+
win_freq_flat = self.win_frequency.flatten()
|
| 641 |
+
|
| 642 |
+
n_total = self.n_neurons
|
| 643 |
+
n_active = int((self.bmu_win_count > 0).sum().item())
|
| 644 |
+
n_dead = n_total - n_active
|
| 645 |
+
|
| 646 |
+
# Estatísticas
|
| 647 |
+
if n_total > 0:
|
| 648 |
+
max_wc = float(win_counts_flat.max().item())
|
| 649 |
+
min_wc = float(win_counts_flat.min().item())
|
| 650 |
+
mean_wc = float(win_counts_flat.mean().item())
|
| 651 |
+
std_wc = float(win_counts_flat.std().item())
|
| 652 |
+
max_wf = float(win_freq_flat.max().item())
|
| 653 |
+
min_wf = float(win_freq_flat.min().item())
|
| 654 |
+
# Uniformidade: 1 - CV (coeficiente de variação)
|
| 655 |
+
cv = float(std_wc / max(mean_wc, 1e-8))
|
| 656 |
+
uniformity = max(0.0, 1.0 - cv)
|
| 657 |
+
else:
|
| 658 |
+
max_wc = min_wc = mean_wc = std_wc = 0.0
|
| 659 |
+
max_wf = min_wf = 0.0
|
| 660 |
+
uniformity = 0.0
|
| 661 |
+
|
| 662 |
+
# Bias de consciência: b_i = γ · (1/N - p_i)
|
| 663 |
+
target_p = 1.0 / n_total
|
| 664 |
+
bias_flat = self.conscience_gamma * (target_p - win_freq_flat)
|
| 665 |
+
bias_mean = float(bias_flat.mean().item())
|
| 666 |
+
bias_std = float(bias_flat.std().item())
|
| 667 |
+
|
| 668 |
+
# Top-20 BMUs mais frequentes
|
| 669 |
+
from collections import Counter
|
| 670 |
+
recent_counter = Counter(self._recent_bmu_flat)
|
| 671 |
+
top20 = dict(recent_counter.most_common(20))
|
| 672 |
+
|
| 673 |
+
# Bottom-20 (mortos ou raros) — últimos em vitórias absolutas
|
| 674 |
+
sorted_indices = torch.argsort(win_counts_flat)
|
| 675 |
+
bottom20_idx = sorted_indices[:20].tolist()
|
| 676 |
+
bottom20 = {
|
| 677 |
+
int(idx): int(self.bmu_win_count.flatten()[idx].item())
|
| 678 |
+
for idx in bottom20_idx
|
| 679 |
+
}
|
| 680 |
+
|
| 681 |
+
return {
|
| 682 |
+
"n_total_neurons": int(n_total),
|
| 683 |
+
"n_active_neurons": int(n_active),
|
| 684 |
+
"n_dead_neurons": int(n_dead),
|
| 685 |
+
"neuron_activation_rate": float(n_active / max(1, n_total)),
|
| 686 |
+
"max_win_count": max_wc,
|
| 687 |
+
"min_win_count": min_wc,
|
| 688 |
+
"mean_win_count": mean_wc,
|
| 689 |
+
"std_win_count": std_wc,
|
| 690 |
+
"max_win_frequency": max_wf,
|
| 691 |
+
"min_win_frequency": min_wf,
|
| 692 |
+
"bmu_distribution_top20": {str(k): int(v) for k, v in top20.items()},
|
| 693 |
+
"bmu_distribution_bottom20": {str(k): int(v) for k, v in bottom20.items()},
|
| 694 |
+
"conscience_bias_mean": bias_mean,
|
| 695 |
+
"conscience_bias_std": bias_std,
|
| 696 |
+
"uniformity_score": uniformity,
|
| 697 |
+
"conscience_gamma": float(self.conscience_gamma),
|
| 698 |
+
"conscience_beta": float(self.conscience_beta),
|
| 699 |
+
"n_recent_bmus_tracked": int(len(self._recent_bmu_flat)),
|
| 700 |
+
}
|
| 701 |
+
|
| 702 |
def update_weights(self, x: torch.Tensor, bmu_idx, accumulate_fisher=False):
|
| 703 |
"""Update Kohonen: ΔW = α·Λ·(x - W) + penalidade EWC em w.
|
| 704 |
|
| 705 |
+
V6.5-V2-metrics-FIX-4 — agora chama update_win_frequency automaticamente
|
| 706 |
+
após o update, garantindo que o conscience mechanism seja atualizado
|
| 707 |
+
a cada amostra processada (sem necessidade de chamada externa).
|
| 708 |
+
|
| 709 |
V6.5-V2-metrics-FIX-3 — Correções matemáticas:
|
| 710 |
1. Floors explícitos em α_t e σ_t (previnem colapso topológico
|
| 711 |
após muitas épocas, quando σ_t→0 degenera o SOM em k-means).
|
|
|
|
| 758 |
|
| 759 |
self.t += 1
|
| 760 |
|
| 761 |
+
# V6.5-V2-metrics-FIX-4 — atualiza conscience mechanism (win frequency)
|
| 762 |
+
# automaticamente após cada update. Isto garante que o bias de consciência
|
| 763 |
+
# seja aplicado corretamente no próximo find_bmu, forçando distribuição
|
| 764 |
+
# uniforme de BMU entre os 864 neurônios.
|
| 765 |
+
try:
|
| 766 |
+
self.update_win_frequency(bmu_idx)
|
| 767 |
+
except Exception:
|
| 768 |
+
# Não deixa falha no conscience quebrar o treino principal
|
| 769 |
+
pass
|
| 770 |
+
|
| 771 |
def finalize_fisher(self):
|
| 772 |
"""Fisher = mean((x_w - W_w)²) sobre samples acumuladas."""
|
| 773 |
cnt = self.fisher_count.clamp(min=1e-8)
|
|
|
|
| 787 |
"""Retorna métricas atuais do SOM para monitoramento."""
|
| 788 |
sigma_t = self.sigma0 * math.exp(-self.t / 1000)
|
| 789 |
alpha_t = self.alpha0 * math.exp(-self.t / 2000)
|
| 790 |
+
# V6.5-V2-metrics-FIX-4 — floors aplicados (consistência com update_weights)
|
| 791 |
+
sigma_t_eff = max(sigma_t, 0.1)
|
| 792 |
+
alpha_t_eff = max(alpha_t, 0.001)
|
| 793 |
+
# V6.5-V2-metrics-FIX-4 — estatísticas do conscience mechanism
|
| 794 |
+
n_total = self.n_neurons
|
| 795 |
+
n_active = int((self.bmu_win_count > 0).sum().item())
|
| 796 |
+
n_dead = n_total - n_active
|
| 797 |
+
win_counts_flat = self.bmu_win_count.flatten().float()
|
| 798 |
return {
|
| 799 |
"t": int(self.t),
|
| 800 |
"sigma_t": float(sigma_t),
|
| 801 |
"alpha_t": float(alpha_t),
|
| 802 |
+
"sigma_t_effective": float(sigma_t_eff),
|
| 803 |
+
"alpha_t_effective": float(alpha_t_eff),
|
| 804 |
"sigma0": float(self.sigma0),
|
| 805 |
"alpha0": float(self.alpha0),
|
| 806 |
"lambda_ewc": float(self.lambda_ewc),
|
|
|
|
| 820 |
"fisher_accum_count": int(self.fisher_count.sum().item()),
|
| 821 |
"weights_norm": float(self.weights.norm().item()),
|
| 822 |
"weights_w_mean": float(self.weights[..., 3].mean().item()),
|
| 823 |
+
# V6.5-V2-metrics-FIX-4 — conscience mechanism status
|
| 824 |
+
"conscience_gamma": float(self.conscience_gamma),
|
| 825 |
+
"conscience_beta": float(self.conscience_beta),
|
| 826 |
+
"n_active_neurons": n_active,
|
| 827 |
+
"n_dead_neurons": n_dead,
|
| 828 |
+
"neuron_activation_rate": float(n_active / max(1, n_total)),
|
| 829 |
+
"bmu_win_count_mean": float(win_counts_flat.mean().item()) if n_total > 0 else 0.0,
|
| 830 |
+
"bmu_win_count_max": float(win_counts_flat.max().item()) if n_total > 0 else 0.0,
|
| 831 |
+
"win_frequency_mean": float(self.win_frequency.mean().item()),
|
| 832 |
+
"win_frequency_max": float(self.win_frequency.max().item()),
|
| 833 |
+
"win_frequency_min": float(self.win_frequency.min().item()),
|
| 834 |
}
|
| 835 |
|
| 836 |
|
|
|
|
| 3556 |
"max_hyp_train_steps": int(self.max_hyp_train_steps),
|
| 3557 |
}
|
| 3558 |
|
| 3559 |
+
# ==================================================================
|
| 3560 |
+
# V6.5-V2-metrics-FIX-4 — Conscience mechanism + Dead neuron revival
|
| 3561 |
+
# ==================================================================
|
| 3562 |
+
# User requirement (latest): "APRIMORAR (em ambas as FASE1 e FASE2):
|
| 3563 |
+
# analisar matematicamente e logicamente a ativação e uso e acesso dos
|
| 3564 |
+
# neurônios (apenas dois estão sendo ativados: neurons_active=2/864)
|
| 3565 |
+
# distribuindo o processamento paralelamente".
|
| 3566 |
+
#
|
| 3567 |
+
# Mathematical formulation (Conscience Mechanism — DeSieno 1988):
|
| 3568 |
+
#
|
| 3569 |
+
# Cada neurônio i mantém win_frequency p_i ∈ [0,1] (EMA, β=0.1).
|
| 3570 |
+
# BMU selection: bmu = argmin_i ( ||W_i - x||² - b_i )
|
| 3571 |
+
# onde b_i = γ · (1/N - p_i) é o conscience bias
|
| 3572 |
+
# γ = 0.1 (default), N = total de neurônios (864)
|
| 3573 |
+
#
|
| 3574 |
+
# Quando p_i > 1/N (neurônio ganha demais): b_i < 0 → distância
|
| 3575 |
+
# efetiva AUMENTA → neurônio é penalizado.
|
| 3576 |
+
# Quando p_i < 1/N (neurônio nunca ganha): b_i > 0 → distância
|
| 3577 |
+
# efetiva DIMINUI → neurônio é favorecido.
|
| 3578 |
+
#
|
| 3579 |
+
# Convergência: p_i → 1/N para todo i (distribuição uniforme de BMU),
|
| 3580 |
+
# garantindo que TODOS os 864 neurônios sejam utilizados.
|
| 3581 |
+
#
|
| 3582 |
+
# Dead neuron revival (complementar):
|
| 3583 |
+
# Se após K amostras um neurônio ainda tem win_count = 0, ele é
|
| 3584 |
+
# reinicializado para uma amostra aleatória do buffer (data-driven)
|
| 3585 |
+
# ou para N(0, 0.1) se buffer vazio. Isto acelera a diversificação
|
| 3586 |
+
# quando o conscience mechanism sozinho não basta.
|
| 3587 |
+
# ------------------------------------------------------------------
|
| 3588 |
+
def revive_dead_neurons(
|
| 3589 |
+
self,
|
| 3590 |
+
dead_threshold: int = 0,
|
| 3591 |
+
use_buffer: bool = True,
|
| 3592 |
+
) -> Dict[str, Any]:
|
| 3593 |
+
"""V6.5-V2-metrics-FIX-4 — Wrapper KLS para som.revive_dead_neurons.
|
| 3594 |
+
|
| 3595 |
+
Usa o buffer_4d atual do KLS como fonte de dados para reinicialização
|
| 3596 |
+
data-driven dos neurônios mortos.
|
| 3597 |
+
|
| 3598 |
+
Args:
|
| 3599 |
+
dead_threshold: neurônios com win_count <= threshold são revividos.
|
| 3600 |
+
use_buffer: se True, usa buffer_4d do KLS como fonte.
|
| 3601 |
+
|
| 3602 |
+
Returns:
|
| 3603 |
+
Dict com status do revival (n_revived, dead_rate_before/after, etc).
|
| 3604 |
+
"""
|
| 3605 |
+
data_buffer = self.buffer_4d if use_buffer else None
|
| 3606 |
+
return self.som.revive_dead_neurons(
|
| 3607 |
+
data_buffer=data_buffer,
|
| 3608 |
+
dead_threshold=dead_threshold,
|
| 3609 |
+
)
|
| 3610 |
+
|
| 3611 |
+
def parallel_neuron_activation_report(self) -> Dict[str, Any]:
|
| 3612 |
+
"""V6.5-V2-metrics-FIX-4 — Wrapper KLS para som.parallel_neuron_activation_report.
|
| 3613 |
+
|
| 3614 |
+
Retorna relatório estruturado de ativação dos 864 neurônios, análogo
|
| 3615 |
+
ao DataFrame do exemplo fornecido pelo usuário, mas otimizado para
|
| 3616 |
+
o grid 4D do SOM.
|
| 3617 |
+
|
| 3618 |
+
Returns:
|
| 3619 |
+
Dict com estatísticas detalhadas (n_active, n_dead, win_count
|
| 3620 |
+
distribution, conscience_bias stats, uniformity_score, etc).
|
| 3621 |
+
"""
|
| 3622 |
+
return self.som.parallel_neuron_activation_report()
|
| 3623 |
+
|
| 3624 |
+
def auto_revive_if_needed(
|
| 3625 |
+
self,
|
| 3626 |
+
dead_rate_threshold: float = 0.5,
|
| 3627 |
+
min_steps_between_revivals: int = 200,
|
| 3628 |
+
) -> Dict[str, Any]:
|
| 3629 |
+
"""V6.5-V2-metrics-FIX-4 — Revive neurônios automaticamente se dead_rate alto.
|
| 3630 |
+
|
| 3631 |
+
Verifica o dead_rate atual e revive neurônios mortos se:
|
| 3632 |
+
- dead_rate > dead_rate_threshold (default 0.5 = 50% mortos)
|
| 3633 |
+
- pelo menos min_steps_between_revivals desde o último revival
|
| 3634 |
+
|
| 3635 |
+
Isto é chamado automaticamente pelo treinador após cada chunk,
|
| 3636 |
+
garantindo que o SOM mantenha distribuição uniforme de BMU ao longo
|
| 3637 |
+
do treino sem intervenção manual.
|
| 3638 |
+
|
| 3639 |
+
Args:
|
| 3640 |
+
dead_rate_threshold: limite para disparar revival (default 0.5).
|
| 3641 |
+
min_steps_between_revivals: cooldown em steps (default 200).
|
| 3642 |
+
|
| 3643 |
+
Returns:
|
| 3644 |
+
Dict com status do revival (ou action="skipped" se não disparou).
|
| 3645 |
+
"""
|
| 3646 |
+
last_revival_step = getattr(self, "_last_revival_step", -min_steps_between_revivals)
|
| 3647 |
+
current_step = self.som.t
|
| 3648 |
+
steps_since_last = current_step - last_revival_step
|
| 3649 |
+
|
| 3650 |
+
# Computa dead rate atual
|
| 3651 |
+
report = self.som.parallel_neuron_activation_report()
|
| 3652 |
+
dead_rate = 1.0 - report["neuron_activation_rate"]
|
| 3653 |
+
|
| 3654 |
+
if dead_rate <= dead_rate_threshold:
|
| 3655 |
+
return {
|
| 3656 |
+
"action": "skipped",
|
| 3657 |
+
"reason": f"dead_rate={dead_rate:.3f} <= threshold={dead_rate_threshold}",
|
| 3658 |
+
"dead_rate": dead_rate,
|
| 3659 |
+
"n_active": report["n_active_neurons"],
|
| 3660 |
+
"n_total": report["n_total_neurons"],
|
| 3661 |
+
}
|
| 3662 |
+
if steps_since_last < min_steps_between_revivals:
|
| 3663 |
+
return {
|
| 3664 |
+
"action": "skipped",
|
| 3665 |
+
"reason": f"cooldown: only {steps_since_last} steps since last revival "
|
| 3666 |
+
f"(need {min_steps_between_revivals})",
|
| 3667 |
+
"dead_rate": dead_rate,
|
| 3668 |
+
"n_active": report["n_active_neurons"],
|
| 3669 |
+
"n_total": report["n_total_neurons"],
|
| 3670 |
+
}
|
| 3671 |
+
|
| 3672 |
+
# Dispara revival
|
| 3673 |
+
revival = self.revive_dead_neurons(
|
| 3674 |
+
dead_threshold=0,
|
| 3675 |
+
use_buffer=True,
|
| 3676 |
+
)
|
| 3677 |
+
self._last_revival_step = current_step
|
| 3678 |
+
revival["action"] = "auto_revived"
|
| 3679 |
+
revival["trigger_dead_rate"] = dead_rate
|
| 3680 |
+
revival["steps_since_last_revival"] = steps_since_last
|
| 3681 |
+
return revival
|
| 3682 |
+
|
| 3683 |
# ==================================================================
|
| 3684 |
# V6.5-V2-metrics — Integração das métricas SOM canônicas
|
| 3685 |
# (QE, TE, Kaski-Lagus, Variância Explicada, Dead Neurons,
|
train_v6_5_v2.py
CHANGED
|
@@ -100,6 +100,27 @@ logger = logging.getLogger("train_v6_5_v2")
|
|
| 100 |
os.environ["V65_ENABLE_STREAMING"] = "1"
|
| 101 |
logger.info(f"[V6.5-V2] V65_ENABLE_STREAMING={os.environ['V65_ENABLE_STREAMING']} (forced)")
|
| 102 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 103 |
sys.path.insert(0, str(SRC_ROOT))
|
| 104 |
|
| 105 |
from bigru_t.utils.xeon_runtime import ( # noqa: E402
|
|
@@ -191,7 +212,11 @@ LOSS_HISTORY_WINDOW = 8
|
|
| 191 |
PUNISHMENT_WINDOW = 12
|
| 192 |
|
| 193 |
# V2-dynamic-memory — Buffer sliding window (evita OOM em treino longo)
|
| 194 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 195 |
|
| 196 |
# Streaming — User requirement: "streaming de 100 em 100 samples"
|
| 197 |
STREAM_BATCH_SIZE = 100
|
|
@@ -205,13 +230,32 @@ MAX_SAMPLES_PUNICAO = 2000
|
|
| 205 |
# User requirement: "não reduzir tempo e não gerar dados sintéticos" +
|
| 206 |
# "todo streaming (FASE1 e da FASE2) deve ter pausa para dar tempo de conclusão
|
| 207 |
# de processamento continuando após conclusão"
|
| 208 |
-
# V6.5-V2-metrics-FIX-
|
| 209 |
-
#
|
| 210 |
-
#
|
| 211 |
-
|
| 212 |
-
|
| 213 |
-
|
| 214 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 215 |
|
| 216 |
# Storage critical
|
| 217 |
STORAGE_CRITICAL_PCT = 90
|
|
@@ -1158,13 +1202,13 @@ def run_fase_conhecimento(
|
|
| 1158 |
})
|
| 1159 |
|
| 1160 |
step += 1
|
| 1161 |
-
time.sleep(
|
| 1162 |
except Exception as e:
|
| 1163 |
logger.error(f"[V6.5-V2] Batch error: {e}")
|
| 1164 |
traceback.print_exc()
|
| 1165 |
continue
|
| 1166 |
|
| 1167 |
-
time.sleep(
|
| 1168 |
# V2-dynamic-memory — Buffer sliding window: trunca para os
|
| 1169 |
# últimos MAX_BUFFER_SIZE amostras após cada chunk.
|
| 1170 |
# Os pesos do SOM já capturam o conhecimento acumulado,
|
|
@@ -1241,6 +1285,31 @@ def run_fase_conhecimento(
|
|
| 1241 |
except Exception as e:
|
| 1242 |
logger.warning(f"[V6.5-V2-metrics-FIX-2] Failed to compute SOM metrics: {e}")
|
| 1243 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1244 |
# Aggressive memory cleanup between chunks
|
| 1245 |
if chunk_idx_global % 2 == 0:
|
| 1246 |
aggressive_memory_cleanup()
|
|
@@ -1253,13 +1322,13 @@ def run_fase_conhecimento(
|
|
| 1253 |
# V6.5-V2-metrics-FIX: pausa pós-processamento para dar tempo
|
| 1254 |
# de conclusão (user requirement: "todo streaming deve ter
|
| 1255 |
# pausa para dar tempo de conclusão de processamento").
|
| 1256 |
-
time.sleep(
|
| 1257 |
except Exception as e:
|
| 1258 |
logger.error(f"[V6.5-V2] Dataset {dataset_name} failed: {e}")
|
| 1259 |
traceback.print_exc()
|
| 1260 |
streaming_failures[dataset_name] += 1
|
| 1261 |
|
| 1262 |
-
time.sleep(
|
| 1263 |
# V6.5-V2-metrics-FIX-2 — Save state after each dataset to preserve
|
| 1264 |
# progress in case the process is killed by container timeout.
|
| 1265 |
# User requirement: "o estado do modelo deve ser contínuo" — saving
|
|
@@ -1340,6 +1409,266 @@ def run_fase_conhecimento(
|
|
| 1340 |
}
|
| 1341 |
|
| 1342 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1343 |
# ============================================================================
|
| 1344 |
# 10. PHASE 2 — TREINAMENTO COM PUNIÇÃO (16 hipóteses × 3 tentativas)
|
| 1345 |
# ============================================================================
|
|
@@ -1606,13 +1935,35 @@ def run_fase_punicão(
|
|
| 1606 |
traceback.print_exc()
|
| 1607 |
|
| 1608 |
step += 1
|
| 1609 |
-
|
|
|
|
|
|
|
|
|
|
| 1610 |
except Exception as e:
|
| 1611 |
logger.error(f"[V6.5-V2] Batch error in PUNIÇÃO: {e}")
|
| 1612 |
traceback.print_exc()
|
| 1613 |
continue
|
| 1614 |
|
| 1615 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1616 |
# V2-dynamic-memory — Buffer sliding window na PUNIÇÃO também
|
| 1617 |
if len(kls.buffer_4d) > MAX_BUFFER_SIZE:
|
| 1618 |
overflow = len(kls.buffer_4d) - MAX_BUFFER_SIZE
|
|
@@ -1629,8 +1980,9 @@ def run_fase_punicão(
|
|
| 1629 |
kls.aggressive_cleanup()
|
| 1630 |
except Exception:
|
| 1631 |
pass
|
| 1632 |
-
# V6.5-V2-metrics-FIX: pausa pós-processamento
|
| 1633 |
-
|
|
|
|
| 1634 |
except Exception as e:
|
| 1635 |
logger.error(f"[V6.5-V2] PUNIÇÃO dataset failed: {e}")
|
| 1636 |
traceback.print_exc()
|
|
@@ -1645,6 +1997,24 @@ def run_fase_punicão(
|
|
| 1645 |
logger.info(f" Hypotheses trainings: {len(hypotheses_log)}")
|
| 1646 |
logger.info(f" Delta applications: {len(delta_applications_log)}")
|
| 1647 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1648 |
return {
|
| 1649 |
"phase": "TREINAMENTO_COM_PUNICAO",
|
| 1650 |
"dataset": PUNICAO_DATASET,
|
|
@@ -1666,6 +2036,8 @@ def run_fase_punicão(
|
|
| 1666 |
),
|
| 1667 |
"som_metric_history": kls.get_som_metric_history(),
|
| 1668 |
"final_v2_state": kls.get_v2_metrics(),
|
|
|
|
|
|
|
| 1669 |
}
|
| 1670 |
|
| 1671 |
|
|
@@ -2108,7 +2480,17 @@ def main() -> int:
|
|
| 2108 |
# + Goose VQ, fornecendo representação compacta do estado do SOM.
|
| 2109 |
# O compressor já sanitiza NaN/Inf internamente (torch.nan_to_num).
|
| 2110 |
enable_vqvae2=True, # REATIVADO (was False in V6.5-V2-metrics-FIX)
|
| 2111 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2112 |
enable_w8a8=False, # W8A8 permanece desabilitado (não essencial para Kohonen)
|
| 2113 |
vqvae2_code_dim=16,
|
| 2114 |
vqvae2_num_codes_top=64,
|
|
|
|
| 100 |
os.environ["V65_ENABLE_STREAMING"] = "1"
|
| 101 |
logger.info(f"[V6.5-V2] V65_ENABLE_STREAMING={os.environ['V65_ENABLE_STREAMING']} (forced)")
|
| 102 |
|
| 103 |
+
# V6.5-V2-metrics-FIX-4 — HF datasets memory optimization (OOM-killer mitigation)
|
| 104 |
+
# User requirement: "o processo vem sendo morto OOM-kiler (Out of memory) devido
|
| 105 |
+
# algum bug de lógica ou falta de otimização que deve ser investigado".
|
| 106 |
+
# Estas variáveis reduzem o consumo de memória do HF datasets library:
|
| 107 |
+
# - HF_DATASETS_DISABLE_IN_MEMORY_CACHE: não cacheia datasets em RAM
|
| 108 |
+
# - HF_DATASETS_OFFLINE=0: permite streaming mas não força cache local
|
| 109 |
+
# - DATASETS_FINGERPRINT_CACHING_DISABLED: skipa fingerprinting (CPU/memory)
|
| 110 |
+
# - TOKENIZERS_PARALLELISM=false: evita spawn de processos paralelos
|
| 111 |
+
# - HF_HUB_DISABLE_TELEMETRY: desabilita telemetria (CPU/network)
|
| 112 |
+
os.environ["HF_DATASETS_DISABLE_IN_MEMORY_CACHE"] = "1"
|
| 113 |
+
os.environ["DATASETS_FINGERPRINT_CACHING_DISABLED"] = "1"
|
| 114 |
+
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
| 115 |
+
os.environ["HF_HUB_DISABLE_TELEMETRY"] = "1"
|
| 116 |
+
os.environ.setdefault("HF_DATASETS_CACHE", "/tmp/hf_datasets_cache_v65")
|
| 117 |
+
# Cria o dir de cache se não existir (limpa cache antigo periodicamente)
|
| 118 |
+
try:
|
| 119 |
+
cache_dir = Path(os.environ["HF_DATASETS_CACHE"])
|
| 120 |
+
cache_dir.mkdir(parents=True, exist_ok=True)
|
| 121 |
+
except Exception:
|
| 122 |
+
pass
|
| 123 |
+
|
| 124 |
sys.path.insert(0, str(SRC_ROOT))
|
| 125 |
|
| 126 |
from bigru_t.utils.xeon_runtime import ( # noqa: E402
|
|
|
|
| 212 |
PUNISHMENT_WINDOW = 12
|
| 213 |
|
| 214 |
# V2-dynamic-memory — Buffer sliding window (evita OOM em treino longo)
|
| 215 |
+
# V6.5-V2-metrics-FIX-4: reduzido de 256 para 128 (OOM-killer mitigation).
|
| 216 |
+
# O buffer_4d só precisa conter amostras recentes para que o SOM compute
|
| 217 |
+
# métricas (QE, TE, KL, VE) — os pesos do SOM já capturam o conhecimento
|
| 218 |
+
# acumulado de todas as amostras históricas.
|
| 219 |
+
MAX_BUFFER_SIZE = 128
|
| 220 |
|
| 221 |
# Streaming — User requirement: "streaming de 100 em 100 samples"
|
| 222 |
STREAM_BATCH_SIZE = 100
|
|
|
|
| 230 |
# User requirement: "não reduzir tempo e não gerar dados sintéticos" +
|
| 231 |
# "todo streaming (FASE1 e da FASE2) deve ter pausa para dar tempo de conclusão
|
| 232 |
# de processamento continuando após conclusão"
|
| 233 |
+
# V6.5-V2-metrics-FIX-4 (latest user requirement):
|
| 234 |
+
# "FASE2 PUNIÇÃO é mais pesada é pode exigir pausas do streaming até
|
| 235 |
+
# concluir o processamento".
|
| 236 |
+
# Pausas separadas para FASE1 (CONHECIMENTO) e FASE2 (PUNITIVA):
|
| 237 |
+
# - FASE1: pausas moderadas (SOM-only, sem hypothesis layer)
|
| 238 |
+
# - FASE2: pausas maiores (16 hipóteses × 30 steps × 3 trials + EWC + revival)
|
| 239 |
+
INTER_BATCH_PAUSE_S_FASE1 = 0.15 # CONHECIMENTO: leve
|
| 240 |
+
INTER_BATCH_PAUSE_S_FASE2 = 0.60 # PUNITIVA: 4x maior (heavier processing)
|
| 241 |
+
INTER_DATASET_PAUSE_S_FASE1 = 0.5
|
| 242 |
+
INTER_DATASET_PAUSE_S_FASE2 = 1.5 # PUNITIVA: 3x maior
|
| 243 |
+
INTER_STREAM_BATCH_PAUSE_S_FASE1 = 0.3
|
| 244 |
+
INTER_STREAM_BATCH_PAUSE_S_FASE2 = 1.2 # PUNITIVA: 4x maior
|
| 245 |
+
POST_PROCESSING_PAUSE_S_FASE1 = 0.4
|
| 246 |
+
POST_PROCESSING_PAUSE_S_FASE2 = 1.5 # PUNITIVA: ~4x maior
|
| 247 |
+
# Compatibilidade (mantém nomes antigos apontando para FASE1 — usados em
|
| 248 |
+
# código legado que não diferencia fases)
|
| 249 |
+
INTER_BATCH_PAUSE_S = INTER_BATCH_PAUSE_S_FASE1
|
| 250 |
+
INTER_DATASET_PAUSE_S = INTER_DATASET_PAUSE_S_FASE1
|
| 251 |
+
INTER_STREAM_BATCH_PAUSE_S = INTER_STREAM_BATCH_PAUSE_S_FASE1
|
| 252 |
+
POST_PROCESSING_PAUSE_S = POST_PROCESSING_PAUSE_S_FASE1
|
| 253 |
+
|
| 254 |
+
# V6.5-V2-metrics-FIX-4 — auto-revive config
|
| 255 |
+
# User requirement: "distribuindo o processamento paralelamente".
|
| 256 |
+
# Quando dead_rate > 50% e passou o cooldown, revive neurônios mortos.
|
| 257 |
+
AUTO_REVIVE_DEAD_RATE_THRESHOLD = 0.5
|
| 258 |
+
AUTO_REVIVE_COOLDOWN_STEPS = 200
|
| 259 |
|
| 260 |
# Storage critical
|
| 261 |
STORAGE_CRITICAL_PCT = 90
|
|
|
|
| 1202 |
})
|
| 1203 |
|
| 1204 |
step += 1
|
| 1205 |
+
time.sleep(INTER_BATCH_PAUSE_S_FASE1)
|
| 1206 |
except Exception as e:
|
| 1207 |
logger.error(f"[V6.5-V2] Batch error: {e}")
|
| 1208 |
traceback.print_exc()
|
| 1209 |
continue
|
| 1210 |
|
| 1211 |
+
time.sleep(INTER_STREAM_BATCH_PAUSE_S_FASE1)
|
| 1212 |
# V2-dynamic-memory — Buffer sliding window: trunca para os
|
| 1213 |
# últimos MAX_BUFFER_SIZE amostras após cada chunk.
|
| 1214 |
# Os pesos do SOM já capturam o conhecimento acumulado,
|
|
|
|
| 1285 |
except Exception as e:
|
| 1286 |
logger.warning(f"[V6.5-V2-metrics-FIX-2] Failed to compute SOM metrics: {e}")
|
| 1287 |
|
| 1288 |
+
# V6.5-V2-metrics-FIX-4 — Auto-revive neurônios mortos
|
| 1289 |
+
# User requirement: "APRIMORAR (...) a ativação e uso e acesso
|
| 1290 |
+
# dos neurônios (apenas dois estão sendo ativados:
|
| 1291 |
+
# neurons_active=2/864) distribuindo o processamento paralelamente".
|
| 1292 |
+
# O conscience mechanism (DeSieno 1988) já força distribuição
|
| 1293 |
+
# uniforme de BMU, mas como safety net adicional, revive
|
| 1294 |
+
# explicitamente neurônios que ainda estão mortos após o cooldown.
|
| 1295 |
+
try:
|
| 1296 |
+
revival = kls.auto_revive_if_needed(
|
| 1297 |
+
dead_rate_threshold=AUTO_REVIVE_DEAD_RATE_THRESHOLD,
|
| 1298 |
+
min_steps_between_revivals=AUTO_REVIVE_COOLDOWN_STEPS,
|
| 1299 |
+
)
|
| 1300 |
+
if revival.get("action") == "auto_revived":
|
| 1301 |
+
logger.info(
|
| 1302 |
+
f"[V6.5-V2-metrics-FIX-4] AUTO-REVIVE triggered: "
|
| 1303 |
+
f"n_revived={revival['n_revived']}/{revival['n_total']}, "
|
| 1304 |
+
f"dead_rate {revival['dead_rate_before']:.3f} → "
|
| 1305 |
+
f"{revival['dead_rate_after']:.3f}, "
|
| 1306 |
+
f"steps_since_last={revival['steps_since_last_revival']}"
|
| 1307 |
+
)
|
| 1308 |
+
except Exception as revive_err:
|
| 1309 |
+
logger.warning(
|
| 1310 |
+
f"[V6.5-V2-metrics-FIX-4] auto_revive_if_needed failed: {revive_err}"
|
| 1311 |
+
)
|
| 1312 |
+
|
| 1313 |
# Aggressive memory cleanup between chunks
|
| 1314 |
if chunk_idx_global % 2 == 0:
|
| 1315 |
aggressive_memory_cleanup()
|
|
|
|
| 1322 |
# V6.5-V2-metrics-FIX: pausa pós-processamento para dar tempo
|
| 1323 |
# de conclusão (user requirement: "todo streaming deve ter
|
| 1324 |
# pausa para dar tempo de conclusão de processamento").
|
| 1325 |
+
time.sleep(POST_PROCESSING_PAUSE_S_FASE1)
|
| 1326 |
except Exception as e:
|
| 1327 |
logger.error(f"[V6.5-V2] Dataset {dataset_name} failed: {e}")
|
| 1328 |
traceback.print_exc()
|
| 1329 |
streaming_failures[dataset_name] += 1
|
| 1330 |
|
| 1331 |
+
time.sleep(INTER_DATASET_PAUSE_S_FASE1)
|
| 1332 |
# V6.5-V2-metrics-FIX-2 — Save state after each dataset to preserve
|
| 1333 |
# progress in case the process is killed by container timeout.
|
| 1334 |
# User requirement: "o estado do modelo deve ser contínuo" — saving
|
|
|
|
| 1409 |
}
|
| 1410 |
|
| 1411 |
|
| 1412 |
+
# ============================================================================
|
| 1413 |
+
# 9.5 — V6.5-V2-metrics-FIX-4: Sumário final da FASE2
|
| 1414 |
+
# ============================================================================
|
| 1415 |
+
def build_fase2_final_summary(
|
| 1416 |
+
kls: KohonenLearningSystemV2,
|
| 1417 |
+
som_metrics_log: List[Dict[str, Any]],
|
| 1418 |
+
punishment_log: List[Dict[str, Any]],
|
| 1419 |
+
hypotheses_log: List[Dict[str, Any]],
|
| 1420 |
+
delta_applications_log: List[Dict[str, Any]],
|
| 1421 |
+
elapsed_s: float,
|
| 1422 |
+
) -> Dict[str, Any]:
|
| 1423 |
+
"""V6.5-V2-metrics-FIX-4 — Constrói sumário final da FASE2 mostrando evolução.
|
| 1424 |
+
|
| 1425 |
+
User requirement: "ao final da FASE2 mostrar evolução de métricas e dos
|
| 1426 |
+
indicadores e da taxa de aprendizagem".
|
| 1427 |
+
|
| 1428 |
+
Extrai trajetória temporal das métricas SOM (QE, TE, KL, VE), indicadores
|
| 1429 |
+
de falha (dead_rate, collapse, stagnation, crossing), taxa de aprendizado
|
| 1430 |
+
(α_t, σ_t) e estatísticas do conscience mechanism (neurons_active,
|
| 1431 |
+
uniformity_score). Compara início vs fim para mostrar evolução.
|
| 1432 |
+
|
| 1433 |
+
Returns:
|
| 1434 |
+
Dict com:
|
| 1435 |
+
- summary_lines: List[str] — linhas formatadas para logger.info
|
| 1436 |
+
- metrics_evolution: Dict com first/last/delta de cada métrica
|
| 1437 |
+
- learning_rate_evolution: Dict com α_t, σ_t no início e fim
|
| 1438 |
+
- indicators_evolution: Dict com indicadores de falha no início e fim
|
| 1439 |
+
- neuron_activation_evolution: Dict com neurons_active no início e fim
|
| 1440 |
+
- punishment_stats: Dict com estatísticas de punição
|
| 1441 |
+
"""
|
| 1442 |
+
summary_lines: List[str] = []
|
| 1443 |
+
|
| 1444 |
+
# 1. Trajetória das métricas principais (QE, TE, KL, VE)
|
| 1445 |
+
def _safe_get(log_list, key, idx):
|
| 1446 |
+
if not log_list or idx >= len(log_list):
|
| 1447 |
+
return 0.0
|
| 1448 |
+
try:
|
| 1449 |
+
return float(log_list[idx].get(key, 0.0))
|
| 1450 |
+
except (TypeError, ValueError):
|
| 1451 |
+
return 0.0
|
| 1452 |
+
|
| 1453 |
+
n_logs = len(som_metrics_log)
|
| 1454 |
+
first_idx = 0
|
| 1455 |
+
last_idx = max(0, n_logs - 1)
|
| 1456 |
+
|
| 1457 |
+
metrics_first = {
|
| 1458 |
+
"QE": _safe_get(som_metrics_log, "quantization_error", first_idx),
|
| 1459 |
+
"TE": _safe_get(som_metrics_log, "topological_error", first_idx),
|
| 1460 |
+
"KL": _safe_get(som_metrics_log, "kaski_lagus_error", first_idx),
|
| 1461 |
+
"VE": _safe_get(som_metrics_log, "explained_variance_share", first_idx),
|
| 1462 |
+
}
|
| 1463 |
+
metrics_last = {
|
| 1464 |
+
"QE": _safe_get(som_metrics_log, "quantization_error", last_idx),
|
| 1465 |
+
"TE": _safe_get(som_metrics_log, "topological_error", last_idx),
|
| 1466 |
+
"KL": _safe_get(som_metrics_log, "kaski_lagus_error", last_idx),
|
| 1467 |
+
"VE": _safe_get(som_metrics_log, "explained_variance_share", last_idx),
|
| 1468 |
+
}
|
| 1469 |
+
metrics_delta = {
|
| 1470 |
+
k: metrics_last[k] - metrics_first[k] for k in metrics_first
|
| 1471 |
+
}
|
| 1472 |
+
|
| 1473 |
+
# 2. Indicadores de falha
|
| 1474 |
+
first_failures = som_metrics_log[first_idx].get("failure_indicators", []) if som_metrics_log else []
|
| 1475 |
+
last_failures = som_metrics_log[last_idx].get("failure_indicators", []) if som_metrics_log else []
|
| 1476 |
+
first_health = som_metrics_log[first_idx].get("overall_health", "unknown") if som_metrics_log else "unknown"
|
| 1477 |
+
last_health = som_metrics_log[last_idx].get("overall_health", "unknown") if som_metrics_log else "unknown"
|
| 1478 |
+
|
| 1479 |
+
# 3. Ativação de neurônios (conscience mechanism)
|
| 1480 |
+
first_neurons_active = _safe_get(som_metrics_log, "n_active_neurons_bmu", first_idx)
|
| 1481 |
+
last_neurons_active = _safe_get(som_metrics_log, "n_active_neurons_bmu", last_idx)
|
| 1482 |
+
first_neurons_total = _safe_get(som_metrics_log, "n_total_neurons_bmu", first_idx) or 864
|
| 1483 |
+
last_neurons_total = _safe_get(som_metrics_log, "n_total_neurons_bmu", last_idx) or 864
|
| 1484 |
+
|
| 1485 |
+
# 4. Relatório paralelo final do SOM (conscience + uniformity)
|
| 1486 |
+
try:
|
| 1487 |
+
parallel_report = kls.parallel_neuron_activation_report()
|
| 1488 |
+
except Exception:
|
| 1489 |
+
parallel_report = {}
|
| 1490 |
+
|
| 1491 |
+
# 5. Taxa de aprendizado (α_t, σ_t) do SOM
|
| 1492 |
+
try:
|
| 1493 |
+
som_metrics = kls.som.get_metrics()
|
| 1494 |
+
alpha_t = som_metrics.get("alpha_t_effective", 0.0)
|
| 1495 |
+
sigma_t = som_metrics.get("sigma_t_effective", 0.0)
|
| 1496 |
+
alpha0 = som_metrics.get("alpha0", 0.5)
|
| 1497 |
+
sigma0 = som_metrics.get("sigma0", 3.0)
|
| 1498 |
+
som_t = som_metrics.get("t", 0)
|
| 1499 |
+
except Exception:
|
| 1500 |
+
alpha_t = sigma_t = alpha0 = sigma0 = som_t = 0.0
|
| 1501 |
+
|
| 1502 |
+
# 6. Estatísticas de punição
|
| 1503 |
+
n_punishments = len(punishment_log)
|
| 1504 |
+
n_train_hyp_calls = len(hypotheses_log)
|
| 1505 |
+
n_delta_apps = len(delta_applications_log)
|
| 1506 |
+
# Taxa de sucesso das aplicações de delta (acc_after > acc_before)
|
| 1507 |
+
delta_success = 0
|
| 1508 |
+
if delta_applications_log:
|
| 1509 |
+
for d in delta_applications_log:
|
| 1510 |
+
try:
|
| 1511 |
+
if float(d.get("acc_after", 0.0)) > float(d.get("acc_before", 0.0)):
|
| 1512 |
+
delta_success += 1
|
| 1513 |
+
except (TypeError, ValueError):
|
| 1514 |
+
pass
|
| 1515 |
+
delta_success_rate = float(delta_success / max(1, n_delta_apps))
|
| 1516 |
+
|
| 1517 |
+
# 7. Hipóteses — evolução da loss
|
| 1518 |
+
if hypotheses_log:
|
| 1519 |
+
loss_first = float(hypotheses_log[0].get("loss_initial", 0.0))
|
| 1520 |
+
loss_last_init = float(hypotheses_log[-1].get("loss_initial", 0.0))
|
| 1521 |
+
loss_last_final = float(hypotheses_log[-1].get("loss_final", 0.0))
|
| 1522 |
+
else:
|
| 1523 |
+
loss_first = loss_last_init = loss_last_final = 0.0
|
| 1524 |
+
|
| 1525 |
+
# ===================== MONTAGEM DAS LINHAS DE SUMÁRIO =====================
|
| 1526 |
+
summary_lines.append(f" Duração total: {elapsed_s:.1f}s")
|
| 1527 |
+
summary_lines.append(f" Logs de métricas computados: {n_logs}")
|
| 1528 |
+
summary_lines.append("")
|
| 1529 |
+
summary_lines.append(" ─── EVOLUÇÃO DAS MÉTRICAS PRINCIPAIS (início → fim) ───")
|
| 1530 |
+
summary_lines.append(
|
| 1531 |
+
f" QE (Quantization Error): {metrics_first['QE']:.4f} → "
|
| 1532 |
+
f"{metrics_last['QE']:.4f} (Δ={metrics_delta['QE']:+.4f})"
|
| 1533 |
+
)
|
| 1534 |
+
summary_lines.append(
|
| 1535 |
+
f" TE (Topological Error) : {metrics_first['TE']:.4f} → "
|
| 1536 |
+
f"{metrics_last['TE']:.4f} (Δ={metrics_delta['TE']:+.4f})"
|
| 1537 |
+
)
|
| 1538 |
+
summary_lines.append(
|
| 1539 |
+
f" KL (Kaski-Lagus) : {metrics_first['KL']:.4f} → "
|
| 1540 |
+
f"{metrics_last['KL']:.4f} (Δ={metrics_delta['KL']:+.4f})"
|
| 1541 |
+
)
|
| 1542 |
+
summary_lines.append(
|
| 1543 |
+
f" VE (Explained Variance): {metrics_first['VE']:.4f} → "
|
| 1544 |
+
f"{metrics_last['VE']:.4f} (Δ={metrics_delta['VE']:+.4f})"
|
| 1545 |
+
)
|
| 1546 |
+
summary_lines.append("")
|
| 1547 |
+
summary_lines.append(" ─── INDICADORES DE FALHA ───")
|
| 1548 |
+
summary_lines.append(
|
| 1549 |
+
f" Overall health: {first_health} → {last_health}"
|
| 1550 |
+
)
|
| 1551 |
+
summary_lines.append(
|
| 1552 |
+
f" Failure indicators (início): {len(first_failures)} — {first_failures[:3]}"
|
| 1553 |
+
)
|
| 1554 |
+
summary_lines.append(
|
| 1555 |
+
f" Failure indicators (fim) : {len(last_failures)} — {last_failures[:3]}"
|
| 1556 |
+
)
|
| 1557 |
+
summary_lines.append("")
|
| 1558 |
+
summary_lines.append(" ─── ATIVAÇÃO DE NEURÔNIOS (Conscience Mechanism) ───")
|
| 1559 |
+
summary_lines.append(
|
| 1560 |
+
f" Neurons ativos (BMU): {int(first_neurons_active)}/{int(first_neurons_total)} "
|
| 1561 |
+
f"→ {int(last_neurons_active)}/{int(last_neurons_total)}"
|
| 1562 |
+
)
|
| 1563 |
+
if parallel_report:
|
| 1564 |
+
summary_lines.append(
|
| 1565 |
+
f" Uniformity score (fim): {parallel_report.get('uniformity_score', 0.0):.4f} "
|
| 1566 |
+
f"(1.0 = perfeitamente uniforme)"
|
| 1567 |
+
)
|
| 1568 |
+
summary_lines.append(
|
| 1569 |
+
f" Conscience bias mean/std: "
|
| 1570 |
+
f"{parallel_report.get('conscience_bias_mean', 0.0):.6f} / "
|
| 1571 |
+
f"{parallel_report.get('conscience_bias_std', 0.0):.6f} "
|
| 1572 |
+
f"(deve tender a 0)"
|
| 1573 |
+
)
|
| 1574 |
+
summary_lines.append(
|
| 1575 |
+
f" Win count max/mean: "
|
| 1576 |
+
f"{parallel_report.get('max_win_count', 0.0):.0f} / "
|
| 1577 |
+
f"{parallel_report.get('mean_win_count', 0.0):.2f}"
|
| 1578 |
+
)
|
| 1579 |
+
summary_lines.append("")
|
| 1580 |
+
summary_lines.append(" ─── TAXA DE APRENDIZADO (Kohonen schedule) ───")
|
| 1581 |
+
summary_lines.append(
|
| 1582 |
+
f" α_t (learning rate): α₀={alpha0:.4f} → α_t={alpha_t:.6f} "
|
| 1583 |
+
f"(floor=0.001, decai em exp(-t/2000))"
|
| 1584 |
+
)
|
| 1585 |
+
summary_lines.append(
|
| 1586 |
+
f" σ_t (neighborhood) : σ₀={sigma0:.4f} → σ_t={sigma_t:.6f} "
|
| 1587 |
+
f"(floor=0.1, decai em exp(-t/1000))"
|
| 1588 |
+
)
|
| 1589 |
+
summary_lines.append(f" SOM t (updates) : {som_t}")
|
| 1590 |
+
summary_lines.append("")
|
| 1591 |
+
summary_lines.append(" ─── ESTATÍSTICAS DE PUNIÇÃO ───")
|
| 1592 |
+
summary_lines.append(f" Total punishment events: {n_punishments}")
|
| 1593 |
+
summary_lines.append(f" Hypotheses trainings : {n_train_hyp_calls}")
|
| 1594 |
+
summary_lines.append(f" Delta applications : {n_delta_apps}")
|
| 1595 |
+
summary_lines.append(
|
| 1596 |
+
f" Delta success rate : {delta_success_rate:.2%} "
|
| 1597 |
+
f"({delta_success}/{n_delta_apps} melhoraram acurácia)"
|
| 1598 |
+
)
|
| 1599 |
+
if hypotheses_log:
|
| 1600 |
+
summary_lines.append(
|
| 1601 |
+
f" Loss inicial primeira chamada: {loss_first:.6f}"
|
| 1602 |
+
)
|
| 1603 |
+
summary_lines.append(
|
| 1604 |
+
f" Loss inicial última chamada : {loss_last_init:.6f} "
|
| 1605 |
+
f"→ final: {loss_last_final:.6f}"
|
| 1606 |
+
)
|
| 1607 |
+
summary_lines.append("")
|
| 1608 |
+
summary_lines.append(" ─── HIPÓTESES (configuração final) ───")
|
| 1609 |
+
try:
|
| 1610 |
+
v2_metrics = kls.get_v2_metrics()
|
| 1611 |
+
summary_lines.append(
|
| 1612 |
+
f" n_hypotheses ativas: {v2_metrics.get('n_hypotheses', 0)} / "
|
| 1613 |
+
f"max {v2_metrics.get('max_n_hypotheses', 0)}"
|
| 1614 |
+
)
|
| 1615 |
+
summary_lines.append(
|
| 1616 |
+
f" hyp_train_steps atual: {v2_metrics.get('hyp_train_steps', 0)}"
|
| 1617 |
+
)
|
| 1618 |
+
summary_lines.append(
|
| 1619 |
+
f" total_hyp_steps_executed: "
|
| 1620 |
+
f"{v2_metrics.get('total_hyp_steps_executed', 0)}"
|
| 1621 |
+
)
|
| 1622 |
+
summary_lines.append(
|
| 1623 |
+
f" n_adaptations dinâmicas: "
|
| 1624 |
+
f"{v2_metrics.get('dynamic_adaptation', {}).get('n_adaptations', 0)}"
|
| 1625 |
+
)
|
| 1626 |
+
summary_lines.append(
|
| 1627 |
+
f" EWC reference set: {v2_metrics.get('ewc_reference_set', False)}"
|
| 1628 |
+
)
|
| 1629 |
+
except Exception:
|
| 1630 |
+
pass
|
| 1631 |
+
|
| 1632 |
+
return {
|
| 1633 |
+
"summary_lines": summary_lines,
|
| 1634 |
+
"metrics_evolution": {
|
| 1635 |
+
"first": metrics_first,
|
| 1636 |
+
"last": metrics_last,
|
| 1637 |
+
"delta": metrics_delta,
|
| 1638 |
+
},
|
| 1639 |
+
"learning_rate_evolution": {
|
| 1640 |
+
"alpha0": alpha0,
|
| 1641 |
+
"alpha_t_final": alpha_t,
|
| 1642 |
+
"sigma0": sigma0,
|
| 1643 |
+
"sigma_t_final": sigma_t,
|
| 1644 |
+
"som_t": som_t,
|
| 1645 |
+
},
|
| 1646 |
+
"indicators_evolution": {
|
| 1647 |
+
"first_health": first_health,
|
| 1648 |
+
"last_health": last_health,
|
| 1649 |
+
"first_failures": first_failures,
|
| 1650 |
+
"last_failures": last_failures,
|
| 1651 |
+
"first_n_failures": len(first_failures),
|
| 1652 |
+
"last_n_failures": len(last_failures),
|
| 1653 |
+
},
|
| 1654 |
+
"neuron_activation_evolution": {
|
| 1655 |
+
"first_active": int(first_neurons_active),
|
| 1656 |
+
"last_active": int(last_neurons_active),
|
| 1657 |
+
"total": int(last_neurons_total),
|
| 1658 |
+
"parallel_report_final": parallel_report,
|
| 1659 |
+
},
|
| 1660 |
+
"punishment_stats": {
|
| 1661 |
+
"n_punishments": n_punishments,
|
| 1662 |
+
"n_train_hyp_calls": n_train_hyp_calls,
|
| 1663 |
+
"n_delta_apps": n_delta_apps,
|
| 1664 |
+
"delta_success_rate": delta_success_rate,
|
| 1665 |
+
"loss_first_initial": loss_first,
|
| 1666 |
+
"loss_last_initial": loss_last_init,
|
| 1667 |
+
"loss_last_final": loss_last_final,
|
| 1668 |
+
},
|
| 1669 |
+
}
|
| 1670 |
+
|
| 1671 |
+
|
| 1672 |
# ============================================================================
|
| 1673 |
# 10. PHASE 2 — TREINAMENTO COM PUNIÇÃO (16 hipóteses × 3 tentativas)
|
| 1674 |
# ============================================================================
|
|
|
|
| 1935 |
traceback.print_exc()
|
| 1936 |
|
| 1937 |
step += 1
|
| 1938 |
+
# V6.5-V2-metrics-FIX-4 — pausa maior na FASE2 (PUNITIVA)
|
| 1939 |
+
# User requirement: "FASE2 PUNIÇÃO é mais pesada é pode exigir
|
| 1940 |
+
# pausas do streaming até concluir o processamento".
|
| 1941 |
+
time.sleep(INTER_BATCH_PAUSE_S_FASE2)
|
| 1942 |
except Exception as e:
|
| 1943 |
logger.error(f"[V6.5-V2] Batch error in PUNIÇÃO: {e}")
|
| 1944 |
traceback.print_exc()
|
| 1945 |
continue
|
| 1946 |
|
| 1947 |
+
# V6.5-V2-metrics-FIX-4 — Auto-revive neurônios mortos na FASE2
|
| 1948 |
+
# (PUNITIVA também deve distribuir processamento paralelamente)
|
| 1949 |
+
try:
|
| 1950 |
+
revival = kls.auto_revive_if_needed(
|
| 1951 |
+
dead_rate_threshold=AUTO_REVIVE_DEAD_RATE_THRESHOLD,
|
| 1952 |
+
min_steps_between_revivals=AUTO_REVIVE_COOLDOWN_STEPS,
|
| 1953 |
+
)
|
| 1954 |
+
if revival.get("action") == "auto_revived":
|
| 1955 |
+
logger.info(
|
| 1956 |
+
f"[V6.5-V2-metrics-FIX-4] PUNIÇÃO AUTO-REVIVE: "
|
| 1957 |
+
f"n_revived={revival['n_revived']}/{revival['n_total']}, "
|
| 1958 |
+
f"dead_rate {revival['dead_rate_before']:.3f} → "
|
| 1959 |
+
f"{revival['dead_rate_after']:.3f}"
|
| 1960 |
+
)
|
| 1961 |
+
except Exception as revive_err:
|
| 1962 |
+
logger.warning(
|
| 1963 |
+
f"[V6.5-V2-metrics-FIX-4] PUNIÇÃO auto_revive failed: {revive_err}"
|
| 1964 |
+
)
|
| 1965 |
+
|
| 1966 |
+
time.sleep(INTER_STREAM_BATCH_PAUSE_S_FASE2)
|
| 1967 |
# V2-dynamic-memory — Buffer sliding window na PUNIÇÃO também
|
| 1968 |
if len(kls.buffer_4d) > MAX_BUFFER_SIZE:
|
| 1969 |
overflow = len(kls.buffer_4d) - MAX_BUFFER_SIZE
|
|
|
|
| 1980 |
kls.aggressive_cleanup()
|
| 1981 |
except Exception:
|
| 1982 |
pass
|
| 1983 |
+
# V6.5-V2-metrics-FIX-4: pausa pós-processamento maior na FASE2
|
| 1984 |
+
# (user requirement: "FASE2 PUNIÇÃO é mais pesada")
|
| 1985 |
+
time.sleep(POST_PROCESSING_PAUSE_S_FASE2)
|
| 1986 |
except Exception as e:
|
| 1987 |
logger.error(f"[V6.5-V2] PUNIÇÃO dataset failed: {e}")
|
| 1988 |
traceback.print_exc()
|
|
|
|
| 1997 |
logger.info(f" Hypotheses trainings: {len(hypotheses_log)}")
|
| 1998 |
logger.info(f" Delta applications: {len(delta_applications_log)}")
|
| 1999 |
|
| 2000 |
+
# V6.5-V2-metrics-FIX-4 — Sumário final da FASE2
|
| 2001 |
+
# User requirement: "ao final da FASE2 mostrar evolução de métricas e dos
|
| 2002 |
+
# indicadores e da taxa de aprendizagem".
|
| 2003 |
+
fase2_summary = build_fase2_final_summary(
|
| 2004 |
+
kls=kls,
|
| 2005 |
+
som_metrics_log=som_metrics_log,
|
| 2006 |
+
punishment_log=punishment_log,
|
| 2007 |
+
hypotheses_log=hypotheses_log,
|
| 2008 |
+
delta_applications_log=delta_applications_log,
|
| 2009 |
+
elapsed_s=t_elapsed,
|
| 2010 |
+
)
|
| 2011 |
+
logger.info("\n" + "=" * 80)
|
| 2012 |
+
logger.info("[V6.5-V2-metrics-FIX-4] FASE 2 — EVOLUÇÃO FINAL DE MÉTRICAS")
|
| 2013 |
+
logger.info("=" * 80)
|
| 2014 |
+
for line in fase2_summary["summary_lines"]:
|
| 2015 |
+
logger.info(line)
|
| 2016 |
+
logger.info("=" * 80 + "\n")
|
| 2017 |
+
|
| 2018 |
return {
|
| 2019 |
"phase": "TREINAMENTO_COM_PUNICAO",
|
| 2020 |
"dataset": PUNICAO_DATASET,
|
|
|
|
| 2036 |
),
|
| 2037 |
"som_metric_history": kls.get_som_metric_history(),
|
| 2038 |
"final_v2_state": kls.get_v2_metrics(),
|
| 2039 |
+
# V6.5-V2-metrics-FIX-4 — sumário final da FASE2
|
| 2040 |
+
"fase2_summary": fase2_summary,
|
| 2041 |
}
|
| 2042 |
|
| 2043 |
|
|
|
|
| 2480 |
# + Goose VQ, fornecendo representação compacta do estado do SOM.
|
| 2481 |
# O compressor já sanitiza NaN/Inf internamente (torch.nan_to_num).
|
| 2482 |
enable_vqvae2=True, # REATIVADO (was False in V6.5-V2-metrics-FIX)
|
| 2483 |
+
# V6.5-V2-metrics-FIX-4: reasoning_engine desabilitado para mitigar
|
| 2484 |
+
# OOM-killer (User requirement implícito: "investigar e corrigir
|
| 2485 |
+
# falhas de lógica e bugs que estejam causando alto consumo de memória
|
| 2486 |
+
# sem distorcer a arquitetura Kohonen"). O ReasoningEngine cria um
|
| 2487 |
+
# ThreadPoolExecutor(4 workers) + ToolAgentCoordinator que consome
|
| 2488 |
+
# ~100-200MB adicionais. Como o reasoning_engine é OPCIONAL e não
|
| 2489 |
+
# afeta o aprendizado do SOM (apenas gera tags <think>/<plan>/<answer>
|
| 2490 |
+
# para predições), desabilitá-lo preserva a arquitetura Kohonen e
|
| 2491 |
+
# libera memória para o streaming de 8000+2000 samples.
|
| 2492 |
+
# Para reativar: mudar para True (requer ≥6GB cgroup).
|
| 2493 |
+
enable_reasoning=False,
|
| 2494 |
enable_w8a8=False, # W8A8 permanece desabilitado (não essencial para Kohonen)
|
| 2495 |
vqvae2_code_dim=16,
|
| 2496 |
vqvae2_num_codes_top=64,
|
v6_5_v2_attention_eval.json
CHANGED
|
@@ -3,18 +3,18 @@
|
|
| 3 |
"user_requirement": "verificar se o mecanismo de atenção está ativo e acessado logicamente funcional",
|
| 4 |
"metrics": {
|
| 5 |
"active": true,
|
| 6 |
-
"n_calls":
|
| 7 |
"n_errors": 0,
|
| 8 |
-
"last_norm_in": 109.
|
| 9 |
-
"last_norm_out":
|
| 10 |
"last_attn_activated": true,
|
| 11 |
-
"last_attn_diff_norm":
|
| 12 |
"n_heads": 8,
|
| 13 |
"logic_functional": true
|
| 14 |
},
|
| 15 |
"active": true,
|
| 16 |
"logic_functional": true,
|
| 17 |
-
"n_calls":
|
| 18 |
"n_errors": 0,
|
| 19 |
"n_heads": 8,
|
| 20 |
"assessment": "PASS"
|
|
|
|
| 3 |
"user_requirement": "verificar se o mecanismo de atenção está ativo e acessado logicamente funcional",
|
| 4 |
"metrics": {
|
| 5 |
"active": true,
|
| 6 |
+
"n_calls": 10000,
|
| 7 |
"n_errors": 0,
|
| 8 |
+
"last_norm_in": 109.84779357910156,
|
| 9 |
+
"last_norm_out": 114.40419006347656,
|
| 10 |
"last_attn_activated": true,
|
| 11 |
+
"last_attn_diff_norm": 114.49571990966797,
|
| 12 |
"n_heads": 8,
|
| 13 |
"logic_functional": true
|
| 14 |
},
|
| 15 |
"active": true,
|
| 16 |
"logic_functional": true,
|
| 17 |
+
"n_calls": 10000,
|
| 18 |
"n_errors": 0,
|
| 19 |
"n_heads": 8,
|
| 20 |
"assessment": "PASS"
|
v6_5_v2_phases_eval.json
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
v6_5_v2_predict_fix_eval.json
CHANGED
|
@@ -6,7 +6,7 @@
|
|
| 6 |
{
|
| 7 |
"query": "o gato dorme na cama",
|
| 8 |
"prediction": "short_text",
|
| 9 |
-
"probability": 0.
|
| 10 |
"is_gato_hardcoded": false,
|
| 11 |
"is_cachorro_hardcoded": false,
|
| 12 |
"is_registry_label": true,
|
|
@@ -15,7 +15,7 @@
|
|
| 15 |
{
|
| 16 |
"query": "calcule dois mais dois",
|
| 17 |
"prediction": "short_text",
|
| 18 |
-
"probability": 0.
|
| 19 |
"is_gato_hardcoded": false,
|
| 20 |
"is_cachorro_hardcoded": false,
|
| 21 |
"is_registry_label": true,
|
|
@@ -24,7 +24,7 @@
|
|
| 24 |
{
|
| 25 |
"query": "qual é a capital do brasil",
|
| 26 |
"prediction": "short_text",
|
| 27 |
-
"probability": 0.
|
| 28 |
"is_gato_hardcoded": false,
|
| 29 |
"is_cachorro_hardcoded": false,
|
| 30 |
"is_registry_label": true,
|
|
@@ -33,7 +33,7 @@
|
|
| 33 |
{
|
| 34 |
"query": "explique o que é uma rede neural",
|
| 35 |
"prediction": "short_text",
|
| 36 |
-
"probability": 0.
|
| 37 |
"is_gato_hardcoded": false,
|
| 38 |
"is_cachorro_hardcoded": false,
|
| 39 |
"is_registry_label": true,
|
|
@@ -42,7 +42,7 @@
|
|
| 42 |
{
|
| 43 |
"query": "olá como você está",
|
| 44 |
"prediction": "short_text",
|
| 45 |
-
"probability": 0.
|
| 46 |
"is_gato_hardcoded": false,
|
| 47 |
"is_cachorro_hardcoded": false,
|
| 48 |
"is_registry_label": true,
|
|
@@ -51,7 +51,7 @@
|
|
| 51 |
{
|
| 52 |
"query": "traduza hello para portugues",
|
| 53 |
"prediction": "short_text",
|
| 54 |
-
"probability": 0.
|
| 55 |
"is_gato_hardcoded": false,
|
| 56 |
"is_cachorro_hardcoded": false,
|
| 57 |
"is_registry_label": true,
|
|
|
|
| 6 |
{
|
| 7 |
"query": "o gato dorme na cama",
|
| 8 |
"prediction": "short_text",
|
| 9 |
+
"probability": 0.443034291267395,
|
| 10 |
"is_gato_hardcoded": false,
|
| 11 |
"is_cachorro_hardcoded": false,
|
| 12 |
"is_registry_label": true,
|
|
|
|
| 15 |
{
|
| 16 |
"query": "calcule dois mais dois",
|
| 17 |
"prediction": "short_text",
|
| 18 |
+
"probability": 0.443034291267395,
|
| 19 |
"is_gato_hardcoded": false,
|
| 20 |
"is_cachorro_hardcoded": false,
|
| 21 |
"is_registry_label": true,
|
|
|
|
| 24 |
{
|
| 25 |
"query": "qual é a capital do brasil",
|
| 26 |
"prediction": "short_text",
|
| 27 |
+
"probability": 0.443034291267395,
|
| 28 |
"is_gato_hardcoded": false,
|
| 29 |
"is_cachorro_hardcoded": false,
|
| 30 |
"is_registry_label": true,
|
|
|
|
| 33 |
{
|
| 34 |
"query": "explique o que é uma rede neural",
|
| 35 |
"prediction": "short_text",
|
| 36 |
+
"probability": 0.443034291267395,
|
| 37 |
"is_gato_hardcoded": false,
|
| 38 |
"is_cachorro_hardcoded": false,
|
| 39 |
"is_registry_label": true,
|
|
|
|
| 42 |
{
|
| 43 |
"query": "olá como você está",
|
| 44 |
"prediction": "short_text",
|
| 45 |
+
"probability": 0.443034291267395,
|
| 46 |
"is_gato_hardcoded": false,
|
| 47 |
"is_cachorro_hardcoded": false,
|
| 48 |
"is_registry_label": true,
|
|
|
|
| 51 |
{
|
| 52 |
"query": "traduza hello para portugues",
|
| 53 |
"prediction": "short_text",
|
| 54 |
+
"probability": 0.443034291267395,
|
| 55 |
"is_gato_hardcoded": false,
|
| 56 |
"is_cachorro_hardcoded": false,
|
| 57 |
"is_registry_label": true,
|
v6_5_v2_report.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
{
|
| 2 |
"version": "V6.5-V2",
|
| 3 |
-
"timestamp": "2026-08-
|
| 4 |
"config": {
|
| 5 |
"som_grid": [
|
| 6 |
6,
|
|
@@ -54,10 +54,10 @@
|
|
| 54 |
"init_done": true
|
| 55 |
},
|
| 56 |
"fp16_benchmark": {
|
| 57 |
-
"best_time_ms":
|
| 58 |
-
"avg_time_ms":
|
| 59 |
-
"best_tflops": 1.
|
| 60 |
-
"avg_tflops": 1.
|
| 61 |
"matrix_size": 4000.0
|
| 62 |
},
|
| 63 |
"v2_verification": {
|
|
@@ -86,13 +86,13 @@
|
|
| 86 |
"fase_1_conhecimento_summary": {
|
| 87 |
"total_samples": 8000,
|
| 88 |
"meta_atingida": true,
|
| 89 |
-
"elapsed_s":
|
| 90 |
"storage_critical_stopped": false
|
| 91 |
},
|
| 92 |
"fase_2_punicão_summary": {
|
| 93 |
"total_samples": 2000,
|
| 94 |
"meta_atingida": true,
|
| 95 |
-
"elapsed_s":
|
| 96 |
"punishment_events": 120,
|
| 97 |
"hypotheses_trainings": 60,
|
| 98 |
"delta_applications": 60,
|
|
@@ -101,7 +101,7 @@
|
|
| 101 |
"attention_eval_summary": {
|
| 102 |
"active": true,
|
| 103 |
"logic_functional": true,
|
| 104 |
-
"n_calls":
|
| 105 |
"assessment": "PASS"
|
| 106 |
},
|
| 107 |
"predict_fix_summary": {
|
|
@@ -114,18 +114,18 @@
|
|
| 114 |
},
|
| 115 |
"user_questions_summary": {
|
| 116 |
"n_with_answer": 3,
|
| 117 |
-
"n_with_think":
|
| 118 |
"answer_rate": 1.0,
|
| 119 |
-
"think_rate":
|
| 120 |
-
"avg_latency_ms":
|
| 121 |
-
"avg_reasoning_length":
|
| 122 |
},
|
| 123 |
"model_states_saved_to": "/home/z/my-project/BiGRU_T_version/v6_5_v2_model_states.pt",
|
| 124 |
"save_info": {
|
| 125 |
"saved": true,
|
| 126 |
"path": "/home/z/my-project/BiGRU_T_version/v6_5_v2_model_states.pt",
|
| 127 |
-
"size_mb": 220.
|
| 128 |
-
"size_gb": 0.
|
| 129 |
"size_status": "OK",
|
| 130 |
"size_within_1gb_limit": true,
|
| 131 |
"reason": "end_of_training_v2",
|
|
@@ -149,16 +149,16 @@
|
|
| 149 |
"training_ready": false,
|
| 150 |
"classifier_trained": true,
|
| 151 |
"ewc_reference_set": true,
|
| 152 |
-
"buffer_size":
|
| 153 |
"total_hyp_steps_executed": 4550,
|
| 154 |
"n_train_hyp_calls": 60,
|
| 155 |
"dynamic_adaptation": {
|
| 156 |
"loss_history_len": 8,
|
| 157 |
"loss_stats": {
|
| 158 |
-
"slope": -0.
|
| 159 |
-
"volatility": 0.
|
| 160 |
-
"mean": 0.
|
| 161 |
-
"std": 0.
|
| 162 |
"n": 8
|
| 163 |
},
|
| 164 |
"punishment_rate": 1.0,
|
|
@@ -189,10 +189,10 @@
|
|
| 189 |
"adapted": false,
|
| 190 |
"rules_fired": [],
|
| 191 |
"loss_stats": {
|
| 192 |
-
"slope": -0.
|
| 193 |
-
"volatility": 0.
|
| 194 |
-
"mean": 0.
|
| 195 |
-
"std": 0.
|
| 196 |
"n": 8
|
| 197 |
},
|
| 198 |
"punishment_rate": 1.0
|
|
@@ -213,10 +213,10 @@
|
|
| 213 |
"adapted": false,
|
| 214 |
"rules_fired": [],
|
| 215 |
"loss_stats": {
|
| 216 |
-
"slope": -0.
|
| 217 |
-
"volatility": 0.
|
| 218 |
-
"mean": 0.
|
| 219 |
-
"std": 0.
|
| 220 |
"n": 8
|
| 221 |
},
|
| 222 |
"punishment_rate": 1.0
|
|
@@ -237,10 +237,10 @@
|
|
| 237 |
"adapted": false,
|
| 238 |
"rules_fired": [],
|
| 239 |
"loss_stats": {
|
| 240 |
-
"slope": -0.
|
| 241 |
-
"volatility": 0.
|
| 242 |
-
"mean": 0.
|
| 243 |
-
"std": 0.
|
| 244 |
"n": 8
|
| 245 |
},
|
| 246 |
"punishment_rate": 1.0
|
|
@@ -261,10 +261,10 @@
|
|
| 261 |
"adapted": false,
|
| 262 |
"rules_fired": [],
|
| 263 |
"loss_stats": {
|
| 264 |
-
"slope": -0.
|
| 265 |
-
"volatility": 0.
|
| 266 |
-
"mean": 0.
|
| 267 |
-
"std": 0.
|
| 268 |
"n": 8
|
| 269 |
},
|
| 270 |
"punishment_rate": 1.0
|
|
@@ -285,10 +285,10 @@
|
|
| 285 |
"adapted": false,
|
| 286 |
"rules_fired": [],
|
| 287 |
"loss_stats": {
|
| 288 |
-
"slope": -0.
|
| 289 |
-
"volatility": 0.
|
| 290 |
-
"mean": 0.
|
| 291 |
-
"std": 0.
|
| 292 |
"n": 8
|
| 293 |
},
|
| 294 |
"punishment_rate": 1.0
|
|
|
|
| 1 |
{
|
| 2 |
"version": "V6.5-V2",
|
| 3 |
+
"timestamp": "2026-08-09T03:12:19.635254",
|
| 4 |
"config": {
|
| 5 |
"som_grid": [
|
| 6 |
6,
|
|
|
|
| 54 |
"init_done": true
|
| 55 |
},
|
| 56 |
"fp16_benchmark": {
|
| 57 |
+
"best_time_ms": 96.71666700160131,
|
| 58 |
+
"avg_time_ms": 98.70659000080195,
|
| 59 |
+
"best_tflops": 1.3234533815963772,
|
| 60 |
+
"avg_tflops": 1.296772586298038,
|
| 61 |
"matrix_size": 4000.0
|
| 62 |
},
|
| 63 |
"v2_verification": {
|
|
|
|
| 86 |
"fase_1_conhecimento_summary": {
|
| 87 |
"total_samples": 8000,
|
| 88 |
"meta_atingida": true,
|
| 89 |
+
"elapsed_s": 460.5420436859131,
|
| 90 |
"storage_critical_stopped": false
|
| 91 |
},
|
| 92 |
"fase_2_punicão_summary": {
|
| 93 |
"total_samples": 2000,
|
| 94 |
"meta_atingida": true,
|
| 95 |
+
"elapsed_s": 728.1188757419586,
|
| 96 |
"punishment_events": 120,
|
| 97 |
"hypotheses_trainings": 60,
|
| 98 |
"delta_applications": 60,
|
|
|
|
| 101 |
"attention_eval_summary": {
|
| 102 |
"active": true,
|
| 103 |
"logic_functional": true,
|
| 104 |
+
"n_calls": 10000,
|
| 105 |
"assessment": "PASS"
|
| 106 |
},
|
| 107 |
"predict_fix_summary": {
|
|
|
|
| 114 |
},
|
| 115 |
"user_questions_summary": {
|
| 116 |
"n_with_answer": 3,
|
| 117 |
+
"n_with_think": 0,
|
| 118 |
"answer_rate": 1.0,
|
| 119 |
+
"think_rate": 0.0,
|
| 120 |
+
"avg_latency_ms": 4.7105153401692705,
|
| 121 |
+
"avg_reasoning_length": 69.0
|
| 122 |
},
|
| 123 |
"model_states_saved_to": "/home/z/my-project/BiGRU_T_version/v6_5_v2_model_states.pt",
|
| 124 |
"save_info": {
|
| 125 |
"saved": true,
|
| 126 |
"path": "/home/z/my-project/BiGRU_T_version/v6_5_v2_model_states.pt",
|
| 127 |
+
"size_mb": 220.15666,
|
| 128 |
+
"size_gb": 0.20503686740994453,
|
| 129 |
"size_status": "OK",
|
| 130 |
"size_within_1gb_limit": true,
|
| 131 |
"reason": "end_of_training_v2",
|
|
|
|
| 149 |
"training_ready": false,
|
| 150 |
"classifier_trained": true,
|
| 151 |
"ewc_reference_set": true,
|
| 152 |
+
"buffer_size": 128,
|
| 153 |
"total_hyp_steps_executed": 4550,
|
| 154 |
"n_train_hyp_calls": 60,
|
| 155 |
"dynamic_adaptation": {
|
| 156 |
"loss_history_len": 8,
|
| 157 |
"loss_stats": {
|
| 158 |
+
"slope": -0.00034456806523459297,
|
| 159 |
+
"volatility": 0.0024739346406965515,
|
| 160 |
+
"mean": 0.6898908242583275,
|
| 161 |
+
"std": 0.0017067448084313731,
|
| 162 |
"n": 8
|
| 163 |
},
|
| 164 |
"punishment_rate": 1.0,
|
|
|
|
| 189 |
"adapted": false,
|
| 190 |
"rules_fired": [],
|
| 191 |
"loss_stats": {
|
| 192 |
+
"slope": -0.00037475568907601494,
|
| 193 |
+
"volatility": 0.0033018073049031742,
|
| 194 |
+
"mean": 0.6912502944469452,
|
| 195 |
+
"std": 0.0022823752717213938,
|
| 196 |
"n": 8
|
| 197 |
},
|
| 198 |
"punishment_rate": 1.0
|
|
|
|
| 213 |
"adapted": false,
|
| 214 |
"rules_fired": [],
|
| 215 |
"loss_stats": {
|
| 216 |
+
"slope": -0.0005504623765037174,
|
| 217 |
+
"volatility": 0.003395094498254662,
|
| 218 |
+
"mean": 0.6910898759961128,
|
| 219 |
+
"std": 0.002346315435793899,
|
| 220 |
"n": 8
|
| 221 |
},
|
| 222 |
"punishment_rate": 1.0
|
|
|
|
| 237 |
"adapted": false,
|
| 238 |
"rules_fired": [],
|
| 239 |
"loss_stats": {
|
| 240 |
+
"slope": -0.0005504623765037174,
|
| 241 |
+
"volatility": 0.003395094498254662,
|
| 242 |
+
"mean": 0.6910898759961128,
|
| 243 |
+
"std": 0.002346315435793899,
|
| 244 |
"n": 8
|
| 245 |
},
|
| 246 |
"punishment_rate": 1.0
|
|
|
|
| 261 |
"adapted": false,
|
| 262 |
"rules_fired": [],
|
| 263 |
"loss_stats": {
|
| 264 |
+
"slope": -0.00034456806523459297,
|
| 265 |
+
"volatility": 0.0024739346406965515,
|
| 266 |
+
"mean": 0.6898908242583275,
|
| 267 |
+
"std": 0.0017067448084313731,
|
| 268 |
"n": 8
|
| 269 |
},
|
| 270 |
"punishment_rate": 1.0
|
|
|
|
| 285 |
"adapted": false,
|
| 286 |
"rules_fired": [],
|
| 287 |
"loss_stats": {
|
| 288 |
+
"slope": -0.00034456806523459297,
|
| 289 |
+
"volatility": 0.0024739346406965515,
|
| 290 |
+
"mean": 0.6898908242583275,
|
| 291 |
+
"std": 0.0017067448084313731,
|
| 292 |
"n": 8
|
| 293 |
},
|
| 294 |
"punishment_rate": 1.0
|
v6_5_v2_user_questions.json
CHANGED
|
@@ -19,16 +19,16 @@
|
|
| 19 |
"system_prompt_used": false,
|
| 20 |
"few_shot_examples": false,
|
| 21 |
"som_prediction": "short_text",
|
| 22 |
-
"reasoning_length":
|
| 23 |
-
"has_think":
|
| 24 |
-
"has_plan":
|
| 25 |
"has_answer": true,
|
| 26 |
-
"has_decompose":
|
| 27 |
-
"think_preview": "
|
| 28 |
-
"answer_preview": "
|
| 29 |
-
"raw_response_preview": "<
|
| 30 |
-
"n_tags":
|
| 31 |
-
"latency_ms":
|
| 32 |
},
|
| 33 |
{
|
| 34 |
"query": "Lula reserva valor",
|
|
@@ -37,16 +37,16 @@
|
|
| 37 |
"system_prompt_used": false,
|
| 38 |
"few_shot_examples": false,
|
| 39 |
"som_prediction": "short_text",
|
| 40 |
-
"reasoning_length":
|
| 41 |
-
"has_think":
|
| 42 |
-
"has_plan":
|
| 43 |
"has_answer": true,
|
| 44 |
-
"has_decompose":
|
| 45 |
-
"think_preview": "
|
| 46 |
-
"answer_preview": "
|
| 47 |
-
"raw_response_preview": "<
|
| 48 |
-
"n_tags":
|
| 49 |
-
"latency_ms":
|
| 50 |
},
|
| 51 |
{
|
| 52 |
"query": "Amazonas força-tarefa vítimas",
|
|
@@ -55,25 +55,25 @@
|
|
| 55 |
"system_prompt_used": false,
|
| 56 |
"few_shot_examples": false,
|
| 57 |
"som_prediction": "short_text",
|
| 58 |
-
"reasoning_length":
|
| 59 |
-
"has_think":
|
| 60 |
-
"has_plan":
|
| 61 |
"has_answer": true,
|
| 62 |
-
"has_decompose":
|
| 63 |
-
"think_preview": "
|
| 64 |
-
"answer_preview": "
|
| 65 |
-
"raw_response_preview": "<
|
| 66 |
-
"n_tags":
|
| 67 |
-
"latency_ms":
|
| 68 |
}
|
| 69 |
],
|
| 70 |
"summary": {
|
| 71 |
"n_with_answer": 3,
|
| 72 |
-
"n_with_think":
|
| 73 |
"answer_rate": 1.0,
|
| 74 |
-
"think_rate":
|
| 75 |
-
"avg_latency_ms":
|
| 76 |
-
"avg_reasoning_length":
|
| 77 |
},
|
| 78 |
"quality_assessment": {
|
| 79 |
"model_not_helped": true,
|
|
|
|
| 19 |
"system_prompt_used": false,
|
| 20 |
"few_shot_examples": false,
|
| 21 |
"som_prediction": "short_text",
|
| 22 |
+
"reasoning_length": 69,
|
| 23 |
+
"has_think": false,
|
| 24 |
+
"has_plan": false,
|
| 25 |
"has_answer": true,
|
| 26 |
+
"has_decompose": false,
|
| 27 |
+
"think_preview": "",
|
| 28 |
+
"answer_preview": "ReasoningEngine disabled. SOM prediction: short_text",
|
| 29 |
+
"raw_response_preview": "<answer>ReasoningEngine disabled. SOM prediction: short_text</answer>",
|
| 30 |
+
"n_tags": 1,
|
| 31 |
+
"latency_ms": 4.777431488037109
|
| 32 |
},
|
| 33 |
{
|
| 34 |
"query": "Lula reserva valor",
|
|
|
|
| 37 |
"system_prompt_used": false,
|
| 38 |
"few_shot_examples": false,
|
| 39 |
"som_prediction": "short_text",
|
| 40 |
+
"reasoning_length": 69,
|
| 41 |
+
"has_think": false,
|
| 42 |
+
"has_plan": false,
|
| 43 |
"has_answer": true,
|
| 44 |
+
"has_decompose": false,
|
| 45 |
+
"think_preview": "",
|
| 46 |
+
"answer_preview": "ReasoningEngine disabled. SOM prediction: short_text",
|
| 47 |
+
"raw_response_preview": "<answer>ReasoningEngine disabled. SOM prediction: short_text</answer>",
|
| 48 |
+
"n_tags": 1,
|
| 49 |
+
"latency_ms": 4.729270935058594
|
| 50 |
},
|
| 51 |
{
|
| 52 |
"query": "Amazonas força-tarefa vítimas",
|
|
|
|
| 55 |
"system_prompt_used": false,
|
| 56 |
"few_shot_examples": false,
|
| 57 |
"som_prediction": "short_text",
|
| 58 |
+
"reasoning_length": 69,
|
| 59 |
+
"has_think": false,
|
| 60 |
+
"has_plan": false,
|
| 61 |
"has_answer": true,
|
| 62 |
+
"has_decompose": false,
|
| 63 |
+
"think_preview": "",
|
| 64 |
+
"answer_preview": "ReasoningEngine disabled. SOM prediction: short_text",
|
| 65 |
+
"raw_response_preview": "<answer>ReasoningEngine disabled. SOM prediction: short_text</answer>",
|
| 66 |
+
"n_tags": 1,
|
| 67 |
+
"latency_ms": 4.624843597412109
|
| 68 |
}
|
| 69 |
],
|
| 70 |
"summary": {
|
| 71 |
"n_with_answer": 3,
|
| 72 |
+
"n_with_think": 0,
|
| 73 |
"answer_rate": 1.0,
|
| 74 |
+
"think_rate": 0.0,
|
| 75 |
+
"avg_latency_ms": 4.7105153401692705,
|
| 76 |
+
"avg_reasoning_length": 69.0
|
| 77 |
},
|
| 78 |
"quality_assessment": {
|
| 79 |
"model_not_helped": true,
|