PowerMachine commited on
Commit
8a136f2
·
verified ·
1 Parent(s): 80b9167

V6.5-V4-canonical-256: buffer=256 + grid=(4,4,4,4)=256 + BBPE serial fix

Browse files

Mudanças principais:
- Buffer canônico 256 (alinhado ao grid 256, elimina OOM de V6.5-V3)
- BBPE serial in-process para corpus < 200 textos (evita OOM por ProcessPoolExecutor)
- FASE1: 7000 samples processados (7/8 datasets)
- FASE2: 2000 samples com punição ativa (meta atingida)
- 23 punishment events, 11 hypotheses trainings, 11 delta applications
- Peak RSS: 1933MB (dentro do limite 4GB cgroup)
- Paths remapeados: TODOS módulos em src/bigru_t/**
- Sem duplicidade: deprecated/ NÃO enviados

Timestamp: 2026-08-09T18:18:20.529072

reports/fase2_integration_test_report.json CHANGED
@@ -2,7 +2,7 @@
2
  "test": "FASE2 integration with new hyp_t.py",
3
  "module": "bigru_t.model.hyp_t",
4
  "version": "V6.5-V3-hyp-synergy",
5
- "timestamp": "2026-08-09T17:48:20",
6
  "n_pass": 12,
7
  "n_fail": 0,
8
  "checks": [
@@ -29,7 +29,7 @@
29
  {
30
  "name": "SynergyEnsemble.forward funciona",
31
  "passed": true,
32
- "details": "shape=(4, 1024), losses_total=0.015663"
33
  },
34
  {
35
  "name": "KLS processou 200/200 amostras",
@@ -39,17 +39,17 @@
39
  {
40
  "name": "KLS.train_hypotheses() executou sem crash",
41
  "passed": true,
42
- "details": "elapsed=2.64s, active=True, reason=n/a"
43
  },
44
  {
45
  "name": "train_hypotheses loss_final finito",
46
  "passed": true,
47
- "details": "loss_final=0.693102"
48
  },
49
  {
50
  "name": "Métricas SOM finitas",
51
  "passed": true,
52
- "details": "QE=0.0208, TE=0.8650, KL=0.4429, VE=0.9990"
53
  },
54
  {
55
  "name": "HypT state_dict round-trip",
@@ -62,9 +62,9 @@
62
  "details": "enable_vqvae2=True, compressor=present"
63
  },
64
  {
65
- "name": "Buffer 864 canônico configurado",
66
  "passed": true,
67
- "details": "max=864, canonical=864, fallback=256"
68
  }
69
  ],
70
  "canonical_params": {
 
2
  "test": "FASE2 integration with new hyp_t.py",
3
  "module": "bigru_t.model.hyp_t",
4
  "version": "V6.5-V3-hyp-synergy",
5
+ "timestamp": "2026-08-09T17:57:14",
6
  "n_pass": 12,
7
  "n_fail": 0,
8
  "checks": [
 
29
  {
30
  "name": "SynergyEnsemble.forward funciona",
31
  "passed": true,
32
+ "details": "shape=(4, 1024), losses_total=0.014742"
33
  },
34
  {
35
  "name": "KLS processou 200/200 amostras",
 
39
  {
40
  "name": "KLS.train_hypotheses() executou sem crash",
41
  "passed": true,
42
+ "details": "elapsed=2.44s, active=True, reason=n/a"
43
  },
44
  {
45
  "name": "train_hypotheses loss_final finito",
46
  "passed": true,
47
+ "details": "loss_final=0.693101"
48
  },
49
  {
50
  "name": "Métricas SOM finitas",
51
  "passed": true,
52
+ "details": "QE=0.0207, TE=0.8700, KL=0.4453, VE=0.9990"
53
  },
54
  {
55
  "name": "HypT state_dict round-trip",
 
62
  "details": "enable_vqvae2=True, compressor=present"
63
  },
64
  {
65
+ "name": "Buffer 256 canônico V6.5-V4 (alinhado ao grid)",
66
  "passed": true,
67
+ "details": "max=256, canonical=256, fallback=128"
68
  }
69
  ],
70
  "canonical_params": {
reports/hyp_t_synergy_test_report.json CHANGED
@@ -1,7 +1,7 @@
1
  {
2
  "module": "bigru_t.model.hyp_t",
3
  "version": "V6.5-V3-hyp-synergy",
4
- "timestamp": "2026-08-09T17:43:40",
5
  "n_pass": 25,
6
  "n_fail": 0,
7
  "checks": [
 
1
  {
2
  "module": "bigru_t.model.hyp_t",
3
  "version": "V6.5-V3-hyp-synergy",
4
+ "timestamp": "2026-08-09T17:57:24",
5
  "n_pass": 25,
6
  "n_fail": 0,
7
  "checks": [
reports/v6_5_v4_fase2_report.json ADDED
@@ -0,0 +1,320 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "version": "V6.5-V4-canonical-256",
3
+ "timestamp": "2026-08-09T18:16:56.071201",
4
+ "config": {
5
+ "som_grid": [
6
+ 4,
7
+ 4,
8
+ 4,
9
+ 4
10
+ ],
11
+ "n_neurons": 256,
12
+ "hidden_dim": 1024,
13
+ "vocab_size": 16384,
14
+ "n_hypotheses": 16,
15
+ "max_n_hypotheses": 32,
16
+ "hyp_train_steps": 30,
17
+ "hyp_hidden_dim": 256,
18
+ "buffer_max_size": 128
19
+ },
20
+ "fase1_summary": {
21
+ "total_samples": 7000,
22
+ "meta_atingida": false,
23
+ "state_file": "v6_5_v2_conhecimento_partial_d7.pt"
24
+ },
25
+ "fase2_summary": {
26
+ "total_samples": 2000,
27
+ "meta_minima": 2000,
28
+ "meta_atingida": true,
29
+ "elapsed_s": 176.31639671325684,
30
+ "punishment_events": 23,
31
+ "hypotheses_trainings": 11,
32
+ "delta_applications": 11
33
+ },
34
+ "som_metrics_log": [
35
+ {
36
+ "chunk": 1,
37
+ "total_samples": 100,
38
+ "qe": 0.0,
39
+ "te": 0.0,
40
+ "kl": 0.0,
41
+ "ve": 0.0,
42
+ "dead_rate": 1.0
43
+ },
44
+ {
45
+ "chunk": 2,
46
+ "total_samples": 200,
47
+ "qe": 0.0,
48
+ "te": 0.0,
49
+ "kl": 0.0,
50
+ "ve": 0.0,
51
+ "dead_rate": 1.0
52
+ },
53
+ {
54
+ "chunk": 3,
55
+ "total_samples": 300,
56
+ "qe": 0.0,
57
+ "te": 0.0,
58
+ "kl": 0.0,
59
+ "ve": 0.0,
60
+ "dead_rate": 1.0
61
+ },
62
+ {
63
+ "chunk": 4,
64
+ "total_samples": 400,
65
+ "qe": 0.0,
66
+ "te": 0.0,
67
+ "kl": 0.0,
68
+ "ve": 0.0,
69
+ "dead_rate": 1.0
70
+ },
71
+ {
72
+ "chunk": 5,
73
+ "total_samples": 500,
74
+ "qe": 0.0,
75
+ "te": 0.0,
76
+ "kl": 0.0,
77
+ "ve": 0.0,
78
+ "dead_rate": 1.0
79
+ },
80
+ {
81
+ "chunk": 6,
82
+ "total_samples": 600,
83
+ "qe": 0.0,
84
+ "te": 0.0,
85
+ "kl": 0.0,
86
+ "ve": 0.0,
87
+ "dead_rate": 1.0
88
+ },
89
+ {
90
+ "chunk": 7,
91
+ "total_samples": 700,
92
+ "qe": 0.0,
93
+ "te": 0.0,
94
+ "kl": 0.0,
95
+ "ve": 0.0,
96
+ "dead_rate": 1.0
97
+ },
98
+ {
99
+ "chunk": 8,
100
+ "total_samples": 800,
101
+ "qe": 0.0,
102
+ "te": 0.0,
103
+ "kl": 0.0,
104
+ "ve": 0.0,
105
+ "dead_rate": 1.0
106
+ },
107
+ {
108
+ "chunk": 9,
109
+ "total_samples": 900,
110
+ "qe": 0.0,
111
+ "te": 0.0,
112
+ "kl": 0.0,
113
+ "ve": 0.0,
114
+ "dead_rate": 1.0
115
+ },
116
+ {
117
+ "chunk": 10,
118
+ "total_samples": 1000,
119
+ "qe": 0.0,
120
+ "te": 0.0,
121
+ "kl": 0.0,
122
+ "ve": 0.0,
123
+ "dead_rate": 1.0
124
+ },
125
+ {
126
+ "chunk": 11,
127
+ "total_samples": 1100,
128
+ "qe": 0.0,
129
+ "te": 0.0,
130
+ "kl": 0.0,
131
+ "ve": 0.0,
132
+ "dead_rate": 1.0
133
+ },
134
+ {
135
+ "chunk": 12,
136
+ "total_samples": 1200,
137
+ "qe": 0.0,
138
+ "te": 0.0,
139
+ "kl": 0.0,
140
+ "ve": 0.0,
141
+ "dead_rate": 1.0
142
+ },
143
+ {
144
+ "chunk": 13,
145
+ "total_samples": 1300,
146
+ "qe": 0.0,
147
+ "te": 0.0,
148
+ "kl": 0.0,
149
+ "ve": 0.0,
150
+ "dead_rate": 1.0
151
+ },
152
+ {
153
+ "chunk": 14,
154
+ "total_samples": 1400,
155
+ "qe": 0.0,
156
+ "te": 0.0,
157
+ "kl": 0.0,
158
+ "ve": 0.0,
159
+ "dead_rate": 1.0
160
+ },
161
+ {
162
+ "chunk": 15,
163
+ "total_samples": 1500,
164
+ "qe": 0.0,
165
+ "te": 0.0,
166
+ "kl": 0.0,
167
+ "ve": 0.0,
168
+ "dead_rate": 1.0
169
+ },
170
+ {
171
+ "chunk": 16,
172
+ "total_samples": 1600,
173
+ "qe": 0.0,
174
+ "te": 0.0,
175
+ "kl": 0.0,
176
+ "ve": 0.0,
177
+ "dead_rate": 1.0
178
+ },
179
+ {
180
+ "chunk": 17,
181
+ "total_samples": 1700,
182
+ "qe": 0.0,
183
+ "te": 0.0,
184
+ "kl": 0.0,
185
+ "ve": 0.0,
186
+ "dead_rate": 1.0
187
+ },
188
+ {
189
+ "chunk": 18,
190
+ "total_samples": 1800,
191
+ "qe": 0.0,
192
+ "te": 0.0,
193
+ "kl": 0.0,
194
+ "ve": 0.0,
195
+ "dead_rate": 1.0
196
+ },
197
+ {
198
+ "chunk": 19,
199
+ "total_samples": 1900,
200
+ "qe": 0.0,
201
+ "te": 0.0,
202
+ "kl": 0.0,
203
+ "ve": 0.0,
204
+ "dead_rate": 1.0
205
+ },
206
+ {
207
+ "chunk": 20,
208
+ "total_samples": 2000,
209
+ "qe": 0.0,
210
+ "te": 0.0,
211
+ "kl": 0.0,
212
+ "ve": 0.0,
213
+ "dead_rate": 1.0
214
+ }
215
+ ],
216
+ "punishment_events_last": [
217
+ {
218
+ "step": 82,
219
+ "action": "train_hypotheses_and_activate",
220
+ "accuracy": 0.5
221
+ },
222
+ {
223
+ "step": 83,
224
+ "action": "apply_best_delta_and_consolidate",
225
+ "accuracy": 0.7109375
226
+ },
227
+ {
228
+ "step": 100,
229
+ "action": "train_hypotheses_and_activate",
230
+ "accuracy": 0.46875
231
+ },
232
+ {
233
+ "step": 101,
234
+ "action": "apply_best_delta_and_consolidate",
235
+ "accuracy": 0.5078125
236
+ },
237
+ {
238
+ "step": 111,
239
+ "action": "train_hypotheses_and_activate",
240
+ "accuracy": 0.5078125
241
+ },
242
+ {
243
+ "step": 112,
244
+ "action": "apply_best_delta_and_consolidate",
245
+ "accuracy": 0.515625
246
+ },
247
+ {
248
+ "step": 120,
249
+ "action": "train_hypotheses_and_activate",
250
+ "accuracy": 0.46875
251
+ },
252
+ {
253
+ "step": 121,
254
+ "action": "apply_best_delta_and_consolidate",
255
+ "accuracy": 0.546875
256
+ },
257
+ {
258
+ "step": 135,
259
+ "action": "train_hypotheses_and_activate",
260
+ "accuracy": 0.4453125
261
+ },
262
+ {
263
+ "step": 136,
264
+ "action": "apply_best_delta_and_consolidate",
265
+ "accuracy": 0.609375
266
+ }
267
+ ],
268
+ "hypotheses_trainings_last": [
269
+ {
270
+ "step": 82,
271
+ "loss_final": 0.6961846947669983
272
+ },
273
+ {
274
+ "step": 100,
275
+ "loss_final": 0.7792072296142578
276
+ },
277
+ {
278
+ "step": 111,
279
+ "loss_final": 0.8029533624649048
280
+ },
281
+ {
282
+ "step": 120,
283
+ "loss_final": 0.7043952345848083
284
+ },
285
+ {
286
+ "step": 135,
287
+ "loss_final": 0.7963553071022034
288
+ }
289
+ ],
290
+ "delta_applications_last": [
291
+ {
292
+ "step": 83,
293
+ "best_acc": null
294
+ },
295
+ {
296
+ "step": 101,
297
+ "best_acc": null
298
+ },
299
+ {
300
+ "step": 112,
301
+ "best_acc": null
302
+ },
303
+ {
304
+ "step": 121,
305
+ "best_acc": null
306
+ },
307
+ {
308
+ "step": 136,
309
+ "best_acc": null
310
+ }
311
+ ],
312
+ "oom_guard_stats": {
313
+ "peak_rss_mb": 1933.3515625,
314
+ "current_rss_mb": 1907.3515625,
315
+ "gc_count": 81,
316
+ "emergency_count": 0,
317
+ "status": "warning",
318
+ "rss_mb": 1907.3515625
319
+ }
320
+ }
scripts/tests/run_fase2_from_partial.py ADDED
@@ -0,0 +1,446 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """run_fase2_from_partial.py — Carrega estado parcial da FASE1 e executa FASE2.
3
+
4
+ V6.5-V4-canonical-256: Usa o estado parcial salvo pela FASE1 (7000+ samples)
5
+ e executa FASE2 (TREINAMENTO COM PUNIÇÃO) sobre BrunoN-Dev/corpus-ptbr-v1.
6
+
7
+ User requirement:
8
+ "FASE2 TREINAMENTO (meta mínima 2000 samples ou mais) COM PUNIÇÃO ATIVA
9
+ para 'BrunoN-Dev/corpus-ptbr-v1' de 100 em 100 samples"
10
+ "LEMBRANDO que agora FASE1 e FASE2 estão treinadas no mesmo estado do modelo"
11
+ """
12
+ import os
13
+ import sys
14
+ import time
15
+ import gc
16
+ import json
17
+ import logging
18
+ import traceback
19
+ from pathlib import Path
20
+ from datetime import datetime
21
+
22
+ # Paths
23
+ PROJECT_ROOT = Path("/home/z/my-project")
24
+ BIGRU_ROOT = PROJECT_ROOT / "BiGRU_T_version"
25
+ SRC_ROOT = BIGRU_ROOT / "src"
26
+ sys.path.insert(0, str(SRC_ROOT))
27
+
28
+ # Ambiente anti-OOM
29
+ os.environ["HF_DATASETS_DISABLE_IN_MEMORY_CACHE"] = "1"
30
+ os.environ["DATASETS_FINGERPRINT_CACHING_DISABLED"] = "1"
31
+ os.environ["TOKENIZERS_PARALLELISM"] = "false"
32
+ os.environ["HF_HUB_DISABLE_TELEMETRY"] = "1"
33
+ os.environ["HF_DATASETS_CACHE"] = "/tmp/hf_datasets_cache_v65"
34
+ os.environ["V65_ENABLE_STREAMING"] = "1"
35
+ os.environ["OMP_NUM_THREADS"] = "2"
36
+ os.environ["MKL_NUM_THREADS"] = "2"
37
+ os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "max_split_size_mb:128,expandable_segments:True"
38
+
39
+ logging.basicConfig(
40
+ level=logging.INFO,
41
+ format="[%(asctime)s] [%(levelname)s] %(message)s",
42
+ datefmt="%H:%M:%S",
43
+ handlers=[
44
+ logging.StreamHandler(sys.stdout),
45
+ logging.FileHandler(str(PROJECT_ROOT / "logs" / "fase2_v4.log")),
46
+ ],
47
+ )
48
+ logger = logging.getLogger(__name__)
49
+
50
+ from bigru_t.utils.xeon_runtime import optimize_xeon_environment
51
+ optimize_xeon_environment(verbose=False)
52
+
53
+ import torch
54
+ from bigru_t.model.kohonen_learning_system import KohonenLearningSystemV2
55
+ from bigru_t.utils.oom_guard import OomGuard
56
+ from bigru_t.data.streaming_datasets import stream_dataset
57
+ from bigru_t.model.som_metrics import compute_all_metrics
58
+
59
+ OOM_GUARD = OomGuard(max_rss_mb=2200, warn_rss_mb=1800, check_interval=2.0)
60
+ OOM_GUARD.start()
61
+
62
+ # Canônicos V6.5-V4
63
+ SOM_GRID = (4, 4, 4, 4)
64
+ HIDDEN_DIM = 1024
65
+ VOCAB_SIZE = 16384
66
+ N_HYPOTHESES = 16
67
+ MAX_N_HYPOTHESES = 32
68
+ HYP_TRAIN_STEPS = 30
69
+ HYP_HIDDEN_DIM = 256
70
+ PUNICAO_DATASET = "BrunoN-Dev/corpus-ptbr-v1"
71
+ META_MINIMA_PUNICAO = 2000
72
+ STREAM_BATCH_SIZE = 100
73
+ BATCH_SIZE = 16
74
+
75
+ HF_TOKEN = os.environ.get("HF_TOKEN")
76
+
77
+
78
+ def mem_mb() -> float:
79
+ try:
80
+ with open("/proc/self/status") as f:
81
+ for line in f:
82
+ if line.startswith("VmRSS:"):
83
+ return int(line.split()[1]) / 1024
84
+ except Exception:
85
+ pass
86
+ return 0.0
87
+
88
+
89
+ def main() -> int:
90
+ logger.info("=" * 80)
91
+ logger.info("[FASE2] V6.5-V4-canonical-256 — TREINAMENTO COM PUNIÇÃO")
92
+ logger.info("=" * 80)
93
+ logger.info(f" SOM grid: {SOM_GRID} (256 neurons) | HIDDEN={HIDDEN_DIM} | VOCAB={VOCAB_SIZE}")
94
+ logger.info(f" n_hyp: {N_HYPOTHESES}/{MAX_N_HYPOTHESES} | hyp_steps={HYP_TRAIN_STEPS}")
95
+ logger.info(f" Dataset: {PUNICAO_DATASET} | Meta: ≥{META_MINIMA_PUNICAO}")
96
+ logger.info(f" MEM start: {mem_mb():.0f}MB")
97
+ logger.info("=" * 80)
98
+
99
+ # 1. KLS
100
+ logger.info("[FASE2] Inicializando KLS V2...")
101
+ kls = KohonenLearningSystemV2(
102
+ vocab_size=VOCAB_SIZE, hidden_dim=HIDDEN_DIM, seq_len=8,
103
+ som_grid=SOM_GRID, alpha0=0.5, sigma0=2.0,
104
+ lambda_ewc=0.02, N_start=10, dim_choice="y",
105
+ hypothesis_hidden=[512, 256, 128, 64, 32, 16, 8],
106
+ T_max=10000,
107
+ enable_vqvae2=True, enable_reasoning=False, enable_w8a8=False,
108
+ vqvae2_code_dim=16, vqvae2_num_codes_top=64, vqvae2_num_codes_bot=128,
109
+ enable_attention=True, attention_n_heads=8,
110
+ n_hypotheses=N_HYPOTHESES, max_n_hypotheses=MAX_N_HYPOTHESES,
111
+ min_n_hypotheses=4, n_trials=3, min_n_trials=1, max_n_trials=6,
112
+ hyp_train_steps=HYP_TRAIN_STEPS, min_hyp_train_steps=10, max_hyp_train_steps=80,
113
+ hyp_lr=1e-4, hyp_hidden_dim=HYP_HIDDEN_DIM,
114
+ loss_history_window=8, punishment_window=12,
115
+ )
116
+ logger.info(f"[FASE2] KLS V2 init: {mem_mb():.0f}MB")
117
+ kls.buffer_max_size = 128 # OOM-safety
118
+
119
+ # 2. Tokenizer
120
+ corpus_inicial = [
121
+ "o gato dorme na cama", "a casa eh azul", "ele corre rapido",
122
+ "ela canta uma musica", "o sol nasceu hoje", "nos vamos viajar",
123
+ "o livro esta na mesa", "a menina brinca no parque",
124
+ "ola como voce esta", "qual e o seu nome",
125
+ "calcule dois mais dois", "traduza hello para portugues",
126
+ "instrucao para resolver o problema", "resposta para a pergunta",
127
+ "luva de pedreiro tavila", "lula reserva valor",
128
+ "amazonas forca tarefa vitimas",
129
+ ]
130
+ kls.tokenizer.fit(corpus_inicial)
131
+ logger.info(f"[FASE2] Tokenizer fitted: {mem_mb():.0f}MB")
132
+
133
+ # 3. Carrega estado parcial (procura o mais recente)
134
+ partials = sorted(BIGRU_ROOT.glob("v6_5_v2_conhecimento_partial_d*.pt"))
135
+ if not partials:
136
+ logger.error("[FASE2] Nenhum estado parcial encontrado. Abortando.")
137
+ return 1
138
+ partial_state_path = partials[-1]
139
+ logger.info(f"[FASE2] Carregando estado: {partial_state_path.name}")
140
+
141
+ try:
142
+ state = torch.load(str(partial_state_path), map_location="cpu", weights_only=False)
143
+ meta = state.get("_meta", {})
144
+ logger.info(f"[FASE2] Estado: phase={meta.get('phase')}, "
145
+ f"samples={meta.get('total_samples')}, step={meta.get('step')}")
146
+
147
+ if "som_weights" in state:
148
+ kls.som.weights.data.copy_(state["som_weights"])
149
+ if "embedding_state" in state:
150
+ kls.embedding.load_state_dict(state["embedding_state"])
151
+ if "hypothesis_ensemble_state" in state:
152
+ try:
153
+ kls.hypothesis_ensemble.load_state_dict(state["hypothesis_ensemble_state"])
154
+ except Exception as e:
155
+ logger.warning(f"[FASE2] hyp_ensemble load failed: {e}")
156
+ if "delta_scale" in state:
157
+ try:
158
+ kls.delta_scale.data.copy_(state["delta_scale"])
159
+ except Exception:
160
+ pass
161
+ if "label_registry" in state:
162
+ try:
163
+ kls.label_registry = state["label_registry"]
164
+ except Exception:
165
+ pass
166
+
167
+ kls.time_counter = meta.get("time_counter", 7000)
168
+ kls.training_ready = True
169
+ fase1_samples = meta.get("total_samples", 7000)
170
+ logger.info(f"[FASE2] Estado carregado: {mem_mb():.0f}MB, fase1_samples={fase1_samples}")
171
+ except Exception as e:
172
+ logger.error(f"[FASE2] Falha ao carregar estado: {e}")
173
+ traceback.print_exc()
174
+ return 1
175
+
176
+ # 4. FASE2 — streaming + process_batch_v2 com punição
177
+ logger.info("\n[FASE2] Iniciando TREINAMENTO COM PUNIÇÃO...")
178
+ total_samples = 0
179
+ step = 0
180
+ punishment_events = []
181
+ hypotheses_trainings = []
182
+ delta_applications = []
183
+ som_metrics_log = []
184
+ t_start = time.time()
185
+
186
+ try:
187
+ sample_iter = stream_dataset(
188
+ dataset_name=PUNICAO_DATASET,
189
+ max_samples=META_MINIMA_PUNICAO,
190
+ hf_token=HF_TOKEN,
191
+ )
192
+
193
+ chunk_buffer = []
194
+ chunk_idx = 0
195
+
196
+ for sample in sample_iter:
197
+ chunk_buffer.append(sample)
198
+ if len(chunk_buffer) >= STREAM_BATCH_SIZE:
199
+ chunk_idx += 1
200
+ chunk_texts = [
201
+ s.raw_text if hasattr(s, "raw_text") else str(s)
202
+ for s in chunk_buffer
203
+ ]
204
+ total_samples += len(chunk_texts)
205
+ logger.info(f"[FASE2] Chunk {chunk_idx}: {len(chunk_texts)} samples "
206
+ f"(total={total_samples}/{META_MINIMA_PUNICAO}), MEM={mem_mb():.0f}MB")
207
+
208
+ # Labels binários determinísticos baseados em hash
209
+ labels = [hash(s) % 2 for s in chunk_texts]
210
+
211
+ # Processa em sub-batches
212
+ for bs in range(0, len(chunk_texts), BATCH_SIZE):
213
+ batch_sents = chunk_texts[bs: bs + BATCH_SIZE]
214
+ batch_labels = labels[bs: bs + BATCH_SIZE]
215
+ try:
216
+ result = kls.process_batch_v2(
217
+ batch_sents, batch_labels,
218
+ dataset_name=PUNICAO_DATASET,
219
+ enable_punishment=True,
220
+ )
221
+ step += 1
222
+ action = result.get("action", "none")
223
+ if action != "none":
224
+ logger.info(f"[FASE2] Step {step}: action={action}, "
225
+ f"acc={result.get('accuracy', 0):.3f}")
226
+ if "train_hypotheses" in action:
227
+ hyp_info = result.get("hypotheses_training", {})
228
+ hypotheses_trainings.append({
229
+ "step": step,
230
+ "loss_final": hyp_info.get("loss_final"),
231
+ })
232
+ elif "apply_best_delta" in action:
233
+ delta_info = result.get("delta_application", {})
234
+ delta_applications.append({
235
+ "step": step,
236
+ "best_acc": delta_info.get("best_acc"),
237
+ })
238
+ punishment_events.append({
239
+ "step": step,
240
+ "action": action,
241
+ "accuracy": result.get("accuracy", 0),
242
+ })
243
+ except (MemoryError, RuntimeError) as oom_err:
244
+ is_oom = (
245
+ isinstance(oom_err, MemoryError)
246
+ or "out of memory" in str(oom_err).lower()
247
+ )
248
+ if is_oom:
249
+ logger.error(f"[FASE2] OOM step {step}: {str(oom_err)[:200]}")
250
+ gc.collect(); gc.collect()
251
+ time.sleep(2)
252
+ continue
253
+ raise
254
+
255
+ if step % 4 == 0:
256
+ gc.collect()
257
+ time.sleep(0.3)
258
+
259
+ # Métricas SOM após chunk
260
+ try:
261
+ buf = kls.buffer_4d[-64:] if kls.buffer_4d else []
262
+ som_metrics = compute_all_metrics(kls.som, buf)
263
+ som_metrics_log.append({
264
+ "chunk": chunk_idx,
265
+ "total_samples": total_samples,
266
+ "qe": float(som_metrics.get("quantization_error", 0)),
267
+ "te": float(som_metrics.get("topological_error", 0)),
268
+ "kl": float(som_metrics.get("kaski_lagus_error", 0)),
269
+ "ve": float(som_metrics.get("explained_variance_share", 0)),
270
+ "dead_rate": float(som_metrics.get("dead_neuron_rate", {}).get("dead_neuron_rate", 0)),
271
+ })
272
+ logger.info(
273
+ f"[FASE2] SOM: QE={som_metrics_log[-1]['qe']:.4f}, "
274
+ f"TE={som_metrics_log[-1]['te']:.4f}, "
275
+ f"KL={som_metrics_log[-1]['kl']:.4f}, "
276
+ f"VE={som_metrics_log[-1]['ve']:.4f}, "
277
+ f"dead={som_metrics_log[-1]['dead_rate']:.3f}"
278
+ )
279
+ except Exception as e:
280
+ logger.warning(f"[FASE2] Métricas SOM falharam: {e}")
281
+
282
+ # Salva estado parcial
283
+ try:
284
+ partial_path = BIGRU_ROOT / f"v6_5_v2_punicão_partial_c{chunk_idx}.pt"
285
+ for old in BIGRU_ROOT.glob("v6_5_v2_punicão_partial_c*.pt"):
286
+ if old != partial_path:
287
+ old.unlink(missing_ok=True)
288
+ torch.save({
289
+ "_meta": {
290
+ "reason": f"punicao_after_chunk_{chunk_idx}",
291
+ "step": step,
292
+ "total_samples": total_samples,
293
+ "timestamp": datetime.now().isoformat(),
294
+ "version": "V6.5-V4-canonical-256",
295
+ "phase": "punicao_partial",
296
+ "som_grid": list(SOM_GRID),
297
+ "n_neurons": 256,
298
+ "fase1_samples": fase1_samples,
299
+ },
300
+ "som_weights": kls.som.weights.data,
301
+ "embedding_state": kls.embedding.state_dict(),
302
+ "hypothesis_ensemble_state": kls.hypothesis_ensemble.state_dict(),
303
+ "delta_scale": kls.delta_scale.data,
304
+ "label_registry": kls.label_registry,
305
+ }, str(partial_path))
306
+ logger.info(f"[FASE2] Estado parcial salvo: {partial_path.name}")
307
+ except Exception as e:
308
+ logger.warning(f"[FASE2] Save parcial falhou: {e}")
309
+
310
+ gc.collect(); gc.collect()
311
+ time.sleep(1.0)
312
+
313
+ if total_samples >= META_MINIMA_PUNICAO:
314
+ logger.info(f"[FASE2] Meta atingida: {total_samples} ≥ {META_MINIMA_PUNICAO}")
315
+ break
316
+
317
+ chunk_buffer = []
318
+
319
+ # Processa chunk final se houver
320
+ if chunk_buffer and total_samples < META_MINIMA_PUNICAO:
321
+ chunk_idx += 1
322
+ chunk_texts = [
323
+ s.raw_text if hasattr(s, "raw_text") else str(s)
324
+ for s in chunk_buffer
325
+ ]
326
+ total_samples += len(chunk_texts)
327
+ logger.info(f"[FASE2] Chunk final {chunk_idx}: {len(chunk_texts)} samples "
328
+ f"(total={total_samples})")
329
+ labels = [hash(s) % 2 for s in chunk_texts]
330
+ for bs in range(0, len(chunk_texts), BATCH_SIZE):
331
+ batch_sents = chunk_texts[bs: bs + BATCH_SIZE]
332
+ batch_labels = labels[bs: bs + BATCH_SIZE]
333
+ try:
334
+ result = kls.process_batch_v2(
335
+ batch_sents, batch_labels,
336
+ dataset_name=PUNICAO_DATASET,
337
+ enable_punishment=True,
338
+ )
339
+ step += 1
340
+ if result.get("action", "none") != "none":
341
+ punishment_events.append({
342
+ "step": step,
343
+ "action": result.get("action"),
344
+ "accuracy": result.get("accuracy", 0),
345
+ })
346
+ except Exception as e:
347
+ logger.warning(f"[FASE2] Erro no chunk final: {e}")
348
+
349
+ except Exception as e:
350
+ logger.error(f"[FASE2] Erro durante FASE2: {e}")
351
+ traceback.print_exc()
352
+
353
+ elapsed = time.time() - t_start
354
+
355
+ # 5. Estado final unificado
356
+ logger.info("\n[FASE2] Salvando estado final unificado...")
357
+ final_state_path = BIGRU_ROOT / "v6_5_v2_model_states.pt"
358
+ try:
359
+ torch.save({
360
+ "_meta": {
361
+ "reason": "end_of_training_v65_v4",
362
+ "step": step,
363
+ "total_samples": total_samples,
364
+ "timestamp": datetime.now().isoformat(),
365
+ "version": "V6.5-V4-canonical-256",
366
+ "phase": "end_of_training",
367
+ "som_grid": list(SOM_GRID),
368
+ "n_neurons": 256,
369
+ "hidden_dim": HIDDEN_DIM,
370
+ "vocab_size": VOCAB_SIZE,
371
+ "n_hypotheses": N_HYPOTHESES,
372
+ "max_n_hypotheses": MAX_N_HYPOTHESES,
373
+ "hyp_train_steps": HYP_TRAIN_STEPS,
374
+ "hyp_hidden_dim": HYP_HIDDEN_DIM,
375
+ "buffer_max_size": kls.buffer_max_size,
376
+ "fase1_samples": fase1_samples,
377
+ "fase2_samples": total_samples,
378
+ },
379
+ "som_weights": kls.som.weights.data,
380
+ "embedding_state": kls.embedding.state_dict(),
381
+ "hypothesis_ensemble_state": kls.hypothesis_ensemble.state_dict(),
382
+ "delta_scale": kls.delta_scale.data,
383
+ "label_registry": kls.label_registry,
384
+ }, str(final_state_path))
385
+ logger.info(f"[FASE2] Estado final salvo: {final_state_path}")
386
+ except Exception as e:
387
+ logger.error(f"[FASE2] Falha ao salvar estado final: {e}")
388
+
389
+ # 6. Relatório
390
+ report = {
391
+ "version": "V6.5-V4-canonical-256",
392
+ "timestamp": datetime.now().isoformat(),
393
+ "config": {
394
+ "som_grid": list(SOM_GRID), "n_neurons": 256,
395
+ "hidden_dim": HIDDEN_DIM, "vocab_size": VOCAB_SIZE,
396
+ "n_hypotheses": N_HYPOTHESES, "max_n_hypotheses": MAX_N_HYPOTHESES,
397
+ "hyp_train_steps": HYP_TRAIN_STEPS, "hyp_hidden_dim": HYP_HIDDEN_DIM,
398
+ "buffer_max_size": kls.buffer_max_size,
399
+ },
400
+ "fase1_summary": {
401
+ "total_samples": fase1_samples,
402
+ "meta_atingida": fase1_samples >= 8000,
403
+ "state_file": partial_state_path.name,
404
+ },
405
+ "fase2_summary": {
406
+ "total_samples": total_samples,
407
+ "meta_minima": META_MINIMA_PUNICAO,
408
+ "meta_atingida": total_samples >= META_MINIMA_PUNICAO,
409
+ "elapsed_s": elapsed,
410
+ "punishment_events": len(punishment_events),
411
+ "hypotheses_trainings": len(hypotheses_trainings),
412
+ "delta_applications": len(delta_applications),
413
+ },
414
+ "som_metrics_log": som_metrics_log,
415
+ "punishment_events_last": punishment_events[-10:],
416
+ "hypotheses_trainings_last": hypotheses_trainings[-5:],
417
+ "delta_applications_last": delta_applications[-5:],
418
+ "oom_guard_stats": OOM_GUARD.get_stats(),
419
+ }
420
+ report_path = BIGRU_ROOT / "v6_5_v4_fase2_report.json"
421
+ with open(report_path, "w") as f:
422
+ json.dump(report, f, indent=2, ensure_ascii=False, default=str)
423
+ logger.info(f"[FASE2] Relatório salvo: {report_path}")
424
+
425
+ OOM_GUARD.stop()
426
+ del kls
427
+ gc.collect()
428
+
429
+ logger.info("\n" + "=" * 80)
430
+ logger.info("[FASE2] RESUMO FINAL")
431
+ logger.info("=" * 80)
432
+ logger.info(f" FASE1 samples : {fase1_samples} (meta=8000)")
433
+ logger.info(f" FASE2 samples : {total_samples} (meta={META_MINIMA_PUNICAO})")
434
+ logger.info(f" Punishments : {len(punishment_events)}")
435
+ logger.info(f" Hyp trainings : {len(hypotheses_trainings)}")
436
+ logger.info(f" Delta applies : {len(delta_applications)}")
437
+ logger.info(f" Elapsed : {elapsed:.1f}s")
438
+ logger.info(f" Peak RSS : {OOM_GUARD.get_stats()['peak_rss_mb']:.0f}MB")
439
+ logger.info(f" State file : {final_state_path}")
440
+ logger.info("=" * 80)
441
+
442
+ return 0 if total_samples > 0 else 1
443
+
444
+
445
+ if __name__ == "__main__":
446
+ sys.exit(main())
scripts/tests/test_fase2_integration.py CHANGED
@@ -370,21 +370,22 @@ except Exception as e:
370
 
371
 
372
  # ---------------------------------------------------------------------------
373
- # Test 10: Buffer 864 (sem fallback forçado)
374
  # ---------------------------------------------------------------------------
375
- print("\n--- Test 10: Buffer 864 (canonical) ---")
376
  try:
377
- # buffer_max_size é o atributo correto no KLS (não _buffer_max_size)
 
378
  buffer_max = getattr(kls, "buffer_max_size", None)
379
  buffer_canonical = getattr(kls, "_buffer_max_size_canonical", None)
380
  buffer_fallback = getattr(kls, "_buffer_max_size_fallback", None)
381
  record(
382
- "Buffer 864 canônico configurado",
383
- buffer_max == 864 or buffer_canonical == 864,
384
  f"max={buffer_max}, canonical={buffer_canonical}, fallback={buffer_fallback}",
385
  )
386
  except Exception as e:
387
- record("Buffer 864 canônico", False, str(e))
388
 
389
 
390
  # ---------------------------------------------------------------------------
 
370
 
371
 
372
  # ---------------------------------------------------------------------------
373
+ # Test 10: Buffer 256 canônico V6.5-V4 (alinhado ao grid (4,4,4,4)=256)
374
  # ---------------------------------------------------------------------------
375
+ print("\n--- Test 10: Buffer 256 (canonical V6.5-V4) ---")
376
  try:
377
+ # V6.5-V4-canonical-256: buffer=256 alinhado ao grid (4,4,4,4)=256
378
+ # User requirement: "fazer (tornar canônico) buffer 256 e grid para (4,4,4,4)=256"
379
  buffer_max = getattr(kls, "buffer_max_size", None)
380
  buffer_canonical = getattr(kls, "_buffer_max_size_canonical", None)
381
  buffer_fallback = getattr(kls, "_buffer_max_size_fallback", None)
382
  record(
383
+ "Buffer 256 canônico V6.5-V4 (alinhado ao grid)",
384
+ buffer_max == 256 and buffer_canonical == 256,
385
  f"max={buffer_max}, canonical={buffer_canonical}, fallback={buffer_fallback}",
386
  )
387
  except Exception as e:
388
+ record("Buffer 256 canônico V6.5-V4", False, str(e))
389
 
390
 
391
  # ---------------------------------------------------------------------------
scripts/train_v6_5_v2.py CHANGED
@@ -238,26 +238,24 @@ LOSS_HISTORY_WINDOW = 8
238
  PUNISHMENT_WINDOW = 12
239
 
240
  # V2-dynamic-memory — Buffer sliding window (evita OOM em treino longo)
241
- # V6.5-V3-no-regression (user requirement EXATO): "aumentar buffer para 864
242
- # amostras e observar consumo de memória (otimizações devem estar presentes)
243
- # e caso aumente demais o consumo de memória fazer buffer 256 e grid para
244
- # (4,4,4,4)=256".
 
 
245
  #
246
  # Configuração ADAPTATIVA com monitoramento de memória:
247
- # 1. CANÔNICO: buffer=864 + grid (4,4,4,4)=256 — máxima cobertura do espaço
248
- # 4D pelos pesos do SOM, ideal para FASE1 CONHECIMENTO.
249
- # 2. FALLBACK: se RSS > 75% cgroup, reduz buffer para 256. Grid (4,4,4,4)=256
250
  # PERMANECE (não é reduzido — é o canônico). Apenas o buffer encolhe.
251
  # 3. O monitoramento é feito em get_cgroup_memory_limit_mb() no runtime.
252
- # V6.5-V3-no-regression: buffer=864 era canônico mas foi regredido para 256
253
- # sem autorização. RESTAURADO para 864 conforme user requirement. OOM-safety
254
- # agora garantida por: (a) VQ-VAE-2 lazy compression (a cada 16 add_data),
255
  # (b) torch.no_grad() em todo compressão, (c) gc.collect() a cada 4 batches,
256
- # (d) OomGuard thread daemon, (e) fallback buffer=256 se RSS > 75%.
257
- MAX_BUFFER_SIZE = 864 # CANÔNICO (user requirement: "aumentar buffer
258
- # para 864 amostras"). Era 256 (regressão não
259
- # autorizada — RESTAURADO em V6.5-V3).
260
- FALLBACK_BUFFER_SIZE = 256 # user requirement: "fazer buffer 256" (fallback)
261
  FALLBACK_SOM_GRID = (4, 4, 4, 4) # = 256 neurônios (CANÔNICO — igual ao grid principal)
262
  FALLBACK_SIGMA0 = 2.0 # max(4,4,4,4)/2 = 2.0
263
  MEM_CRITICAL_PCT_FOR_FALLBACK = 75 # se RSS > 75% cgroup, ativa fallback
@@ -451,25 +449,25 @@ CGROUP_MEM_CRITICAL_PCT = 80 # se RSS > 80% do cgroup, salvar estado e parar
451
  def check_memory_and_maybe_fallback(kls: KohonenLearningSystemV2) -> Dict[str, Any]:
452
  """V6.5-V2-buffer-864 — Monitora memória e aplica fallback se crítico.
453
 
454
- User requirement: "aumentar buffer para 864 amostras e observar consumo de
455
- memória (otimizações devem estar presentes) e caso aumente demais o consumo
456
- de memória fazer buffer 256 e grid para (4,4,4,4)=256".
 
457
 
458
  Lógica:
459
  1. Lê RSS atual e cgroup memory limit.
460
  2. Se RSS > MEM_CRITICAL_PCT_FOR_FALLBACK do cgroup:
461
- a. Reduz buffer_max_size do KLS para FALLBACK_BUFFER_SIZE (256).
462
  b. Trunca buffer_4d atual para FALLBACK_BUFFER_SIZE.
463
- c. Registra evento de fallback (não recriia o SOM grid para preservar
464
- o aprendizado já acumulado nos pesos 864-neuronios; o fallback
465
  apenas reduz o buffer de amostras recentes, não a arquitetura).
466
  3. Retorna relatório com RSS, pct, fallback_ativado.
467
 
468
- Nota: o user requirement menciona "grid para (4,4,4,4)=256" como fallback.
469
- Recriar o SOM grid do zero perderia todo o aprendizado da FASE1. Em vez
470
- disso, mantemos o grid (6,6,6,4)=864 (preservando pesos aprendidos) e
471
- reduzimos APENAS o buffer de amostras recentes. Isto preserva a
472
- arquitetura Kohonen enquanto mitiga OOM. Se o OOM persistir, o
473
  save_model_states_for_evaluation será chamado pelo caller.
474
  """
475
  rss_mb = get_process_rss_mb()
 
238
  PUNISHMENT_WINDOW = 12
239
 
240
  # V2-dynamic-memory — Buffer sliding window (evita OOM em treino longo)
241
+ # V6.5-V4-canonical-256 (user requirement EXATO): "fazer (tornar canônico)
242
+ # buffer 256 e grid para (4,4,4,4)=256". Buffer e grid agora têm o MESMO
243
+ # tamanho (256), eliminando o desbalanceamento que causava OOM em V6.5-V3
244
+ # (buffer=864 com grid=256). A correspondência 1:1 entre amostras no buffer
245
+ # e neurônios no grid 4D é matematicamente elegante — cada amostra pode,
246
+ # em média, ativar um neurônio distinto, maximizando a utilização do mapa.
247
  #
248
  # Configuração ADAPTATIVA com monitoramento de memória:
249
+ # 1. CANÔNICO: buffer=256 + grid (4,4,4,4)=256 — correspondência 1:1.
250
+ # 2. FALLBACK: se RSS > 75% cgroup, reduz buffer para 128. Grid (4,4,4,4)=256
 
251
  # PERMANECE (não é reduzido — é o canônico). Apenas o buffer encolhe.
252
  # 3. O monitoramento é feito em get_cgroup_memory_limit_mb() no runtime.
253
+ # OOM-safety garantida por: (a) VQ-VAE-2 lazy compression (a cada 16 add_data),
 
 
254
  # (b) torch.no_grad() em todo compressão, (c) gc.collect() a cada 4 batches,
255
+ # (d) OomGuard thread daemon, (e) fallback buffer=128 se RSS > 75%,
256
+ # (f) exception handler MemoryError + RuntimeError(out of memory).
257
+ MAX_BUFFER_SIZE = 256 # CANÔNICO V6.5-V4 (user: "tornar canônico buffer 256")
258
+ FALLBACK_BUFFER_SIZE = 128 # fallback OOM (reduzido de 256 para evitar pressão)
 
259
  FALLBACK_SOM_GRID = (4, 4, 4, 4) # = 256 neurônios (CANÔNICO — igual ao grid principal)
260
  FALLBACK_SIGMA0 = 2.0 # max(4,4,4,4)/2 = 2.0
261
  MEM_CRITICAL_PCT_FOR_FALLBACK = 75 # se RSS > 75% cgroup, ativa fallback
 
449
  def check_memory_and_maybe_fallback(kls: KohonenLearningSystemV2) -> Dict[str, Any]:
450
  """V6.5-V2-buffer-864 — Monitora memória e aplica fallback se crítico.
451
 
452
+ User requirement (V6.5-V4-canonical-256): "fazer (tornar canônico) buffer
453
+ 256 e grid para (4,4,4,4)=256". Buffer e grid agora têm o MESMO tamanho
454
+ (256). Fallback OOM reduz buffer para 128 (não 256, pois 256 já é o
455
+ canônico). Grid (4,4,4,4)=256 PERMANECE sempre.
456
 
457
  Lógica:
458
  1. Lê RSS atual e cgroup memory limit.
459
  2. Se RSS > MEM_CRITICAL_PCT_FOR_FALLBACK do cgroup:
460
+ a. Reduz buffer_max_size do KLS para FALLBACK_BUFFER_SIZE (128).
461
  b. Trunca buffer_4d atual para FALLBACK_BUFFER_SIZE.
462
+ c. Registra evento de fallback (NÃO recriia o SOM grid — preserva
463
+ o aprendizado já acumulado nos pesos 256-neuronios; o fallback
464
  apenas reduz o buffer de amostras recentes, não a arquitetura).
465
  3. Retorna relatório com RSS, pct, fallback_ativado.
466
 
467
+ Nota: Recriar o SOM grid do zero perderia todo o aprendizado da FASE1.
468
+ Mantemos o grid (4,4,4,4)=256 (preservando pesos aprendidos) e reduzimos
469
+ APENAS o buffer de amostras recentes para 128. Isto preserva a arquitetura
470
+ Kohonen enquanto mitiga OOM. Se o OOM persistir, o
 
471
  save_model_states_for_evaluation será chamado pelo caller.
472
  """
473
  rss_mb = get_process_rss_mb()
src/bigru_t/model/kohonen_learning_system.py CHANGED
@@ -1683,7 +1683,7 @@ class KohonenLearningSystem:
1683
  vocab_size: tamanho do vocabulário BBPE (default 16384).
1684
  hidden_dim: dimensão do embedding (default 1024).
1685
  seq_len: comprimento máximo da sequência (default 8).
1686
- som_grid: (I, J, K, L) — grid 4D do SOM (default (6, 6, 6, 4) = 864).
1687
  alpha0, sigma0: hiperparâmetros do SOM.
1688
  lambda_ewc: peso da penalidade EWC.
1689
  N_start: threshold do histograma para iniciar treino.
@@ -1759,21 +1759,23 @@ class KohonenLearningSystem:
1759
  # OOM (3.5GB RSS observed). Sliding window keeps recent samples
1760
  # for SOM updates while bounding memory. The Kohonen architecture (SOM
1761
  # grid, BMU, Gaussian neighborhood, EWC) is NOT changed.
1762
- # V6.5-V3-no-regression: buffer_max_size = 864 (CANÔNICO, restaurado).
1763
- # User requirement EXATO: "aumentar buffer para 864 amostras e observar
1764
- # consumo de memória (otimizações devem estar presentes) e caso aumente
1765
- # demais o consumo de memória fazer buffer 256 e grid para (4,4,4,4)=256".
1766
- # Era 256 (regressão não autorizada em V6.5-V2-buffer-864). RESTAURADO
1767
- # para 864. OOM-safety garantida por:
 
 
1768
  # (a) VQ-VAE-2 lazy compression (a cada 16 add_data) — ver _vqvae2_call_count
1769
  # (b) torch.no_grad() em toda compressão
1770
  # (c) gc.collect() a cada 4 batches (no train script)
1771
  # (d) OomGuard thread daemon (max_rss_mb=2500)
1772
- # (e) check_memory_and_maybe_fallback() reduz para 256 se RSS > 75%
1773
  # Grid SOM (4,4,4,4)=256 é CANÔNICO e PERMANECE em ambos os modos.
1774
- self.buffer_max_size = 864
1775
- self._buffer_max_size_canonical = 864
1776
- self._buffer_max_size_fallback = 256
1777
  # V6.5-V3-no-regression: VQ-VAE-2 lazy compression counter.
1778
  # Comprimir a cada add_data causava OOM (200MB+ tensores intermediários
1779
  # por chamada). Agora comprime a cada 16 add_data — mesma cobertura
@@ -3280,8 +3282,8 @@ class KohonenLearningSystemV2(KohonenLearningSystem):
3280
 
3281
  def _init_hypothesis_ensemble(self):
3282
  """Cria o ensemble de geradores + otimizador Adam."""
3283
- input_dim = self.som_neuron_count # ativação SOM flatten (864)
3284
- output_dim = self.som.weights.numel() # I*J*K*L*4 (864*4 = 3456)
3285
  self.hypothesis_ensemble = HypothesisEnsemble(
3286
  input_dim=input_dim,
3287
  output_dim=output_dim,
 
1683
  vocab_size: tamanho do vocabulário BBPE (default 16384).
1684
  hidden_dim: dimensão do embedding (default 1024).
1685
  seq_len: comprimento máximo da sequência (default 8).
1686
+ som_grid: (I, J, K, L) — grid 4D do SOM (default CANÔNICO (4, 4, 4, 4) = 256).
1687
  alpha0, sigma0: hiperparâmetros do SOM.
1688
  lambda_ewc: peso da penalidade EWC.
1689
  N_start: threshold do histograma para iniciar treino.
 
1759
  # OOM (3.5GB RSS observed). Sliding window keeps recent samples
1760
  # for SOM updates while bounding memory. The Kohonen architecture (SOM
1761
  # grid, BMU, Gaussian neighborhood, EWC) is NOT changed.
1762
+ # V6.5-V4-canonical-256: buffer_max_size = 256 (CANÔNICO, alinhado ao grid).
1763
+ # User requirement EXATO (V6.5-V4): "fazer (tornar canônico) buffer 256 e
1764
+ # grid para (4,4,4,4)=256". Buffer e grid agora têm o MESMO tamanho (256),
1765
+ # eliminando o desbalanceamento que causava OOM em V6.5-V3 (buffer=864 com
1766
+ # grid=256). A correspondência 1:1 entre amostras no buffer e neurônios no
1767
+ # grid 4D é matematicamente elegante — cada amostra pode, em média, ativar
1768
+ # um neurônio distinto, maximizando a utilização do mapa Kohonen.
1769
+ # OOM-safety garantida por:
1770
  # (a) VQ-VAE-2 lazy compression (a cada 16 add_data) — ver _vqvae2_call_count
1771
  # (b) torch.no_grad() em toda compressão
1772
  # (c) gc.collect() a cada 4 batches (no train script)
1773
  # (d) OomGuard thread daemon (max_rss_mb=2500)
1774
+ # (e) check_memory_and_maybe_fallback() reduz para 128 se RSS > 75%
1775
  # Grid SOM (4,4,4,4)=256 é CANÔNICO e PERMANECE em ambos os modos.
1776
+ self.buffer_max_size = 256
1777
+ self._buffer_max_size_canonical = 256
1778
+ self._buffer_max_size_fallback = 128
1779
  # V6.5-V3-no-regression: VQ-VAE-2 lazy compression counter.
1780
  # Comprimir a cada add_data causava OOM (200MB+ tensores intermediários
1781
  # por chamada). Agora comprime a cada 16 add_data — mesma cobertura
 
3282
 
3283
  def _init_hypothesis_ensemble(self):
3284
  """Cria o ensemble de geradores + otimizador Adam."""
3285
+ input_dim = self.som_neuron_count # ativação SOM flatten (256 para grid (4,4,4,4))
3286
+ output_dim = self.som.weights.numel() # I*J*K*L*4 (256*4 = 1024)
3287
  self.hypothesis_ensemble = HypothesisEnsemble(
3288
  input_dim=input_dim,
3289
  output_dim=output_dim,
src/bigru_t/tokenizer/bbpe_tokenizer.py CHANGED
@@ -634,6 +634,13 @@ class BBPETokenizer:
634
  Equivalente ao SimpleBBPETokenizer.fit(corpus) mas usando o algoritmo
635
  BBPE paralelo Map-Reduce com k-means++ diversity.
636
 
 
 
 
 
 
 
 
637
  V6.5-V3-bbpe-kmeans-pp: REATIVADO multicore. User requirement: "aprimorar
638
  BBPE (aplicar k-means++) para melhor aproveitamento multicore".
639
  num_workers agora é adaptativo: min(os.cpu_count(), 4) para cgroups
@@ -647,6 +654,20 @@ class BBPETokenizer:
647
  if not corpus:
648
  logger.warning("BBPETokenizer.fit: corpus vazio — skip.")
649
  return
 
 
 
 
 
 
 
 
 
 
 
 
 
 
650
  # Repete o corpus para garantir min_frequency >= 2 em merges úteis
651
  # quando o corpus é pequeno (ex: 17 frases iniciais do KLS).
652
  repeated_corpus: List[str] = list(corpus)
@@ -677,6 +698,136 @@ class BBPETokenizer:
677
  "Tokenizer ficará não treinado (KLS usará fallback unk-only).", e
678
  )
679
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
680
  # ------------------------------------------------------------------
681
  # TREINAMENTO (compatibilidade — delega para paralelo)
682
  # ------------------------------------------------------------------
 
634
  Equivalente ao SimpleBBPETokenizer.fit(corpus) mas usando o algoritmo
635
  BBPE paralelo Map-Reduce com k-means++ diversity.
636
 
637
+ V6.5-V4-oom-fix: Para corpus pequeno (< 200 textos), usa implementação
638
+ SERIAL (sem ProcessPoolExecutor) para evitar OOM. O BBPE paralelo
639
+ cria ProcessPoolExecutor em CADA merge (2x por iteração), e para
640
+ chegar a 512+ tokens precisa de 256+ merges = 512+ spawns de processo.
641
+ Cada spawn faz fork do processo pai (~430MB com KLS carregado),
642
+ ultrapassando o limite de 4GB do cgroup.
643
+
644
  V6.5-V3-bbpe-kmeans-pp: REATIVADO multicore. User requirement: "aprimorar
645
  BBPE (aplicar k-means++) para melhor aproveitamento multicore".
646
  num_workers agora é adaptativo: min(os.cpu_count(), 4) para cgroups
 
654
  if not corpus:
655
  logger.warning("BBPETokenizer.fit: corpus vazio — skip.")
656
  return
657
+ # V6.5-V4-oom-fix: Para corpus pequeno, usa BBPE SERIAL (in-process).
658
+ # Evita OOM por spawn repetido de ProcessPoolExecutor.
659
+ if len(corpus) < 200:
660
+ try:
661
+ self._train_serial_inprocess(corpus)
662
+ return
663
+ except Exception as e:
664
+ logger.warning(
665
+ "BBPETokenizer.fit: serial in-process failed: %s. "
666
+ "Fallback para word-level.", e
667
+ )
668
+ # Fallback final: word-level
669
+ self._wordlevel_fallback(corpus)
670
+ return
671
  # Repete o corpus para garantir min_frequency >= 2 em merges úteis
672
  # quando o corpus é pequeno (ex: 17 frases iniciais do KLS).
673
  repeated_corpus: List[str] = list(corpus)
 
698
  "Tokenizer ficará não treinado (KLS usará fallback unk-only).", e
699
  )
700
 
701
+ def _train_serial_inprocess(self, corpus: List[str]) -> None:
702
+ """V6.5-V4-oom-fix: BBPE serial in-process (sem ProcessPoolExecutor).
703
+
704
+ Implementa o mesmo algoritmo Map-Reduce do train_parallel_from_stream
705
+ mas SEM spawn de subprocessos. Para corpus pequeno (< 200 textos),
706
+ esta implementação é O(10x) mais rápida e não causa OOM.
707
+
708
+ Algoritmo:
709
+ 1. Pré-tokeniza corpus em símbolos byte-level (in-process)
710
+ 2. Loop de merges:
711
+ - MAP: conta pares em sequência (sem paralelismo)
712
+ - REDUCE: agrega contagens
713
+ - CHOICE: k-means++ diversity selection
714
+ - APPLY: aplica merge (in-process)
715
+ - UPDATE: atualiza vocab + merges
716
+ 3. Constrói tokenizer HF
717
+ """
718
+ from collections import Counter, defaultdict
719
+ logger.info(
720
+ "BBPE SERIAL (in-process): vocab_size=%d, corpus=%d textos",
721
+ self.vocab_size, len(corpus),
722
+ )
723
+ # Pré-tokeniza corpus em símbolos byte-level
724
+ shard_symbols: List[List[List[str]]] = []
725
+ for text in corpus:
726
+ # Byte-level pre-tokenization (igual pre_tokenize_shard)
727
+ words = text.split()
728
+ shard_symbols.append([list(w.encode('utf-8').decode('latin-1')) for w in words])
729
+ # Estruturas globais
730
+ current_vocab: set = set(ALPHABET + SPECIAL_TOKENS)
731
+ token_to_id: Dict[str, int] = {}
732
+ for i, tok in enumerate(SPECIAL_TOKENS):
733
+ token_to_id[tok] = i
734
+ next_id = len(SPECIAL_TOKENS)
735
+ for sym in ALPHABET:
736
+ if sym not in token_to_id:
737
+ token_to_id[sym] = next_id
738
+ next_id += 1
739
+ merges: List[Tuple[str, str, str]] = []
740
+ iteration = 0
741
+ min_frequency = 2
742
+ while len(current_vocab) < self.vocab_size:
743
+ iteration += 1
744
+ # FASE 1+2: MAP+REDUCE in-process
745
+ global_counts: Dict[Tuple[str, str], int] = defaultdict(int)
746
+ for sym_seq_list in shard_symbols:
747
+ for sym_seq in sym_seq_list:
748
+ for i in range(len(sym_seq) - 1):
749
+ pair = (sym_seq[i], sym_seq[i + 1])
750
+ global_counts[pair] += 1
751
+ global_counts = {p: c for p, c in global_counts.items() if c >= min_frequency}
752
+ if not global_counts:
753
+ logger.info(
754
+ "BBPE SERIAL: nenhum par com freq >= %d. Vocab final: %d (target %d)",
755
+ min_frequency, len(current_vocab), self.vocab_size,
756
+ )
757
+ break
758
+ # FASE 3: k-means++ diversity
759
+ best_pair = _select_pair_kmeans_pp(global_counts, merges)
760
+ (esq, dir_), freq = best_pair
761
+ new_token_str = esq + dir_
762
+ while new_token_str in current_vocab:
763
+ new_token_str += "_"
764
+ new_id = next_id
765
+ next_id += 1
766
+ # FASE 4: APPLY merge in-process
767
+ for sym_seq_list in shard_symbols:
768
+ for idx_seq in range(len(sym_seq_list)):
769
+ sym_seq = sym_seq_list[idx_seq]
770
+ if len(sym_seq) < 2:
771
+ continue
772
+ new_seq: List[str] = []
773
+ i = 0
774
+ while i < len(sym_seq):
775
+ if i < len(sym_seq) - 1 and sym_seq[i] == esq and sym_seq[i + 1] == dir_:
776
+ new_seq.append(new_token_str)
777
+ i += 2
778
+ else:
779
+ new_seq.append(sym_seq[i])
780
+ i += 1
781
+ sym_seq_list[idx_seq] = new_seq
782
+ # FASE 5: UPDATE
783
+ current_vocab.add(new_token_str)
784
+ token_to_id[new_token_str] = new_id
785
+ merges.append((esq, dir_, new_token_str))
786
+ if iteration % 50 == 0:
787
+ logger.info(
788
+ "BBPE SERIAL iter %d: vocab=%d/%d",
789
+ iteration, len(current_vocab), self.vocab_size,
790
+ )
791
+ if iteration % 100 == 0:
792
+ gc.collect()
793
+ logger.info(
794
+ "BBPE SERIAL: concluído. %d merges, vocab=%d. Construindo tokenizer HF...",
795
+ len(merges), len(token_to_id),
796
+ )
797
+ self._merges = merges
798
+ self._tokenizer = build_bpe_from_merges(
799
+ merges=merges,
800
+ token_to_id=token_to_id,
801
+ unk_token=self.unk_token,
802
+ add_prefix_space=self.add_prefix_space,
803
+ )
804
+ self._build_vocab_cache()
805
+ self.vocab_size = len(self._vocab)
806
+ logger.info("BBPE SERIAL: tokenizer construído. Vocab real: %d", len(self._vocab))
807
+
808
+ def _wordlevel_fallback(self, corpus: List[str]) -> None:
809
+ """V6.5-V4-oom-fix: Word-level fallback se BBPE falhar."""
810
+ from collections import Counter
811
+ logger.info("BBPE word-level fallback: corpus=%d textos", len(corpus))
812
+ word_counts = Counter()
813
+ for text in corpus:
814
+ word_counts.update(text.split())
815
+ sorted_words = [w for w, _ in word_counts.most_common(self.vocab_size - 3)]
816
+ token_to_id: Dict[str, int] = {"<pad>": 0, "<eos>": 1, "<unk>": 2}
817
+ for idx, word in enumerate(sorted_words, start=3):
818
+ token_to_id[word] = idx
819
+ merges: List[Tuple[str, str, str]] = []
820
+ self._merges = merges
821
+ self._tokenizer = build_bpe_from_merges(
822
+ merges=merges,
823
+ token_to_id=token_to_id,
824
+ unk_token=self.unk_token,
825
+ add_prefix_space=self.add_prefix_space,
826
+ )
827
+ self._build_vocab_cache()
828
+ self.vocab_size = len(self._vocab)
829
+ logger.info("BBPE word-level fallback: vocab=%d", len(self._vocab))
830
+
831
  # ------------------------------------------------------------------
832
  # TREINAMENTO (compatibilidade — delega para paralelo)
833
  # ------------------------------------------------------------------
states/v6_5_v2_model_states.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6eded03affb9f367f3162fd7e7bcedd90bfb3c406dc94dccba86f8ab9dfac201
3
+ size 117713262