V6.5-V4-canonical-256: buffer=256 + grid=(4,4,4,4)=256 + BBPE serial fix
Browse filesMudanças principais:
- Buffer canônico 256 (alinhado ao grid 256, elimina OOM de V6.5-V3)
- BBPE serial in-process para corpus < 200 textos (evita OOM por ProcessPoolExecutor)
- FASE1: 7000 samples processados (7/8 datasets)
- FASE2: 2000 samples com punição ativa (meta atingida)
- 23 punishment events, 11 hypotheses trainings, 11 delta applications
- Peak RSS: 1933MB (dentro do limite 4GB cgroup)
- Paths remapeados: TODOS módulos em src/bigru_t/**
- Sem duplicidade: deprecated/ NÃO enviados
Timestamp: 2026-08-09T18:18:20.529072
- reports/fase2_integration_test_report.json +7 -7
- reports/hyp_t_synergy_test_report.json +1 -1
- reports/v6_5_v4_fase2_report.json +320 -0
- scripts/tests/run_fase2_from_partial.py +446 -0
- scripts/tests/test_fase2_integration.py +7 -6
- scripts/train_v6_5_v2.py +24 -26
- src/bigru_t/model/kohonen_learning_system.py +15 -13
- src/bigru_t/tokenizer/bbpe_tokenizer.py +151 -0
- states/v6_5_v2_model_states.pt +3 -0
reports/fase2_integration_test_report.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
| 2 |
"test": "FASE2 integration with new hyp_t.py",
|
| 3 |
"module": "bigru_t.model.hyp_t",
|
| 4 |
"version": "V6.5-V3-hyp-synergy",
|
| 5 |
-
"timestamp": "2026-08-09T17:
|
| 6 |
"n_pass": 12,
|
| 7 |
"n_fail": 0,
|
| 8 |
"checks": [
|
|
@@ -29,7 +29,7 @@
|
|
| 29 |
{
|
| 30 |
"name": "SynergyEnsemble.forward funciona",
|
| 31 |
"passed": true,
|
| 32 |
-
"details": "shape=(4, 1024), losses_total=0.
|
| 33 |
},
|
| 34 |
{
|
| 35 |
"name": "KLS processou 200/200 amostras",
|
|
@@ -39,17 +39,17 @@
|
|
| 39 |
{
|
| 40 |
"name": "KLS.train_hypotheses() executou sem crash",
|
| 41 |
"passed": true,
|
| 42 |
-
"details": "elapsed=2.
|
| 43 |
},
|
| 44 |
{
|
| 45 |
"name": "train_hypotheses loss_final finito",
|
| 46 |
"passed": true,
|
| 47 |
-
"details": "loss_final=0.
|
| 48 |
},
|
| 49 |
{
|
| 50 |
"name": "Métricas SOM finitas",
|
| 51 |
"passed": true,
|
| 52 |
-
"details": "QE=0.
|
| 53 |
},
|
| 54 |
{
|
| 55 |
"name": "HypT state_dict round-trip",
|
|
@@ -62,9 +62,9 @@
|
|
| 62 |
"details": "enable_vqvae2=True, compressor=present"
|
| 63 |
},
|
| 64 |
{
|
| 65 |
-
"name": "Buffer
|
| 66 |
"passed": true,
|
| 67 |
-
"details": "max=
|
| 68 |
}
|
| 69 |
],
|
| 70 |
"canonical_params": {
|
|
|
|
| 2 |
"test": "FASE2 integration with new hyp_t.py",
|
| 3 |
"module": "bigru_t.model.hyp_t",
|
| 4 |
"version": "V6.5-V3-hyp-synergy",
|
| 5 |
+
"timestamp": "2026-08-09T17:57:14",
|
| 6 |
"n_pass": 12,
|
| 7 |
"n_fail": 0,
|
| 8 |
"checks": [
|
|
|
|
| 29 |
{
|
| 30 |
"name": "SynergyEnsemble.forward funciona",
|
| 31 |
"passed": true,
|
| 32 |
+
"details": "shape=(4, 1024), losses_total=0.014742"
|
| 33 |
},
|
| 34 |
{
|
| 35 |
"name": "KLS processou 200/200 amostras",
|
|
|
|
| 39 |
{
|
| 40 |
"name": "KLS.train_hypotheses() executou sem crash",
|
| 41 |
"passed": true,
|
| 42 |
+
"details": "elapsed=2.44s, active=True, reason=n/a"
|
| 43 |
},
|
| 44 |
{
|
| 45 |
"name": "train_hypotheses loss_final finito",
|
| 46 |
"passed": true,
|
| 47 |
+
"details": "loss_final=0.693101"
|
| 48 |
},
|
| 49 |
{
|
| 50 |
"name": "Métricas SOM finitas",
|
| 51 |
"passed": true,
|
| 52 |
+
"details": "QE=0.0207, TE=0.8700, KL=0.4453, VE=0.9990"
|
| 53 |
},
|
| 54 |
{
|
| 55 |
"name": "HypT state_dict round-trip",
|
|
|
|
| 62 |
"details": "enable_vqvae2=True, compressor=present"
|
| 63 |
},
|
| 64 |
{
|
| 65 |
+
"name": "Buffer 256 canônico V6.5-V4 (alinhado ao grid)",
|
| 66 |
"passed": true,
|
| 67 |
+
"details": "max=256, canonical=256, fallback=128"
|
| 68 |
}
|
| 69 |
],
|
| 70 |
"canonical_params": {
|
reports/hyp_t_synergy_test_report.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
| 1 |
{
|
| 2 |
"module": "bigru_t.model.hyp_t",
|
| 3 |
"version": "V6.5-V3-hyp-synergy",
|
| 4 |
-
"timestamp": "2026-08-09T17:
|
| 5 |
"n_pass": 25,
|
| 6 |
"n_fail": 0,
|
| 7 |
"checks": [
|
|
|
|
| 1 |
{
|
| 2 |
"module": "bigru_t.model.hyp_t",
|
| 3 |
"version": "V6.5-V3-hyp-synergy",
|
| 4 |
+
"timestamp": "2026-08-09T17:57:24",
|
| 5 |
"n_pass": 25,
|
| 6 |
"n_fail": 0,
|
| 7 |
"checks": [
|
reports/v6_5_v4_fase2_report.json
ADDED
|
@@ -0,0 +1,320 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"version": "V6.5-V4-canonical-256",
|
| 3 |
+
"timestamp": "2026-08-09T18:16:56.071201",
|
| 4 |
+
"config": {
|
| 5 |
+
"som_grid": [
|
| 6 |
+
4,
|
| 7 |
+
4,
|
| 8 |
+
4,
|
| 9 |
+
4
|
| 10 |
+
],
|
| 11 |
+
"n_neurons": 256,
|
| 12 |
+
"hidden_dim": 1024,
|
| 13 |
+
"vocab_size": 16384,
|
| 14 |
+
"n_hypotheses": 16,
|
| 15 |
+
"max_n_hypotheses": 32,
|
| 16 |
+
"hyp_train_steps": 30,
|
| 17 |
+
"hyp_hidden_dim": 256,
|
| 18 |
+
"buffer_max_size": 128
|
| 19 |
+
},
|
| 20 |
+
"fase1_summary": {
|
| 21 |
+
"total_samples": 7000,
|
| 22 |
+
"meta_atingida": false,
|
| 23 |
+
"state_file": "v6_5_v2_conhecimento_partial_d7.pt"
|
| 24 |
+
},
|
| 25 |
+
"fase2_summary": {
|
| 26 |
+
"total_samples": 2000,
|
| 27 |
+
"meta_minima": 2000,
|
| 28 |
+
"meta_atingida": true,
|
| 29 |
+
"elapsed_s": 176.31639671325684,
|
| 30 |
+
"punishment_events": 23,
|
| 31 |
+
"hypotheses_trainings": 11,
|
| 32 |
+
"delta_applications": 11
|
| 33 |
+
},
|
| 34 |
+
"som_metrics_log": [
|
| 35 |
+
{
|
| 36 |
+
"chunk": 1,
|
| 37 |
+
"total_samples": 100,
|
| 38 |
+
"qe": 0.0,
|
| 39 |
+
"te": 0.0,
|
| 40 |
+
"kl": 0.0,
|
| 41 |
+
"ve": 0.0,
|
| 42 |
+
"dead_rate": 1.0
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"chunk": 2,
|
| 46 |
+
"total_samples": 200,
|
| 47 |
+
"qe": 0.0,
|
| 48 |
+
"te": 0.0,
|
| 49 |
+
"kl": 0.0,
|
| 50 |
+
"ve": 0.0,
|
| 51 |
+
"dead_rate": 1.0
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"chunk": 3,
|
| 55 |
+
"total_samples": 300,
|
| 56 |
+
"qe": 0.0,
|
| 57 |
+
"te": 0.0,
|
| 58 |
+
"kl": 0.0,
|
| 59 |
+
"ve": 0.0,
|
| 60 |
+
"dead_rate": 1.0
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"chunk": 4,
|
| 64 |
+
"total_samples": 400,
|
| 65 |
+
"qe": 0.0,
|
| 66 |
+
"te": 0.0,
|
| 67 |
+
"kl": 0.0,
|
| 68 |
+
"ve": 0.0,
|
| 69 |
+
"dead_rate": 1.0
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"chunk": 5,
|
| 73 |
+
"total_samples": 500,
|
| 74 |
+
"qe": 0.0,
|
| 75 |
+
"te": 0.0,
|
| 76 |
+
"kl": 0.0,
|
| 77 |
+
"ve": 0.0,
|
| 78 |
+
"dead_rate": 1.0
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
"chunk": 6,
|
| 82 |
+
"total_samples": 600,
|
| 83 |
+
"qe": 0.0,
|
| 84 |
+
"te": 0.0,
|
| 85 |
+
"kl": 0.0,
|
| 86 |
+
"ve": 0.0,
|
| 87 |
+
"dead_rate": 1.0
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"chunk": 7,
|
| 91 |
+
"total_samples": 700,
|
| 92 |
+
"qe": 0.0,
|
| 93 |
+
"te": 0.0,
|
| 94 |
+
"kl": 0.0,
|
| 95 |
+
"ve": 0.0,
|
| 96 |
+
"dead_rate": 1.0
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"chunk": 8,
|
| 100 |
+
"total_samples": 800,
|
| 101 |
+
"qe": 0.0,
|
| 102 |
+
"te": 0.0,
|
| 103 |
+
"kl": 0.0,
|
| 104 |
+
"ve": 0.0,
|
| 105 |
+
"dead_rate": 1.0
|
| 106 |
+
},
|
| 107 |
+
{
|
| 108 |
+
"chunk": 9,
|
| 109 |
+
"total_samples": 900,
|
| 110 |
+
"qe": 0.0,
|
| 111 |
+
"te": 0.0,
|
| 112 |
+
"kl": 0.0,
|
| 113 |
+
"ve": 0.0,
|
| 114 |
+
"dead_rate": 1.0
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"chunk": 10,
|
| 118 |
+
"total_samples": 1000,
|
| 119 |
+
"qe": 0.0,
|
| 120 |
+
"te": 0.0,
|
| 121 |
+
"kl": 0.0,
|
| 122 |
+
"ve": 0.0,
|
| 123 |
+
"dead_rate": 1.0
|
| 124 |
+
},
|
| 125 |
+
{
|
| 126 |
+
"chunk": 11,
|
| 127 |
+
"total_samples": 1100,
|
| 128 |
+
"qe": 0.0,
|
| 129 |
+
"te": 0.0,
|
| 130 |
+
"kl": 0.0,
|
| 131 |
+
"ve": 0.0,
|
| 132 |
+
"dead_rate": 1.0
|
| 133 |
+
},
|
| 134 |
+
{
|
| 135 |
+
"chunk": 12,
|
| 136 |
+
"total_samples": 1200,
|
| 137 |
+
"qe": 0.0,
|
| 138 |
+
"te": 0.0,
|
| 139 |
+
"kl": 0.0,
|
| 140 |
+
"ve": 0.0,
|
| 141 |
+
"dead_rate": 1.0
|
| 142 |
+
},
|
| 143 |
+
{
|
| 144 |
+
"chunk": 13,
|
| 145 |
+
"total_samples": 1300,
|
| 146 |
+
"qe": 0.0,
|
| 147 |
+
"te": 0.0,
|
| 148 |
+
"kl": 0.0,
|
| 149 |
+
"ve": 0.0,
|
| 150 |
+
"dead_rate": 1.0
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"chunk": 14,
|
| 154 |
+
"total_samples": 1400,
|
| 155 |
+
"qe": 0.0,
|
| 156 |
+
"te": 0.0,
|
| 157 |
+
"kl": 0.0,
|
| 158 |
+
"ve": 0.0,
|
| 159 |
+
"dead_rate": 1.0
|
| 160 |
+
},
|
| 161 |
+
{
|
| 162 |
+
"chunk": 15,
|
| 163 |
+
"total_samples": 1500,
|
| 164 |
+
"qe": 0.0,
|
| 165 |
+
"te": 0.0,
|
| 166 |
+
"kl": 0.0,
|
| 167 |
+
"ve": 0.0,
|
| 168 |
+
"dead_rate": 1.0
|
| 169 |
+
},
|
| 170 |
+
{
|
| 171 |
+
"chunk": 16,
|
| 172 |
+
"total_samples": 1600,
|
| 173 |
+
"qe": 0.0,
|
| 174 |
+
"te": 0.0,
|
| 175 |
+
"kl": 0.0,
|
| 176 |
+
"ve": 0.0,
|
| 177 |
+
"dead_rate": 1.0
|
| 178 |
+
},
|
| 179 |
+
{
|
| 180 |
+
"chunk": 17,
|
| 181 |
+
"total_samples": 1700,
|
| 182 |
+
"qe": 0.0,
|
| 183 |
+
"te": 0.0,
|
| 184 |
+
"kl": 0.0,
|
| 185 |
+
"ve": 0.0,
|
| 186 |
+
"dead_rate": 1.0
|
| 187 |
+
},
|
| 188 |
+
{
|
| 189 |
+
"chunk": 18,
|
| 190 |
+
"total_samples": 1800,
|
| 191 |
+
"qe": 0.0,
|
| 192 |
+
"te": 0.0,
|
| 193 |
+
"kl": 0.0,
|
| 194 |
+
"ve": 0.0,
|
| 195 |
+
"dead_rate": 1.0
|
| 196 |
+
},
|
| 197 |
+
{
|
| 198 |
+
"chunk": 19,
|
| 199 |
+
"total_samples": 1900,
|
| 200 |
+
"qe": 0.0,
|
| 201 |
+
"te": 0.0,
|
| 202 |
+
"kl": 0.0,
|
| 203 |
+
"ve": 0.0,
|
| 204 |
+
"dead_rate": 1.0
|
| 205 |
+
},
|
| 206 |
+
{
|
| 207 |
+
"chunk": 20,
|
| 208 |
+
"total_samples": 2000,
|
| 209 |
+
"qe": 0.0,
|
| 210 |
+
"te": 0.0,
|
| 211 |
+
"kl": 0.0,
|
| 212 |
+
"ve": 0.0,
|
| 213 |
+
"dead_rate": 1.0
|
| 214 |
+
}
|
| 215 |
+
],
|
| 216 |
+
"punishment_events_last": [
|
| 217 |
+
{
|
| 218 |
+
"step": 82,
|
| 219 |
+
"action": "train_hypotheses_and_activate",
|
| 220 |
+
"accuracy": 0.5
|
| 221 |
+
},
|
| 222 |
+
{
|
| 223 |
+
"step": 83,
|
| 224 |
+
"action": "apply_best_delta_and_consolidate",
|
| 225 |
+
"accuracy": 0.7109375
|
| 226 |
+
},
|
| 227 |
+
{
|
| 228 |
+
"step": 100,
|
| 229 |
+
"action": "train_hypotheses_and_activate",
|
| 230 |
+
"accuracy": 0.46875
|
| 231 |
+
},
|
| 232 |
+
{
|
| 233 |
+
"step": 101,
|
| 234 |
+
"action": "apply_best_delta_and_consolidate",
|
| 235 |
+
"accuracy": 0.5078125
|
| 236 |
+
},
|
| 237 |
+
{
|
| 238 |
+
"step": 111,
|
| 239 |
+
"action": "train_hypotheses_and_activate",
|
| 240 |
+
"accuracy": 0.5078125
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"step": 112,
|
| 244 |
+
"action": "apply_best_delta_and_consolidate",
|
| 245 |
+
"accuracy": 0.515625
|
| 246 |
+
},
|
| 247 |
+
{
|
| 248 |
+
"step": 120,
|
| 249 |
+
"action": "train_hypotheses_and_activate",
|
| 250 |
+
"accuracy": 0.46875
|
| 251 |
+
},
|
| 252 |
+
{
|
| 253 |
+
"step": 121,
|
| 254 |
+
"action": "apply_best_delta_and_consolidate",
|
| 255 |
+
"accuracy": 0.546875
|
| 256 |
+
},
|
| 257 |
+
{
|
| 258 |
+
"step": 135,
|
| 259 |
+
"action": "train_hypotheses_and_activate",
|
| 260 |
+
"accuracy": 0.4453125
|
| 261 |
+
},
|
| 262 |
+
{
|
| 263 |
+
"step": 136,
|
| 264 |
+
"action": "apply_best_delta_and_consolidate",
|
| 265 |
+
"accuracy": 0.609375
|
| 266 |
+
}
|
| 267 |
+
],
|
| 268 |
+
"hypotheses_trainings_last": [
|
| 269 |
+
{
|
| 270 |
+
"step": 82,
|
| 271 |
+
"loss_final": 0.6961846947669983
|
| 272 |
+
},
|
| 273 |
+
{
|
| 274 |
+
"step": 100,
|
| 275 |
+
"loss_final": 0.7792072296142578
|
| 276 |
+
},
|
| 277 |
+
{
|
| 278 |
+
"step": 111,
|
| 279 |
+
"loss_final": 0.8029533624649048
|
| 280 |
+
},
|
| 281 |
+
{
|
| 282 |
+
"step": 120,
|
| 283 |
+
"loss_final": 0.7043952345848083
|
| 284 |
+
},
|
| 285 |
+
{
|
| 286 |
+
"step": 135,
|
| 287 |
+
"loss_final": 0.7963553071022034
|
| 288 |
+
}
|
| 289 |
+
],
|
| 290 |
+
"delta_applications_last": [
|
| 291 |
+
{
|
| 292 |
+
"step": 83,
|
| 293 |
+
"best_acc": null
|
| 294 |
+
},
|
| 295 |
+
{
|
| 296 |
+
"step": 101,
|
| 297 |
+
"best_acc": null
|
| 298 |
+
},
|
| 299 |
+
{
|
| 300 |
+
"step": 112,
|
| 301 |
+
"best_acc": null
|
| 302 |
+
},
|
| 303 |
+
{
|
| 304 |
+
"step": 121,
|
| 305 |
+
"best_acc": null
|
| 306 |
+
},
|
| 307 |
+
{
|
| 308 |
+
"step": 136,
|
| 309 |
+
"best_acc": null
|
| 310 |
+
}
|
| 311 |
+
],
|
| 312 |
+
"oom_guard_stats": {
|
| 313 |
+
"peak_rss_mb": 1933.3515625,
|
| 314 |
+
"current_rss_mb": 1907.3515625,
|
| 315 |
+
"gc_count": 81,
|
| 316 |
+
"emergency_count": 0,
|
| 317 |
+
"status": "warning",
|
| 318 |
+
"rss_mb": 1907.3515625
|
| 319 |
+
}
|
| 320 |
+
}
|
scripts/tests/run_fase2_from_partial.py
ADDED
|
@@ -0,0 +1,446 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""run_fase2_from_partial.py — Carrega estado parcial da FASE1 e executa FASE2.
|
| 3 |
+
|
| 4 |
+
V6.5-V4-canonical-256: Usa o estado parcial salvo pela FASE1 (7000+ samples)
|
| 5 |
+
e executa FASE2 (TREINAMENTO COM PUNIÇÃO) sobre BrunoN-Dev/corpus-ptbr-v1.
|
| 6 |
+
|
| 7 |
+
User requirement:
|
| 8 |
+
"FASE2 TREINAMENTO (meta mínima 2000 samples ou mais) COM PUNIÇÃO ATIVA
|
| 9 |
+
para 'BrunoN-Dev/corpus-ptbr-v1' de 100 em 100 samples"
|
| 10 |
+
"LEMBRANDO que agora FASE1 e FASE2 estão treinadas no mesmo estado do modelo"
|
| 11 |
+
"""
|
| 12 |
+
import os
|
| 13 |
+
import sys
|
| 14 |
+
import time
|
| 15 |
+
import gc
|
| 16 |
+
import json
|
| 17 |
+
import logging
|
| 18 |
+
import traceback
|
| 19 |
+
from pathlib import Path
|
| 20 |
+
from datetime import datetime
|
| 21 |
+
|
| 22 |
+
# Paths
|
| 23 |
+
PROJECT_ROOT = Path("/home/z/my-project")
|
| 24 |
+
BIGRU_ROOT = PROJECT_ROOT / "BiGRU_T_version"
|
| 25 |
+
SRC_ROOT = BIGRU_ROOT / "src"
|
| 26 |
+
sys.path.insert(0, str(SRC_ROOT))
|
| 27 |
+
|
| 28 |
+
# Ambiente anti-OOM
|
| 29 |
+
os.environ["HF_DATASETS_DISABLE_IN_MEMORY_CACHE"] = "1"
|
| 30 |
+
os.environ["DATASETS_FINGERPRINT_CACHING_DISABLED"] = "1"
|
| 31 |
+
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
| 32 |
+
os.environ["HF_HUB_DISABLE_TELEMETRY"] = "1"
|
| 33 |
+
os.environ["HF_DATASETS_CACHE"] = "/tmp/hf_datasets_cache_v65"
|
| 34 |
+
os.environ["V65_ENABLE_STREAMING"] = "1"
|
| 35 |
+
os.environ["OMP_NUM_THREADS"] = "2"
|
| 36 |
+
os.environ["MKL_NUM_THREADS"] = "2"
|
| 37 |
+
os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "max_split_size_mb:128,expandable_segments:True"
|
| 38 |
+
|
| 39 |
+
logging.basicConfig(
|
| 40 |
+
level=logging.INFO,
|
| 41 |
+
format="[%(asctime)s] [%(levelname)s] %(message)s",
|
| 42 |
+
datefmt="%H:%M:%S",
|
| 43 |
+
handlers=[
|
| 44 |
+
logging.StreamHandler(sys.stdout),
|
| 45 |
+
logging.FileHandler(str(PROJECT_ROOT / "logs" / "fase2_v4.log")),
|
| 46 |
+
],
|
| 47 |
+
)
|
| 48 |
+
logger = logging.getLogger(__name__)
|
| 49 |
+
|
| 50 |
+
from bigru_t.utils.xeon_runtime import optimize_xeon_environment
|
| 51 |
+
optimize_xeon_environment(verbose=False)
|
| 52 |
+
|
| 53 |
+
import torch
|
| 54 |
+
from bigru_t.model.kohonen_learning_system import KohonenLearningSystemV2
|
| 55 |
+
from bigru_t.utils.oom_guard import OomGuard
|
| 56 |
+
from bigru_t.data.streaming_datasets import stream_dataset
|
| 57 |
+
from bigru_t.model.som_metrics import compute_all_metrics
|
| 58 |
+
|
| 59 |
+
OOM_GUARD = OomGuard(max_rss_mb=2200, warn_rss_mb=1800, check_interval=2.0)
|
| 60 |
+
OOM_GUARD.start()
|
| 61 |
+
|
| 62 |
+
# Canônicos V6.5-V4
|
| 63 |
+
SOM_GRID = (4, 4, 4, 4)
|
| 64 |
+
HIDDEN_DIM = 1024
|
| 65 |
+
VOCAB_SIZE = 16384
|
| 66 |
+
N_HYPOTHESES = 16
|
| 67 |
+
MAX_N_HYPOTHESES = 32
|
| 68 |
+
HYP_TRAIN_STEPS = 30
|
| 69 |
+
HYP_HIDDEN_DIM = 256
|
| 70 |
+
PUNICAO_DATASET = "BrunoN-Dev/corpus-ptbr-v1"
|
| 71 |
+
META_MINIMA_PUNICAO = 2000
|
| 72 |
+
STREAM_BATCH_SIZE = 100
|
| 73 |
+
BATCH_SIZE = 16
|
| 74 |
+
|
| 75 |
+
HF_TOKEN = os.environ.get("HF_TOKEN")
|
| 76 |
+
|
| 77 |
+
|
| 78 |
+
def mem_mb() -> float:
|
| 79 |
+
try:
|
| 80 |
+
with open("/proc/self/status") as f:
|
| 81 |
+
for line in f:
|
| 82 |
+
if line.startswith("VmRSS:"):
|
| 83 |
+
return int(line.split()[1]) / 1024
|
| 84 |
+
except Exception:
|
| 85 |
+
pass
|
| 86 |
+
return 0.0
|
| 87 |
+
|
| 88 |
+
|
| 89 |
+
def main() -> int:
|
| 90 |
+
logger.info("=" * 80)
|
| 91 |
+
logger.info("[FASE2] V6.5-V4-canonical-256 — TREINAMENTO COM PUNIÇÃO")
|
| 92 |
+
logger.info("=" * 80)
|
| 93 |
+
logger.info(f" SOM grid: {SOM_GRID} (256 neurons) | HIDDEN={HIDDEN_DIM} | VOCAB={VOCAB_SIZE}")
|
| 94 |
+
logger.info(f" n_hyp: {N_HYPOTHESES}/{MAX_N_HYPOTHESES} | hyp_steps={HYP_TRAIN_STEPS}")
|
| 95 |
+
logger.info(f" Dataset: {PUNICAO_DATASET} | Meta: ≥{META_MINIMA_PUNICAO}")
|
| 96 |
+
logger.info(f" MEM start: {mem_mb():.0f}MB")
|
| 97 |
+
logger.info("=" * 80)
|
| 98 |
+
|
| 99 |
+
# 1. KLS
|
| 100 |
+
logger.info("[FASE2] Inicializando KLS V2...")
|
| 101 |
+
kls = KohonenLearningSystemV2(
|
| 102 |
+
vocab_size=VOCAB_SIZE, hidden_dim=HIDDEN_DIM, seq_len=8,
|
| 103 |
+
som_grid=SOM_GRID, alpha0=0.5, sigma0=2.0,
|
| 104 |
+
lambda_ewc=0.02, N_start=10, dim_choice="y",
|
| 105 |
+
hypothesis_hidden=[512, 256, 128, 64, 32, 16, 8],
|
| 106 |
+
T_max=10000,
|
| 107 |
+
enable_vqvae2=True, enable_reasoning=False, enable_w8a8=False,
|
| 108 |
+
vqvae2_code_dim=16, vqvae2_num_codes_top=64, vqvae2_num_codes_bot=128,
|
| 109 |
+
enable_attention=True, attention_n_heads=8,
|
| 110 |
+
n_hypotheses=N_HYPOTHESES, max_n_hypotheses=MAX_N_HYPOTHESES,
|
| 111 |
+
min_n_hypotheses=4, n_trials=3, min_n_trials=1, max_n_trials=6,
|
| 112 |
+
hyp_train_steps=HYP_TRAIN_STEPS, min_hyp_train_steps=10, max_hyp_train_steps=80,
|
| 113 |
+
hyp_lr=1e-4, hyp_hidden_dim=HYP_HIDDEN_DIM,
|
| 114 |
+
loss_history_window=8, punishment_window=12,
|
| 115 |
+
)
|
| 116 |
+
logger.info(f"[FASE2] KLS V2 init: {mem_mb():.0f}MB")
|
| 117 |
+
kls.buffer_max_size = 128 # OOM-safety
|
| 118 |
+
|
| 119 |
+
# 2. Tokenizer
|
| 120 |
+
corpus_inicial = [
|
| 121 |
+
"o gato dorme na cama", "a casa eh azul", "ele corre rapido",
|
| 122 |
+
"ela canta uma musica", "o sol nasceu hoje", "nos vamos viajar",
|
| 123 |
+
"o livro esta na mesa", "a menina brinca no parque",
|
| 124 |
+
"ola como voce esta", "qual e o seu nome",
|
| 125 |
+
"calcule dois mais dois", "traduza hello para portugues",
|
| 126 |
+
"instrucao para resolver o problema", "resposta para a pergunta",
|
| 127 |
+
"luva de pedreiro tavila", "lula reserva valor",
|
| 128 |
+
"amazonas forca tarefa vitimas",
|
| 129 |
+
]
|
| 130 |
+
kls.tokenizer.fit(corpus_inicial)
|
| 131 |
+
logger.info(f"[FASE2] Tokenizer fitted: {mem_mb():.0f}MB")
|
| 132 |
+
|
| 133 |
+
# 3. Carrega estado parcial (procura o mais recente)
|
| 134 |
+
partials = sorted(BIGRU_ROOT.glob("v6_5_v2_conhecimento_partial_d*.pt"))
|
| 135 |
+
if not partials:
|
| 136 |
+
logger.error("[FASE2] Nenhum estado parcial encontrado. Abortando.")
|
| 137 |
+
return 1
|
| 138 |
+
partial_state_path = partials[-1]
|
| 139 |
+
logger.info(f"[FASE2] Carregando estado: {partial_state_path.name}")
|
| 140 |
+
|
| 141 |
+
try:
|
| 142 |
+
state = torch.load(str(partial_state_path), map_location="cpu", weights_only=False)
|
| 143 |
+
meta = state.get("_meta", {})
|
| 144 |
+
logger.info(f"[FASE2] Estado: phase={meta.get('phase')}, "
|
| 145 |
+
f"samples={meta.get('total_samples')}, step={meta.get('step')}")
|
| 146 |
+
|
| 147 |
+
if "som_weights" in state:
|
| 148 |
+
kls.som.weights.data.copy_(state["som_weights"])
|
| 149 |
+
if "embedding_state" in state:
|
| 150 |
+
kls.embedding.load_state_dict(state["embedding_state"])
|
| 151 |
+
if "hypothesis_ensemble_state" in state:
|
| 152 |
+
try:
|
| 153 |
+
kls.hypothesis_ensemble.load_state_dict(state["hypothesis_ensemble_state"])
|
| 154 |
+
except Exception as e:
|
| 155 |
+
logger.warning(f"[FASE2] hyp_ensemble load failed: {e}")
|
| 156 |
+
if "delta_scale" in state:
|
| 157 |
+
try:
|
| 158 |
+
kls.delta_scale.data.copy_(state["delta_scale"])
|
| 159 |
+
except Exception:
|
| 160 |
+
pass
|
| 161 |
+
if "label_registry" in state:
|
| 162 |
+
try:
|
| 163 |
+
kls.label_registry = state["label_registry"]
|
| 164 |
+
except Exception:
|
| 165 |
+
pass
|
| 166 |
+
|
| 167 |
+
kls.time_counter = meta.get("time_counter", 7000)
|
| 168 |
+
kls.training_ready = True
|
| 169 |
+
fase1_samples = meta.get("total_samples", 7000)
|
| 170 |
+
logger.info(f"[FASE2] Estado carregado: {mem_mb():.0f}MB, fase1_samples={fase1_samples}")
|
| 171 |
+
except Exception as e:
|
| 172 |
+
logger.error(f"[FASE2] Falha ao carregar estado: {e}")
|
| 173 |
+
traceback.print_exc()
|
| 174 |
+
return 1
|
| 175 |
+
|
| 176 |
+
# 4. FASE2 — streaming + process_batch_v2 com punição
|
| 177 |
+
logger.info("\n[FASE2] Iniciando TREINAMENTO COM PUNIÇÃO...")
|
| 178 |
+
total_samples = 0
|
| 179 |
+
step = 0
|
| 180 |
+
punishment_events = []
|
| 181 |
+
hypotheses_trainings = []
|
| 182 |
+
delta_applications = []
|
| 183 |
+
som_metrics_log = []
|
| 184 |
+
t_start = time.time()
|
| 185 |
+
|
| 186 |
+
try:
|
| 187 |
+
sample_iter = stream_dataset(
|
| 188 |
+
dataset_name=PUNICAO_DATASET,
|
| 189 |
+
max_samples=META_MINIMA_PUNICAO,
|
| 190 |
+
hf_token=HF_TOKEN,
|
| 191 |
+
)
|
| 192 |
+
|
| 193 |
+
chunk_buffer = []
|
| 194 |
+
chunk_idx = 0
|
| 195 |
+
|
| 196 |
+
for sample in sample_iter:
|
| 197 |
+
chunk_buffer.append(sample)
|
| 198 |
+
if len(chunk_buffer) >= STREAM_BATCH_SIZE:
|
| 199 |
+
chunk_idx += 1
|
| 200 |
+
chunk_texts = [
|
| 201 |
+
s.raw_text if hasattr(s, "raw_text") else str(s)
|
| 202 |
+
for s in chunk_buffer
|
| 203 |
+
]
|
| 204 |
+
total_samples += len(chunk_texts)
|
| 205 |
+
logger.info(f"[FASE2] Chunk {chunk_idx}: {len(chunk_texts)} samples "
|
| 206 |
+
f"(total={total_samples}/{META_MINIMA_PUNICAO}), MEM={mem_mb():.0f}MB")
|
| 207 |
+
|
| 208 |
+
# Labels binários determinísticos baseados em hash
|
| 209 |
+
labels = [hash(s) % 2 for s in chunk_texts]
|
| 210 |
+
|
| 211 |
+
# Processa em sub-batches
|
| 212 |
+
for bs in range(0, len(chunk_texts), BATCH_SIZE):
|
| 213 |
+
batch_sents = chunk_texts[bs: bs + BATCH_SIZE]
|
| 214 |
+
batch_labels = labels[bs: bs + BATCH_SIZE]
|
| 215 |
+
try:
|
| 216 |
+
result = kls.process_batch_v2(
|
| 217 |
+
batch_sents, batch_labels,
|
| 218 |
+
dataset_name=PUNICAO_DATASET,
|
| 219 |
+
enable_punishment=True,
|
| 220 |
+
)
|
| 221 |
+
step += 1
|
| 222 |
+
action = result.get("action", "none")
|
| 223 |
+
if action != "none":
|
| 224 |
+
logger.info(f"[FASE2] Step {step}: action={action}, "
|
| 225 |
+
f"acc={result.get('accuracy', 0):.3f}")
|
| 226 |
+
if "train_hypotheses" in action:
|
| 227 |
+
hyp_info = result.get("hypotheses_training", {})
|
| 228 |
+
hypotheses_trainings.append({
|
| 229 |
+
"step": step,
|
| 230 |
+
"loss_final": hyp_info.get("loss_final"),
|
| 231 |
+
})
|
| 232 |
+
elif "apply_best_delta" in action:
|
| 233 |
+
delta_info = result.get("delta_application", {})
|
| 234 |
+
delta_applications.append({
|
| 235 |
+
"step": step,
|
| 236 |
+
"best_acc": delta_info.get("best_acc"),
|
| 237 |
+
})
|
| 238 |
+
punishment_events.append({
|
| 239 |
+
"step": step,
|
| 240 |
+
"action": action,
|
| 241 |
+
"accuracy": result.get("accuracy", 0),
|
| 242 |
+
})
|
| 243 |
+
except (MemoryError, RuntimeError) as oom_err:
|
| 244 |
+
is_oom = (
|
| 245 |
+
isinstance(oom_err, MemoryError)
|
| 246 |
+
or "out of memory" in str(oom_err).lower()
|
| 247 |
+
)
|
| 248 |
+
if is_oom:
|
| 249 |
+
logger.error(f"[FASE2] OOM step {step}: {str(oom_err)[:200]}")
|
| 250 |
+
gc.collect(); gc.collect()
|
| 251 |
+
time.sleep(2)
|
| 252 |
+
continue
|
| 253 |
+
raise
|
| 254 |
+
|
| 255 |
+
if step % 4 == 0:
|
| 256 |
+
gc.collect()
|
| 257 |
+
time.sleep(0.3)
|
| 258 |
+
|
| 259 |
+
# Métricas SOM após chunk
|
| 260 |
+
try:
|
| 261 |
+
buf = kls.buffer_4d[-64:] if kls.buffer_4d else []
|
| 262 |
+
som_metrics = compute_all_metrics(kls.som, buf)
|
| 263 |
+
som_metrics_log.append({
|
| 264 |
+
"chunk": chunk_idx,
|
| 265 |
+
"total_samples": total_samples,
|
| 266 |
+
"qe": float(som_metrics.get("quantization_error", 0)),
|
| 267 |
+
"te": float(som_metrics.get("topological_error", 0)),
|
| 268 |
+
"kl": float(som_metrics.get("kaski_lagus_error", 0)),
|
| 269 |
+
"ve": float(som_metrics.get("explained_variance_share", 0)),
|
| 270 |
+
"dead_rate": float(som_metrics.get("dead_neuron_rate", {}).get("dead_neuron_rate", 0)),
|
| 271 |
+
})
|
| 272 |
+
logger.info(
|
| 273 |
+
f"[FASE2] SOM: QE={som_metrics_log[-1]['qe']:.4f}, "
|
| 274 |
+
f"TE={som_metrics_log[-1]['te']:.4f}, "
|
| 275 |
+
f"KL={som_metrics_log[-1]['kl']:.4f}, "
|
| 276 |
+
f"VE={som_metrics_log[-1]['ve']:.4f}, "
|
| 277 |
+
f"dead={som_metrics_log[-1]['dead_rate']:.3f}"
|
| 278 |
+
)
|
| 279 |
+
except Exception as e:
|
| 280 |
+
logger.warning(f"[FASE2] Métricas SOM falharam: {e}")
|
| 281 |
+
|
| 282 |
+
# Salva estado parcial
|
| 283 |
+
try:
|
| 284 |
+
partial_path = BIGRU_ROOT / f"v6_5_v2_punicão_partial_c{chunk_idx}.pt"
|
| 285 |
+
for old in BIGRU_ROOT.glob("v6_5_v2_punicão_partial_c*.pt"):
|
| 286 |
+
if old != partial_path:
|
| 287 |
+
old.unlink(missing_ok=True)
|
| 288 |
+
torch.save({
|
| 289 |
+
"_meta": {
|
| 290 |
+
"reason": f"punicao_after_chunk_{chunk_idx}",
|
| 291 |
+
"step": step,
|
| 292 |
+
"total_samples": total_samples,
|
| 293 |
+
"timestamp": datetime.now().isoformat(),
|
| 294 |
+
"version": "V6.5-V4-canonical-256",
|
| 295 |
+
"phase": "punicao_partial",
|
| 296 |
+
"som_grid": list(SOM_GRID),
|
| 297 |
+
"n_neurons": 256,
|
| 298 |
+
"fase1_samples": fase1_samples,
|
| 299 |
+
},
|
| 300 |
+
"som_weights": kls.som.weights.data,
|
| 301 |
+
"embedding_state": kls.embedding.state_dict(),
|
| 302 |
+
"hypothesis_ensemble_state": kls.hypothesis_ensemble.state_dict(),
|
| 303 |
+
"delta_scale": kls.delta_scale.data,
|
| 304 |
+
"label_registry": kls.label_registry,
|
| 305 |
+
}, str(partial_path))
|
| 306 |
+
logger.info(f"[FASE2] Estado parcial salvo: {partial_path.name}")
|
| 307 |
+
except Exception as e:
|
| 308 |
+
logger.warning(f"[FASE2] Save parcial falhou: {e}")
|
| 309 |
+
|
| 310 |
+
gc.collect(); gc.collect()
|
| 311 |
+
time.sleep(1.0)
|
| 312 |
+
|
| 313 |
+
if total_samples >= META_MINIMA_PUNICAO:
|
| 314 |
+
logger.info(f"[FASE2] Meta atingida: {total_samples} ≥ {META_MINIMA_PUNICAO}")
|
| 315 |
+
break
|
| 316 |
+
|
| 317 |
+
chunk_buffer = []
|
| 318 |
+
|
| 319 |
+
# Processa chunk final se houver
|
| 320 |
+
if chunk_buffer and total_samples < META_MINIMA_PUNICAO:
|
| 321 |
+
chunk_idx += 1
|
| 322 |
+
chunk_texts = [
|
| 323 |
+
s.raw_text if hasattr(s, "raw_text") else str(s)
|
| 324 |
+
for s in chunk_buffer
|
| 325 |
+
]
|
| 326 |
+
total_samples += len(chunk_texts)
|
| 327 |
+
logger.info(f"[FASE2] Chunk final {chunk_idx}: {len(chunk_texts)} samples "
|
| 328 |
+
f"(total={total_samples})")
|
| 329 |
+
labels = [hash(s) % 2 for s in chunk_texts]
|
| 330 |
+
for bs in range(0, len(chunk_texts), BATCH_SIZE):
|
| 331 |
+
batch_sents = chunk_texts[bs: bs + BATCH_SIZE]
|
| 332 |
+
batch_labels = labels[bs: bs + BATCH_SIZE]
|
| 333 |
+
try:
|
| 334 |
+
result = kls.process_batch_v2(
|
| 335 |
+
batch_sents, batch_labels,
|
| 336 |
+
dataset_name=PUNICAO_DATASET,
|
| 337 |
+
enable_punishment=True,
|
| 338 |
+
)
|
| 339 |
+
step += 1
|
| 340 |
+
if result.get("action", "none") != "none":
|
| 341 |
+
punishment_events.append({
|
| 342 |
+
"step": step,
|
| 343 |
+
"action": result.get("action"),
|
| 344 |
+
"accuracy": result.get("accuracy", 0),
|
| 345 |
+
})
|
| 346 |
+
except Exception as e:
|
| 347 |
+
logger.warning(f"[FASE2] Erro no chunk final: {e}")
|
| 348 |
+
|
| 349 |
+
except Exception as e:
|
| 350 |
+
logger.error(f"[FASE2] Erro durante FASE2: {e}")
|
| 351 |
+
traceback.print_exc()
|
| 352 |
+
|
| 353 |
+
elapsed = time.time() - t_start
|
| 354 |
+
|
| 355 |
+
# 5. Estado final unificado
|
| 356 |
+
logger.info("\n[FASE2] Salvando estado final unificado...")
|
| 357 |
+
final_state_path = BIGRU_ROOT / "v6_5_v2_model_states.pt"
|
| 358 |
+
try:
|
| 359 |
+
torch.save({
|
| 360 |
+
"_meta": {
|
| 361 |
+
"reason": "end_of_training_v65_v4",
|
| 362 |
+
"step": step,
|
| 363 |
+
"total_samples": total_samples,
|
| 364 |
+
"timestamp": datetime.now().isoformat(),
|
| 365 |
+
"version": "V6.5-V4-canonical-256",
|
| 366 |
+
"phase": "end_of_training",
|
| 367 |
+
"som_grid": list(SOM_GRID),
|
| 368 |
+
"n_neurons": 256,
|
| 369 |
+
"hidden_dim": HIDDEN_DIM,
|
| 370 |
+
"vocab_size": VOCAB_SIZE,
|
| 371 |
+
"n_hypotheses": N_HYPOTHESES,
|
| 372 |
+
"max_n_hypotheses": MAX_N_HYPOTHESES,
|
| 373 |
+
"hyp_train_steps": HYP_TRAIN_STEPS,
|
| 374 |
+
"hyp_hidden_dim": HYP_HIDDEN_DIM,
|
| 375 |
+
"buffer_max_size": kls.buffer_max_size,
|
| 376 |
+
"fase1_samples": fase1_samples,
|
| 377 |
+
"fase2_samples": total_samples,
|
| 378 |
+
},
|
| 379 |
+
"som_weights": kls.som.weights.data,
|
| 380 |
+
"embedding_state": kls.embedding.state_dict(),
|
| 381 |
+
"hypothesis_ensemble_state": kls.hypothesis_ensemble.state_dict(),
|
| 382 |
+
"delta_scale": kls.delta_scale.data,
|
| 383 |
+
"label_registry": kls.label_registry,
|
| 384 |
+
}, str(final_state_path))
|
| 385 |
+
logger.info(f"[FASE2] Estado final salvo: {final_state_path}")
|
| 386 |
+
except Exception as e:
|
| 387 |
+
logger.error(f"[FASE2] Falha ao salvar estado final: {e}")
|
| 388 |
+
|
| 389 |
+
# 6. Relatório
|
| 390 |
+
report = {
|
| 391 |
+
"version": "V6.5-V4-canonical-256",
|
| 392 |
+
"timestamp": datetime.now().isoformat(),
|
| 393 |
+
"config": {
|
| 394 |
+
"som_grid": list(SOM_GRID), "n_neurons": 256,
|
| 395 |
+
"hidden_dim": HIDDEN_DIM, "vocab_size": VOCAB_SIZE,
|
| 396 |
+
"n_hypotheses": N_HYPOTHESES, "max_n_hypotheses": MAX_N_HYPOTHESES,
|
| 397 |
+
"hyp_train_steps": HYP_TRAIN_STEPS, "hyp_hidden_dim": HYP_HIDDEN_DIM,
|
| 398 |
+
"buffer_max_size": kls.buffer_max_size,
|
| 399 |
+
},
|
| 400 |
+
"fase1_summary": {
|
| 401 |
+
"total_samples": fase1_samples,
|
| 402 |
+
"meta_atingida": fase1_samples >= 8000,
|
| 403 |
+
"state_file": partial_state_path.name,
|
| 404 |
+
},
|
| 405 |
+
"fase2_summary": {
|
| 406 |
+
"total_samples": total_samples,
|
| 407 |
+
"meta_minima": META_MINIMA_PUNICAO,
|
| 408 |
+
"meta_atingida": total_samples >= META_MINIMA_PUNICAO,
|
| 409 |
+
"elapsed_s": elapsed,
|
| 410 |
+
"punishment_events": len(punishment_events),
|
| 411 |
+
"hypotheses_trainings": len(hypotheses_trainings),
|
| 412 |
+
"delta_applications": len(delta_applications),
|
| 413 |
+
},
|
| 414 |
+
"som_metrics_log": som_metrics_log,
|
| 415 |
+
"punishment_events_last": punishment_events[-10:],
|
| 416 |
+
"hypotheses_trainings_last": hypotheses_trainings[-5:],
|
| 417 |
+
"delta_applications_last": delta_applications[-5:],
|
| 418 |
+
"oom_guard_stats": OOM_GUARD.get_stats(),
|
| 419 |
+
}
|
| 420 |
+
report_path = BIGRU_ROOT / "v6_5_v4_fase2_report.json"
|
| 421 |
+
with open(report_path, "w") as f:
|
| 422 |
+
json.dump(report, f, indent=2, ensure_ascii=False, default=str)
|
| 423 |
+
logger.info(f"[FASE2] Relatório salvo: {report_path}")
|
| 424 |
+
|
| 425 |
+
OOM_GUARD.stop()
|
| 426 |
+
del kls
|
| 427 |
+
gc.collect()
|
| 428 |
+
|
| 429 |
+
logger.info("\n" + "=" * 80)
|
| 430 |
+
logger.info("[FASE2] RESUMO FINAL")
|
| 431 |
+
logger.info("=" * 80)
|
| 432 |
+
logger.info(f" FASE1 samples : {fase1_samples} (meta=8000)")
|
| 433 |
+
logger.info(f" FASE2 samples : {total_samples} (meta={META_MINIMA_PUNICAO})")
|
| 434 |
+
logger.info(f" Punishments : {len(punishment_events)}")
|
| 435 |
+
logger.info(f" Hyp trainings : {len(hypotheses_trainings)}")
|
| 436 |
+
logger.info(f" Delta applies : {len(delta_applications)}")
|
| 437 |
+
logger.info(f" Elapsed : {elapsed:.1f}s")
|
| 438 |
+
logger.info(f" Peak RSS : {OOM_GUARD.get_stats()['peak_rss_mb']:.0f}MB")
|
| 439 |
+
logger.info(f" State file : {final_state_path}")
|
| 440 |
+
logger.info("=" * 80)
|
| 441 |
+
|
| 442 |
+
return 0 if total_samples > 0 else 1
|
| 443 |
+
|
| 444 |
+
|
| 445 |
+
if __name__ == "__main__":
|
| 446 |
+
sys.exit(main())
|
scripts/tests/test_fase2_integration.py
CHANGED
|
@@ -370,21 +370,22 @@ except Exception as e:
|
|
| 370 |
|
| 371 |
|
| 372 |
# ---------------------------------------------------------------------------
|
| 373 |
-
# Test 10: Buffer
|
| 374 |
# ---------------------------------------------------------------------------
|
| 375 |
-
print("\n--- Test 10: Buffer
|
| 376 |
try:
|
| 377 |
-
#
|
|
|
|
| 378 |
buffer_max = getattr(kls, "buffer_max_size", None)
|
| 379 |
buffer_canonical = getattr(kls, "_buffer_max_size_canonical", None)
|
| 380 |
buffer_fallback = getattr(kls, "_buffer_max_size_fallback", None)
|
| 381 |
record(
|
| 382 |
-
"Buffer
|
| 383 |
-
buffer_max ==
|
| 384 |
f"max={buffer_max}, canonical={buffer_canonical}, fallback={buffer_fallback}",
|
| 385 |
)
|
| 386 |
except Exception as e:
|
| 387 |
-
record("Buffer
|
| 388 |
|
| 389 |
|
| 390 |
# ---------------------------------------------------------------------------
|
|
|
|
| 370 |
|
| 371 |
|
| 372 |
# ---------------------------------------------------------------------------
|
| 373 |
+
# Test 10: Buffer 256 canônico V6.5-V4 (alinhado ao grid (4,4,4,4)=256)
|
| 374 |
# ---------------------------------------------------------------------------
|
| 375 |
+
print("\n--- Test 10: Buffer 256 (canonical V6.5-V4) ---")
|
| 376 |
try:
|
| 377 |
+
# V6.5-V4-canonical-256: buffer=256 alinhado ao grid (4,4,4,4)=256
|
| 378 |
+
# User requirement: "fazer (tornar canônico) buffer 256 e grid para (4,4,4,4)=256"
|
| 379 |
buffer_max = getattr(kls, "buffer_max_size", None)
|
| 380 |
buffer_canonical = getattr(kls, "_buffer_max_size_canonical", None)
|
| 381 |
buffer_fallback = getattr(kls, "_buffer_max_size_fallback", None)
|
| 382 |
record(
|
| 383 |
+
"Buffer 256 canônico V6.5-V4 (alinhado ao grid)",
|
| 384 |
+
buffer_max == 256 and buffer_canonical == 256,
|
| 385 |
f"max={buffer_max}, canonical={buffer_canonical}, fallback={buffer_fallback}",
|
| 386 |
)
|
| 387 |
except Exception as e:
|
| 388 |
+
record("Buffer 256 canônico V6.5-V4", False, str(e))
|
| 389 |
|
| 390 |
|
| 391 |
# ---------------------------------------------------------------------------
|
scripts/train_v6_5_v2.py
CHANGED
|
@@ -238,26 +238,24 @@ LOSS_HISTORY_WINDOW = 8
|
|
| 238 |
PUNISHMENT_WINDOW = 12
|
| 239 |
|
| 240 |
# V2-dynamic-memory — Buffer sliding window (evita OOM em treino longo)
|
| 241 |
-
# V6.5-
|
| 242 |
-
#
|
| 243 |
-
#
|
| 244 |
-
# (
|
|
|
|
|
|
|
| 245 |
#
|
| 246 |
# Configuração ADAPTATIVA com monitoramento de memória:
|
| 247 |
-
# 1. CANÔNICO: buffer=
|
| 248 |
-
#
|
| 249 |
-
# 2. FALLBACK: se RSS > 75% cgroup, reduz buffer para 256. Grid (4,4,4,4)=256
|
| 250 |
# PERMANECE (não é reduzido — é o canônico). Apenas o buffer encolhe.
|
| 251 |
# 3. O monitoramento é feito em get_cgroup_memory_limit_mb() no runtime.
|
| 252 |
-
#
|
| 253 |
-
# sem autorização. RESTAURADO para 864 conforme user requirement. OOM-safety
|
| 254 |
-
# agora garantida por: (a) VQ-VAE-2 lazy compression (a cada 16 add_data),
|
| 255 |
# (b) torch.no_grad() em todo compressão, (c) gc.collect() a cada 4 batches,
|
| 256 |
-
# (d) OomGuard thread daemon, (e) fallback buffer=
|
| 257 |
-
|
| 258 |
-
|
| 259 |
-
|
| 260 |
-
FALLBACK_BUFFER_SIZE = 256 # user requirement: "fazer buffer 256" (fallback)
|
| 261 |
FALLBACK_SOM_GRID = (4, 4, 4, 4) # = 256 neurônios (CANÔNICO — igual ao grid principal)
|
| 262 |
FALLBACK_SIGMA0 = 2.0 # max(4,4,4,4)/2 = 2.0
|
| 263 |
MEM_CRITICAL_PCT_FOR_FALLBACK = 75 # se RSS > 75% cgroup, ativa fallback
|
|
@@ -451,25 +449,25 @@ CGROUP_MEM_CRITICAL_PCT = 80 # se RSS > 80% do cgroup, salvar estado e parar
|
|
| 451 |
def check_memory_and_maybe_fallback(kls: KohonenLearningSystemV2) -> Dict[str, Any]:
|
| 452 |
"""V6.5-V2-buffer-864 — Monitora memória e aplica fallback se crítico.
|
| 453 |
|
| 454 |
-
User requirement: "
|
| 455 |
-
|
| 456 |
-
|
|
|
|
| 457 |
|
| 458 |
Lógica:
|
| 459 |
1. Lê RSS atual e cgroup memory limit.
|
| 460 |
2. Se RSS > MEM_CRITICAL_PCT_FOR_FALLBACK do cgroup:
|
| 461 |
-
a. Reduz buffer_max_size do KLS para FALLBACK_BUFFER_SIZE (
|
| 462 |
b. Trunca buffer_4d atual para FALLBACK_BUFFER_SIZE.
|
| 463 |
-
c. Registra evento de fallback (
|
| 464 |
-
o aprendizado já acumulado nos pesos
|
| 465 |
apenas reduz o buffer de amostras recentes, não a arquitetura).
|
| 466 |
3. Retorna relatório com RSS, pct, fallback_ativado.
|
| 467 |
|
| 468 |
-
Nota: o
|
| 469 |
-
|
| 470 |
-
|
| 471 |
-
|
| 472 |
-
arquitetura Kohonen enquanto mitiga OOM. Se o OOM persistir, o
|
| 473 |
save_model_states_for_evaluation será chamado pelo caller.
|
| 474 |
"""
|
| 475 |
rss_mb = get_process_rss_mb()
|
|
|
|
| 238 |
PUNISHMENT_WINDOW = 12
|
| 239 |
|
| 240 |
# V2-dynamic-memory — Buffer sliding window (evita OOM em treino longo)
|
| 241 |
+
# V6.5-V4-canonical-256 (user requirement EXATO): "fazer (tornar canônico)
|
| 242 |
+
# buffer 256 e grid para (4,4,4,4)=256". Buffer e grid agora têm o MESMO
|
| 243 |
+
# tamanho (256), eliminando o desbalanceamento que causava OOM em V6.5-V3
|
| 244 |
+
# (buffer=864 com grid=256). A correspondência 1:1 entre amostras no buffer
|
| 245 |
+
# e neurônios no grid 4D é matematicamente elegante — cada amostra pode,
|
| 246 |
+
# em média, ativar um neurônio distinto, maximizando a utilização do mapa.
|
| 247 |
#
|
| 248 |
# Configuração ADAPTATIVA com monitoramento de memória:
|
| 249 |
+
# 1. CANÔNICO: buffer=256 + grid (4,4,4,4)=256 — correspondência 1:1.
|
| 250 |
+
# 2. FALLBACK: se RSS > 75% cgroup, reduz buffer para 128. Grid (4,4,4,4)=256
|
|
|
|
| 251 |
# PERMANECE (não é reduzido — é o canônico). Apenas o buffer encolhe.
|
| 252 |
# 3. O monitoramento é feito em get_cgroup_memory_limit_mb() no runtime.
|
| 253 |
+
# OOM-safety garantida por: (a) VQ-VAE-2 lazy compression (a cada 16 add_data),
|
|
|
|
|
|
|
| 254 |
# (b) torch.no_grad() em todo compressão, (c) gc.collect() a cada 4 batches,
|
| 255 |
+
# (d) OomGuard thread daemon, (e) fallback buffer=128 se RSS > 75%,
|
| 256 |
+
# (f) exception handler MemoryError + RuntimeError(out of memory).
|
| 257 |
+
MAX_BUFFER_SIZE = 256 # CANÔNICO V6.5-V4 (user: "tornar canônico buffer 256")
|
| 258 |
+
FALLBACK_BUFFER_SIZE = 128 # fallback OOM (reduzido de 256 para evitar pressão)
|
|
|
|
| 259 |
FALLBACK_SOM_GRID = (4, 4, 4, 4) # = 256 neurônios (CANÔNICO — igual ao grid principal)
|
| 260 |
FALLBACK_SIGMA0 = 2.0 # max(4,4,4,4)/2 = 2.0
|
| 261 |
MEM_CRITICAL_PCT_FOR_FALLBACK = 75 # se RSS > 75% cgroup, ativa fallback
|
|
|
|
| 449 |
def check_memory_and_maybe_fallback(kls: KohonenLearningSystemV2) -> Dict[str, Any]:
|
| 450 |
"""V6.5-V2-buffer-864 — Monitora memória e aplica fallback se crítico.
|
| 451 |
|
| 452 |
+
User requirement (V6.5-V4-canonical-256): "fazer (tornar canônico) buffer
|
| 453 |
+
256 e grid para (4,4,4,4)=256". Buffer e grid agora têm o MESMO tamanho
|
| 454 |
+
(256). Fallback OOM reduz buffer para 128 (não 256, pois 256 já é o
|
| 455 |
+
canônico). Grid (4,4,4,4)=256 PERMANECE sempre.
|
| 456 |
|
| 457 |
Lógica:
|
| 458 |
1. Lê RSS atual e cgroup memory limit.
|
| 459 |
2. Se RSS > MEM_CRITICAL_PCT_FOR_FALLBACK do cgroup:
|
| 460 |
+
a. Reduz buffer_max_size do KLS para FALLBACK_BUFFER_SIZE (128).
|
| 461 |
b. Trunca buffer_4d atual para FALLBACK_BUFFER_SIZE.
|
| 462 |
+
c. Registra evento de fallback (NÃO recriia o SOM grid — preserva
|
| 463 |
+
o aprendizado já acumulado nos pesos 256-neuronios; o fallback
|
| 464 |
apenas reduz o buffer de amostras recentes, não a arquitetura).
|
| 465 |
3. Retorna relatório com RSS, pct, fallback_ativado.
|
| 466 |
|
| 467 |
+
Nota: Recriar o SOM grid do zero perderia todo o aprendizado da FASE1.
|
| 468 |
+
Mantemos o grid (4,4,4,4)=256 (preservando pesos aprendidos) e reduzimos
|
| 469 |
+
APENAS o buffer de amostras recentes para 128. Isto preserva a arquitetura
|
| 470 |
+
Kohonen enquanto mitiga OOM. Se o OOM persistir, o
|
|
|
|
| 471 |
save_model_states_for_evaluation será chamado pelo caller.
|
| 472 |
"""
|
| 473 |
rss_mb = get_process_rss_mb()
|
src/bigru_t/model/kohonen_learning_system.py
CHANGED
|
@@ -1683,7 +1683,7 @@ class KohonenLearningSystem:
|
|
| 1683 |
vocab_size: tamanho do vocabulário BBPE (default 16384).
|
| 1684 |
hidden_dim: dimensão do embedding (default 1024).
|
| 1685 |
seq_len: comprimento máximo da sequência (default 8).
|
| 1686 |
-
som_grid: (I, J, K, L) — grid 4D do SOM (default (
|
| 1687 |
alpha0, sigma0: hiperparâmetros do SOM.
|
| 1688 |
lambda_ewc: peso da penalidade EWC.
|
| 1689 |
N_start: threshold do histograma para iniciar treino.
|
|
@@ -1759,21 +1759,23 @@ class KohonenLearningSystem:
|
|
| 1759 |
# OOM (3.5GB RSS observed). Sliding window keeps recent samples
|
| 1760 |
# for SOM updates while bounding memory. The Kohonen architecture (SOM
|
| 1761 |
# grid, BMU, Gaussian neighborhood, EWC) is NOT changed.
|
| 1762 |
-
# V6.5-
|
| 1763 |
-
# User requirement EXATO: "
|
| 1764 |
-
#
|
| 1765 |
-
#
|
| 1766 |
-
#
|
| 1767 |
-
#
|
|
|
|
|
|
|
| 1768 |
# (a) VQ-VAE-2 lazy compression (a cada 16 add_data) — ver _vqvae2_call_count
|
| 1769 |
# (b) torch.no_grad() em toda compressão
|
| 1770 |
# (c) gc.collect() a cada 4 batches (no train script)
|
| 1771 |
# (d) OomGuard thread daemon (max_rss_mb=2500)
|
| 1772 |
-
# (e) check_memory_and_maybe_fallback() reduz para
|
| 1773 |
# Grid SOM (4,4,4,4)=256 é CANÔNICO e PERMANECE em ambos os modos.
|
| 1774 |
-
self.buffer_max_size =
|
| 1775 |
-
self._buffer_max_size_canonical =
|
| 1776 |
-
self._buffer_max_size_fallback =
|
| 1777 |
# V6.5-V3-no-regression: VQ-VAE-2 lazy compression counter.
|
| 1778 |
# Comprimir a cada add_data causava OOM (200MB+ tensores intermediários
|
| 1779 |
# por chamada). Agora comprime a cada 16 add_data — mesma cobertura
|
|
@@ -3280,8 +3282,8 @@ class KohonenLearningSystemV2(KohonenLearningSystem):
|
|
| 3280 |
|
| 3281 |
def _init_hypothesis_ensemble(self):
|
| 3282 |
"""Cria o ensemble de geradores + otimizador Adam."""
|
| 3283 |
-
input_dim = self.som_neuron_count # ativação SOM flatten (
|
| 3284 |
-
output_dim = self.som.weights.numel() # I*J*K*L*4 (
|
| 3285 |
self.hypothesis_ensemble = HypothesisEnsemble(
|
| 3286 |
input_dim=input_dim,
|
| 3287 |
output_dim=output_dim,
|
|
|
|
| 1683 |
vocab_size: tamanho do vocabulário BBPE (default 16384).
|
| 1684 |
hidden_dim: dimensão do embedding (default 1024).
|
| 1685 |
seq_len: comprimento máximo da sequência (default 8).
|
| 1686 |
+
som_grid: (I, J, K, L) — grid 4D do SOM (default CANÔNICO (4, 4, 4, 4) = 256).
|
| 1687 |
alpha0, sigma0: hiperparâmetros do SOM.
|
| 1688 |
lambda_ewc: peso da penalidade EWC.
|
| 1689 |
N_start: threshold do histograma para iniciar treino.
|
|
|
|
| 1759 |
# OOM (3.5GB RSS observed). Sliding window keeps recent samples
|
| 1760 |
# for SOM updates while bounding memory. The Kohonen architecture (SOM
|
| 1761 |
# grid, BMU, Gaussian neighborhood, EWC) is NOT changed.
|
| 1762 |
+
# V6.5-V4-canonical-256: buffer_max_size = 256 (CANÔNICO, alinhado ao grid).
|
| 1763 |
+
# User requirement EXATO (V6.5-V4): "fazer (tornar canônico) buffer 256 e
|
| 1764 |
+
# grid para (4,4,4,4)=256". Buffer e grid agora têm o MESMO tamanho (256),
|
| 1765 |
+
# eliminando o desbalanceamento que causava OOM em V6.5-V3 (buffer=864 com
|
| 1766 |
+
# grid=256). A correspondência 1:1 entre amostras no buffer e neurônios no
|
| 1767 |
+
# grid 4D é matematicamente elegante — cada amostra pode, em média, ativar
|
| 1768 |
+
# um neurônio distinto, maximizando a utilização do mapa Kohonen.
|
| 1769 |
+
# OOM-safety garantida por:
|
| 1770 |
# (a) VQ-VAE-2 lazy compression (a cada 16 add_data) — ver _vqvae2_call_count
|
| 1771 |
# (b) torch.no_grad() em toda compressão
|
| 1772 |
# (c) gc.collect() a cada 4 batches (no train script)
|
| 1773 |
# (d) OomGuard thread daemon (max_rss_mb=2500)
|
| 1774 |
+
# (e) check_memory_and_maybe_fallback() reduz para 128 se RSS > 75%
|
| 1775 |
# Grid SOM (4,4,4,4)=256 é CANÔNICO e PERMANECE em ambos os modos.
|
| 1776 |
+
self.buffer_max_size = 256
|
| 1777 |
+
self._buffer_max_size_canonical = 256
|
| 1778 |
+
self._buffer_max_size_fallback = 128
|
| 1779 |
# V6.5-V3-no-regression: VQ-VAE-2 lazy compression counter.
|
| 1780 |
# Comprimir a cada add_data causava OOM (200MB+ tensores intermediários
|
| 1781 |
# por chamada). Agora comprime a cada 16 add_data — mesma cobertura
|
|
|
|
| 3282 |
|
| 3283 |
def _init_hypothesis_ensemble(self):
|
| 3284 |
"""Cria o ensemble de geradores + otimizador Adam."""
|
| 3285 |
+
input_dim = self.som_neuron_count # ativação SOM flatten (256 para grid (4,4,4,4))
|
| 3286 |
+
output_dim = self.som.weights.numel() # I*J*K*L*4 (256*4 = 1024)
|
| 3287 |
self.hypothesis_ensemble = HypothesisEnsemble(
|
| 3288 |
input_dim=input_dim,
|
| 3289 |
output_dim=output_dim,
|
src/bigru_t/tokenizer/bbpe_tokenizer.py
CHANGED
|
@@ -634,6 +634,13 @@ class BBPETokenizer:
|
|
| 634 |
Equivalente ao SimpleBBPETokenizer.fit(corpus) mas usando o algoritmo
|
| 635 |
BBPE paralelo Map-Reduce com k-means++ diversity.
|
| 636 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 637 |
V6.5-V3-bbpe-kmeans-pp: REATIVADO multicore. User requirement: "aprimorar
|
| 638 |
BBPE (aplicar k-means++) para melhor aproveitamento multicore".
|
| 639 |
num_workers agora é adaptativo: min(os.cpu_count(), 4) para cgroups
|
|
@@ -647,6 +654,20 @@ class BBPETokenizer:
|
|
| 647 |
if not corpus:
|
| 648 |
logger.warning("BBPETokenizer.fit: corpus vazio — skip.")
|
| 649 |
return
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 650 |
# Repete o corpus para garantir min_frequency >= 2 em merges úteis
|
| 651 |
# quando o corpus é pequeno (ex: 17 frases iniciais do KLS).
|
| 652 |
repeated_corpus: List[str] = list(corpus)
|
|
@@ -677,6 +698,136 @@ class BBPETokenizer:
|
|
| 677 |
"Tokenizer ficará não treinado (KLS usará fallback unk-only).", e
|
| 678 |
)
|
| 679 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 680 |
# ------------------------------------------------------------------
|
| 681 |
# TREINAMENTO (compatibilidade — delega para paralelo)
|
| 682 |
# ------------------------------------------------------------------
|
|
|
|
| 634 |
Equivalente ao SimpleBBPETokenizer.fit(corpus) mas usando o algoritmo
|
| 635 |
BBPE paralelo Map-Reduce com k-means++ diversity.
|
| 636 |
|
| 637 |
+
V6.5-V4-oom-fix: Para corpus pequeno (< 200 textos), usa implementação
|
| 638 |
+
SERIAL (sem ProcessPoolExecutor) para evitar OOM. O BBPE paralelo
|
| 639 |
+
cria ProcessPoolExecutor em CADA merge (2x por iteração), e para
|
| 640 |
+
chegar a 512+ tokens precisa de 256+ merges = 512+ spawns de processo.
|
| 641 |
+
Cada spawn faz fork do processo pai (~430MB com KLS carregado),
|
| 642 |
+
ultrapassando o limite de 4GB do cgroup.
|
| 643 |
+
|
| 644 |
V6.5-V3-bbpe-kmeans-pp: REATIVADO multicore. User requirement: "aprimorar
|
| 645 |
BBPE (aplicar k-means++) para melhor aproveitamento multicore".
|
| 646 |
num_workers agora é adaptativo: min(os.cpu_count(), 4) para cgroups
|
|
|
|
| 654 |
if not corpus:
|
| 655 |
logger.warning("BBPETokenizer.fit: corpus vazio — skip.")
|
| 656 |
return
|
| 657 |
+
# V6.5-V4-oom-fix: Para corpus pequeno, usa BBPE SERIAL (in-process).
|
| 658 |
+
# Evita OOM por spawn repetido de ProcessPoolExecutor.
|
| 659 |
+
if len(corpus) < 200:
|
| 660 |
+
try:
|
| 661 |
+
self._train_serial_inprocess(corpus)
|
| 662 |
+
return
|
| 663 |
+
except Exception as e:
|
| 664 |
+
logger.warning(
|
| 665 |
+
"BBPETokenizer.fit: serial in-process failed: %s. "
|
| 666 |
+
"Fallback para word-level.", e
|
| 667 |
+
)
|
| 668 |
+
# Fallback final: word-level
|
| 669 |
+
self._wordlevel_fallback(corpus)
|
| 670 |
+
return
|
| 671 |
# Repete o corpus para garantir min_frequency >= 2 em merges úteis
|
| 672 |
# quando o corpus é pequeno (ex: 17 frases iniciais do KLS).
|
| 673 |
repeated_corpus: List[str] = list(corpus)
|
|
|
|
| 698 |
"Tokenizer ficará não treinado (KLS usará fallback unk-only).", e
|
| 699 |
)
|
| 700 |
|
| 701 |
+
def _train_serial_inprocess(self, corpus: List[str]) -> None:
|
| 702 |
+
"""V6.5-V4-oom-fix: BBPE serial in-process (sem ProcessPoolExecutor).
|
| 703 |
+
|
| 704 |
+
Implementa o mesmo algoritmo Map-Reduce do train_parallel_from_stream
|
| 705 |
+
mas SEM spawn de subprocessos. Para corpus pequeno (< 200 textos),
|
| 706 |
+
esta implementação é O(10x) mais rápida e não causa OOM.
|
| 707 |
+
|
| 708 |
+
Algoritmo:
|
| 709 |
+
1. Pré-tokeniza corpus em símbolos byte-level (in-process)
|
| 710 |
+
2. Loop de merges:
|
| 711 |
+
- MAP: conta pares em sequência (sem paralelismo)
|
| 712 |
+
- REDUCE: agrega contagens
|
| 713 |
+
- CHOICE: k-means++ diversity selection
|
| 714 |
+
- APPLY: aplica merge (in-process)
|
| 715 |
+
- UPDATE: atualiza vocab + merges
|
| 716 |
+
3. Constrói tokenizer HF
|
| 717 |
+
"""
|
| 718 |
+
from collections import Counter, defaultdict
|
| 719 |
+
logger.info(
|
| 720 |
+
"BBPE SERIAL (in-process): vocab_size=%d, corpus=%d textos",
|
| 721 |
+
self.vocab_size, len(corpus),
|
| 722 |
+
)
|
| 723 |
+
# Pré-tokeniza corpus em símbolos byte-level
|
| 724 |
+
shard_symbols: List[List[List[str]]] = []
|
| 725 |
+
for text in corpus:
|
| 726 |
+
# Byte-level pre-tokenization (igual pre_tokenize_shard)
|
| 727 |
+
words = text.split()
|
| 728 |
+
shard_symbols.append([list(w.encode('utf-8').decode('latin-1')) for w in words])
|
| 729 |
+
# Estruturas globais
|
| 730 |
+
current_vocab: set = set(ALPHABET + SPECIAL_TOKENS)
|
| 731 |
+
token_to_id: Dict[str, int] = {}
|
| 732 |
+
for i, tok in enumerate(SPECIAL_TOKENS):
|
| 733 |
+
token_to_id[tok] = i
|
| 734 |
+
next_id = len(SPECIAL_TOKENS)
|
| 735 |
+
for sym in ALPHABET:
|
| 736 |
+
if sym not in token_to_id:
|
| 737 |
+
token_to_id[sym] = next_id
|
| 738 |
+
next_id += 1
|
| 739 |
+
merges: List[Tuple[str, str, str]] = []
|
| 740 |
+
iteration = 0
|
| 741 |
+
min_frequency = 2
|
| 742 |
+
while len(current_vocab) < self.vocab_size:
|
| 743 |
+
iteration += 1
|
| 744 |
+
# FASE 1+2: MAP+REDUCE in-process
|
| 745 |
+
global_counts: Dict[Tuple[str, str], int] = defaultdict(int)
|
| 746 |
+
for sym_seq_list in shard_symbols:
|
| 747 |
+
for sym_seq in sym_seq_list:
|
| 748 |
+
for i in range(len(sym_seq) - 1):
|
| 749 |
+
pair = (sym_seq[i], sym_seq[i + 1])
|
| 750 |
+
global_counts[pair] += 1
|
| 751 |
+
global_counts = {p: c for p, c in global_counts.items() if c >= min_frequency}
|
| 752 |
+
if not global_counts:
|
| 753 |
+
logger.info(
|
| 754 |
+
"BBPE SERIAL: nenhum par com freq >= %d. Vocab final: %d (target %d)",
|
| 755 |
+
min_frequency, len(current_vocab), self.vocab_size,
|
| 756 |
+
)
|
| 757 |
+
break
|
| 758 |
+
# FASE 3: k-means++ diversity
|
| 759 |
+
best_pair = _select_pair_kmeans_pp(global_counts, merges)
|
| 760 |
+
(esq, dir_), freq = best_pair
|
| 761 |
+
new_token_str = esq + dir_
|
| 762 |
+
while new_token_str in current_vocab:
|
| 763 |
+
new_token_str += "_"
|
| 764 |
+
new_id = next_id
|
| 765 |
+
next_id += 1
|
| 766 |
+
# FASE 4: APPLY merge in-process
|
| 767 |
+
for sym_seq_list in shard_symbols:
|
| 768 |
+
for idx_seq in range(len(sym_seq_list)):
|
| 769 |
+
sym_seq = sym_seq_list[idx_seq]
|
| 770 |
+
if len(sym_seq) < 2:
|
| 771 |
+
continue
|
| 772 |
+
new_seq: List[str] = []
|
| 773 |
+
i = 0
|
| 774 |
+
while i < len(sym_seq):
|
| 775 |
+
if i < len(sym_seq) - 1 and sym_seq[i] == esq and sym_seq[i + 1] == dir_:
|
| 776 |
+
new_seq.append(new_token_str)
|
| 777 |
+
i += 2
|
| 778 |
+
else:
|
| 779 |
+
new_seq.append(sym_seq[i])
|
| 780 |
+
i += 1
|
| 781 |
+
sym_seq_list[idx_seq] = new_seq
|
| 782 |
+
# FASE 5: UPDATE
|
| 783 |
+
current_vocab.add(new_token_str)
|
| 784 |
+
token_to_id[new_token_str] = new_id
|
| 785 |
+
merges.append((esq, dir_, new_token_str))
|
| 786 |
+
if iteration % 50 == 0:
|
| 787 |
+
logger.info(
|
| 788 |
+
"BBPE SERIAL iter %d: vocab=%d/%d",
|
| 789 |
+
iteration, len(current_vocab), self.vocab_size,
|
| 790 |
+
)
|
| 791 |
+
if iteration % 100 == 0:
|
| 792 |
+
gc.collect()
|
| 793 |
+
logger.info(
|
| 794 |
+
"BBPE SERIAL: concluído. %d merges, vocab=%d. Construindo tokenizer HF...",
|
| 795 |
+
len(merges), len(token_to_id),
|
| 796 |
+
)
|
| 797 |
+
self._merges = merges
|
| 798 |
+
self._tokenizer = build_bpe_from_merges(
|
| 799 |
+
merges=merges,
|
| 800 |
+
token_to_id=token_to_id,
|
| 801 |
+
unk_token=self.unk_token,
|
| 802 |
+
add_prefix_space=self.add_prefix_space,
|
| 803 |
+
)
|
| 804 |
+
self._build_vocab_cache()
|
| 805 |
+
self.vocab_size = len(self._vocab)
|
| 806 |
+
logger.info("BBPE SERIAL: tokenizer construído. Vocab real: %d", len(self._vocab))
|
| 807 |
+
|
| 808 |
+
def _wordlevel_fallback(self, corpus: List[str]) -> None:
|
| 809 |
+
"""V6.5-V4-oom-fix: Word-level fallback se BBPE falhar."""
|
| 810 |
+
from collections import Counter
|
| 811 |
+
logger.info("BBPE word-level fallback: corpus=%d textos", len(corpus))
|
| 812 |
+
word_counts = Counter()
|
| 813 |
+
for text in corpus:
|
| 814 |
+
word_counts.update(text.split())
|
| 815 |
+
sorted_words = [w for w, _ in word_counts.most_common(self.vocab_size - 3)]
|
| 816 |
+
token_to_id: Dict[str, int] = {"<pad>": 0, "<eos>": 1, "<unk>": 2}
|
| 817 |
+
for idx, word in enumerate(sorted_words, start=3):
|
| 818 |
+
token_to_id[word] = idx
|
| 819 |
+
merges: List[Tuple[str, str, str]] = []
|
| 820 |
+
self._merges = merges
|
| 821 |
+
self._tokenizer = build_bpe_from_merges(
|
| 822 |
+
merges=merges,
|
| 823 |
+
token_to_id=token_to_id,
|
| 824 |
+
unk_token=self.unk_token,
|
| 825 |
+
add_prefix_space=self.add_prefix_space,
|
| 826 |
+
)
|
| 827 |
+
self._build_vocab_cache()
|
| 828 |
+
self.vocab_size = len(self._vocab)
|
| 829 |
+
logger.info("BBPE word-level fallback: vocab=%d", len(self._vocab))
|
| 830 |
+
|
| 831 |
# ------------------------------------------------------------------
|
| 832 |
# TREINAMENTO (compatibilidade — delega para paralelo)
|
| 833 |
# ------------------------------------------------------------------
|
states/v6_5_v2_model_states.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6eded03affb9f367f3162fd7e7bcedd90bfb3c406dc94dccba86f8ab9dfac201
|
| 3 |
+
size 117713262
|