Add files using upload-large-folder tool
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- _sweep_logs/master.log +12 -0
- _sweep_logs/nohup.out +90 -0
- _sweep_logs/topk_k30_s1_barrier.log +30 -0
- _sweep_logs/topk_k30_s2_barrier.log +30 -0
- _sweep_logs/topk_k40_s1_mc.log +43 -0
- _sweep_logs/topk_k40_s2_barrier.log +30 -0
- _sweep_logs/topk_k45_s1_barrier.log +30 -0
- _sweep_logs/topk_k45_s1_mc.log +43 -0
- _sweep_logs/topk_k45_s2_mc.log +43 -0
- _sweep_logs/topk_k48_s2_mc.log +43 -0
- _sweep_logs/topk_k50_s1_barrier.log +30 -0
- _sweep_logs/topk_k50_s1_mc.log +43 -0
- _sweep_logs/topk_k50_s2_barrier.log +30 -0
- altmin_perturbed_omp_k48_s43_c0.30/results.json +23 -0
- altmin_perturbed_omp_k48_s43_c0.40/results.json +23 -0
- basin_radius/results_seed42.json +104 -0
- basin_radius/results_seed43.json +104 -0
- basin_radius/results_seed44.json +104 -0
- frozen_dec_k48_s1/results.json +26 -0
- oracle_init_smoke/cfg.json +1 -0
- oracle_init_smoke/ckpts/114688/activation_scaler.json +1 -0
- oracle_init_smoke/ckpts/114688/cfg.json +1 -0
- oracle_init_smoke/ckpts/114688/runner_config.json +59 -0
- oracle_init_smoke/ckpts/225280/activation_scaler.json +1 -0
- oracle_init_smoke/ckpts/225280/cfg.json +1 -0
- oracle_init_smoke/ckpts/225280/runner_config.json +59 -0
- oracle_init_smoke/ckpts/335872/activation_scaler.json +1 -0
- oracle_init_smoke/ckpts/335872/cfg.json +1 -0
- oracle_init_smoke/ckpts/335872/runner_config.json +59 -0
- oracle_init_smoke/ckpts/446464/activation_scaler.json +1 -0
- oracle_init_smoke/ckpts/446464/cfg.json +1 -0
- oracle_init_smoke/ckpts/446464/runner_config.json +59 -0
- oracle_init_smoke/ckpts/557056/activation_scaler.json +1 -0
- oracle_init_smoke/ckpts/557056/cfg.json +1 -0
- oracle_init_smoke/ckpts/557056/runner_config.json +59 -0
- oracle_init_smoke/ckpts/667648/activation_scaler.json +1 -0
- oracle_init_smoke/ckpts/667648/cfg.json +1 -0
- oracle_init_smoke/ckpts/667648/runner_config.json +59 -0
- oracle_init_smoke/ckpts/778240/activation_scaler.json +1 -0
- oracle_init_smoke/ckpts/778240/cfg.json +1 -0
- oracle_init_smoke/ckpts/778240/runner_config.json +59 -0
- oracle_init_smoke/ckpts/892928/activation_scaler.json +1 -0
- oracle_init_smoke/ckpts/892928/cfg.json +1 -0
- oracle_init_smoke/ckpts/892928/runner_config.json +59 -0
- oracle_init_smoke/ckpts/final_1003520/activation_scaler.json +1 -0
- oracle_init_smoke/ckpts/final_1003520/cfg.json +1 -0
- oracle_init_smoke/ckpts/final_1003520/runner_config.json +59 -0
- oracle_init_smoke/eval_stats.json +15 -0
- oracle_init_smoke/meta.json +8 -0
- oracle_init_smoke/runner_config.json +59 -0
_sweep_logs/master.log
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
===== topk_k30_s1 (k=30) =====
|
| 2 |
+
===== topk_k30_s2 (k=30) =====
|
| 3 |
+
===== topk_k35_s2 (k=35) =====
|
| 4 |
+
===== topk_k40_s1 (k=40) =====
|
| 5 |
+
===== topk_k40_s2 (k=40) =====
|
| 6 |
+
===== topk_k45_s1 (k=45) =====
|
| 7 |
+
===== topk_k45_s2 (k=45) =====
|
| 8 |
+
===== topk_k48_s2 (k=48) =====
|
| 9 |
+
===== topk_k50_s1 (k=50) =====
|
| 10 |
+
===== topk_k50_s2 (k=50) =====
|
| 11 |
+
===== EXTRA topk_k35_s1 (k=35) mc =====
|
| 12 |
+
ALL_BARRIER_MC_DONE
|
_sweep_logs/nohup.out
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
===== topk_k30_s1 (k=30) =====
|
| 2 |
+
|
| 3 |
+
wrote runs/topk_k30_s1/barrier.json
|
| 4 |
+
|
| 5 |
+
summary: MSE(α=0) = 0.05222 -> MSE(α=1) = 0.13120
|
| 6 |
+
max over path = 0.13120 argmax α = 1.000
|
| 7 |
+
|
| 8 |
+
summary: peak=0.13200 argmax α=1.000
|
| 9 |
+
endpoints: 0.05251, 0.13200
|
| 10 |
+
===== topk_k30_s2 (k=30) =====
|
| 11 |
+
|
| 12 |
+
wrote runs/topk_k30_s2/barrier.json
|
| 13 |
+
|
| 14 |
+
summary: MSE(α=0) = 0.05386 -> MSE(α=1) = 0.13341
|
| 15 |
+
max over path = 0.13341 argmax α = 1.000
|
| 16 |
+
|
| 17 |
+
summary: peak=0.13334 argmax α=1.000
|
| 18 |
+
endpoints: 0.05225, 0.13334
|
| 19 |
+
===== topk_k35_s2 (k=35) =====
|
| 20 |
+
|
| 21 |
+
wrote runs/topk_k35_s2/barrier.json
|
| 22 |
+
|
| 23 |
+
summary: MSE(α=0) = 0.04577 -> MSE(α=1) = 0.07856
|
| 24 |
+
max over path = 0.07856 argmax α = 1.000
|
| 25 |
+
|
| 26 |
+
summary: peak=0.07992 argmax α=1.000
|
| 27 |
+
endpoints: 0.04629, 0.07992
|
| 28 |
+
===== topk_k40_s1 (k=40) =====
|
| 29 |
+
|
| 30 |
+
wrote runs/topk_k40_s1/barrier.json
|
| 31 |
+
|
| 32 |
+
summary: MSE(α=0) = 0.03803 -> MSE(α=1) = 0.04402
|
| 33 |
+
max over path = 0.05656 argmax α = 0.450
|
| 34 |
+
|
| 35 |
+
summary: peak=0.05201 argmax α=0.400
|
| 36 |
+
endpoints: 0.03904, 0.04542
|
| 37 |
+
===== topk_k40_s2 (k=40) =====
|
| 38 |
+
|
| 39 |
+
wrote runs/topk_k40_s2/barrier.json
|
| 40 |
+
|
| 41 |
+
summary: MSE(α=0) = 0.03702 -> MSE(α=1) = 0.04158
|
| 42 |
+
max over path = 0.05404 argmax α = 0.450
|
| 43 |
+
|
| 44 |
+
summary: peak=0.05049 argmax α=0.400
|
| 45 |
+
endpoints: 0.03750, 0.04304
|
| 46 |
+
===== topk_k45_s1 (k=45) =====
|
| 47 |
+
|
| 48 |
+
wrote runs/topk_k45_s1/barrier.json
|
| 49 |
+
|
| 50 |
+
summary: MSE(α=0) = 0.03002 -> MSE(α=1) = 0.02165
|
| 51 |
+
max over path = 0.06657 argmax α = 0.500
|
| 52 |
+
|
| 53 |
+
summary: peak=0.04251 argmax α=0.400
|
| 54 |
+
endpoints: 0.02965, 0.02010
|
| 55 |
+
===== topk_k45_s2 (k=45) =====
|
| 56 |
+
|
| 57 |
+
wrote runs/topk_k45_s2/barrier.json
|
| 58 |
+
|
| 59 |
+
summary: MSE(α=0) = 0.03035 -> MSE(α=1) = 0.02334
|
| 60 |
+
max over path = 0.06798 argmax α = 0.500
|
| 61 |
+
|
| 62 |
+
summary: peak=0.04420 argmax α=0.450
|
| 63 |
+
endpoints: 0.02988, 0.02408
|
| 64 |
+
===== topk_k48_s2 (k=48) =====
|
| 65 |
+
|
| 66 |
+
summary: peak=0.03999 argmax α=0.400
|
| 67 |
+
endpoints: 0.02708, 0.01338
|
| 68 |
+
===== topk_k50_s1 (k=50) =====
|
| 69 |
+
|
| 70 |
+
wrote runs/topk_k50_s1/barrier.json
|
| 71 |
+
|
| 72 |
+
summary: MSE(α=0) = 0.02412 -> MSE(α=1) = 0.00932
|
| 73 |
+
max over path = 0.05774 argmax α = 0.500
|
| 74 |
+
|
| 75 |
+
summary: peak=0.03846 argmax α=0.450
|
| 76 |
+
endpoints: 0.02413, 0.01036
|
| 77 |
+
===== topk_k50_s2 (k=50) =====
|
| 78 |
+
|
| 79 |
+
wrote runs/topk_k50_s2/barrier.json
|
| 80 |
+
|
| 81 |
+
summary: MSE(α=0) = 0.02492 -> MSE(α=1) = 0.01075
|
| 82 |
+
max over path = 0.06035 argmax α = 0.450
|
| 83 |
+
|
| 84 |
+
summary: peak=0.03823 argmax α=0.350
|
| 85 |
+
endpoints: 0.02410, 0.00993
|
| 86 |
+
===== EXTRA topk_k35_s1 (k=35) mc =====
|
| 87 |
+
|
| 88 |
+
summary: peak=0.07593 argmax α=1.000
|
| 89 |
+
endpoints: 0.04496, 0.07593
|
| 90 |
+
ALL_BARRIER_MC_DONE
|
_sweep_logs/topk_k30_s1_barrier.log
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
barrier eval: run=runs/topk_k30_s1 k=30 n=5000 alphas=21
|
| 3 |
+
|
| 4 |
+
Hungarian alignment: mean cos = 0.6010 (0.6010 per scipy)
|
| 5 |
+
α=0.000 MSE=0.05222
|
| 6 |
+
α=0.050 MSE=0.05257
|
| 7 |
+
α=0.100 MSE=0.05410
|
| 8 |
+
α=0.150 MSE=0.05670
|
| 9 |
+
α=0.200 MSE=0.06035
|
| 10 |
+
α=0.250 MSE=0.06481
|
| 11 |
+
α=0.300 MSE=0.06998
|
| 12 |
+
α=0.350 MSE=0.07554
|
| 13 |
+
α=0.400 MSE=0.08108
|
| 14 |
+
α=0.450 MSE=0.08637
|
| 15 |
+
α=0.500 MSE=0.09153
|
| 16 |
+
α=0.550 MSE=0.09655
|
| 17 |
+
α=0.600 MSE=0.10156
|
| 18 |
+
α=0.650 MSE=0.10641
|
| 19 |
+
α=0.700 MSE=0.11103
|
| 20 |
+
α=0.750 MSE=0.11534
|
| 21 |
+
α=0.800 MSE=0.11899
|
| 22 |
+
α=0.850 MSE=0.12208
|
| 23 |
+
α=0.900 MSE=0.12484
|
| 24 |
+
α=0.950 MSE=0.12781
|
| 25 |
+
α=1.000 MSE=0.13120
|
| 26 |
+
|
| 27 |
+
wrote runs/topk_k30_s1/barrier.json
|
| 28 |
+
|
| 29 |
+
summary: MSE(α=0) = 0.05222 -> MSE(α=1) = 0.13120
|
| 30 |
+
max over path = 0.13120 argmax α = 1.000
|
_sweep_logs/topk_k30_s2_barrier.log
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
barrier eval: run=runs/topk_k30_s2 k=30 n=5000 alphas=21
|
| 3 |
+
|
| 4 |
+
Hungarian alignment: mean cos = 0.5997 (0.5997 per scipy)
|
| 5 |
+
α=0.000 MSE=0.05386
|
| 6 |
+
α=0.050 MSE=0.05409
|
| 7 |
+
α=0.100 MSE=0.05553
|
| 8 |
+
α=0.150 MSE=0.05806
|
| 9 |
+
α=0.200 MSE=0.06159
|
| 10 |
+
α=0.250 MSE=0.06596
|
| 11 |
+
α=0.300 MSE=0.07100
|
| 12 |
+
α=0.350 MSE=0.07646
|
| 13 |
+
α=0.400 MSE=0.08206
|
| 14 |
+
α=0.450 MSE=0.08748
|
| 15 |
+
α=0.500 MSE=0.09287
|
| 16 |
+
α=0.550 MSE=0.09813
|
| 17 |
+
α=0.600 MSE=0.10334
|
| 18 |
+
α=0.650 MSE=0.10837
|
| 19 |
+
α=0.700 MSE=0.11316
|
| 20 |
+
α=0.750 MSE=0.11754
|
| 21 |
+
α=0.800 MSE=0.12121
|
| 22 |
+
α=0.850 MSE=0.12434
|
| 23 |
+
α=0.900 MSE=0.12713
|
| 24 |
+
α=0.950 MSE=0.13001
|
| 25 |
+
α=1.000 MSE=0.13341
|
| 26 |
+
|
| 27 |
+
wrote runs/topk_k30_s2/barrier.json
|
| 28 |
+
|
| 29 |
+
summary: MSE(α=0) = 0.05386 -> MSE(α=1) = 0.13341
|
| 30 |
+
max over path = 0.13341 argmax α = 1.000
|
_sweep_logs/topk_k40_s1_mc.log
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
mode connectivity: run=runs/topk_k40_s1 k=40 train α=[0.2, 0.4, 0.5, 0.6, 0.8]
|
| 3 |
+
|
| 4 |
+
optimizing midpoint dictionary (80 steps, batch 128)...
|
| 5 |
+
step 1 meanMSE=0.05600 [0.20:0.0436 0.40:0.0578 0.50:0.0611 0.60:0.0611 0.80:0.0564]
|
| 6 |
+
step 10 meanMSE=0.05563 [0.20:0.0486 0.40:0.0597 0.50:0.0603 0.60:0.0581 0.80:0.0515]
|
| 7 |
+
step 20 meanMSE=0.04781 [0.20:0.0420 0.40:0.0526 0.50:0.0525 0.60:0.0498 0.80:0.0423]
|
| 8 |
+
step 30 meanMSE=0.05105 [0.20:0.0463 0.40:0.0557 0.50:0.0553 0.60:0.0523 0.80:0.0456]
|
| 9 |
+
step 40 meanMSE=0.05143 [0.20:0.0469 0.40:0.0551 0.50:0.0550 0.60:0.0528 0.80:0.0473]
|
| 10 |
+
step 50 meanMSE=0.04723 [0.20:0.0418 0.40:0.0516 0.50:0.0517 0.60:0.0494 0.80:0.0417]
|
| 11 |
+
step 60 meanMSE=0.05595 [0.20:0.0500 0.40:0.0590 0.50:0.0592 0.60:0.0571 0.80:0.0545]
|
| 12 |
+
step 70 meanMSE=0.04839 [0.20:0.0472 0.40:0.0533 0.50:0.0521 0.60:0.0487 0.80:0.0407]
|
| 13 |
+
step 80 meanMSE=0.04619 [0.20:0.0445 0.40:0.0507 0.50:0.0499 0.60:0.0465 0.80:0.0395]
|
| 14 |
+
|
| 15 |
+
optimised in 13.0 sec
|
| 16 |
+
|
| 17 |
+
evaluating optimised curve (21 αs × 5000 samples)
|
| 18 |
+
α=0.000 MSE=0.03904
|
| 19 |
+
α=0.050 MSE=0.03948
|
| 20 |
+
α=0.100 MSE=0.04142
|
| 21 |
+
α=0.150 MSE=0.04389
|
| 22 |
+
α=0.200 MSE=0.04638
|
| 23 |
+
α=0.250 MSE=0.04857
|
| 24 |
+
α=0.300 MSE=0.05028
|
| 25 |
+
α=0.350 MSE=0.05146
|
| 26 |
+
α=0.400 MSE=0.05201
|
| 27 |
+
α=0.450 MSE=0.05191
|
| 28 |
+
α=0.500 MSE=0.05122
|
| 29 |
+
α=0.550 MSE=0.04995
|
| 30 |
+
α=0.600 MSE=0.04826
|
| 31 |
+
α=0.650 MSE=0.04631
|
| 32 |
+
α=0.700 MSE=0.04447
|
| 33 |
+
α=0.750 MSE=0.04308
|
| 34 |
+
α=0.800 MSE=0.04241
|
| 35 |
+
α=0.850 MSE=0.04253
|
| 36 |
+
α=0.900 MSE=0.04318
|
| 37 |
+
α=0.950 MSE=0.04408
|
| 38 |
+
α=1.000 MSE=0.04542
|
| 39 |
+
|
| 40 |
+
wrote runs/topk_k40_s1/mode_connectivity.json
|
| 41 |
+
|
| 42 |
+
summary: peak=0.05201 argmax α=0.400
|
| 43 |
+
endpoints: 0.03904, 0.04542
|
_sweep_logs/topk_k40_s2_barrier.log
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
barrier eval: run=runs/topk_k40_s2 k=40 n=5000 alphas=21
|
| 3 |
+
|
| 4 |
+
Hungarian alignment: mean cos = 0.6576 (0.6576 per scipy)
|
| 5 |
+
α=0.000 MSE=0.03702
|
| 6 |
+
α=0.050 MSE=0.03730
|
| 7 |
+
α=0.100 MSE=0.03833
|
| 8 |
+
α=0.150 MSE=0.04001
|
| 9 |
+
α=0.200 MSE=0.04229
|
| 10 |
+
α=0.250 MSE=0.04503
|
| 11 |
+
α=0.300 MSE=0.04802
|
| 12 |
+
α=0.350 MSE=0.05084
|
| 13 |
+
α=0.400 MSE=0.05300
|
| 14 |
+
α=0.450 MSE=0.05404
|
| 15 |
+
α=0.500 MSE=0.05397
|
| 16 |
+
α=0.550 MSE=0.05314
|
| 17 |
+
α=0.600 MSE=0.05162
|
| 18 |
+
α=0.650 MSE=0.04985
|
| 19 |
+
α=0.700 MSE=0.04784
|
| 20 |
+
α=0.750 MSE=0.04588
|
| 21 |
+
α=0.800 MSE=0.04399
|
| 22 |
+
α=0.850 MSE=0.04239
|
| 23 |
+
α=0.900 MSE=0.04129
|
| 24 |
+
α=0.950 MSE=0.04095
|
| 25 |
+
α=1.000 MSE=0.04158
|
| 26 |
+
|
| 27 |
+
wrote runs/topk_k40_s2/barrier.json
|
| 28 |
+
|
| 29 |
+
summary: MSE(α=0) = 0.03702 -> MSE(α=1) = 0.04158
|
| 30 |
+
max over path = 0.05404 argmax α = 0.450
|
_sweep_logs/topk_k45_s1_barrier.log
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
barrier eval: run=runs/topk_k45_s1 k=45 n=5000 alphas=21
|
| 3 |
+
|
| 4 |
+
Hungarian alignment: mean cos = 0.6868 (0.6868 per scipy)
|
| 5 |
+
α=0.000 MSE=0.03002
|
| 6 |
+
α=0.050 MSE=0.03061
|
| 7 |
+
α=0.100 MSE=0.03255
|
| 8 |
+
α=0.150 MSE=0.03566
|
| 9 |
+
α=0.200 MSE=0.03990
|
| 10 |
+
α=0.250 MSE=0.04514
|
| 11 |
+
α=0.300 MSE=0.05113
|
| 12 |
+
α=0.350 MSE=0.05728
|
| 13 |
+
α=0.400 MSE=0.06258
|
| 14 |
+
α=0.450 MSE=0.06588
|
| 15 |
+
α=0.500 MSE=0.06657
|
| 16 |
+
α=0.550 MSE=0.06490
|
| 17 |
+
α=0.600 MSE=0.06119
|
| 18 |
+
α=0.650 MSE=0.05584
|
| 19 |
+
α=0.700 MSE=0.04892
|
| 20 |
+
α=0.750 MSE=0.04164
|
| 21 |
+
α=0.800 MSE=0.03480
|
| 22 |
+
α=0.850 MSE=0.02899
|
| 23 |
+
α=0.900 MSE=0.02470
|
| 24 |
+
α=0.950 MSE=0.02214
|
| 25 |
+
α=1.000 MSE=0.02165
|
| 26 |
+
|
| 27 |
+
wrote runs/topk_k45_s1/barrier.json
|
| 28 |
+
|
| 29 |
+
summary: MSE(α=0) = 0.03002 -> MSE(α=1) = 0.02165
|
| 30 |
+
max over path = 0.06657 argmax α = 0.500
|
_sweep_logs/topk_k45_s1_mc.log
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
mode connectivity: run=runs/topk_k45_s1 k=45 train α=[0.2, 0.4, 0.5, 0.6, 0.8]
|
| 3 |
+
|
| 4 |
+
optimizing midpoint dictionary (80 steps, batch 128)...
|
| 5 |
+
step 1 meanMSE=0.05412 [0.20:0.0382 0.40:0.0623 0.50:0.0674 0.60:0.0636 0.80:0.0391]
|
| 6 |
+
step 10 meanMSE=0.04365 [0.20:0.0362 0.40:0.0474 0.50:0.0489 0.60:0.0481 0.80:0.0377]
|
| 7 |
+
step 20 meanMSE=0.04038 [0.20:0.0373 0.40:0.0472 0.50:0.0464 0.60:0.0417 0.80:0.0293]
|
| 8 |
+
step 30 meanMSE=0.04035 [0.20:0.0373 0.40:0.0464 0.50:0.0453 0.60:0.0414 0.80:0.0314]
|
| 9 |
+
step 40 meanMSE=0.04231 [0.20:0.0401 0.40:0.0482 0.50:0.0471 0.60:0.0430 0.80:0.0331]
|
| 10 |
+
step 50 meanMSE=0.04177 [0.20:0.0402 0.40:0.0487 0.50:0.0478 0.60:0.0429 0.80:0.0292]
|
| 11 |
+
step 60 meanMSE=0.04048 [0.20:0.0378 0.40:0.0461 0.50:0.0456 0.60:0.0422 0.80:0.0306]
|
| 12 |
+
step 70 meanMSE=0.03406 [0.20:0.0327 0.40:0.0394 0.50:0.0392 0.60:0.0357 0.80:0.0234]
|
| 13 |
+
step 80 meanMSE=0.04230 [0.20:0.0406 0.40:0.0468 0.50:0.0462 0.60:0.0435 0.80:0.0344]
|
| 14 |
+
|
| 15 |
+
optimised in 15.7 sec
|
| 16 |
+
|
| 17 |
+
evaluating optimised curve (21 αs × 5000 samples)
|
| 18 |
+
α=0.000 MSE=0.02965
|
| 19 |
+
α=0.050 MSE=0.03000
|
| 20 |
+
α=0.100 MSE=0.03170
|
| 21 |
+
α=0.150 MSE=0.03396
|
| 22 |
+
α=0.200 MSE=0.03634
|
| 23 |
+
α=0.250 MSE=0.03857
|
| 24 |
+
α=0.300 MSE=0.04046
|
| 25 |
+
α=0.350 MSE=0.04184
|
| 26 |
+
α=0.400 MSE=0.04251
|
| 27 |
+
α=0.450 MSE=0.04244
|
| 28 |
+
α=0.500 MSE=0.04160
|
| 29 |
+
α=0.550 MSE=0.04011
|
| 30 |
+
α=0.600 MSE=0.03800
|
| 31 |
+
α=0.650 MSE=0.03541
|
| 32 |
+
α=0.700 MSE=0.03256
|
| 33 |
+
α=0.750 MSE=0.02964
|
| 34 |
+
α=0.800 MSE=0.02683
|
| 35 |
+
α=0.850 MSE=0.02429
|
| 36 |
+
α=0.900 MSE=0.02207
|
| 37 |
+
α=0.950 MSE=0.02045
|
| 38 |
+
α=1.000 MSE=0.02010
|
| 39 |
+
|
| 40 |
+
wrote runs/topk_k45_s1/mode_connectivity.json
|
| 41 |
+
|
| 42 |
+
summary: peak=0.04251 argmax α=0.400
|
| 43 |
+
endpoints: 0.02965, 0.02010
|
_sweep_logs/topk_k45_s2_mc.log
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
mode connectivity: run=runs/topk_k45_s2 k=45 train α=[0.2, 0.4, 0.5, 0.6, 0.8]
|
| 3 |
+
|
| 4 |
+
optimizing midpoint dictionary (80 steps, batch 128)...
|
| 5 |
+
step 1 meanMSE=0.05019 [0.20:0.0415 0.40:0.0622 0.50:0.0635 0.60:0.0562 0.80:0.0275]
|
| 6 |
+
step 10 meanMSE=0.03718 [0.20:0.0361 0.40:0.0443 0.50:0.0429 0.60:0.0384 0.80:0.0242]
|
| 7 |
+
step 20 meanMSE=0.04184 [0.20:0.0359 0.40:0.0470 0.50:0.0472 0.60:0.0445 0.80:0.0345]
|
| 8 |
+
step 30 meanMSE=0.04091 [0.20:0.0369 0.40:0.0466 0.50:0.0463 0.60:0.0427 0.80:0.0321]
|
| 9 |
+
step 40 meanMSE=0.03408 [0.20:0.0324 0.40:0.0405 0.50:0.0396 0.60:0.0351 0.80:0.0228]
|
| 10 |
+
step 50 meanMSE=0.03986 [0.20:0.0386 0.40:0.0465 0.50:0.0452 0.60:0.0409 0.80:0.0281]
|
| 11 |
+
step 60 meanMSE=0.03756 [0.20:0.0387 0.40:0.0441 0.50:0.0423 0.60:0.0376 0.80:0.0253]
|
| 12 |
+
step 70 meanMSE=0.03755 [0.20:0.0356 0.40:0.0425 0.50:0.0417 0.60:0.0383 0.80:0.0297]
|
| 13 |
+
step 80 meanMSE=0.03776 [0.20:0.0385 0.40:0.0435 0.50:0.0423 0.60:0.0383 0.80:0.0262]
|
| 14 |
+
|
| 15 |
+
optimised in 15.6 sec
|
| 16 |
+
|
| 17 |
+
evaluating optimised curve (21 αs × 5000 samples)
|
| 18 |
+
α=0.000 MSE=0.02988
|
| 19 |
+
α=0.050 MSE=0.03029
|
| 20 |
+
α=0.100 MSE=0.03216
|
| 21 |
+
α=0.150 MSE=0.03468
|
| 22 |
+
α=0.200 MSE=0.03731
|
| 23 |
+
α=0.250 MSE=0.03977
|
| 24 |
+
α=0.300 MSE=0.04187
|
| 25 |
+
α=0.350 MSE=0.04334
|
| 26 |
+
α=0.400 MSE=0.04415
|
| 27 |
+
α=0.450 MSE=0.04420
|
| 28 |
+
α=0.500 MSE=0.04354
|
| 29 |
+
α=0.550 MSE=0.04225
|
| 30 |
+
α=0.600 MSE=0.04033
|
| 31 |
+
α=0.650 MSE=0.03791
|
| 32 |
+
α=0.700 MSE=0.03530
|
| 33 |
+
α=0.750 MSE=0.03261
|
| 34 |
+
α=0.800 MSE=0.03006
|
| 35 |
+
α=0.850 MSE=0.02778
|
| 36 |
+
α=0.900 MSE=0.02578
|
| 37 |
+
α=0.950 MSE=0.02422
|
| 38 |
+
α=1.000 MSE=0.02408
|
| 39 |
+
|
| 40 |
+
wrote runs/topk_k45_s2/mode_connectivity.json
|
| 41 |
+
|
| 42 |
+
summary: peak=0.04420 argmax α=0.450
|
| 43 |
+
endpoints: 0.02988, 0.02408
|
_sweep_logs/topk_k48_s2_mc.log
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
mode connectivity: run=runs/topk_k48_s2 k=48 train α=[0.2, 0.4, 0.5, 0.6, 0.8]
|
| 3 |
+
|
| 4 |
+
optimizing midpoint dictionary (80 steps, batch 128)...
|
| 5 |
+
step 1 meanMSE=0.04885 [0.20:0.0373 0.40:0.0593 0.50:0.0621 0.60:0.0562 0.80:0.0293]
|
| 6 |
+
step 10 meanMSE=0.03331 [0.20:0.0332 0.40:0.0413 0.50:0.0398 0.60:0.0346 0.80:0.0176]
|
| 7 |
+
step 20 meanMSE=0.02877 [0.20:0.0288 0.40:0.0365 0.50:0.0346 0.60:0.0296 0.80:0.0144]
|
| 8 |
+
step 30 meanMSE=0.03208 [0.20:0.0326 0.40:0.0393 0.50:0.0375 0.60:0.0328 0.80:0.0182]
|
| 9 |
+
step 40 meanMSE=0.03851 [0.20:0.0376 0.40:0.0461 0.50:0.0448 0.60:0.0395 0.80:0.0246]
|
| 10 |
+
step 50 meanMSE=0.03263 [0.20:0.0313 0.40:0.0386 0.50:0.0377 0.60:0.0339 0.80:0.0216]
|
| 11 |
+
step 60 meanMSE=0.03791 [0.20:0.0366 0.40:0.0436 0.50:0.0430 0.60:0.0394 0.80:0.0270]
|
| 12 |
+
step 70 meanMSE=0.03371 [0.20:0.0348 0.40:0.0404 0.50:0.0388 0.60:0.0343 0.80:0.0202]
|
| 13 |
+
step 80 meanMSE=0.03407 [0.20:0.0333 0.40:0.0400 0.50:0.0393 0.60:0.0356 0.80:0.0222]
|
| 14 |
+
|
| 15 |
+
optimised in 16.9 sec
|
| 16 |
+
|
| 17 |
+
evaluating optimised curve (21 αs × 5000 samples)
|
| 18 |
+
α=0.000 MSE=0.02708
|
| 19 |
+
α=0.050 MSE=0.02740
|
| 20 |
+
α=0.100 MSE=0.02909
|
| 21 |
+
α=0.150 MSE=0.03136
|
| 22 |
+
α=0.200 MSE=0.03374
|
| 23 |
+
α=0.250 MSE=0.03599
|
| 24 |
+
α=0.300 MSE=0.03788
|
| 25 |
+
α=0.350 MSE=0.03925
|
| 26 |
+
α=0.400 MSE=0.03999
|
| 27 |
+
α=0.450 MSE=0.03998
|
| 28 |
+
α=0.500 MSE=0.03920
|
| 29 |
+
α=0.550 MSE=0.03772
|
| 30 |
+
α=0.600 MSE=0.03554
|
| 31 |
+
α=0.650 MSE=0.03279
|
| 32 |
+
α=0.700 MSE=0.02966
|
| 33 |
+
α=0.750 MSE=0.02631
|
| 34 |
+
α=0.800 MSE=0.02289
|
| 35 |
+
α=0.850 MSE=0.01957
|
| 36 |
+
α=0.900 MSE=0.01650
|
| 37 |
+
α=0.950 MSE=0.01408
|
| 38 |
+
α=1.000 MSE=0.01338
|
| 39 |
+
|
| 40 |
+
wrote runs/topk_k48_s2/mode_connectivity.json
|
| 41 |
+
|
| 42 |
+
summary: peak=0.03999 argmax α=0.400
|
| 43 |
+
endpoints: 0.02708, 0.01338
|
_sweep_logs/topk_k50_s1_barrier.log
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
barrier eval: run=runs/topk_k50_s1 k=50 n=5000 alphas=21
|
| 3 |
+
|
| 4 |
+
Hungarian alignment: mean cos = 0.7114 (0.7114 per scipy)
|
| 5 |
+
α=0.000 MSE=0.02412
|
| 6 |
+
α=0.050 MSE=0.02465
|
| 7 |
+
α=0.100 MSE=0.02638
|
| 8 |
+
α=0.150 MSE=0.02917
|
| 9 |
+
α=0.200 MSE=0.03299
|
| 10 |
+
α=0.250 MSE=0.03777
|
| 11 |
+
α=0.300 MSE=0.04328
|
| 12 |
+
α=0.350 MSE=0.04895
|
| 13 |
+
α=0.400 MSE=0.05391
|
| 14 |
+
α=0.450 MSE=0.05709
|
| 15 |
+
α=0.500 MSE=0.05774
|
| 16 |
+
α=0.550 MSE=0.05586
|
| 17 |
+
α=0.600 MSE=0.05166
|
| 18 |
+
α=0.650 MSE=0.04569
|
| 19 |
+
α=0.700 MSE=0.03836
|
| 20 |
+
α=0.750 MSE=0.03068
|
| 21 |
+
α=0.800 MSE=0.02352
|
| 22 |
+
α=0.850 MSE=0.01745
|
| 23 |
+
α=0.900 MSE=0.01289
|
| 24 |
+
α=0.950 MSE=0.01013
|
| 25 |
+
α=1.000 MSE=0.00932
|
| 26 |
+
|
| 27 |
+
wrote runs/topk_k50_s1/barrier.json
|
| 28 |
+
|
| 29 |
+
summary: MSE(α=0) = 0.02412 -> MSE(α=1) = 0.00932
|
| 30 |
+
max over path = 0.05774 argmax α = 0.500
|
_sweep_logs/topk_k50_s1_mc.log
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
mode connectivity: run=runs/topk_k50_s1 k=50 train α=[0.2, 0.4, 0.5, 0.6, 0.8]
|
| 3 |
+
|
| 4 |
+
optimizing midpoint dictionary (80 steps, batch 128)...
|
| 5 |
+
step 1 meanMSE=0.04677 [0.20:0.0357 0.40:0.0571 0.50:0.0609 0.60:0.0543 0.80:0.0259]
|
| 6 |
+
step 10 meanMSE=0.02857 [0.20:0.0274 0.40:0.0351 0.50:0.0343 0.60:0.0301 0.80:0.0160]
|
| 7 |
+
step 20 meanMSE=0.02857 [0.20:0.0291 0.40:0.0366 0.50:0.0344 0.60:0.0292 0.80:0.0135]
|
| 8 |
+
step 30 meanMSE=0.03326 [0.20:0.0298 0.40:0.0391 0.50:0.0389 0.60:0.0357 0.80:0.0229]
|
| 9 |
+
step 40 meanMSE=0.03114 [0.20:0.0318 0.40:0.0389 0.50:0.0369 0.60:0.0318 0.80:0.0164]
|
| 10 |
+
step 50 meanMSE=0.03590 [0.20:0.0338 0.40:0.0428 0.50:0.0420 0.60:0.0376 0.80:0.0233]
|
| 11 |
+
step 60 meanMSE=0.03094 [0.20:0.0300 0.40:0.0372 0.50:0.0365 0.60:0.0326 0.80:0.0184]
|
| 12 |
+
step 70 meanMSE=0.03008 [0.20:0.0301 0.40:0.0367 0.50:0.0358 0.60:0.0316 0.80:0.0162]
|
| 13 |
+
step 80 meanMSE=0.02948 [0.20:0.0270 0.40:0.0348 0.50:0.0349 0.60:0.0317 0.80:0.0190]
|
| 14 |
+
|
| 15 |
+
optimised in 18.1 sec
|
| 16 |
+
|
| 17 |
+
evaluating optimised curve (21 αs × 5000 samples)
|
| 18 |
+
α=0.000 MSE=0.02413
|
| 19 |
+
α=0.050 MSE=0.02446
|
| 20 |
+
α=0.100 MSE=0.02617
|
| 21 |
+
α=0.150 MSE=0.02854
|
| 22 |
+
α=0.200 MSE=0.03113
|
| 23 |
+
α=0.250 MSE=0.03361
|
| 24 |
+
α=0.300 MSE=0.03572
|
| 25 |
+
α=0.350 MSE=0.03732
|
| 26 |
+
α=0.400 MSE=0.03828
|
| 27 |
+
α=0.450 MSE=0.03846
|
| 28 |
+
α=0.500 MSE=0.03783
|
| 29 |
+
α=0.550 MSE=0.03636
|
| 30 |
+
α=0.600 MSE=0.03418
|
| 31 |
+
α=0.650 MSE=0.03141
|
| 32 |
+
α=0.700 MSE=0.02818
|
| 33 |
+
α=0.750 MSE=0.02462
|
| 34 |
+
α=0.800 MSE=0.02094
|
| 35 |
+
α=0.850 MSE=0.01723
|
| 36 |
+
α=0.900 MSE=0.01385
|
| 37 |
+
α=0.950 MSE=0.01122
|
| 38 |
+
α=1.000 MSE=0.01036
|
| 39 |
+
|
| 40 |
+
wrote runs/topk_k50_s1/mode_connectivity.json
|
| 41 |
+
|
| 42 |
+
summary: peak=0.03846 argmax α=0.450
|
| 43 |
+
endpoints: 0.02413, 0.01036
|
_sweep_logs/topk_k50_s2_barrier.log
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
barrier eval: run=runs/topk_k50_s2 k=50 n=5000 alphas=21
|
| 3 |
+
|
| 4 |
+
Hungarian alignment: mean cos = 0.7091 (0.7091 per scipy)
|
| 5 |
+
α=0.000 MSE=0.02492
|
| 6 |
+
α=0.050 MSE=0.02552
|
| 7 |
+
α=0.100 MSE=0.02761
|
| 8 |
+
α=0.150 MSE=0.03107
|
| 9 |
+
α=0.200 MSE=0.03573
|
| 10 |
+
α=0.250 MSE=0.04131
|
| 11 |
+
α=0.300 MSE=0.04735
|
| 12 |
+
α=0.350 MSE=0.05317
|
| 13 |
+
α=0.400 MSE=0.05783
|
| 14 |
+
α=0.450 MSE=0.06035
|
| 15 |
+
α=0.500 MSE=0.06025
|
| 16 |
+
α=0.550 MSE=0.05780
|
| 17 |
+
α=0.600 MSE=0.05331
|
| 18 |
+
α=0.650 MSE=0.04725
|
| 19 |
+
α=0.700 MSE=0.03984
|
| 20 |
+
α=0.750 MSE=0.03205
|
| 21 |
+
α=0.800 MSE=0.02483
|
| 22 |
+
α=0.850 MSE=0.01874
|
| 23 |
+
α=0.900 MSE=0.01422
|
| 24 |
+
α=0.950 MSE=0.01150
|
| 25 |
+
α=1.000 MSE=0.01075
|
| 26 |
+
|
| 27 |
+
wrote runs/topk_k50_s2/barrier.json
|
| 28 |
+
|
| 29 |
+
summary: MSE(α=0) = 0.02492 -> MSE(α=1) = 0.01075
|
| 30 |
+
max over path = 0.06035 argmax α = 0.450
|
altmin_perturbed_omp_k48_s43_c0.30/results.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"init": "perturbed",
|
| 3 |
+
"coding": "omp",
|
| 4 |
+
"k": 48,
|
| 5 |
+
"n_samples": 10000000,
|
| 6 |
+
"lr": 0.0003,
|
| 7 |
+
"seed": 43,
|
| 8 |
+
"eval": {
|
| 9 |
+
"true_l0": 34.493215,
|
| 10 |
+
"sae_l0": 38.94516,
|
| 11 |
+
"dead_latents": 0,
|
| 12 |
+
"shrinkage": 0.9630266931152344,
|
| 13 |
+
"explained_variance": 0.9091709596347162,
|
| 14 |
+
"mcc": 0.9172072410583496,
|
| 15 |
+
"uniqueness": 0.9921875,
|
| 16 |
+
"classification/precision": 0.6697086691856384,
|
| 17 |
+
"classification/recall": 0.6960408687591553,
|
| 18 |
+
"classification/f1_score": 0.6599588394165039,
|
| 19 |
+
"classification/accuracy": 0.9984580278396606
|
| 20 |
+
},
|
| 21 |
+
"dec_fidelity_final": 0.9184098839759827,
|
| 22 |
+
"seconds": 3053.9301381111145
|
| 23 |
+
}
|
altmin_perturbed_omp_k48_s43_c0.40/results.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"init": "perturbed",
|
| 3 |
+
"coding": "omp",
|
| 4 |
+
"k": 48,
|
| 5 |
+
"n_samples": 10000000,
|
| 6 |
+
"lr": 0.0003,
|
| 7 |
+
"seed": 43,
|
| 8 |
+
"eval": {
|
| 9 |
+
"true_l0": 34.493215,
|
| 10 |
+
"sae_l0": 39.43884,
|
| 11 |
+
"dead_latents": 0,
|
| 12 |
+
"shrinkage": 0.9721412866210938,
|
| 13 |
+
"explained_variance": 0.9291685954011948,
|
| 14 |
+
"mcc": 0.9540121555328369,
|
| 15 |
+
"uniqueness": 0.99951171875,
|
| 16 |
+
"classification/precision": 0.7348754405975342,
|
| 17 |
+
"classification/recall": 0.750920832157135,
|
| 18 |
+
"classification/f1_score": 0.7241045236587524,
|
| 19 |
+
"classification/accuracy": 0.9988424777984619
|
| 20 |
+
},
|
| 21 |
+
"dec_fidelity_final": 0.9541122913360596,
|
| 22 |
+
"seconds": 3065.881599664688
|
| 23 |
+
}
|
basin_radius/results_seed42.json
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"k": 48,
|
| 3 |
+
"n_steps": 500,
|
| 4 |
+
"n_samples": 2048000,
|
| 5 |
+
"seed": 42,
|
| 6 |
+
"results": [
|
| 7 |
+
{
|
| 8 |
+
"target_cos": 0.99,
|
| 9 |
+
"fid_init": 0.9900000095367432,
|
| 10 |
+
"fid_final": 0.9861536026000977,
|
| 11 |
+
"delta": -0.003846406936645508,
|
| 12 |
+
"converging": false,
|
| 13 |
+
"seconds": 448.9438157081604
|
| 14 |
+
},
|
| 15 |
+
{
|
| 16 |
+
"target_cos": 0.97,
|
| 17 |
+
"fid_init": 0.9700000286102295,
|
| 18 |
+
"fid_final": 0.9848216772079468,
|
| 19 |
+
"delta": 0.014821648597717285,
|
| 20 |
+
"converging": true,
|
| 21 |
+
"seconds": 452.4573805332184
|
| 22 |
+
},
|
| 23 |
+
{
|
| 24 |
+
"target_cos": 0.95,
|
| 25 |
+
"fid_init": 0.9500000476837158,
|
| 26 |
+
"fid_final": 0.9826305508613586,
|
| 27 |
+
"delta": 0.03263050317764282,
|
| 28 |
+
"converging": true,
|
| 29 |
+
"seconds": 453.744690656662
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"target_cos": 0.9,
|
| 33 |
+
"fid_init": 0.9000000953674316,
|
| 34 |
+
"fid_final": 0.9756519198417664,
|
| 35 |
+
"delta": 0.07565182447433472,
|
| 36 |
+
"converging": true,
|
| 37 |
+
"seconds": 455.57620120048523
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"target_cos": 0.85,
|
| 41 |
+
"fid_init": 0.8500000238418579,
|
| 42 |
+
"fid_final": 0.9676114320755005,
|
| 43 |
+
"delta": 0.11761140823364258,
|
| 44 |
+
"converging": true,
|
| 45 |
+
"seconds": 456.31761837005615
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"target_cos": 0.8,
|
| 49 |
+
"fid_init": 0.8000000715255737,
|
| 50 |
+
"fid_final": 0.9589360356330872,
|
| 51 |
+
"delta": 0.15893596410751343,
|
| 52 |
+
"converging": true,
|
| 53 |
+
"seconds": 455.580677986145
|
| 54 |
+
},
|
| 55 |
+
{
|
| 56 |
+
"target_cos": 0.7,
|
| 57 |
+
"fid_init": 0.7118338346481323,
|
| 58 |
+
"fid_final": 0.8864246010780334,
|
| 59 |
+
"delta": 0.17459076642990112,
|
| 60 |
+
"converging": true,
|
| 61 |
+
"seconds": 455.87197256088257
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"target_cos": 0.6,
|
| 65 |
+
"fid_init": 0.7552973031997681,
|
| 66 |
+
"fid_final": 0.866296648979187,
|
| 67 |
+
"delta": 0.11099934577941895,
|
| 68 |
+
"converging": true,
|
| 69 |
+
"seconds": 455.4593770503998
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"target_cos": 0.5,
|
| 73 |
+
"fid_init": 0.7880702614784241,
|
| 74 |
+
"fid_final": 0.8552908897399902,
|
| 75 |
+
"delta": 0.06722062826156616,
|
| 76 |
+
"converging": true,
|
| 77 |
+
"seconds": 455.7541358470917
|
| 78 |
+
},
|
| 79 |
+
{
|
| 80 |
+
"target_cos": 0.3,
|
| 81 |
+
"fid_init": 0.8184905052185059,
|
| 82 |
+
"fid_final": 0.8534693717956543,
|
| 83 |
+
"delta": 0.03497886657714844,
|
| 84 |
+
"converging": true,
|
| 85 |
+
"seconds": 456.12142276763916
|
| 86 |
+
},
|
| 87 |
+
{
|
| 88 |
+
"target_cos": 0.2,
|
| 89 |
+
"fid_init": 0.8192972540855408,
|
| 90 |
+
"fid_final": 0.8383699655532837,
|
| 91 |
+
"delta": 0.01907271146774292,
|
| 92 |
+
"converging": true,
|
| 93 |
+
"seconds": 455.7897198200226
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"target_cos": 0.1,
|
| 97 |
+
"fid_init": 0.8211536407470703,
|
| 98 |
+
"fid_final": 0.8282573223114014,
|
| 99 |
+
"delta": 0.007103681564331055,
|
| 100 |
+
"converging": true,
|
| 101 |
+
"seconds": 455.4574017524719
|
| 102 |
+
}
|
| 103 |
+
]
|
| 104 |
+
}
|
basin_radius/results_seed43.json
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"k": 48,
|
| 3 |
+
"n_steps": 500,
|
| 4 |
+
"n_samples": 2048000,
|
| 5 |
+
"seed": 43,
|
| 6 |
+
"results": [
|
| 7 |
+
{
|
| 8 |
+
"target_cos": 0.99,
|
| 9 |
+
"fid_init": 0.9900000095367432,
|
| 10 |
+
"fid_final": 0.9860057830810547,
|
| 11 |
+
"delta": -0.0039942264556884766,
|
| 12 |
+
"converging": false,
|
| 13 |
+
"seconds": 448.4635183811188
|
| 14 |
+
},
|
| 15 |
+
{
|
| 16 |
+
"target_cos": 0.97,
|
| 17 |
+
"fid_init": 0.9700000286102295,
|
| 18 |
+
"fid_final": 0.9846746921539307,
|
| 19 |
+
"delta": 0.014674663543701172,
|
| 20 |
+
"converging": true,
|
| 21 |
+
"seconds": 447.87573170661926
|
| 22 |
+
},
|
| 23 |
+
{
|
| 24 |
+
"target_cos": 0.95,
|
| 25 |
+
"fid_init": 0.9500000476837158,
|
| 26 |
+
"fid_final": 0.9829118251800537,
|
| 27 |
+
"delta": 0.03291177749633789,
|
| 28 |
+
"converging": true,
|
| 29 |
+
"seconds": 448.0157024860382
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"target_cos": 0.9,
|
| 33 |
+
"fid_init": 0.9000000953674316,
|
| 34 |
+
"fid_final": 0.9769527316093445,
|
| 35 |
+
"delta": 0.07695263624191284,
|
| 36 |
+
"converging": true,
|
| 37 |
+
"seconds": 448.6988363265991
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"target_cos": 0.85,
|
| 41 |
+
"fid_init": 0.8500000238418579,
|
| 42 |
+
"fid_final": 0.969549298286438,
|
| 43 |
+
"delta": 0.11954927444458008,
|
| 44 |
+
"converging": true,
|
| 45 |
+
"seconds": 448.3461797237396
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"target_cos": 0.8,
|
| 49 |
+
"fid_init": 0.8000000715255737,
|
| 50 |
+
"fid_final": 0.9612268209457397,
|
| 51 |
+
"delta": 0.16122674942016602,
|
| 52 |
+
"converging": true,
|
| 53 |
+
"seconds": 448.9183044433594
|
| 54 |
+
},
|
| 55 |
+
{
|
| 56 |
+
"target_cos": 0.7,
|
| 57 |
+
"fid_init": 0.7000000476837158,
|
| 58 |
+
"fid_final": 0.9416828155517578,
|
| 59 |
+
"delta": 0.241682767868042,
|
| 60 |
+
"converging": true,
|
| 61 |
+
"seconds": 449.0006408691406
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"target_cos": 0.6,
|
| 65 |
+
"fid_init": 0.6000000238418579,
|
| 66 |
+
"fid_final": 0.9099850654602051,
|
| 67 |
+
"delta": 0.30998504161834717,
|
| 68 |
+
"converging": true,
|
| 69 |
+
"seconds": 449.1651108264923
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"target_cos": 0.5,
|
| 73 |
+
"fid_init": 0.5,
|
| 74 |
+
"fid_final": 0.8409985303878784,
|
| 75 |
+
"delta": 0.3409985303878784,
|
| 76 |
+
"converging": true,
|
| 77 |
+
"seconds": 450.0478971004486
|
| 78 |
+
},
|
| 79 |
+
{
|
| 80 |
+
"target_cos": 0.3,
|
| 81 |
+
"fid_init": 0.30000001192092896,
|
| 82 |
+
"fid_final": 0.41315793991088867,
|
| 83 |
+
"delta": 0.11315792798995972,
|
| 84 |
+
"converging": true,
|
| 85 |
+
"seconds": 448.98216104507446
|
| 86 |
+
},
|
| 87 |
+
{
|
| 88 |
+
"target_cos": 0.2,
|
| 89 |
+
"fid_init": 0.20000237226486206,
|
| 90 |
+
"fid_final": 0.2386893630027771,
|
| 91 |
+
"delta": 0.03868699073791504,
|
| 92 |
+
"converging": true,
|
| 93 |
+
"seconds": 449.53068566322327
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"target_cos": 0.1,
|
| 97 |
+
"fid_init": 0.14860697090625763,
|
| 98 |
+
"fid_final": 0.21956323087215424,
|
| 99 |
+
"delta": 0.0709562599658966,
|
| 100 |
+
"converging": true,
|
| 101 |
+
"seconds": 450.1033709049225
|
| 102 |
+
}
|
| 103 |
+
]
|
| 104 |
+
}
|
basin_radius/results_seed44.json
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"k": 48,
|
| 3 |
+
"n_steps": 500,
|
| 4 |
+
"n_samples": 2048000,
|
| 5 |
+
"seed": 44,
|
| 6 |
+
"results": [
|
| 7 |
+
{
|
| 8 |
+
"target_cos": 0.99,
|
| 9 |
+
"fid_init": 0.9900000095367432,
|
| 10 |
+
"fid_final": 0.9861334562301636,
|
| 11 |
+
"delta": -0.00386655330657959,
|
| 12 |
+
"converging": false,
|
| 13 |
+
"seconds": 458.3089339733124
|
| 14 |
+
},
|
| 15 |
+
{
|
| 16 |
+
"target_cos": 0.97,
|
| 17 |
+
"fid_init": 0.9700000286102295,
|
| 18 |
+
"fid_final": 0.9848356246948242,
|
| 19 |
+
"delta": 0.014835596084594727,
|
| 20 |
+
"converging": true,
|
| 21 |
+
"seconds": 461.82796478271484
|
| 22 |
+
},
|
| 23 |
+
{
|
| 24 |
+
"target_cos": 0.95,
|
| 25 |
+
"fid_init": 0.9500000476837158,
|
| 26 |
+
"fid_final": 0.983006477355957,
|
| 27 |
+
"delta": 0.03300642967224121,
|
| 28 |
+
"converging": true,
|
| 29 |
+
"seconds": 461.8896379470825
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"target_cos": 0.9,
|
| 33 |
+
"fid_init": 0.9000000953674316,
|
| 34 |
+
"fid_final": 0.9769294261932373,
|
| 35 |
+
"delta": 0.07692933082580566,
|
| 36 |
+
"converging": true,
|
| 37 |
+
"seconds": 460.7876327037811
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"target_cos": 0.85,
|
| 41 |
+
"fid_init": 0.8500000238418579,
|
| 42 |
+
"fid_final": 0.9694505333900452,
|
| 43 |
+
"delta": 0.11945050954818726,
|
| 44 |
+
"converging": true,
|
| 45 |
+
"seconds": 461.25571632385254
|
| 46 |
+
},
|
| 47 |
+
{
|
| 48 |
+
"target_cos": 0.8,
|
| 49 |
+
"fid_init": 0.8000000715255737,
|
| 50 |
+
"fid_final": 0.9610999822616577,
|
| 51 |
+
"delta": 0.16109991073608398,
|
| 52 |
+
"converging": true,
|
| 53 |
+
"seconds": 465.52943754196167
|
| 54 |
+
},
|
| 55 |
+
{
|
| 56 |
+
"target_cos": 0.7,
|
| 57 |
+
"fid_init": 0.7000000476837158,
|
| 58 |
+
"fid_final": 0.9415371417999268,
|
| 59 |
+
"delta": 0.24153709411621094,
|
| 60 |
+
"converging": true,
|
| 61 |
+
"seconds": 461.8532736301422
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"target_cos": 0.6,
|
| 65 |
+
"fid_init": 0.6000000238418579,
|
| 66 |
+
"fid_final": 0.9099147319793701,
|
| 67 |
+
"delta": 0.3099147081375122,
|
| 68 |
+
"converging": true,
|
| 69 |
+
"seconds": 463.27074575424194
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"target_cos": 0.5,
|
| 73 |
+
"fid_init": 0.5,
|
| 74 |
+
"fid_final": 0.8409147262573242,
|
| 75 |
+
"delta": 0.3409147262573242,
|
| 76 |
+
"converging": true,
|
| 77 |
+
"seconds": 464.4487407207489
|
| 78 |
+
},
|
| 79 |
+
{
|
| 80 |
+
"target_cos": 0.3,
|
| 81 |
+
"fid_init": 0.30000001192092896,
|
| 82 |
+
"fid_final": 0.4143797755241394,
|
| 83 |
+
"delta": 0.11437976360321045,
|
| 84 |
+
"converging": true,
|
| 85 |
+
"seconds": 463.4620110988617
|
| 86 |
+
},
|
| 87 |
+
{
|
| 88 |
+
"target_cos": 0.2,
|
| 89 |
+
"fid_init": 0.20000238716602325,
|
| 90 |
+
"fid_final": 0.24059367179870605,
|
| 91 |
+
"delta": 0.0405912846326828,
|
| 92 |
+
"converging": true,
|
| 93 |
+
"seconds": 457.25708651542664
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"target_cos": 0.1,
|
| 97 |
+
"fid_init": 0.14846540987491608,
|
| 98 |
+
"fid_final": 0.22045353055000305,
|
| 99 |
+
"delta": 0.07198812067508698,
|
| 100 |
+
"converging": true,
|
| 101 |
+
"seconds": 459.9436514377594
|
| 102 |
+
}
|
| 103 |
+
]
|
| 104 |
+
}
|
frozen_dec_k48_s1/results.json
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"k": 48,
|
| 3 |
+
"n_samples": 15000000,
|
| 4 |
+
"lr": 0.0003,
|
| 5 |
+
"seed": 1,
|
| 6 |
+
"support_overlap_init": {
|
| 7 |
+
"jaccard": 0.4592665173012477,
|
| 8 |
+
"recall": 0.6161499023437506
|
| 9 |
+
},
|
| 10 |
+
"support_overlap_final": {
|
| 11 |
+
"jaccard": 0.5475213736319267,
|
| 12 |
+
"recall": 0.6813151041666667
|
| 13 |
+
},
|
| 14 |
+
"dec_fidelity_init": 1.0,
|
| 15 |
+
"dec_fidelity_final": 1.0,
|
| 16 |
+
"eval": {
|
| 17 |
+
"true_l0": 34.46586,
|
| 18 |
+
"sae_l0": 47.98879,
|
| 19 |
+
"dead_latents": 0,
|
| 20 |
+
"shrinkage": 0.9727815844726563,
|
| 21 |
+
"explained_variance": 0.9287085583622611,
|
| 22 |
+
"mcc": 1.0,
|
| 23 |
+
"uniqueness": 1.0
|
| 24 |
+
},
|
| 25 |
+
"seconds": 160.0245246887207
|
| 26 |
+
}
|
oracle_init_smoke/cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"normalize_activations": "none", "dtype": "float32", "device": "cuda", "d_in": 768, "rescale_acts_by_decoder_norm": true, "apply_b_dec_to_input": true, "d_sae": 16384, "k": 48, "metadata": {"sae_lens_version": "6.43.0", "sae_lens_training_version": "6.43.0"}, "reshape_activations": "none", "architecture": "topk"}
|
oracle_init_smoke/ckpts/114688/activation_scaler.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"scaling_factor": 0.9888814189820523}
|
oracle_init_smoke/ckpts/114688/cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"d_in": 768, "d_sae": 16384, "dtype": "float32", "device": "cuda", "apply_b_dec_to_input": true, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.43.0", "sae_lens_training_version": "6.43.0"}, "decoder_init_norm": 0.1, "k": 48, "use_sparse_activations": false, "aux_loss_coefficient": 1.0, "rescale_acts_by_decoder_norm": true, "architecture": "topk"}
|
oracle_init_smoke/ckpts/114688/runner_config.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"synthetic_model": "decoderesearch/synth-sae-bench-16k-v1",
|
| 3 |
+
"sae": {
|
| 4 |
+
"d_in": 768,
|
| 5 |
+
"d_sae": 16384,
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"device": "cuda",
|
| 8 |
+
"apply_b_dec_to_input": true,
|
| 9 |
+
"normalize_activations": "expected_average_only_in",
|
| 10 |
+
"reshape_activations": "none",
|
| 11 |
+
"metadata": {
|
| 12 |
+
"sae_lens_version": "6.43.0",
|
| 13 |
+
"sae_lens_training_version": "6.43.0"
|
| 14 |
+
},
|
| 15 |
+
"decoder_init_norm": 0.1,
|
| 16 |
+
"k": 48,
|
| 17 |
+
"use_sparse_activations": false,
|
| 18 |
+
"aux_loss_coefficient": 1.0,
|
| 19 |
+
"rescale_acts_by_decoder_norm": true,
|
| 20 |
+
"architecture": "topk"
|
| 21 |
+
},
|
| 22 |
+
"training_samples": 1000000,
|
| 23 |
+
"batch_size": 4096,
|
| 24 |
+
"lr": 0.0003,
|
| 25 |
+
"lr_warm_up_steps": 0,
|
| 26 |
+
"lr_decay_steps": 0,
|
| 27 |
+
"lr_scheduler_name": "constant",
|
| 28 |
+
"lr_end": 2.9999999999999997e-05,
|
| 29 |
+
"adam_beta1": 0.9,
|
| 30 |
+
"adam_beta2": 0.999,
|
| 31 |
+
"n_restart_cycles": 1,
|
| 32 |
+
"device": "cuda",
|
| 33 |
+
"autocast_sae": false,
|
| 34 |
+
"autocast_data": false,
|
| 35 |
+
"n_checkpoints": 8,
|
| 36 |
+
"checkpoint_path": "runs/oracle_init_smoke/ckpts",
|
| 37 |
+
"save_final_checkpoint": true,
|
| 38 |
+
"output_path": "runs/oracle_init_smoke",
|
| 39 |
+
"save_synthetic_model": false,
|
| 40 |
+
"eval_frequency": 200000,
|
| 41 |
+
"eval_samples": 200000,
|
| 42 |
+
"run_final_eval": true,
|
| 43 |
+
"dead_feature_window": 1000,
|
| 44 |
+
"feature_sampling_window": 2000,
|
| 45 |
+
"n_batches_for_norm_estimate": 1000,
|
| 46 |
+
"sae_lens_version": "6.43.0",
|
| 47 |
+
"logger": {
|
| 48 |
+
"log_to_wandb": true,
|
| 49 |
+
"log_activations_store_to_wandb": false,
|
| 50 |
+
"log_optimizer_state_to_wandb": false,
|
| 51 |
+
"log_weights_to_wandb": true,
|
| 52 |
+
"wandb_project": "sae_lens_training",
|
| 53 |
+
"wandb_id": null,
|
| 54 |
+
"run_name": "synthetic-topk-16384-LR-0.0003",
|
| 55 |
+
"wandb_entity": null,
|
| 56 |
+
"wandb_log_frequency": 10,
|
| 57 |
+
"eval_every_n_wandb_logs": 100
|
| 58 |
+
}
|
| 59 |
+
}
|
oracle_init_smoke/ckpts/225280/activation_scaler.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"scaling_factor": 0.9888814189820523}
|
oracle_init_smoke/ckpts/225280/cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"d_in": 768, "d_sae": 16384, "dtype": "float32", "device": "cuda", "apply_b_dec_to_input": true, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.43.0", "sae_lens_training_version": "6.43.0"}, "decoder_init_norm": 0.1, "k": 48, "use_sparse_activations": false, "aux_loss_coefficient": 1.0, "rescale_acts_by_decoder_norm": true, "architecture": "topk"}
|
oracle_init_smoke/ckpts/225280/runner_config.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"synthetic_model": "decoderesearch/synth-sae-bench-16k-v1",
|
| 3 |
+
"sae": {
|
| 4 |
+
"d_in": 768,
|
| 5 |
+
"d_sae": 16384,
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"device": "cuda",
|
| 8 |
+
"apply_b_dec_to_input": true,
|
| 9 |
+
"normalize_activations": "expected_average_only_in",
|
| 10 |
+
"reshape_activations": "none",
|
| 11 |
+
"metadata": {
|
| 12 |
+
"sae_lens_version": "6.43.0",
|
| 13 |
+
"sae_lens_training_version": "6.43.0"
|
| 14 |
+
},
|
| 15 |
+
"decoder_init_norm": 0.1,
|
| 16 |
+
"k": 48,
|
| 17 |
+
"use_sparse_activations": false,
|
| 18 |
+
"aux_loss_coefficient": 1.0,
|
| 19 |
+
"rescale_acts_by_decoder_norm": true,
|
| 20 |
+
"architecture": "topk"
|
| 21 |
+
},
|
| 22 |
+
"training_samples": 1000000,
|
| 23 |
+
"batch_size": 4096,
|
| 24 |
+
"lr": 0.0003,
|
| 25 |
+
"lr_warm_up_steps": 0,
|
| 26 |
+
"lr_decay_steps": 0,
|
| 27 |
+
"lr_scheduler_name": "constant",
|
| 28 |
+
"lr_end": 2.9999999999999997e-05,
|
| 29 |
+
"adam_beta1": 0.9,
|
| 30 |
+
"adam_beta2": 0.999,
|
| 31 |
+
"n_restart_cycles": 1,
|
| 32 |
+
"device": "cuda",
|
| 33 |
+
"autocast_sae": false,
|
| 34 |
+
"autocast_data": false,
|
| 35 |
+
"n_checkpoints": 8,
|
| 36 |
+
"checkpoint_path": "runs/oracle_init_smoke/ckpts",
|
| 37 |
+
"save_final_checkpoint": true,
|
| 38 |
+
"output_path": "runs/oracle_init_smoke",
|
| 39 |
+
"save_synthetic_model": false,
|
| 40 |
+
"eval_frequency": 200000,
|
| 41 |
+
"eval_samples": 200000,
|
| 42 |
+
"run_final_eval": true,
|
| 43 |
+
"dead_feature_window": 1000,
|
| 44 |
+
"feature_sampling_window": 2000,
|
| 45 |
+
"n_batches_for_norm_estimate": 1000,
|
| 46 |
+
"sae_lens_version": "6.43.0",
|
| 47 |
+
"logger": {
|
| 48 |
+
"log_to_wandb": true,
|
| 49 |
+
"log_activations_store_to_wandb": false,
|
| 50 |
+
"log_optimizer_state_to_wandb": false,
|
| 51 |
+
"log_weights_to_wandb": true,
|
| 52 |
+
"wandb_project": "sae_lens_training",
|
| 53 |
+
"wandb_id": null,
|
| 54 |
+
"run_name": "synthetic-topk-16384-LR-0.0003",
|
| 55 |
+
"wandb_entity": null,
|
| 56 |
+
"wandb_log_frequency": 10,
|
| 57 |
+
"eval_every_n_wandb_logs": 100
|
| 58 |
+
}
|
| 59 |
+
}
|
oracle_init_smoke/ckpts/335872/activation_scaler.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"scaling_factor": 0.9888814189820523}
|
oracle_init_smoke/ckpts/335872/cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"d_in": 768, "d_sae": 16384, "dtype": "float32", "device": "cuda", "apply_b_dec_to_input": true, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.43.0", "sae_lens_training_version": "6.43.0"}, "decoder_init_norm": 0.1, "k": 48, "use_sparse_activations": false, "aux_loss_coefficient": 1.0, "rescale_acts_by_decoder_norm": true, "architecture": "topk"}
|
oracle_init_smoke/ckpts/335872/runner_config.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"synthetic_model": "decoderesearch/synth-sae-bench-16k-v1",
|
| 3 |
+
"sae": {
|
| 4 |
+
"d_in": 768,
|
| 5 |
+
"d_sae": 16384,
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"device": "cuda",
|
| 8 |
+
"apply_b_dec_to_input": true,
|
| 9 |
+
"normalize_activations": "expected_average_only_in",
|
| 10 |
+
"reshape_activations": "none",
|
| 11 |
+
"metadata": {
|
| 12 |
+
"sae_lens_version": "6.43.0",
|
| 13 |
+
"sae_lens_training_version": "6.43.0"
|
| 14 |
+
},
|
| 15 |
+
"decoder_init_norm": 0.1,
|
| 16 |
+
"k": 48,
|
| 17 |
+
"use_sparse_activations": false,
|
| 18 |
+
"aux_loss_coefficient": 1.0,
|
| 19 |
+
"rescale_acts_by_decoder_norm": true,
|
| 20 |
+
"architecture": "topk"
|
| 21 |
+
},
|
| 22 |
+
"training_samples": 1000000,
|
| 23 |
+
"batch_size": 4096,
|
| 24 |
+
"lr": 0.0003,
|
| 25 |
+
"lr_warm_up_steps": 0,
|
| 26 |
+
"lr_decay_steps": 0,
|
| 27 |
+
"lr_scheduler_name": "constant",
|
| 28 |
+
"lr_end": 2.9999999999999997e-05,
|
| 29 |
+
"adam_beta1": 0.9,
|
| 30 |
+
"adam_beta2": 0.999,
|
| 31 |
+
"n_restart_cycles": 1,
|
| 32 |
+
"device": "cuda",
|
| 33 |
+
"autocast_sae": false,
|
| 34 |
+
"autocast_data": false,
|
| 35 |
+
"n_checkpoints": 8,
|
| 36 |
+
"checkpoint_path": "runs/oracle_init_smoke/ckpts",
|
| 37 |
+
"save_final_checkpoint": true,
|
| 38 |
+
"output_path": "runs/oracle_init_smoke",
|
| 39 |
+
"save_synthetic_model": false,
|
| 40 |
+
"eval_frequency": 200000,
|
| 41 |
+
"eval_samples": 200000,
|
| 42 |
+
"run_final_eval": true,
|
| 43 |
+
"dead_feature_window": 1000,
|
| 44 |
+
"feature_sampling_window": 2000,
|
| 45 |
+
"n_batches_for_norm_estimate": 1000,
|
| 46 |
+
"sae_lens_version": "6.43.0",
|
| 47 |
+
"logger": {
|
| 48 |
+
"log_to_wandb": true,
|
| 49 |
+
"log_activations_store_to_wandb": false,
|
| 50 |
+
"log_optimizer_state_to_wandb": false,
|
| 51 |
+
"log_weights_to_wandb": true,
|
| 52 |
+
"wandb_project": "sae_lens_training",
|
| 53 |
+
"wandb_id": null,
|
| 54 |
+
"run_name": "synthetic-topk-16384-LR-0.0003",
|
| 55 |
+
"wandb_entity": null,
|
| 56 |
+
"wandb_log_frequency": 10,
|
| 57 |
+
"eval_every_n_wandb_logs": 100
|
| 58 |
+
}
|
| 59 |
+
}
|
oracle_init_smoke/ckpts/446464/activation_scaler.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"scaling_factor": 0.9888814189820523}
|
oracle_init_smoke/ckpts/446464/cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"d_in": 768, "d_sae": 16384, "dtype": "float32", "device": "cuda", "apply_b_dec_to_input": true, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.43.0", "sae_lens_training_version": "6.43.0"}, "decoder_init_norm": 0.1, "k": 48, "use_sparse_activations": false, "aux_loss_coefficient": 1.0, "rescale_acts_by_decoder_norm": true, "architecture": "topk"}
|
oracle_init_smoke/ckpts/446464/runner_config.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"synthetic_model": "decoderesearch/synth-sae-bench-16k-v1",
|
| 3 |
+
"sae": {
|
| 4 |
+
"d_in": 768,
|
| 5 |
+
"d_sae": 16384,
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"device": "cuda",
|
| 8 |
+
"apply_b_dec_to_input": true,
|
| 9 |
+
"normalize_activations": "expected_average_only_in",
|
| 10 |
+
"reshape_activations": "none",
|
| 11 |
+
"metadata": {
|
| 12 |
+
"sae_lens_version": "6.43.0",
|
| 13 |
+
"sae_lens_training_version": "6.43.0"
|
| 14 |
+
},
|
| 15 |
+
"decoder_init_norm": 0.1,
|
| 16 |
+
"k": 48,
|
| 17 |
+
"use_sparse_activations": false,
|
| 18 |
+
"aux_loss_coefficient": 1.0,
|
| 19 |
+
"rescale_acts_by_decoder_norm": true,
|
| 20 |
+
"architecture": "topk"
|
| 21 |
+
},
|
| 22 |
+
"training_samples": 1000000,
|
| 23 |
+
"batch_size": 4096,
|
| 24 |
+
"lr": 0.0003,
|
| 25 |
+
"lr_warm_up_steps": 0,
|
| 26 |
+
"lr_decay_steps": 0,
|
| 27 |
+
"lr_scheduler_name": "constant",
|
| 28 |
+
"lr_end": 2.9999999999999997e-05,
|
| 29 |
+
"adam_beta1": 0.9,
|
| 30 |
+
"adam_beta2": 0.999,
|
| 31 |
+
"n_restart_cycles": 1,
|
| 32 |
+
"device": "cuda",
|
| 33 |
+
"autocast_sae": false,
|
| 34 |
+
"autocast_data": false,
|
| 35 |
+
"n_checkpoints": 8,
|
| 36 |
+
"checkpoint_path": "runs/oracle_init_smoke/ckpts",
|
| 37 |
+
"save_final_checkpoint": true,
|
| 38 |
+
"output_path": "runs/oracle_init_smoke",
|
| 39 |
+
"save_synthetic_model": false,
|
| 40 |
+
"eval_frequency": 200000,
|
| 41 |
+
"eval_samples": 200000,
|
| 42 |
+
"run_final_eval": true,
|
| 43 |
+
"dead_feature_window": 1000,
|
| 44 |
+
"feature_sampling_window": 2000,
|
| 45 |
+
"n_batches_for_norm_estimate": 1000,
|
| 46 |
+
"sae_lens_version": "6.43.0",
|
| 47 |
+
"logger": {
|
| 48 |
+
"log_to_wandb": true,
|
| 49 |
+
"log_activations_store_to_wandb": false,
|
| 50 |
+
"log_optimizer_state_to_wandb": false,
|
| 51 |
+
"log_weights_to_wandb": true,
|
| 52 |
+
"wandb_project": "sae_lens_training",
|
| 53 |
+
"wandb_id": null,
|
| 54 |
+
"run_name": "synthetic-topk-16384-LR-0.0003",
|
| 55 |
+
"wandb_entity": null,
|
| 56 |
+
"wandb_log_frequency": 10,
|
| 57 |
+
"eval_every_n_wandb_logs": 100
|
| 58 |
+
}
|
| 59 |
+
}
|
oracle_init_smoke/ckpts/557056/activation_scaler.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"scaling_factor": 0.9888814189820523}
|
oracle_init_smoke/ckpts/557056/cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"d_in": 768, "d_sae": 16384, "dtype": "float32", "device": "cuda", "apply_b_dec_to_input": true, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.43.0", "sae_lens_training_version": "6.43.0"}, "decoder_init_norm": 0.1, "k": 48, "use_sparse_activations": false, "aux_loss_coefficient": 1.0, "rescale_acts_by_decoder_norm": true, "architecture": "topk"}
|
oracle_init_smoke/ckpts/557056/runner_config.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"synthetic_model": "decoderesearch/synth-sae-bench-16k-v1",
|
| 3 |
+
"sae": {
|
| 4 |
+
"d_in": 768,
|
| 5 |
+
"d_sae": 16384,
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"device": "cuda",
|
| 8 |
+
"apply_b_dec_to_input": true,
|
| 9 |
+
"normalize_activations": "expected_average_only_in",
|
| 10 |
+
"reshape_activations": "none",
|
| 11 |
+
"metadata": {
|
| 12 |
+
"sae_lens_version": "6.43.0",
|
| 13 |
+
"sae_lens_training_version": "6.43.0"
|
| 14 |
+
},
|
| 15 |
+
"decoder_init_norm": 0.1,
|
| 16 |
+
"k": 48,
|
| 17 |
+
"use_sparse_activations": false,
|
| 18 |
+
"aux_loss_coefficient": 1.0,
|
| 19 |
+
"rescale_acts_by_decoder_norm": true,
|
| 20 |
+
"architecture": "topk"
|
| 21 |
+
},
|
| 22 |
+
"training_samples": 1000000,
|
| 23 |
+
"batch_size": 4096,
|
| 24 |
+
"lr": 0.0003,
|
| 25 |
+
"lr_warm_up_steps": 0,
|
| 26 |
+
"lr_decay_steps": 0,
|
| 27 |
+
"lr_scheduler_name": "constant",
|
| 28 |
+
"lr_end": 2.9999999999999997e-05,
|
| 29 |
+
"adam_beta1": 0.9,
|
| 30 |
+
"adam_beta2": 0.999,
|
| 31 |
+
"n_restart_cycles": 1,
|
| 32 |
+
"device": "cuda",
|
| 33 |
+
"autocast_sae": false,
|
| 34 |
+
"autocast_data": false,
|
| 35 |
+
"n_checkpoints": 8,
|
| 36 |
+
"checkpoint_path": "runs/oracle_init_smoke/ckpts",
|
| 37 |
+
"save_final_checkpoint": true,
|
| 38 |
+
"output_path": "runs/oracle_init_smoke",
|
| 39 |
+
"save_synthetic_model": false,
|
| 40 |
+
"eval_frequency": 200000,
|
| 41 |
+
"eval_samples": 200000,
|
| 42 |
+
"run_final_eval": true,
|
| 43 |
+
"dead_feature_window": 1000,
|
| 44 |
+
"feature_sampling_window": 2000,
|
| 45 |
+
"n_batches_for_norm_estimate": 1000,
|
| 46 |
+
"sae_lens_version": "6.43.0",
|
| 47 |
+
"logger": {
|
| 48 |
+
"log_to_wandb": true,
|
| 49 |
+
"log_activations_store_to_wandb": false,
|
| 50 |
+
"log_optimizer_state_to_wandb": false,
|
| 51 |
+
"log_weights_to_wandb": true,
|
| 52 |
+
"wandb_project": "sae_lens_training",
|
| 53 |
+
"wandb_id": null,
|
| 54 |
+
"run_name": "synthetic-topk-16384-LR-0.0003",
|
| 55 |
+
"wandb_entity": null,
|
| 56 |
+
"wandb_log_frequency": 10,
|
| 57 |
+
"eval_every_n_wandb_logs": 100
|
| 58 |
+
}
|
| 59 |
+
}
|
oracle_init_smoke/ckpts/667648/activation_scaler.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"scaling_factor": 0.9888814189820523}
|
oracle_init_smoke/ckpts/667648/cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"d_in": 768, "d_sae": 16384, "dtype": "float32", "device": "cuda", "apply_b_dec_to_input": true, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.43.0", "sae_lens_training_version": "6.43.0"}, "decoder_init_norm": 0.1, "k": 48, "use_sparse_activations": false, "aux_loss_coefficient": 1.0, "rescale_acts_by_decoder_norm": true, "architecture": "topk"}
|
oracle_init_smoke/ckpts/667648/runner_config.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"synthetic_model": "decoderesearch/synth-sae-bench-16k-v1",
|
| 3 |
+
"sae": {
|
| 4 |
+
"d_in": 768,
|
| 5 |
+
"d_sae": 16384,
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"device": "cuda",
|
| 8 |
+
"apply_b_dec_to_input": true,
|
| 9 |
+
"normalize_activations": "expected_average_only_in",
|
| 10 |
+
"reshape_activations": "none",
|
| 11 |
+
"metadata": {
|
| 12 |
+
"sae_lens_version": "6.43.0",
|
| 13 |
+
"sae_lens_training_version": "6.43.0"
|
| 14 |
+
},
|
| 15 |
+
"decoder_init_norm": 0.1,
|
| 16 |
+
"k": 48,
|
| 17 |
+
"use_sparse_activations": false,
|
| 18 |
+
"aux_loss_coefficient": 1.0,
|
| 19 |
+
"rescale_acts_by_decoder_norm": true,
|
| 20 |
+
"architecture": "topk"
|
| 21 |
+
},
|
| 22 |
+
"training_samples": 1000000,
|
| 23 |
+
"batch_size": 4096,
|
| 24 |
+
"lr": 0.0003,
|
| 25 |
+
"lr_warm_up_steps": 0,
|
| 26 |
+
"lr_decay_steps": 0,
|
| 27 |
+
"lr_scheduler_name": "constant",
|
| 28 |
+
"lr_end": 2.9999999999999997e-05,
|
| 29 |
+
"adam_beta1": 0.9,
|
| 30 |
+
"adam_beta2": 0.999,
|
| 31 |
+
"n_restart_cycles": 1,
|
| 32 |
+
"device": "cuda",
|
| 33 |
+
"autocast_sae": false,
|
| 34 |
+
"autocast_data": false,
|
| 35 |
+
"n_checkpoints": 8,
|
| 36 |
+
"checkpoint_path": "runs/oracle_init_smoke/ckpts",
|
| 37 |
+
"save_final_checkpoint": true,
|
| 38 |
+
"output_path": "runs/oracle_init_smoke",
|
| 39 |
+
"save_synthetic_model": false,
|
| 40 |
+
"eval_frequency": 200000,
|
| 41 |
+
"eval_samples": 200000,
|
| 42 |
+
"run_final_eval": true,
|
| 43 |
+
"dead_feature_window": 1000,
|
| 44 |
+
"feature_sampling_window": 2000,
|
| 45 |
+
"n_batches_for_norm_estimate": 1000,
|
| 46 |
+
"sae_lens_version": "6.43.0",
|
| 47 |
+
"logger": {
|
| 48 |
+
"log_to_wandb": true,
|
| 49 |
+
"log_activations_store_to_wandb": false,
|
| 50 |
+
"log_optimizer_state_to_wandb": false,
|
| 51 |
+
"log_weights_to_wandb": true,
|
| 52 |
+
"wandb_project": "sae_lens_training",
|
| 53 |
+
"wandb_id": null,
|
| 54 |
+
"run_name": "synthetic-topk-16384-LR-0.0003",
|
| 55 |
+
"wandb_entity": null,
|
| 56 |
+
"wandb_log_frequency": 10,
|
| 57 |
+
"eval_every_n_wandb_logs": 100
|
| 58 |
+
}
|
| 59 |
+
}
|
oracle_init_smoke/ckpts/778240/activation_scaler.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"scaling_factor": 0.9888814189820523}
|
oracle_init_smoke/ckpts/778240/cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"d_in": 768, "d_sae": 16384, "dtype": "float32", "device": "cuda", "apply_b_dec_to_input": true, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.43.0", "sae_lens_training_version": "6.43.0"}, "decoder_init_norm": 0.1, "k": 48, "use_sparse_activations": false, "aux_loss_coefficient": 1.0, "rescale_acts_by_decoder_norm": true, "architecture": "topk"}
|
oracle_init_smoke/ckpts/778240/runner_config.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"synthetic_model": "decoderesearch/synth-sae-bench-16k-v1",
|
| 3 |
+
"sae": {
|
| 4 |
+
"d_in": 768,
|
| 5 |
+
"d_sae": 16384,
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"device": "cuda",
|
| 8 |
+
"apply_b_dec_to_input": true,
|
| 9 |
+
"normalize_activations": "expected_average_only_in",
|
| 10 |
+
"reshape_activations": "none",
|
| 11 |
+
"metadata": {
|
| 12 |
+
"sae_lens_version": "6.43.0",
|
| 13 |
+
"sae_lens_training_version": "6.43.0"
|
| 14 |
+
},
|
| 15 |
+
"decoder_init_norm": 0.1,
|
| 16 |
+
"k": 48,
|
| 17 |
+
"use_sparse_activations": false,
|
| 18 |
+
"aux_loss_coefficient": 1.0,
|
| 19 |
+
"rescale_acts_by_decoder_norm": true,
|
| 20 |
+
"architecture": "topk"
|
| 21 |
+
},
|
| 22 |
+
"training_samples": 1000000,
|
| 23 |
+
"batch_size": 4096,
|
| 24 |
+
"lr": 0.0003,
|
| 25 |
+
"lr_warm_up_steps": 0,
|
| 26 |
+
"lr_decay_steps": 0,
|
| 27 |
+
"lr_scheduler_name": "constant",
|
| 28 |
+
"lr_end": 2.9999999999999997e-05,
|
| 29 |
+
"adam_beta1": 0.9,
|
| 30 |
+
"adam_beta2": 0.999,
|
| 31 |
+
"n_restart_cycles": 1,
|
| 32 |
+
"device": "cuda",
|
| 33 |
+
"autocast_sae": false,
|
| 34 |
+
"autocast_data": false,
|
| 35 |
+
"n_checkpoints": 8,
|
| 36 |
+
"checkpoint_path": "runs/oracle_init_smoke/ckpts",
|
| 37 |
+
"save_final_checkpoint": true,
|
| 38 |
+
"output_path": "runs/oracle_init_smoke",
|
| 39 |
+
"save_synthetic_model": false,
|
| 40 |
+
"eval_frequency": 200000,
|
| 41 |
+
"eval_samples": 200000,
|
| 42 |
+
"run_final_eval": true,
|
| 43 |
+
"dead_feature_window": 1000,
|
| 44 |
+
"feature_sampling_window": 2000,
|
| 45 |
+
"n_batches_for_norm_estimate": 1000,
|
| 46 |
+
"sae_lens_version": "6.43.0",
|
| 47 |
+
"logger": {
|
| 48 |
+
"log_to_wandb": true,
|
| 49 |
+
"log_activations_store_to_wandb": false,
|
| 50 |
+
"log_optimizer_state_to_wandb": false,
|
| 51 |
+
"log_weights_to_wandb": true,
|
| 52 |
+
"wandb_project": "sae_lens_training",
|
| 53 |
+
"wandb_id": null,
|
| 54 |
+
"run_name": "synthetic-topk-16384-LR-0.0003",
|
| 55 |
+
"wandb_entity": null,
|
| 56 |
+
"wandb_log_frequency": 10,
|
| 57 |
+
"eval_every_n_wandb_logs": 100
|
| 58 |
+
}
|
| 59 |
+
}
|
oracle_init_smoke/ckpts/892928/activation_scaler.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"scaling_factor": 0.9888814189820523}
|
oracle_init_smoke/ckpts/892928/cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"d_in": 768, "d_sae": 16384, "dtype": "float32", "device": "cuda", "apply_b_dec_to_input": true, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.43.0", "sae_lens_training_version": "6.43.0"}, "decoder_init_norm": 0.1, "k": 48, "use_sparse_activations": false, "aux_loss_coefficient": 1.0, "rescale_acts_by_decoder_norm": true, "architecture": "topk"}
|
oracle_init_smoke/ckpts/892928/runner_config.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"synthetic_model": "decoderesearch/synth-sae-bench-16k-v1",
|
| 3 |
+
"sae": {
|
| 4 |
+
"d_in": 768,
|
| 5 |
+
"d_sae": 16384,
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"device": "cuda",
|
| 8 |
+
"apply_b_dec_to_input": true,
|
| 9 |
+
"normalize_activations": "expected_average_only_in",
|
| 10 |
+
"reshape_activations": "none",
|
| 11 |
+
"metadata": {
|
| 12 |
+
"sae_lens_version": "6.43.0",
|
| 13 |
+
"sae_lens_training_version": "6.43.0"
|
| 14 |
+
},
|
| 15 |
+
"decoder_init_norm": 0.1,
|
| 16 |
+
"k": 48,
|
| 17 |
+
"use_sparse_activations": false,
|
| 18 |
+
"aux_loss_coefficient": 1.0,
|
| 19 |
+
"rescale_acts_by_decoder_norm": true,
|
| 20 |
+
"architecture": "topk"
|
| 21 |
+
},
|
| 22 |
+
"training_samples": 1000000,
|
| 23 |
+
"batch_size": 4096,
|
| 24 |
+
"lr": 0.0003,
|
| 25 |
+
"lr_warm_up_steps": 0,
|
| 26 |
+
"lr_decay_steps": 0,
|
| 27 |
+
"lr_scheduler_name": "constant",
|
| 28 |
+
"lr_end": 2.9999999999999997e-05,
|
| 29 |
+
"adam_beta1": 0.9,
|
| 30 |
+
"adam_beta2": 0.999,
|
| 31 |
+
"n_restart_cycles": 1,
|
| 32 |
+
"device": "cuda",
|
| 33 |
+
"autocast_sae": false,
|
| 34 |
+
"autocast_data": false,
|
| 35 |
+
"n_checkpoints": 8,
|
| 36 |
+
"checkpoint_path": "runs/oracle_init_smoke/ckpts",
|
| 37 |
+
"save_final_checkpoint": true,
|
| 38 |
+
"output_path": "runs/oracle_init_smoke",
|
| 39 |
+
"save_synthetic_model": false,
|
| 40 |
+
"eval_frequency": 200000,
|
| 41 |
+
"eval_samples": 200000,
|
| 42 |
+
"run_final_eval": true,
|
| 43 |
+
"dead_feature_window": 1000,
|
| 44 |
+
"feature_sampling_window": 2000,
|
| 45 |
+
"n_batches_for_norm_estimate": 1000,
|
| 46 |
+
"sae_lens_version": "6.43.0",
|
| 47 |
+
"logger": {
|
| 48 |
+
"log_to_wandb": true,
|
| 49 |
+
"log_activations_store_to_wandb": false,
|
| 50 |
+
"log_optimizer_state_to_wandb": false,
|
| 51 |
+
"log_weights_to_wandb": true,
|
| 52 |
+
"wandb_project": "sae_lens_training",
|
| 53 |
+
"wandb_id": null,
|
| 54 |
+
"run_name": "synthetic-topk-16384-LR-0.0003",
|
| 55 |
+
"wandb_entity": null,
|
| 56 |
+
"wandb_log_frequency": 10,
|
| 57 |
+
"eval_every_n_wandb_logs": 100
|
| 58 |
+
}
|
| 59 |
+
}
|
oracle_init_smoke/ckpts/final_1003520/activation_scaler.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"scaling_factor": null}
|
oracle_init_smoke/ckpts/final_1003520/cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"d_in": 768, "d_sae": 16384, "dtype": "float32", "device": "cuda", "apply_b_dec_to_input": true, "normalize_activations": "none", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.43.0", "sae_lens_training_version": "6.43.0"}, "decoder_init_norm": 0.1, "k": 48, "use_sparse_activations": false, "aux_loss_coefficient": 1.0, "rescale_acts_by_decoder_norm": true, "architecture": "topk"}
|
oracle_init_smoke/ckpts/final_1003520/runner_config.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"synthetic_model": "decoderesearch/synth-sae-bench-16k-v1",
|
| 3 |
+
"sae": {
|
| 4 |
+
"d_in": 768,
|
| 5 |
+
"d_sae": 16384,
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"device": "cuda",
|
| 8 |
+
"apply_b_dec_to_input": true,
|
| 9 |
+
"normalize_activations": "none",
|
| 10 |
+
"reshape_activations": "none",
|
| 11 |
+
"metadata": {
|
| 12 |
+
"sae_lens_version": "6.43.0",
|
| 13 |
+
"sae_lens_training_version": "6.43.0"
|
| 14 |
+
},
|
| 15 |
+
"decoder_init_norm": 0.1,
|
| 16 |
+
"k": 48,
|
| 17 |
+
"use_sparse_activations": false,
|
| 18 |
+
"aux_loss_coefficient": 1.0,
|
| 19 |
+
"rescale_acts_by_decoder_norm": true,
|
| 20 |
+
"architecture": "topk"
|
| 21 |
+
},
|
| 22 |
+
"training_samples": 1000000,
|
| 23 |
+
"batch_size": 4096,
|
| 24 |
+
"lr": 0.0003,
|
| 25 |
+
"lr_warm_up_steps": 0,
|
| 26 |
+
"lr_decay_steps": 0,
|
| 27 |
+
"lr_scheduler_name": "constant",
|
| 28 |
+
"lr_end": 2.9999999999999997e-05,
|
| 29 |
+
"adam_beta1": 0.9,
|
| 30 |
+
"adam_beta2": 0.999,
|
| 31 |
+
"n_restart_cycles": 1,
|
| 32 |
+
"device": "cuda",
|
| 33 |
+
"autocast_sae": false,
|
| 34 |
+
"autocast_data": false,
|
| 35 |
+
"n_checkpoints": 8,
|
| 36 |
+
"checkpoint_path": "runs/oracle_init_smoke/ckpts",
|
| 37 |
+
"save_final_checkpoint": true,
|
| 38 |
+
"output_path": "runs/oracle_init_smoke",
|
| 39 |
+
"save_synthetic_model": false,
|
| 40 |
+
"eval_frequency": 200000,
|
| 41 |
+
"eval_samples": 200000,
|
| 42 |
+
"run_final_eval": true,
|
| 43 |
+
"dead_feature_window": 1000,
|
| 44 |
+
"feature_sampling_window": 2000,
|
| 45 |
+
"n_batches_for_norm_estimate": 1000,
|
| 46 |
+
"sae_lens_version": "6.43.0",
|
| 47 |
+
"logger": {
|
| 48 |
+
"log_to_wandb": true,
|
| 49 |
+
"log_activations_store_to_wandb": false,
|
| 50 |
+
"log_optimizer_state_to_wandb": false,
|
| 51 |
+
"log_weights_to_wandb": true,
|
| 52 |
+
"wandb_project": "sae_lens_training",
|
| 53 |
+
"wandb_id": null,
|
| 54 |
+
"run_name": "synthetic-topk-16384-LR-0.0003",
|
| 55 |
+
"wandb_entity": null,
|
| 56 |
+
"wandb_log_frequency": 10,
|
| 57 |
+
"eval_every_n_wandb_logs": 100
|
| 58 |
+
}
|
| 59 |
+
}
|
oracle_init_smoke/eval_stats.json
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"true_l0": 34.46811,
|
| 3 |
+
"sae_l0": 47.999505,
|
| 4 |
+
"dead_latents": 0,
|
| 5 |
+
"shrinkage": 0.9734420861816406,
|
| 6 |
+
"explained_variance": 0.9196613902159156,
|
| 7 |
+
"mcc": 0.990073561668396,
|
| 8 |
+
"uniqueness": 1.0,
|
| 9 |
+
"classification": {
|
| 10 |
+
"precision": 0.5700647234916687,
|
| 11 |
+
"recall": 0.91277015209198,
|
| 12 |
+
"f1_score": 0.6972928047180176,
|
| 13 |
+
"accuracy": 0.998992383480072
|
| 14 |
+
}
|
| 15 |
+
}
|
oracle_init_smoke/meta.json
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"k": 48,
|
| 3 |
+
"seed": 1,
|
| 4 |
+
"n_samples": 1000000,
|
| 5 |
+
"cos_init": 1.0,
|
| 6 |
+
"cos_final": 0.990073561668396,
|
| 7 |
+
"seconds": 1409.8078198432922
|
| 8 |
+
}
|
oracle_init_smoke/runner_config.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"synthetic_model": "decoderesearch/synth-sae-bench-16k-v1",
|
| 3 |
+
"sae": {
|
| 4 |
+
"d_in": 768,
|
| 5 |
+
"d_sae": 16384,
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"device": "cuda",
|
| 8 |
+
"apply_b_dec_to_input": true,
|
| 9 |
+
"normalize_activations": "none",
|
| 10 |
+
"reshape_activations": "none",
|
| 11 |
+
"metadata": {
|
| 12 |
+
"sae_lens_version": "6.43.0",
|
| 13 |
+
"sae_lens_training_version": "6.43.0"
|
| 14 |
+
},
|
| 15 |
+
"decoder_init_norm": 0.1,
|
| 16 |
+
"k": 48,
|
| 17 |
+
"use_sparse_activations": false,
|
| 18 |
+
"aux_loss_coefficient": 1.0,
|
| 19 |
+
"rescale_acts_by_decoder_norm": true,
|
| 20 |
+
"architecture": "topk"
|
| 21 |
+
},
|
| 22 |
+
"training_samples": 1000000,
|
| 23 |
+
"batch_size": 4096,
|
| 24 |
+
"lr": 0.0003,
|
| 25 |
+
"lr_warm_up_steps": 0,
|
| 26 |
+
"lr_decay_steps": 0,
|
| 27 |
+
"lr_scheduler_name": "constant",
|
| 28 |
+
"lr_end": 2.9999999999999997e-05,
|
| 29 |
+
"adam_beta1": 0.9,
|
| 30 |
+
"adam_beta2": 0.999,
|
| 31 |
+
"n_restart_cycles": 1,
|
| 32 |
+
"device": "cuda",
|
| 33 |
+
"autocast_sae": false,
|
| 34 |
+
"autocast_data": false,
|
| 35 |
+
"n_checkpoints": 8,
|
| 36 |
+
"checkpoint_path": "runs/oracle_init_smoke/ckpts",
|
| 37 |
+
"save_final_checkpoint": true,
|
| 38 |
+
"output_path": "runs/oracle_init_smoke",
|
| 39 |
+
"save_synthetic_model": false,
|
| 40 |
+
"eval_frequency": 200000,
|
| 41 |
+
"eval_samples": 200000,
|
| 42 |
+
"run_final_eval": true,
|
| 43 |
+
"dead_feature_window": 1000,
|
| 44 |
+
"feature_sampling_window": 2000,
|
| 45 |
+
"n_batches_for_norm_estimate": 1000,
|
| 46 |
+
"sae_lens_version": "6.43.0",
|
| 47 |
+
"logger": {
|
| 48 |
+
"log_to_wandb": true,
|
| 49 |
+
"log_activations_store_to_wandb": false,
|
| 50 |
+
"log_optimizer_state_to_wandb": false,
|
| 51 |
+
"log_weights_to_wandb": true,
|
| 52 |
+
"wandb_project": "sae_lens_training",
|
| 53 |
+
"wandb_id": null,
|
| 54 |
+
"run_name": "synthetic-topk-16384-LR-0.0003",
|
| 55 |
+
"wandb_entity": null,
|
| 56 |
+
"wandb_log_frequency": 10,
|
| 57 |
+
"eval_every_n_wandb_logs": 100
|
| 58 |
+
}
|
| 59 |
+
}
|