Upload 288 files
Browse filesPer-attention head SAEs for GPT-2 small.
This view is limited to 50 files because it contains too many changes. See raw diff
- gpt2-small_L0_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L0_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L0_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L10_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L10_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
- gpt2-small_L11_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
- gpt2-small_L11_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
gpt2-small_L0_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3d31762b7cabb032794cc894a130279f89bafe42b6283786bf6a6d3fda6a0ff8
|
| 3 |
+
size 1060520
|
gpt2-small_L0_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 0, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H0_2048_z"}
|
gpt2-small_L0_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:aed3581387beb655ac2fdadbffab7eef8d091a60d5785079d0dd68ab16efcac4
|
| 3 |
+
size 1060528
|
gpt2-small_L0_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 10, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H10_2048_z"}
|
gpt2-small_L0_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0179b44141b53ff885a30b3bef30e6ecae0515c1cf2674173fe92fdc551c7671
|
| 3 |
+
size 1060528
|
gpt2-small_L0_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 11, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H11_2048_z"}
|
gpt2-small_L0_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b997b5b0e91cad93a03167f28fbd207b0958f3efc02fdd2d6c3dbdc71684287f
|
| 3 |
+
size 1060520
|
gpt2-small_L0_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 1, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H1_2048_z"}
|
gpt2-small_L0_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4e181ec24c0a5386c297aaebc06f8ec7658aa98cfeb0b9e710cfa72bc8f35566
|
| 3 |
+
size 1060520
|
gpt2-small_L0_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 2, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H2_2048_z"}
|
gpt2-small_L0_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:42e96ccc94619475787fe0fc1ff0218dd7e725edcbe4f6326fe17a325d6c00ff
|
| 3 |
+
size 1060520
|
gpt2-small_L0_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 3, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H3_2048_z"}
|
gpt2-small_L0_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c78d5dca4516076a0d72f103b721181389823e84d1a7ecaf4900c26426fcf629
|
| 3 |
+
size 1060520
|
gpt2-small_L0_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 4, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H4_2048_z"}
|
gpt2-small_L0_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8737d63fcecbc34cbb50ff0e28bb10dad9db601eb40d8ed02d7adf8aaf7d7bbc
|
| 3 |
+
size 1060520
|
gpt2-small_L0_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 5, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H5_2048_z"}
|
gpt2-small_L0_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9cbd4b3165531eeb2e10e29b7f59a758badf742863d9691c307e712dd21d26aa
|
| 3 |
+
size 1060520
|
gpt2-small_L0_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 6, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H6_2048_z"}
|
gpt2-small_L0_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b30fede14c9eb20665f1db7b90d654cfde8caaec5f97402f55863386a459483b
|
| 3 |
+
size 1060520
|
gpt2-small_L0_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 7, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H7_2048_z"}
|
gpt2-small_L0_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:25632dee721b4a2f39c7e3dc33a6feab1f9a8801ab96250772b602937003e60c
|
| 3 |
+
size 1060520
|
gpt2-small_L0_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 8, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H8_2048_z"}
|
gpt2-small_L0_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1ad42723234a1e2005fdc5cdbae3eafca6863741b7fcce0d877655ffa6b1c96f
|
| 3 |
+
size 1060520
|
gpt2-small_L0_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 9, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H9_2048_z"}
|
gpt2-small_L10_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:537d6163e03194e60b64e8fe02a9280bb4edc70736e995e038a0a68565571d01
|
| 3 |
+
size 1060528
|
gpt2-small_L10_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 0, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H0_2048_z"}
|
gpt2-small_L10_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7fd536192b2c6271f73c5bdc81ede4a4adaff06636b678d4100b5ac446b94404
|
| 3 |
+
size 1060536
|
gpt2-small_L10_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 10, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H10_2048_z"}
|
gpt2-small_L10_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e2baf79a674a1c7a7b3a7ea3730b7bd920c9e5c89270fc57aa86804155958324
|
| 3 |
+
size 1060536
|
gpt2-small_L10_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 11, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H11_2048_z"}
|
gpt2-small_L10_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8ba5a8676df520ac811b2128a527b25b321606430727a7c2fdf6e8debb8505f1
|
| 3 |
+
size 1060528
|
gpt2-small_L10_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 1, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H1_2048_z"}
|
gpt2-small_L10_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4759273ca0b794e7828e06f4ac7da60a7b118a730cc3e37deeb1d02168d35c7d
|
| 3 |
+
size 1060528
|
gpt2-small_L10_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 2, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H2_2048_z"}
|
gpt2-small_L10_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d5d79a1f0e5d25d997267b8a2cbd8699a25f97ebfebd592fd75180b82c729cf1
|
| 3 |
+
size 1060528
|
gpt2-small_L10_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 3, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H3_2048_z"}
|
gpt2-small_L10_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:03ed8d7ae732dd63c462f0fd5f17b6da98b2bc5cfec5edc409e58d6883ee1086
|
| 3 |
+
size 1060528
|
gpt2-small_L10_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 4, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H4_2048_z"}
|
gpt2-small_L10_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:af1e9c9e8a5d64714afefb5a9aa474f26936e9331d95338f1e3761214855b1cb
|
| 3 |
+
size 1060528
|
gpt2-small_L10_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 5, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H5_2048_z"}
|
gpt2-small_L10_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7367a54a98f921ef1e87314b688f477c46fe9589d5a7821eedbb1fe056d0926f
|
| 3 |
+
size 1060528
|
gpt2-small_L10_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 6, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H6_2048_z"}
|
gpt2-small_L10_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8daaeda74a44c0b4428e043e2d7f87b106297be9d5bb4e97f6bea230f770ae4e
|
| 3 |
+
size 1060528
|
gpt2-small_L10_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 7, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H7_2048_z"}
|
gpt2-small_L10_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c757d4bf89a3193632b27750631bdb62553b26a7da48d68ad3c4da074d5dd55a
|
| 3 |
+
size 1060528
|
gpt2-small_L10_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 8, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H8_2048_z"}
|
gpt2-small_L10_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:447595578a4d51319837d6fb88799de201d65a8a6e24a96c68e254d34a011a17
|
| 3 |
+
size 1060528
|
gpt2-small_L10_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 9, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H9_2048_z"}
|
gpt2-small_L11_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:04153d14ce3cce9c7bb1efc4841732143d760916b3d37bd9ac35ddccf8dc6c14
|
| 3 |
+
size 1060528
|
gpt2-small_L11_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 11, "head": 0, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.11.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L11H0_2048_z"}
|