danf0 commited on
Commit
91df278
·
verified ·
1 Parent(s): 92374ea

Upload 288 files

Browse files

Per-attention head SAEs for GPT-2 small.

This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. gpt2-small_L0_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  2. gpt2-small_L0_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  3. gpt2-small_L0_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  4. gpt2-small_L0_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  5. gpt2-small_L0_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  6. gpt2-small_L0_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  7. gpt2-small_L0_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  8. gpt2-small_L0_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  9. gpt2-small_L0_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  10. gpt2-small_L0_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  11. gpt2-small_L0_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  12. gpt2-small_L0_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  13. gpt2-small_L0_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  14. gpt2-small_L0_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  15. gpt2-small_L0_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  16. gpt2-small_L0_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  17. gpt2-small_L0_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  18. gpt2-small_L0_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  19. gpt2-small_L0_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  20. gpt2-small_L0_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  21. gpt2-small_L0_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  22. gpt2-small_L0_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  23. gpt2-small_L0_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  24. gpt2-small_L0_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  25. gpt2-small_L10_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  26. gpt2-small_L10_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  27. gpt2-small_L10_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  28. gpt2-small_L10_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  29. gpt2-small_L10_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  30. gpt2-small_L10_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  31. gpt2-small_L10_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  32. gpt2-small_L10_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  33. gpt2-small_L10_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  34. gpt2-small_L10_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  35. gpt2-small_L10_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  36. gpt2-small_L10_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  37. gpt2-small_L10_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  38. gpt2-small_L10_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  39. gpt2-small_L10_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  40. gpt2-small_L10_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  41. gpt2-small_L10_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  42. gpt2-small_L10_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  43. gpt2-small_L10_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  44. gpt2-small_L10_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  45. gpt2-small_L10_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  46. gpt2-small_L10_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  47. gpt2-small_L10_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  48. gpt2-small_L10_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
  49. gpt2-small_L11_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt +3 -0
  50. gpt2-small_L11_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json +1 -0
gpt2-small_L0_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3d31762b7cabb032794cc894a130279f89bafe42b6283786bf6a6d3fda6a0ff8
3
+ size 1060520
gpt2-small_L0_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 0, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H0_2048_z"}
gpt2-small_L0_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:aed3581387beb655ac2fdadbffab7eef8d091a60d5785079d0dd68ab16efcac4
3
+ size 1060528
gpt2-small_L0_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 10, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H10_2048_z"}
gpt2-small_L0_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0179b44141b53ff885a30b3bef30e6ecae0515c1cf2674173fe92fdc551c7671
3
+ size 1060528
gpt2-small_L0_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 11, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H11_2048_z"}
gpt2-small_L0_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b997b5b0e91cad93a03167f28fbd207b0958f3efc02fdd2d6c3dbdc71684287f
3
+ size 1060520
gpt2-small_L0_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 1, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H1_2048_z"}
gpt2-small_L0_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4e181ec24c0a5386c297aaebc06f8ec7658aa98cfeb0b9e710cfa72bc8f35566
3
+ size 1060520
gpt2-small_L0_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 2, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H2_2048_z"}
gpt2-small_L0_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:42e96ccc94619475787fe0fc1ff0218dd7e725edcbe4f6326fe17a325d6c00ff
3
+ size 1060520
gpt2-small_L0_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 3, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H3_2048_z"}
gpt2-small_L0_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c78d5dca4516076a0d72f103b721181389823e84d1a7ecaf4900c26426fcf629
3
+ size 1060520
gpt2-small_L0_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 4, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H4_2048_z"}
gpt2-small_L0_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8737d63fcecbc34cbb50ff0e28bb10dad9db601eb40d8ed02d7adf8aaf7d7bbc
3
+ size 1060520
gpt2-small_L0_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 5, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H5_2048_z"}
gpt2-small_L0_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9cbd4b3165531eeb2e10e29b7f59a758badf742863d9691c307e712dd21d26aa
3
+ size 1060520
gpt2-small_L0_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 6, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H6_2048_z"}
gpt2-small_L0_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b30fede14c9eb20665f1db7b90d654cfde8caaec5f97402f55863386a459483b
3
+ size 1060520
gpt2-small_L0_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 7, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H7_2048_z"}
gpt2-small_L0_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:25632dee721b4a2f39c7e3dc33a6feab1f9a8801ab96250772b602937003e60c
3
+ size 1060520
gpt2-small_L0_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 8, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H8_2048_z"}
gpt2-small_L0_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ad42723234a1e2005fdc5cdbae3eafca6863741b7fcce0d877655ffa6b1c96f
3
+ size 1060520
gpt2-small_L0_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 0, "head": 9, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.0.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L0H9_2048_z"}
gpt2-small_L10_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:537d6163e03194e60b64e8fe02a9280bb4edc70736e995e038a0a68565571d01
3
+ size 1060528
gpt2-small_L10_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 0, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H0_2048_z"}
gpt2-small_L10_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7fd536192b2c6271f73c5bdc81ede4a4adaff06636b678d4100b5ac446b94404
3
+ size 1060536
gpt2-small_L10_H10_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 10, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H10_2048_z"}
gpt2-small_L10_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e2baf79a674a1c7a7b3a7ea3730b7bd920c9e5c89270fc57aa86804155958324
3
+ size 1060536
gpt2-small_L10_H11_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 11, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H11_2048_z"}
gpt2-small_L10_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8ba5a8676df520ac811b2128a527b25b321606430727a7c2fdf6e8debb8505f1
3
+ size 1060528
gpt2-small_L10_H1_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 1, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H1_2048_z"}
gpt2-small_L10_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4759273ca0b794e7828e06f4ac7da60a7b118a730cc3e37deeb1d02168d35c7d
3
+ size 1060528
gpt2-small_L10_H2_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 2, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H2_2048_z"}
gpt2-small_L10_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d5d79a1f0e5d25d997267b8a2cbd8699a25f97ebfebd592fd75180b82c729cf1
3
+ size 1060528
gpt2-small_L10_H3_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 3, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H3_2048_z"}
gpt2-small_L10_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:03ed8d7ae732dd63c462f0fd5f17b6da98b2bc5cfec5edc409e58d6883ee1086
3
+ size 1060528
gpt2-small_L10_H4_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 4, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H4_2048_z"}
gpt2-small_L10_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:af1e9c9e8a5d64714afefb5a9aa474f26936e9331d95338f1e3761214855b1cb
3
+ size 1060528
gpt2-small_L10_H5_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 5, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H5_2048_z"}
gpt2-small_L10_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7367a54a98f921ef1e87314b688f477c46fe9589d5a7821eedbb1fe056d0926f
3
+ size 1060528
gpt2-small_L10_H6_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 6, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H6_2048_z"}
gpt2-small_L10_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8daaeda74a44c0b4428e043e2d7f87b106297be9d5bb4e97f6bea230f770ae4e
3
+ size 1060528
gpt2-small_L10_H7_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 7, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H7_2048_z"}
gpt2-small_L10_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c757d4bf89a3193632b27750631bdb62553b26a7da48d68ad3c4da074d5dd55a
3
+ size 1060528
gpt2-small_L10_H8_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 8, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H8_2048_z"}
gpt2-small_L10_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:447595578a4d51319837d6fb88799de201d65a8a6e24a96c68e254d34a011a17
3
+ size 1060528
gpt2-small_L10_H9_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 10, "head": 9, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.10.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L10H9_2048_z"}
gpt2-small_L11_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:04153d14ce3cce9c7bb1efc4841732143d760916b3d37bd9ac35ddccf8dc6c14
3
+ size 1060528
gpt2-small_L11_H0_z_lr1.20e-03_l11.00e+00_ds2048_nt2000000000_bs4096_dc1.00e-06_rsanthropic_rie25000_nr4_sl64_c1_v9_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"seed": 49, "batch_size": 4096, "buffer_mult": 384, "lr": 0.0012, "num_tokens": 2000000000, "l1_coeff": 1.0, "beta1": 0.9, "beta2": 0.99, "dict_mult": 32, "seq_len": 64, "contiguous": 1, "enc_dtype": "fp32", "model_name": "gpt2-small", "site": "z", "device": "cuda", "reinit": "reinit", "concat_heads": false, "per_head": true, "layer": 11, "head": 0, "resample_scheme": "anthropic", "anthropic_neuron_resample_scale": 0.2, "dead_direction_cutoff": 1e-06, "re_init_every": 25000, "anthropic_resample_last": 12500, "resample_factor": 0.01, "num_resamples": 4, "wandb_project_name": "attention-head-saes", "wandb_entity": "danf", "save_state_dict_every": 50000, "b_dec_init": "zeros", "sched_type": "cosine_warmup", "sched_epochs": 1000, "sched_lr_factor": 0.1, "sched_warmup_epochs": 1000, "sched_finish": true, "anthropic_resample_batches": 100, "eval_every": 1000, "test": false, "model_batch_size": 512, "buffer_size": 1572864, "buffer_batches": 24576, "act_name": "blocks.11.attn.hook_z", "act_size": 64, "dict_size": 2048, "name": "gpt2-small_L11H0_2048_z"}