Upload folder using huggingface_hub
Browse filesThis view is limited to 50 files because it contains too many changes.
See raw diff
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_0/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_0/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_1/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_1/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_2/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_2/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_3/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_3/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_4/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_4/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_5/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_5/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_0/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_0/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_308/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_308/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_3088/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_3088/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_30881/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_30881/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_97/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_97/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_976/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_976/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_9765/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_9765/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_0/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_0/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_308/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_308/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_3088/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_3088/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_30881/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_30881/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_97/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_97/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_976/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_976/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_9765/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_9765/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_0/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_0/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_308/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_308/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_3088/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_3088/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_30881/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_30881/config.json +26 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_97/ae.pt +3 -0
- gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_97/config.json +26 -0
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_0/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:208e1a9dfa6a72da57408cdf09d5e956bdc73d22cb40fcef6ac276e0f7c0f40f
|
3 |
+
size 1208232744
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_0/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": 97656,
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 20,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_1/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:b62256e9d5194b29791199a91bda4b7a57f1fe2053d9f33a00e2bac6a4103fca
|
3 |
+
size 1208232744
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_1/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": 97656,
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 40,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_2/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:f1806bd2410e9b3624cd75b80660b0c9b5e1b4de5e573fb7e01f0fadb47827eb
|
3 |
+
size 1208232744
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_2/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": 97656,
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 80,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_3/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:70d494ac3a55427b1c8a71862177a6207389dedeb5216272cb225773fa18108a
|
3 |
+
size 1208232744
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_3/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": 97656,
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 160,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_4/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:097f6be632e37ed5ad2db3bb71275074af73381414aed4c958d8c981ba57ff0b
|
3 |
+
size 1208232744
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_4/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": 97656,
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 320,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_5/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:f08b5edb8d4f08d5624f3f2ed57099068d5793e56a6a0b995b89ad008a0158d2
|
3 |
+
size 1208232744
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12/trainer_5/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": 97656,
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 640,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_0/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:502a714e42542d494218c7cf9ba4eb50e44f5d991e21f7380afa481874e98186
|
3 |
+
size 1208232760
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_0/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "0",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 20,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_308/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:1a26798a8bcb7a1651ce9ecea96e8a3c028d085c6e076deb10716d8ac54a7e34
|
3 |
+
size 1208232776
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_308/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "308",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 20,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_3088/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:b4c1e18c0ebec34395a355c549668618fae357fbf8a711941e2c6a19218a167b
|
3 |
+
size 1208232848
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_3088/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "3088",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 20,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_30881/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:9ad2b8ba91efe15d2b0a3b50df6e7d72f20693ff85187e6affecf24c99b3cdaf
|
3 |
+
size 1208233048
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_30881/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "30881",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 20,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_97/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:1fdc5bfe07945e0e7c747785f6c0f7697e31de06247271407f63f4b43989ce7f
|
3 |
+
size 1208232768
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_97/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "97",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 20,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_976/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:e6b27d354e0c1291a14f2237c2179480763c807b10d357479287996d6bd4a32d
|
3 |
+
size 1208232776
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_976/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "976",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 20,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_9765/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:42673cf80ea69ba2a20937fade9cb5229e8e9c2473622a93179b73bd04f66192
|
3 |
+
size 1208232848
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_0_step_9765/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "9765",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 20,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_0/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:502a714e42542d494218c7cf9ba4eb50e44f5d991e21f7380afa481874e98186
|
3 |
+
size 1208232760
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_0/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "0",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 40,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_308/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:6f4970ab95412f34dc6d2a51e1a44a2ba646f4fc3db7b66e78d0f8457e5f3de3
|
3 |
+
size 1208232776
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_308/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "308",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 40,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_3088/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:68bfc21858c287b3469700526b53b45af93be77ac98f10ae6823ccd93c6b2656
|
3 |
+
size 1208232848
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_3088/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "3088",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 40,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_30881/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:2aa60e665467bcc939083c498b0185bb764f393b12f3e809b83eed112aebaca1
|
3 |
+
size 1208233048
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_30881/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "30881",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 40,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_97/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:e977b4a56cc05dff066aeb2814bf968ef19b6ee2597a916859336d280e94fa4c
|
3 |
+
size 1208232768
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_97/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "97",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 40,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_976/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:4f23b1538dc9bca7beffba37ea52766db4fa837f3efaf6020db6a12cc7c8d1a2
|
3 |
+
size 1208232776
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_976/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "976",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 40,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_9765/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:9488ef8aa60b2f1e26f0ae852cc39c41ea54755c1b7213af3e3d8b917f894615
|
3 |
+
size 1208232848
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_1_step_9765/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "9765",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 40,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_0/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:502a714e42542d494218c7cf9ba4eb50e44f5d991e21f7380afa481874e98186
|
3 |
+
size 1208232760
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_0/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "0",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 80,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_308/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:d5ef1da96268137de3a086f3cc7ad0eb2648ef398566ce418b9d905e07fece03
|
3 |
+
size 1208232776
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_308/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "308",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 80,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_3088/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:33aecee459f09e9fb9e136a885097e0c54ab6c0fb0f09dbc5b4c3f04d9fc3e5b
|
3 |
+
size 1208232848
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_3088/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "3088",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 80,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_30881/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:fb384a6445ac3ab10ab4c7118e2849d4db2053a51661d5c13333515ba02740c3
|
3 |
+
size 1208233048
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_30881/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "30881",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 80,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_97/ae.pt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:a9116ed6e64a6bba8b45edda4ca2f67c0e783aaf950a6c00f2bc149e513b1132
|
3 |
+
size 1208232768
|
gemma-2-2b_topk_width-2pow16_date-1109/resid_post_layer_12_checkpoints/trainer_2_step_97/config.json
ADDED
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"trainer": {
|
3 |
+
"trainer_class": "TrainerTopK",
|
4 |
+
"dict_class": "AutoEncoderTopK",
|
5 |
+
"lr": 0.0001,
|
6 |
+
"steps": "97",
|
7 |
+
"seed": 0,
|
8 |
+
"activation_dim": 2304,
|
9 |
+
"dict_size": 65536,
|
10 |
+
"k": 80,
|
11 |
+
"device": "cuda:1",
|
12 |
+
"layer": 12,
|
13 |
+
"lm_name": "google/gemma-2-2b",
|
14 |
+
"wandb_name": "TopKTrainer-google/gemma-2-2b-resid_post_layer_12",
|
15 |
+
"submodule_name": "resid_post_layer_12"
|
16 |
+
},
|
17 |
+
"buffer": {
|
18 |
+
"d_submodule": 2304,
|
19 |
+
"io": "out",
|
20 |
+
"n_ctxs": 2048,
|
21 |
+
"ctx_len": 128,
|
22 |
+
"refresh_batch_size": 24,
|
23 |
+
"out_batch_size": 4096,
|
24 |
+
"device": "cuda:1"
|
25 |
+
}
|
26 |
+
}
|