Tristan Thrush commited on
Commit
e25f6c6
1 Parent(s): 88149f5

Training in progress, epoch 0

Browse files
Files changed (27) hide show
  1. .gitignore +1 -0
  2. config.json +39 -0
  3. logs/1672680026.553644/events.out.tfevents.1672680026.tristan-olm-training-a100-80.97506.1 +3 -0
  4. logs/1672680410.3099709/events.out.tfevents.1672680410.tristan-olm-training-a100-80.101195.1 +3 -0
  5. logs/1672680510.7320256/events.out.tfevents.1672680510.tristan-olm-training-a100-80.104556.1 +3 -0
  6. logs/1672680710.2332163/events.out.tfevents.1672680710.tristan-olm-training-a100-80.107943.1 +3 -0
  7. logs/1672681285.8309555/events.out.tfevents.1672681285.tristan-olm-training-a100-80.111576.1 +3 -0
  8. logs/1672681495.1712687/events.out.tfevents.1672681495.tristan-olm-training-a100-80.115679.1 +3 -0
  9. logs/1672681775.8689537/events.out.tfevents.1672681775.tristan-olm-training-a100-80.119314.1 +3 -0
  10. logs/1672682182.0067658/events.out.tfevents.1672682182.tristan-olm-training-a100-80.123038.1 +3 -0
  11. logs/1672705969.2600806/events.out.tfevents.1672705969.tristan-olm-training-a100-80.138319.1 +3 -0
  12. logs/events.out.tfevents.1672680026.tristan-olm-training-a100-80.97506.0 +3 -0
  13. logs/events.out.tfevents.1672680410.tristan-olm-training-a100-80.101195.0 +3 -0
  14. logs/events.out.tfevents.1672680510.tristan-olm-training-a100-80.104556.0 +3 -0
  15. logs/events.out.tfevents.1672680710.tristan-olm-training-a100-80.107943.0 +3 -0
  16. logs/events.out.tfevents.1672681285.tristan-olm-training-a100-80.111576.0 +3 -0
  17. logs/events.out.tfevents.1672681495.tristan-olm-training-a100-80.115679.0 +3 -0
  18. logs/events.out.tfevents.1672681775.tristan-olm-training-a100-80.119314.0 +3 -0
  19. logs/events.out.tfevents.1672682181.tristan-olm-training-a100-80.123038.0 +3 -0
  20. logs/events.out.tfevents.1672705969.tristan-olm-training-a100-80.138319.0 +3 -0
  21. merges.txt +0 -0
  22. pytorch_model.bin +3 -0
  23. special_tokens_map.json +15 -0
  24. tokenizer.json +0 -0
  25. tokenizer_config.json +23 -0
  26. training_args.bin +3 -0
  27. vocab.json +0 -0
.gitignore ADDED
@@ -0,0 +1 @@
 
1
+ checkpoint-*/
config.json ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "gpt2",
3
+ "activation_function": "gelu_new",
4
+ "architectures": [
5
+ "GPT2LMHeadModel"
6
+ ],
7
+ "attn_pdrop": 0.1,
8
+ "bos_token_id": 50256,
9
+ "embd_pdrop": 0.1,
10
+ "eos_token_id": 50256,
11
+ "initializer_range": 0.02,
12
+ "layer_norm_epsilon": 1e-05,
13
+ "model_type": "gpt2",
14
+ "n_ctx": 1024,
15
+ "n_embd": 768,
16
+ "n_head": 12,
17
+ "n_inner": null,
18
+ "n_layer": 12,
19
+ "n_positions": 1024,
20
+ "reorder_and_upcast_attn": false,
21
+ "resid_pdrop": 0.1,
22
+ "scale_attn_by_inverse_layer_idx": false,
23
+ "scale_attn_weights": true,
24
+ "summary_activation": null,
25
+ "summary_first_dropout": 0.1,
26
+ "summary_proj_to_labels": true,
27
+ "summary_type": "cls_index",
28
+ "summary_use_proj": true,
29
+ "task_specific_params": {
30
+ "text-generation": {
31
+ "do_sample": true,
32
+ "max_length": 50
33
+ }
34
+ },
35
+ "torch_dtype": "float32",
36
+ "transformers_version": "4.24.0",
37
+ "use_cache": true,
38
+ "vocab_size": 50265
39
+ }
logs/1672680026.553644/events.out.tfevents.1672680026.tristan-olm-training-a100-80.97506.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5d55cada293cf7937c658b03319d262b1fd1eb2af8bf4bd340395b0520397531
3
+ size 5500
logs/1672680410.3099709/events.out.tfevents.1672680410.tristan-olm-training-a100-80.101195.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:85710cd20c6303dc637ccd0e516aa8761ac82884da22a39245f18521e8701750
3
+ size 5500
logs/1672680510.7320256/events.out.tfevents.1672680510.tristan-olm-training-a100-80.104556.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a6cb00f5120cc48672175bc3e42486f40d63b3f2ff2c862efc69674d1185844f
3
+ size 5500
logs/1672680710.2332163/events.out.tfevents.1672680710.tristan-olm-training-a100-80.107943.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:337045cd794d084162abe53f2e58094502ca24493691feaa08602ec919a7c47c
3
+ size 5495
logs/1672681285.8309555/events.out.tfevents.1672681285.tristan-olm-training-a100-80.111576.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1de7177520ca325b4744bce7bc6eb78f769749bc1c08314286f06abaf73fabb6
3
+ size 5495
logs/1672681495.1712687/events.out.tfevents.1672681495.tristan-olm-training-a100-80.115679.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eaa3057c09d085f8c84bbf006a6ea84c78f837e3387666c576bb2f5e25970919
3
+ size 5495
logs/1672681775.8689537/events.out.tfevents.1672681775.tristan-olm-training-a100-80.119314.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:86129e43b3d84334b2892400c9cab0525c6d576de0a463b6214ee96f9692c928
3
+ size 5495
logs/1672682182.0067658/events.out.tfevents.1672682182.tristan-olm-training-a100-80.123038.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:feb5479d16803853694f36094e4a3e6915852d16e6403834e481f52225b80d40
3
+ size 5483
logs/1672705969.2600806/events.out.tfevents.1672705969.tristan-olm-training-a100-80.138319.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b7044c9e849377b1f8035578063fd93fa61054e87e5d459587d8356964bb7011
3
+ size 5483
logs/events.out.tfevents.1672680026.tristan-olm-training-a100-80.97506.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6a4a9ae3520d3eb6c49d8e36ff71e24f171b4e740d9305f001dc1302f799399b
3
+ size 4133
logs/events.out.tfevents.1672680410.tristan-olm-training-a100-80.101195.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d8eae39626cef36bf5317d4b114d72e6ce5d8c6bfa020f419ede147c6fab3606
3
+ size 3980
logs/events.out.tfevents.1672680510.tristan-olm-training-a100-80.104556.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ed7f17024d92fdaf0594b449450071258bd1a674f3211cdbcb53b3bcc625f31d
3
+ size 3980
logs/events.out.tfevents.1672680710.tristan-olm-training-a100-80.107943.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5f9ce39d22ac949f9815fc93dd0a30d7b3b708e7cad2f5cdc629498c5a7b79c0
3
+ size 4128
logs/events.out.tfevents.1672681285.tristan-olm-training-a100-80.111576.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f81d6a5b9483b8a84cb8d49b66d019389dd70609cb94a0c0effa2e8c156ef87f
3
+ size 3974
logs/events.out.tfevents.1672681495.tristan-olm-training-a100-80.115679.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8f12719bff2c5cbc76ff8445d8f99867d2bcf8e16af08e20029d3a73f1eeef6a
3
+ size 3974
logs/events.out.tfevents.1672681775.tristan-olm-training-a100-80.119314.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1f5e660e61ca28c40e55515e892f81a60e57ecd378cc4dd27def2bf91e4572fa
3
+ size 3974
logs/events.out.tfevents.1672682181.tristan-olm-training-a100-80.123038.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:98e328b81155b5cd8429bf8b81ba9f931e2e58080fd765bd5f38e3f00e154b6b
3
+ size 19814
logs/events.out.tfevents.1672705969.tristan-olm-training-a100-80.138319.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f7c4e23503db43e1ac594bc100af0a6598bb9e319a493f940fbf7c600258168a
3
+ size 47148
merges.txt ADDED
The diff for this file is too large to render. See raw diff
pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:46c0e28bb45961c113c876b056dbbde32f8f4e9eb4c514c4f92ed324a3a32697
3
+ size 510422589
special_tokens_map.json ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": "<s>",
3
+ "cls_token": "<s>",
4
+ "eos_token": "</s>",
5
+ "mask_token": {
6
+ "content": "<mask>",
7
+ "lstrip": true,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false
11
+ },
12
+ "pad_token": "<pad>",
13
+ "sep_token": "</s>",
14
+ "unk_token": "<unk>"
15
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
tokenizer_config.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "bos_token": "<s>",
4
+ "cls_token": "<s>",
5
+ "eos_token": "</s>",
6
+ "errors": "replace",
7
+ "mask_token": {
8
+ "__type": "AddedToken",
9
+ "content": "<mask>",
10
+ "lstrip": true,
11
+ "normalized": false,
12
+ "rstrip": false,
13
+ "single_word": false
14
+ },
15
+ "model_max_length": 512,
16
+ "name_or_path": "Tristan/olm-tokenizer",
17
+ "pad_token": "<pad>",
18
+ "sep_token": "</s>",
19
+ "special_tokens_map_file": null,
20
+ "tokenizer_class": "RobertaTokenizer",
21
+ "trim_offsets": true,
22
+ "unk_token": "<unk>"
23
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:01df519a89a223188624b16e2f3b882b77a5f58fe4d896d0f81d46e593acf45b
3
+ size 3387
vocab.json ADDED
The diff for this file is too large to render. See raw diff