Transformers
PyTorch
Graphcore
English
groupbert
Generated from Trainer
Inference Endpoints
Ivan Chelombiev commited on
Commit
b7b6a69
1 Parent(s): b1d6998

Training in progress, step 500

Browse files
Files changed (4) hide show
  1. config.json +5 -6
  2. ipu_config.json +1 -1
  3. pytorch_model.bin +1 -1
  4. training_args.bin +2 -2
config.json CHANGED
@@ -1,20 +1,19 @@
1
  {
2
- "_name_or_path": "run9-phase2",
3
  "architectures": [
4
- "PipelinedGroupBertForPreTraining"
5
  ],
6
- "attention_probs_dropout_prob": 0.1,
7
  "classifier_dropout": null,
8
  "conv_group_size": 16,
9
  "conv_kernel_size": 7,
10
  "ffn_groups": 4,
11
- "gradient_checkpointing": false,
12
  "hidden_act": "gelu",
13
- "hidden_dropout_prob": 0.1,
14
  "hidden_size": 768,
15
  "initializer_range": 0.02,
16
  "intermediate_size": 3072,
17
- "layer_norm_eps": 0.001,
18
  "max_num_masked_tokens": 96,
19
  "max_position_embeddings": 512,
20
  "model_type": "groupbert",
 
1
  {
2
+ "_name_or_path": "./run9",
3
  "architectures": [
4
+ "PoptorchPipelinedGroupBertForPreTraining"
5
  ],
6
+ "attention_probs_dropout_prob": 0.0,
7
  "classifier_dropout": null,
8
  "conv_group_size": 16,
9
  "conv_kernel_size": 7,
10
  "ffn_groups": 4,
 
11
  "hidden_act": "gelu",
12
+ "hidden_dropout_prob": 0.0,
13
  "hidden_size": 768,
14
  "initializer_range": 0.02,
15
  "intermediate_size": 3072,
16
+ "layer_norm_eps": 1e-12,
17
  "max_num_masked_tokens": 96,
18
  "max_position_embeddings": 512,
19
  "model_type": "groupbert",
ipu_config.json CHANGED
@@ -17,7 +17,7 @@
17
  ],
18
  "matmul_proportion": 0.22,
19
  "optimizer_state_offchip": false,
20
- "optimum_version": "1.1.1",
21
  "output_mode": "final",
22
  "recompute_checkpoint_every_layer": true,
23
  "replicated_tensor_sharding": true,
 
17
  ],
18
  "matmul_proportion": 0.22,
19
  "optimizer_state_offchip": false,
20
+ "optimum_version": "1.4.1",
21
  "output_mode": "final",
22
  "recompute_checkpoint_every_layer": true,
23
  "replicated_tensor_sharding": true,
pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4f3b913dc3f9b1fb1db37b58adc647d4ea8b0a7c266f2eb9aad5e22575be3ac3
3
  size 321956797
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8a406baa7c032227f8ac5eb3f882b8312f8b52f36f38288c34332d2c7d5c9c15
3
  size 321956797
training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f02572ddce2f8070fa4cd8025303dab3b529c6cfee97cf22db3aff4aecdb3a21
3
- size 2927
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7fcbe6e7f558e3ec8b775c0b6efc35b5d905da365ad127cd316fdd70ecef2c91
3
+ size 2863