CodeIsAbstract commited on
Commit
ebc7090
·
verified ·
1 Parent(s): 99574d3

Training in progress, step 77000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2a187dee5bccd16e94b5ab10c8dee4691190193326142bcafb73a0a6eb058d93
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4f64707cb242aae96a30c6b449fcc99592355f500c9224203686d1d890be2992
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4576b4611184440e8e89546d8dcec0928f3f749f9bd7cd9939158be70b4eeb74
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:10adaaa97072ac420e324e8f48de65dca77077c9e183034ad6853c28043a73f7
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:375646414b14bb81be31b058ed8af728ea3d95a2e32dca6c4303eaf631e462ed
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:754331f5e21dff730a141fc4f1c359fb757055cf0e71500d04c1537884521087
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4abf2f313f27bccac25e9c8e915c70451aa8583b1fd70999a373943c42410d5b
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e31d3f2dd7778628805fe0f91bc420be77c7d5b49ef9506f643542e757c4a470
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.36363636363636365,
6
  "eval_steps": 1000,
7
- "global_step": 76000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -5936,10 +5936,88 @@
5936
  "eval_samples_per_second": 77.062,
5937
  "eval_steps_per_second": 19.266,
5938
  "step": 76000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5939
  }
5940
  ],
5941
  "logging_steps": 100,
5942
- "max_steps": 110000,
5943
  "num_input_tokens_seen": 0,
5944
  "num_train_epochs": 9223372036854775807,
5945
  "save_steps": 4000,
@@ -5950,12 +6028,12 @@
5950
  "should_evaluate": false,
5951
  "should_log": false,
5952
  "should_save": true,
5953
- "should_training_stop": false
5954
  },
5955
  "attributes": {}
5956
  }
5957
  },
5958
- "total_flos": 1.893508155703296e+18,
5959
  "train_batch_size": 22,
5960
  "trial_name": null,
5961
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.012987012987012988,
6
  "eval_steps": 1000,
7
+ "global_step": 77000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
5936
  "eval_samples_per_second": 77.062,
5937
  "eval_steps_per_second": 19.266,
5938
  "step": 76000
5939
+ },
5940
+ {
5941
+ "epoch": 0.0012987012987012987,
5942
+ "grad_norm": 0.16393087804317474,
5943
+ "learning_rate": 3.391192895308981e-07,
5944
+ "loss": 2.6057958984375,
5945
+ "step": 76100
5946
+ },
5947
+ {
5948
+ "epoch": 0.0025974025974025974,
5949
+ "grad_norm": 0.1581353396177292,
5950
+ "learning_rate": 2.6802681023424534e-07,
5951
+ "loss": 2.6045050048828124,
5952
+ "step": 76200
5953
+ },
5954
+ {
5955
+ "epoch": 0.003896103896103896,
5956
+ "grad_norm": 0.16540294885635376,
5957
+ "learning_rate": 2.0528552419035728e-07,
5958
+ "loss": 2.6286279296875,
5959
+ "step": 76300
5960
+ },
5961
+ {
5962
+ "epoch": 0.005194805194805195,
5963
+ "grad_norm": 0.1575717329978943,
5964
+ "learning_rate": 1.508964798905832e-07,
5965
+ "loss": 2.624171142578125,
5966
+ "step": 76400
5967
+ },
5968
+ {
5969
+ "epoch": 0.006493506493506494,
5970
+ "grad_norm": 0.15712368488311768,
5971
+ "learning_rate": 1.0486058624897821e-07,
5972
+ "loss": 2.590557861328125,
5973
+ "step": 76500
5974
+ },
5975
+ {
5976
+ "epoch": 0.007792207792207792,
5977
+ "grad_norm": 0.14502283930778503,
5978
+ "learning_rate": 6.717861258720426e-08,
5979
+ "loss": 2.5785467529296877,
5980
+ "step": 76600
5981
+ },
5982
+ {
5983
+ "epoch": 0.00909090909090909,
5984
+ "grad_norm": 0.16336509585380554,
5985
+ "learning_rate": 3.7851188621707e-08,
5986
+ "loss": 2.6053866577148437,
5987
+ "step": 76700
5988
+ },
5989
+ {
5990
+ "epoch": 0.01038961038961039,
5991
+ "grad_norm": 0.1576337367296219,
5992
+ "learning_rate": 1.6878804453168694e-08,
5993
+ "loss": 2.595892028808594,
5994
+ "step": 76800
5995
+ },
5996
+ {
5997
+ "epoch": 0.011688311688311689,
5998
+ "grad_norm": 0.15049995481967926,
5999
+ "learning_rate": 4.261810558348067e-09,
6000
+ "loss": 2.629699401855469,
6001
+ "step": 76900
6002
+ },
6003
+ {
6004
+ "epoch": 0.012987012987012988,
6005
+ "grad_norm": 0.1851549744606018,
6006
+ "learning_rate": 4.177841961272577e-13,
6007
+ "loss": 2.5818411254882814,
6008
+ "step": 77000
6009
+ },
6010
+ {
6011
+ "epoch": 0.012987012987012988,
6012
+ "eval_loss": 3.022578001022339,
6013
+ "eval_runtime": 7.458,
6014
+ "eval_samples_per_second": 77.233,
6015
+ "eval_steps_per_second": 19.308,
6016
+ "step": 77000
6017
  }
6018
  ],
6019
  "logging_steps": 100,
6020
+ "max_steps": 77000,
6021
  "num_input_tokens_seen": 0,
6022
  "num_train_epochs": 9223372036854775807,
6023
  "save_steps": 4000,
 
6028
  "should_evaluate": false,
6029
  "should_log": false,
6030
  "should_save": true,
6031
+ "should_training_stop": true
6032
  },
6033
  "attributes": {}
6034
  }
6035
  },
6036
+ "total_flos": 1.918422736699392e+18,
6037
  "train_batch_size": 22,
6038
  "trial_name": null,
6039
  "trial_params": null
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ef8c80f5d40636555b8c09ec97785c8964e446b8d757c55eb76e5b7454841939
3
  size 5201
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:de39541c4d3c2c8cd3db89cc5a7ea96a1526b7e8be52cafa957494d7daabed0f
3
  size 5201