CodeIsAbstract commited on
Commit
984b983
·
verified ·
1 Parent(s): 6168bac

Training in progress, step 64000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4a3559f22e44a8a648518fcea3f8d97b85f98d79f16bfe2745da3c24e542ca94
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4f1052fc44f17677882e9c85aa46ba7cf0ce95d12039334b20adedd2169a62e5
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:db7a3532ebb932b79a0341be80d04d64f85c81470da944cf60a0f259b00b3e37
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3fc31a21e830e5240ed3d4e54481bf756f6f044cd5580281c9cff0c9b8736b9e
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c4fb849fc4647980403a7edf5a9af1697f444456121347b0f9e1834ac737e420
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:907db0cce0a5bafbb1926a59c3e73634f18337e15182c3dda71e7ef66bbb3df1
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2a97b2afb49bff977115e649631e564b8b83f6947e290d5e2212d7d11b04765d
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1845127fa7eca8c5b502f7e5468e1855ba6951691315fb77854957ecb9da8539
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.21818181818181817,
6
  "eval_steps": 1000,
7
- "global_step": 60000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -4688,6 +4688,318 @@
4688
  "eval_samples_per_second": 77.714,
4689
  "eval_steps_per_second": 19.429,
4690
  "step": 60000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4691
  }
4692
  ],
4693
  "logging_steps": 100,
@@ -4707,7 +5019,7 @@
4707
  "attributes": {}
4708
  }
4709
  },
4710
- "total_flos": 1.49487485976576e+18,
4711
  "train_batch_size": 22,
4712
  "trial_name": null,
4713
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.2545454545454545,
6
  "eval_steps": 1000,
7
+ "global_step": 64000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
4688
  "eval_samples_per_second": 77.714,
4689
  "eval_steps_per_second": 19.429,
4690
  "step": 60000
4691
+ },
4692
+ {
4693
+ "epoch": 0.2190909090909091,
4694
+ "grad_norm": 0.16105052828788757,
4695
+ "learning_rate": 0.0004284064105413123,
4696
+ "loss": 2.6820413208007814,
4697
+ "step": 60100
4698
+ },
4699
+ {
4700
+ "epoch": 0.22,
4701
+ "grad_norm": 0.17260073125362396,
4702
+ "learning_rate": 0.00042699149336866053,
4703
+ "loss": 2.705378112792969,
4704
+ "step": 60200
4705
+ },
4706
+ {
4707
+ "epoch": 0.22090909090909092,
4708
+ "grad_norm": 0.1929374486207962,
4709
+ "learning_rate": 0.0004255771733313727,
4710
+ "loss": 2.6583901977539064,
4711
+ "step": 60300
4712
+ },
4713
+ {
4714
+ "epoch": 0.22181818181818183,
4715
+ "grad_norm": 0.15301187336444855,
4716
+ "learning_rate": 0.00042416346199714895,
4717
+ "loss": 2.6433419799804687,
4718
+ "step": 60400
4719
+ },
4720
+ {
4721
+ "epoch": 0.22272727272727272,
4722
+ "grad_norm": 0.16559265553951263,
4723
+ "learning_rate": 0.0004227503709287107,
4724
+ "loss": 2.6537457275390626,
4725
+ "step": 60500
4726
+ },
4727
+ {
4728
+ "epoch": 0.22363636363636363,
4729
+ "grad_norm": 0.16312868893146515,
4730
+ "learning_rate": 0.0004213379116837065,
4731
+ "loss": 2.642680969238281,
4732
+ "step": 60600
4733
+ },
4734
+ {
4735
+ "epoch": 0.22454545454545455,
4736
+ "grad_norm": 0.1596183180809021,
4737
+ "learning_rate": 0.00041992609581461714,
4738
+ "loss": 2.6479568481445312,
4739
+ "step": 60700
4740
+ },
4741
+ {
4742
+ "epoch": 0.22545454545454546,
4743
+ "grad_norm": 0.1691562980413437,
4744
+ "learning_rate": 0.00041851493486866093,
4745
+ "loss": 2.668114013671875,
4746
+ "step": 60800
4747
+ },
4748
+ {
4749
+ "epoch": 0.22636363636363635,
4750
+ "grad_norm": 0.17783690989017487,
4751
+ "learning_rate": 0.00041710444038770006,
4752
+ "loss": 2.6744537353515625,
4753
+ "step": 60900
4754
+ },
4755
+ {
4756
+ "epoch": 0.22727272727272727,
4757
+ "grad_norm": 0.14821407198905945,
4758
+ "learning_rate": 0.00041569462390814507,
4759
+ "loss": 2.6778204345703127,
4760
+ "step": 61000
4761
+ },
4762
+ {
4763
+ "epoch": 0.22727272727272727,
4764
+ "eval_loss": 3.0541512966156006,
4765
+ "eval_runtime": 7.4401,
4766
+ "eval_samples_per_second": 77.419,
4767
+ "eval_steps_per_second": 19.355,
4768
+ "step": 61000
4769
+ },
4770
+ {
4771
+ "epoch": 0.22818181818181818,
4772
+ "grad_norm": 0.15560130774974823,
4773
+ "learning_rate": 0.00041428549696086213,
4774
+ "loss": 2.6814599609375,
4775
+ "step": 61100
4776
+ },
4777
+ {
4778
+ "epoch": 0.2290909090909091,
4779
+ "grad_norm": 0.16080529987812042,
4780
+ "learning_rate": 0.0004128770710710768,
4781
+ "loss": 2.6637908935546877,
4782
+ "step": 61200
4783
+ },
4784
+ {
4785
+ "epoch": 0.23,
4786
+ "grad_norm": 0.15518587827682495,
4787
+ "learning_rate": 0.0004114693577582812,
4788
+ "loss": 2.6588839721679687,
4789
+ "step": 61300
4790
+ },
4791
+ {
4792
+ "epoch": 0.2309090909090909,
4793
+ "grad_norm": 0.16887351870536804,
4794
+ "learning_rate": 0.000410062368536139,
4795
+ "loss": 2.674110107421875,
4796
+ "step": 61400
4797
+ },
4798
+ {
4799
+ "epoch": 0.2318181818181818,
4800
+ "grad_norm": 0.1643567532300949,
4801
+ "learning_rate": 0.0004086561149123919,
4802
+ "loss": 2.6627960205078125,
4803
+ "step": 61500
4804
+ },
4805
+ {
4806
+ "epoch": 0.23272727272727273,
4807
+ "grad_norm": 0.17247343063354492,
4808
+ "learning_rate": 0.0004072506083887647,
4809
+ "loss": 2.680544738769531,
4810
+ "step": 61600
4811
+ },
4812
+ {
4813
+ "epoch": 0.23363636363636364,
4814
+ "grad_norm": 0.17303307354450226,
4815
+ "learning_rate": 0.0004058458604608721,
4816
+ "loss": 2.6711715698242187,
4817
+ "step": 61700
4818
+ },
4819
+ {
4820
+ "epoch": 0.23454545454545456,
4821
+ "grad_norm": 0.15602687001228333,
4822
+ "learning_rate": 0.0004044418826181241,
4823
+ "loss": 2.6567123413085936,
4824
+ "step": 61800
4825
+ },
4826
+ {
4827
+ "epoch": 0.23545454545454544,
4828
+ "grad_norm": 0.17260140180587769,
4829
+ "learning_rate": 0.0004030386863436319,
4830
+ "loss": 2.6413006591796875,
4831
+ "step": 61900
4832
+ },
4833
+ {
4834
+ "epoch": 0.23636363636363636,
4835
+ "grad_norm": 0.15273793041706085,
4836
+ "learning_rate": 0.000401636283114115,
4837
+ "loss": 2.6535302734375,
4838
+ "step": 62000
4839
+ },
4840
+ {
4841
+ "epoch": 0.23636363636363636,
4842
+ "eval_loss": 3.0504443645477295,
4843
+ "eval_runtime": 7.4706,
4844
+ "eval_samples_per_second": 77.102,
4845
+ "eval_steps_per_second": 19.276,
4846
+ "step": 62000
4847
+ },
4848
+ {
4849
+ "epoch": 0.23727272727272727,
4850
+ "grad_norm": 0.1516442894935608,
4851
+ "learning_rate": 0.0004002346843998058,
4852
+ "loss": 2.65826416015625,
4853
+ "step": 62100
4854
+ },
4855
+ {
4856
+ "epoch": 0.2381818181818182,
4857
+ "grad_norm": 0.16887111961841583,
4858
+ "learning_rate": 0.0003988339016643571,
4859
+ "loss": 2.65544189453125,
4860
+ "step": 62200
4861
+ },
4862
+ {
4863
+ "epoch": 0.2390909090909091,
4864
+ "grad_norm": 0.17878231406211853,
4865
+ "learning_rate": 0.00039743394636474735,
4866
+ "loss": 2.6390304565429688,
4867
+ "step": 62300
4868
+ },
4869
+ {
4870
+ "epoch": 0.24,
4871
+ "grad_norm": 0.14858455955982208,
4872
+ "learning_rate": 0.00039603482995118793,
4873
+ "loss": 2.6456539916992186,
4874
+ "step": 62400
4875
+ },
4876
+ {
4877
+ "epoch": 0.2409090909090909,
4878
+ "grad_norm": 0.16330905258655548,
4879
+ "learning_rate": 0.0003946365638670289,
4880
+ "loss": 2.6405880737304686,
4881
+ "step": 62500
4882
+ },
4883
+ {
4884
+ "epoch": 0.24181818181818182,
4885
+ "grad_norm": 0.17122872173786163,
4886
+ "learning_rate": 0.00039323915954866506,
4887
+ "loss": 2.6345315551757813,
4888
+ "step": 62600
4889
+ },
4890
+ {
4891
+ "epoch": 0.24272727272727274,
4892
+ "grad_norm": 0.15139397978782654,
4893
+ "learning_rate": 0.00039184262842544346,
4894
+ "loss": 2.617703857421875,
4895
+ "step": 62700
4896
+ },
4897
+ {
4898
+ "epoch": 0.24363636363636362,
4899
+ "grad_norm": 0.14694419503211975,
4900
+ "learning_rate": 0.0003904469819195686,
4901
+ "loss": 2.6260647583007812,
4902
+ "step": 62800
4903
+ },
4904
+ {
4905
+ "epoch": 0.24454545454545454,
4906
+ "grad_norm": 0.1512179970741272,
4907
+ "learning_rate": 0.00038905223144601044,
4908
+ "loss": 2.6326007080078124,
4909
+ "step": 62900
4910
+ },
4911
+ {
4912
+ "epoch": 0.24545454545454545,
4913
+ "grad_norm": 0.1644383668899536,
4914
+ "learning_rate": 0.00038765838841240964,
4915
+ "loss": 2.655694885253906,
4916
+ "step": 63000
4917
+ },
4918
+ {
4919
+ "epoch": 0.24545454545454545,
4920
+ "eval_loss": 3.050816774368286,
4921
+ "eval_runtime": 7.4114,
4922
+ "eval_samples_per_second": 77.718,
4923
+ "eval_steps_per_second": 19.43,
4924
+ "step": 63000
4925
+ },
4926
+ {
4927
+ "epoch": 0.24636363636363637,
4928
+ "grad_norm": 0.16079971194267273,
4929
+ "learning_rate": 0.00038626546421898547,
4930
+ "loss": 2.6361544799804686,
4931
+ "step": 63100
4932
+ },
4933
+ {
4934
+ "epoch": 0.24727272727272728,
4935
+ "grad_norm": 0.16880013048648834,
4936
+ "learning_rate": 0.0003848734702584417,
4937
+ "loss": 2.6428652954101564,
4938
+ "step": 63200
4939
+ },
4940
+ {
4941
+ "epoch": 0.24818181818181817,
4942
+ "grad_norm": 0.15799115598201752,
4943
+ "learning_rate": 0.00038348241791587377,
4944
+ "loss": 2.63120361328125,
4945
+ "step": 63300
4946
+ },
4947
+ {
4948
+ "epoch": 0.24909090909090909,
4949
+ "grad_norm": 0.1738392412662506,
4950
+ "learning_rate": 0.00038209231856867594,
4951
+ "loss": 2.6429544067382813,
4952
+ "step": 63400
4953
+ },
4954
+ {
4955
+ "epoch": 0.25,
4956
+ "grad_norm": 0.16007262468338013,
4957
+ "learning_rate": 0.0003807031835864474,
4958
+ "loss": 2.6346502685546875,
4959
+ "step": 63500
4960
+ },
4961
+ {
4962
+ "epoch": 0.2509090909090909,
4963
+ "grad_norm": 0.16581037640571594,
4964
+ "learning_rate": 0.0003793150243309002,
4965
+ "loss": 2.625074462890625,
4966
+ "step": 63600
4967
+ },
4968
+ {
4969
+ "epoch": 0.25181818181818183,
4970
+ "grad_norm": 0.14953260123729706,
4971
+ "learning_rate": 0.00037792785215576605,
4972
+ "loss": 2.6320794677734374,
4973
+ "step": 63700
4974
+ },
4975
+ {
4976
+ "epoch": 0.25272727272727274,
4977
+ "grad_norm": 0.15426383912563324,
4978
+ "learning_rate": 0.0003765416784067028,
4979
+ "loss": 2.6317816162109375,
4980
+ "step": 63800
4981
+ },
4982
+ {
4983
+ "epoch": 0.25363636363636366,
4984
+ "grad_norm": 0.1912180483341217,
4985
+ "learning_rate": 0.0003751565144212027,
4986
+ "loss": 2.63588623046875,
4987
+ "step": 63900
4988
+ },
4989
+ {
4990
+ "epoch": 0.2545454545454545,
4991
+ "grad_norm": 0.1402943879365921,
4992
+ "learning_rate": 0.0003737723715284992,
4993
+ "loss": 2.644476013183594,
4994
+ "step": 64000
4995
+ },
4996
+ {
4997
+ "epoch": 0.2545454545454545,
4998
+ "eval_loss": 3.0472707748413086,
4999
+ "eval_runtime": 7.4723,
5000
+ "eval_samples_per_second": 77.085,
5001
+ "eval_steps_per_second": 19.271,
5002
+ "step": 64000
5003
  }
5004
  ],
5005
  "logging_steps": 100,
 
5019
  "attributes": {}
5020
  }
5021
  },
5022
+ "total_flos": 1.594533183750144e+18,
5023
  "train_batch_size": 22,
5024
  "trial_name": null,
5025
  "trial_params": null