CodeIsAbstract commited on
Commit
560097b
·
verified ·
1 Parent(s): c7bdb76

Training in progress, step 10000, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ebb0d1344d85e1fc40cd8989c7c6db8bce3a550c89c85b825448ecbec3b9d6c1
3
  size 1600779
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ff5f848615af7284182bf34bb526cf66a4611210963f90a45f3c9021871e1d85
3
  size 1600779
last-checkpoint/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2abc969de4ce645f9d76eb977581465612165dd82ab373f63378a5f647630661
3
  size 621149155
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2b4c790553f3e4092f7857a64225265bb38246a5d8a25fe76b1107d5f0509e1f
3
  size 621149155
last-checkpoint/rng_state_0.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a2f17c67fa754158f3c03c5f9856c1aa1a0efb60e8b2ba7ff0933cd7e615deac
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:189ca5f500a151c520561b912184d219fb4a14e37e0545ece3a7916f40a7b6f6
3
  size 14469
last-checkpoint/rng_state_1.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:418ff55ed0dbe835354a2c0bf4053deea4eb6d628f7fb25addc88d7dd433e1a0
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:585e6d762d92f384f163db69802c96c5f70ee11af325886531da1e25dbdd3c46
3
  size 14469
last-checkpoint/rng_state_2.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7a35f8395aebcabd2d2ff704dcd0a2516f18acacf15eaa6f6de4d80c3f5a64ac
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:23411dc088849538602b2c97c02e11f26fb822081e4a284d2b43ea9891ec5924
3
  size 14469
last-checkpoint/rng_state_3.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:20538af5e8a8c67556fafaef54a8d5e99ae679aaa736d860c04c1a2e37c3c03c
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a96d850263b64569ae2096b8d72bc91a5ebdd71deb02d3267c85822dd6891262
3
  size 14469
last-checkpoint/rng_state_4.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7766e34119d4055998fe3273163a12ee2907d627abe9c4714c8b112cc18d1fb5
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:14acad3fcc328519114bc488b787139326de9b018d7a082a9a787fbad2c61513
3
  size 14469
last-checkpoint/rng_state_5.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b86ea660b800de8c8f56b525f1c8ec70de64ab61f7f595cc6d7a9eec27f020ed
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4d9d4bb8960e987a9f4826d17f0ffbf80325efd81a2b4fbccc2aba1b74b8cb51
3
  size 14469
last-checkpoint/rng_state_6.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:afea3cb817289360580454c91b5d591d50af9204d37e43c3818d916f106c73a9
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ca2326a5fbe6bf3c05da0cff924b9b40c864d3c0c3f624c13f7edec4700ef239
3
  size 14469
last-checkpoint/rng_state_7.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2f608549c3734c6d52376589878069a475672b0ca9e5e32be42d15b90c85b8d1
3
  size 14469
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:df1ed186896b3430710bec5fd2c1b85c275cbb3ef74ad87064fdc0302116db5c
3
  size 14469
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e9567d76a2efaa946d19913fd9fe6f6f70f6f7101d7dcb8866c09743534f2d4d
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:833776bf99a0597a4bf38df86be38e8ac151736eba1d084c9cbb84cab1caf4e9
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.16,
6
  "eval_steps": 1000,
7
- "global_step": 8000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
@@ -640,6 +640,164 @@
640
  "eval_samples_per_second": 38.613,
641
  "eval_steps_per_second": 0.618,
642
  "step": 8000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
643
  }
644
  ],
645
  "logging_steps": 100,
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.2,
6
  "eval_steps": 1000,
7
+ "global_step": 10000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": false,
10
  "is_world_process_zero": true,
 
640
  "eval_samples_per_second": 38.613,
641
  "eval_steps_per_second": 0.618,
642
  "step": 8000
643
+ },
644
+ {
645
+ "epoch": 0.162,
646
+ "grad_norm": 1.4030160903930664,
647
+ "learning_rate": 0.0019322148017972016,
648
+ "loss": 32.8396,
649
+ "step": 8100
650
+ },
651
+ {
652
+ "epoch": 0.164,
653
+ "grad_norm": 2.127406597137451,
654
+ "learning_rate": 0.001929800831168135,
655
+ "loss": 32.9587,
656
+ "step": 8200
657
+ },
658
+ {
659
+ "epoch": 0.166,
660
+ "grad_norm": 1.437472939491272,
661
+ "learning_rate": 0.001927346188038576,
662
+ "loss": 33.9238,
663
+ "step": 8300
664
+ },
665
+ {
666
+ "epoch": 0.168,
667
+ "grad_norm": 1.5702464580535889,
668
+ "learning_rate": 0.0019248509797825672,
669
+ "loss": 33.8778,
670
+ "step": 8400
671
+ },
672
+ {
673
+ "epoch": 0.17,
674
+ "grad_norm": 1.6056277751922607,
675
+ "learning_rate": 0.0019223153155486009,
676
+ "loss": 33.6902,
677
+ "step": 8500
678
+ },
679
+ {
680
+ "epoch": 0.172,
681
+ "grad_norm": 1.6342443227767944,
682
+ "learning_rate": 0.0019197393062548454,
683
+ "loss": 33.2568,
684
+ "step": 8600
685
+ },
686
+ {
687
+ "epoch": 0.174,
688
+ "grad_norm": 1.5210442543029785,
689
+ "learning_rate": 0.0019171230645842923,
690
+ "loss": 33.2864,
691
+ "step": 8700
692
+ },
693
+ {
694
+ "epoch": 0.176,
695
+ "grad_norm": 2.7911336421966553,
696
+ "learning_rate": 0.0019144667049798272,
697
+ "loss": 32.8163,
698
+ "step": 8800
699
+ },
700
+ {
701
+ "epoch": 0.178,
702
+ "grad_norm": 1.7785000801086426,
703
+ "learning_rate": 0.0019117703436392253,
704
+ "loss": 32.6067,
705
+ "step": 8900
706
+ },
707
+ {
708
+ "epoch": 0.18,
709
+ "grad_norm": 1.7542123794555664,
710
+ "learning_rate": 0.001909034098510066,
711
+ "loss": 33.2589,
712
+ "step": 9000
713
+ },
714
+ {
715
+ "epoch": 0.18,
716
+ "eval_accuracy": 0.2880078277886497,
717
+ "eval_loss": 33.28087615966797,
718
+ "eval_runtime": 3.2266,
719
+ "eval_samples_per_second": 38.74,
720
+ "eval_steps_per_second": 0.62,
721
+ "step": 9000
722
+ },
723
+ {
724
+ "epoch": 0.182,
725
+ "grad_norm": 1.2243378162384033,
726
+ "learning_rate": 0.001906258089284576,
727
+ "loss": 33.5476,
728
+ "step": 9100
729
+ },
730
+ {
731
+ "epoch": 0.184,
732
+ "grad_norm": 1.503909707069397,
733
+ "learning_rate": 0.0019034424373943915,
734
+ "loss": 33.7592,
735
+ "step": 9200
736
+ },
737
+ {
738
+ "epoch": 0.186,
739
+ "grad_norm": 1.6347205638885498,
740
+ "learning_rate": 0.0019005872660052478,
741
+ "loss": 33.2045,
742
+ "step": 9300
743
+ },
744
+ {
745
+ "epoch": 0.188,
746
+ "grad_norm": 1.5243566036224365,
747
+ "learning_rate": 0.001897692700011591,
748
+ "loss": 32.841,
749
+ "step": 9400
750
+ },
751
+ {
752
+ "epoch": 0.19,
753
+ "grad_norm": 1.8256958723068237,
754
+ "learning_rate": 0.0018947588660311143,
755
+ "loss": 32.7556,
756
+ "step": 9500
757
+ },
758
+ {
759
+ "epoch": 0.192,
760
+ "grad_norm": 1.5017831325531006,
761
+ "learning_rate": 0.0018917858923992211,
762
+ "loss": 32.2029,
763
+ "step": 9600
764
+ },
765
+ {
766
+ "epoch": 0.194,
767
+ "grad_norm": 2.1232316493988037,
768
+ "learning_rate": 0.0018887739091634085,
769
+ "loss": 32.9246,
770
+ "step": 9700
771
+ },
772
+ {
773
+ "epoch": 0.196,
774
+ "grad_norm": 2.373900890350342,
775
+ "learning_rate": 0.0018857230480775807,
776
+ "loss": 33.4374,
777
+ "step": 9800
778
+ },
779
+ {
780
+ "epoch": 0.198,
781
+ "grad_norm": 1.6649442911148071,
782
+ "learning_rate": 0.0018826334425962855,
783
+ "loss": 33.1861,
784
+ "step": 9900
785
+ },
786
+ {
787
+ "epoch": 0.2,
788
+ "grad_norm": 2.6526882648468018,
789
+ "learning_rate": 0.0018795052278688753,
790
+ "loss": 33.3808,
791
+ "step": 10000
792
+ },
793
+ {
794
+ "epoch": 0.2,
795
+ "eval_accuracy": 0.2909138943248532,
796
+ "eval_loss": 33.03311538696289,
797
+ "eval_runtime": 3.1811,
798
+ "eval_samples_per_second": 39.294,
799
+ "eval_steps_per_second": 0.629,
800
+ "step": 10000
801
  }
802
  ],
803
  "logging_steps": 100,