CodeIsAbstract commited on
Commit
3e5e9d7
·
verified ·
1 Parent(s): 1576a6d

Training in progress, step 72000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e65c628bc21a488cc995a999f4ebe043aded7453145c4be01688da0234dc825a
3
  size 469337272
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bc5ba4b46dd67e2a49a16bd82facc11222fece8f84dc62617b2e908b04f6c169
3
  size 469337272
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4a4b726e70e9f1b2d482d4b906ff82dfb782045da636caaeed5e73095f82fd90
3
  size 938825803
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:be2468544f1374c838edfcc2826cfea73f333398467978c2e92f3248e38f729a
3
  size 938825803
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3cab5a61edbb7c8a2083270aa3aceaa772762aaf6c804e164b525561621e196f
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a37fbaec01f56c5e37b1a1afe260103cefbac9e7d19fccf924630793e4ba7a84
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ee827426ba5e510531f97228be83448aeb6e47b62607221c379bc9913032527f
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:899e7c047ffeacb320a881df994af73296a60acae9cbaa8b910682541dc7eed6
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.2909090909090909,
6
  "eval_steps": 1000,
7
- "global_step": 68000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -5312,6 +5312,318 @@
5312
  "eval_samples_per_second": 77.454,
5313
  "eval_steps_per_second": 19.363,
5314
  "step": 68000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5315
  }
5316
  ],
5317
  "logging_steps": 100,
@@ -5331,7 +5643,7 @@
5331
  "attributes": {}
5332
  }
5333
  },
5334
- "total_flos": 1.694191507734528e+18,
5335
  "train_batch_size": 22,
5336
  "trial_name": null,
5337
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.32727272727272727,
6
  "eval_steps": 1000,
7
+ "global_step": 72000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
5312
  "eval_samples_per_second": 77.454,
5313
  "eval_steps_per_second": 19.363,
5314
  "step": 68000
5315
+ },
5316
+ {
5317
+ "epoch": 0.2918181818181818,
5318
+ "grad_norm": 0.18501834571361542,
5319
+ "learning_rate": 0.0003180402541276048,
5320
+ "loss": 2.650793151855469,
5321
+ "step": 68100
5322
+ },
5323
+ {
5324
+ "epoch": 0.2927272727272727,
5325
+ "grad_norm": 0.14561575651168823,
5326
+ "learning_rate": 0.00031670910433338273,
5327
+ "loss": 2.6040188598632814,
5328
+ "step": 68200
5329
+ },
5330
+ {
5331
+ "epoch": 0.29363636363636364,
5332
+ "grad_norm": 0.16025520861148834,
5333
+ "learning_rate": 0.00031537945367235393,
5334
+ "loss": 2.6409112548828126,
5335
+ "step": 68300
5336
+ },
5337
+ {
5338
+ "epoch": 0.29454545454545455,
5339
+ "grad_norm": 0.15475542843341827,
5340
+ "learning_rate": 0.00031405131301970936,
5341
+ "loss": 2.6054486083984374,
5342
+ "step": 68400
5343
+ },
5344
+ {
5345
+ "epoch": 0.29545454545454547,
5346
+ "grad_norm": 0.15857645869255066,
5347
+ "learning_rate": 0.0003127246932382891,
5348
+ "loss": 2.616719055175781,
5349
+ "step": 68500
5350
+ },
5351
+ {
5352
+ "epoch": 0.2963636363636364,
5353
+ "grad_norm": 0.19392845034599304,
5354
+ "learning_rate": 0.0003113996051784945,
5355
+ "loss": 2.60544921875,
5356
+ "step": 68600
5357
+ },
5358
+ {
5359
+ "epoch": 0.2972727272727273,
5360
+ "grad_norm": 0.19967029988765717,
5361
+ "learning_rate": 0.00031007605967819896,
5362
+ "loss": 2.6308856201171875,
5363
+ "step": 68700
5364
+ },
5365
+ {
5366
+ "epoch": 0.29818181818181816,
5367
+ "grad_norm": 0.19477345049381256,
5368
+ "learning_rate": 0.00030875406756265887,
5369
+ "loss": 2.6351129150390626,
5370
+ "step": 68800
5371
+ },
5372
+ {
5373
+ "epoch": 0.2990909090909091,
5374
+ "grad_norm": 0.15118417143821716,
5375
+ "learning_rate": 0.000307433639644426,
5376
+ "loss": 2.614127197265625,
5377
+ "step": 68900
5378
+ },
5379
+ {
5380
+ "epoch": 0.3,
5381
+ "grad_norm": 0.15570956468582153,
5382
+ "learning_rate": 0.0003061147867232582,
5383
+ "loss": 2.608861389160156,
5384
+ "step": 69000
5385
+ },
5386
+ {
5387
+ "epoch": 0.3,
5388
+ "eval_loss": 3.0342400074005127,
5389
+ "eval_runtime": 7.4253,
5390
+ "eval_samples_per_second": 77.572,
5391
+ "eval_steps_per_second": 19.393,
5392
+ "step": 69000
5393
+ },
5394
+ {
5395
+ "epoch": 0.3009090909090909,
5396
+ "grad_norm": 0.1571355015039444,
5397
+ "learning_rate": 0.0003047975195860318,
5398
+ "loss": 2.606195983886719,
5399
+ "step": 69100
5400
+ },
5401
+ {
5402
+ "epoch": 0.3018181818181818,
5403
+ "grad_norm": 0.15814636647701263,
5404
+ "learning_rate": 0.0003034818490066527,
5405
+ "loss": 2.5954779052734374,
5406
+ "step": 69200
5407
+ },
5408
+ {
5409
+ "epoch": 0.30272727272727273,
5410
+ "grad_norm": 0.14323082566261292,
5411
+ "learning_rate": 0.00030216778574596883,
5412
+ "loss": 2.6024984741210937,
5413
+ "step": 69300
5414
+ },
5415
+ {
5416
+ "epoch": 0.30363636363636365,
5417
+ "grad_norm": 0.17691482603549957,
5418
+ "learning_rate": 0.00030085534055168184,
5419
+ "loss": 2.602333984375,
5420
+ "step": 69400
5421
+ },
5422
+ {
5423
+ "epoch": 0.30454545454545456,
5424
+ "grad_norm": 0.17705939710140228,
5425
+ "learning_rate": 0.00029954452415825896,
5426
+ "loss": 2.628582763671875,
5427
+ "step": 69500
5428
+ },
5429
+ {
5430
+ "epoch": 0.3054545454545455,
5431
+ "grad_norm": 0.16018672287464142,
5432
+ "learning_rate": 0.00029823534728684587,
5433
+ "loss": 2.617008056640625,
5434
+ "step": 69600
5435
+ },
5436
+ {
5437
+ "epoch": 0.30636363636363634,
5438
+ "grad_norm": 0.16868868470191956,
5439
+ "learning_rate": 0.0002969278206451786,
5440
+ "loss": 2.597875671386719,
5441
+ "step": 69700
5442
+ },
5443
+ {
5444
+ "epoch": 0.30727272727272725,
5445
+ "grad_norm": 0.1678052544593811,
5446
+ "learning_rate": 0.0002956219549274957,
5447
+ "loss": 2.59896728515625,
5448
+ "step": 69800
5449
+ },
5450
+ {
5451
+ "epoch": 0.30818181818181817,
5452
+ "grad_norm": 0.15835554897785187,
5453
+ "learning_rate": 0.00029431776081445104,
5454
+ "loss": 2.6005072021484374,
5455
+ "step": 69900
5456
+ },
5457
+ {
5458
+ "epoch": 0.3090909090909091,
5459
+ "grad_norm": 0.148189514875412,
5460
+ "learning_rate": 0.0002930152489730271,
5461
+ "loss": 2.596640625,
5462
+ "step": 70000
5463
+ },
5464
+ {
5465
+ "epoch": 0.3090909090909091,
5466
+ "eval_loss": 3.034733772277832,
5467
+ "eval_runtime": 7.4478,
5468
+ "eval_samples_per_second": 77.338,
5469
+ "eval_steps_per_second": 19.335,
5470
+ "step": 70000
5471
+ },
5472
+ {
5473
+ "epoch": 0.31,
5474
+ "grad_norm": 0.1630752682685852,
5475
+ "learning_rate": 0.0002917144300564461,
5476
+ "loss": 2.6199655151367187,
5477
+ "step": 70100
5478
+ },
5479
+ {
5480
+ "epoch": 0.3109090909090909,
5481
+ "grad_norm": 0.16336363554000854,
5482
+ "learning_rate": 0.00029041531470408465,
5483
+ "loss": 2.595935974121094,
5484
+ "step": 70200
5485
+ },
5486
+ {
5487
+ "epoch": 0.3118181818181818,
5488
+ "grad_norm": 0.19114582240581512,
5489
+ "learning_rate": 0.00028911791354138574,
5490
+ "loss": 2.6245462036132814,
5491
+ "step": 70300
5492
+ },
5493
+ {
5494
+ "epoch": 0.31272727272727274,
5495
+ "grad_norm": 0.15327058732509613,
5496
+ "learning_rate": 0.0002878222371797716,
5497
+ "loss": 2.606686706542969,
5498
+ "step": 70400
5499
+ },
5500
+ {
5501
+ "epoch": 0.31363636363636366,
5502
+ "grad_norm": 0.16627787053585052,
5503
+ "learning_rate": 0.00028652829621655793,
5504
+ "loss": 2.6142779541015626,
5505
+ "step": 70500
5506
+ },
5507
+ {
5508
+ "epoch": 0.3145454545454546,
5509
+ "grad_norm": 0.17076851427555084,
5510
+ "learning_rate": 0.0002852361012348663,
5511
+ "loss": 2.6306427001953123,
5512
+ "step": 70600
5513
+ },
5514
+ {
5515
+ "epoch": 0.31545454545454543,
5516
+ "grad_norm": 0.15905171632766724,
5517
+ "learning_rate": 0.0002839456628035381,
5518
+ "loss": 2.646249084472656,
5519
+ "step": 70700
5520
+ },
5521
+ {
5522
+ "epoch": 0.31636363636363635,
5523
+ "grad_norm": 0.15757545828819275,
5524
+ "learning_rate": 0.000282656991477048,
5525
+ "loss": 2.6342782592773437,
5526
+ "step": 70800
5527
+ },
5528
+ {
5529
+ "epoch": 0.31727272727272726,
5530
+ "grad_norm": 0.1566353142261505,
5531
+ "learning_rate": 0.000281370097795417,
5532
+ "loss": 2.5926519775390626,
5533
+ "step": 70900
5534
+ },
5535
+ {
5536
+ "epoch": 0.3181818181818182,
5537
+ "grad_norm": 0.1587604433298111,
5538
+ "learning_rate": 0.0002800849922841273,
5539
+ "loss": 2.62255615234375,
5540
+ "step": 71000
5541
+ },
5542
+ {
5543
+ "epoch": 0.3181818181818182,
5544
+ "eval_loss": 3.0347187519073486,
5545
+ "eval_runtime": 7.4419,
5546
+ "eval_samples_per_second": 77.4,
5547
+ "eval_steps_per_second": 19.35,
5548
+ "step": 71000
5549
+ },
5550
+ {
5551
+ "epoch": 0.3190909090909091,
5552
+ "grad_norm": 0.15599116683006287,
5553
+ "learning_rate": 0.0002788016854540356,
5554
+ "loss": 2.6368426513671874,
5555
+ "step": 71100
5556
+ },
5557
+ {
5558
+ "epoch": 0.32,
5559
+ "grad_norm": 0.15771758556365967,
5560
+ "learning_rate": 0.0002775201878012872,
5561
+ "loss": 2.6378125,
5562
+ "step": 71200
5563
+ },
5564
+ {
5565
+ "epoch": 0.3209090909090909,
5566
+ "grad_norm": 0.16494904458522797,
5567
+ "learning_rate": 0.0002762405098072303,
5568
+ "loss": 2.6372637939453125,
5569
+ "step": 71300
5570
+ },
5571
+ {
5572
+ "epoch": 0.32181818181818184,
5573
+ "grad_norm": 0.18344129621982574,
5574
+ "learning_rate": 0.0002749626619383296,
5575
+ "loss": 2.6274960327148436,
5576
+ "step": 71400
5577
+ },
5578
+ {
5579
+ "epoch": 0.32272727272727275,
5580
+ "grad_norm": 0.19829513132572174,
5581
+ "learning_rate": 0.00027368665464608174,
5582
+ "loss": 2.652212829589844,
5583
+ "step": 71500
5584
+ },
5585
+ {
5586
+ "epoch": 0.3236363636363636,
5587
+ "grad_norm": 0.170766219496727,
5588
+ "learning_rate": 0.0002724124983669292,
5589
+ "loss": 2.6236361694335937,
5590
+ "step": 71600
5591
+ },
5592
+ {
5593
+ "epoch": 0.3245454545454545,
5594
+ "grad_norm": 0.15507978200912476,
5595
+ "learning_rate": 0.0002711402035221751,
5596
+ "loss": 2.612099304199219,
5597
+ "step": 71700
5598
+ },
5599
+ {
5600
+ "epoch": 0.32545454545454544,
5601
+ "grad_norm": 0.17562313377857208,
5602
+ "learning_rate": 0.00026986978051789804,
5603
+ "loss": 2.651109313964844,
5604
+ "step": 71800
5605
+ },
5606
+ {
5607
+ "epoch": 0.32636363636363636,
5608
+ "grad_norm": 0.16349823772907257,
5609
+ "learning_rate": 0.0002686012397448662,
5610
+ "loss": 2.652076416015625,
5611
+ "step": 71900
5612
+ },
5613
+ {
5614
+ "epoch": 0.32727272727272727,
5615
+ "grad_norm": 0.1562015861272812,
5616
+ "learning_rate": 0.00026733459157845385,
5617
+ "loss": 2.6174331665039063,
5618
+ "step": 72000
5619
+ },
5620
+ {
5621
+ "epoch": 0.32727272727272727,
5622
+ "eval_loss": 3.030982255935669,
5623
+ "eval_runtime": 7.4103,
5624
+ "eval_samples_per_second": 77.729,
5625
+ "eval_steps_per_second": 19.432,
5626
+ "step": 72000
5627
  }
5628
  ],
5629
  "logging_steps": 100,
 
5643
  "attributes": {}
5644
  }
5645
  },
5646
+ "total_flos": 1.793849831718912e+18,
5647
  "train_batch_size": 22,
5648
  "trial_name": null,
5649
  "trial_params": null