CodeIsAbstract commited on
Commit
42ad56a
·
verified ·
1 Parent(s): e7f48ee

Training in progress, step 6000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dedf215e9b506136c4eafaa5ca983fab54ed4eb2af03d61c3f95ec65d05da787
3
  size 579824888
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:82a943209b657b15ff00004efcaa88da19c9f1fd4603f11587c023fc47eda605
3
  size 579824888
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fa05fea595e28a91c765e7e95378a3f84bc9106fc2e6146f83186626bae9e711
3
  size 1159794763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:13ae1bff671c06244b8db21ff756e241dc09da71f9f9dc4f2ec99c071e13b99f
3
  size 1159794763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9e06dfbf969388fb27efc692631d298abe0d1c57738a0c64598fe664b1ac46b1
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:15bd918144b9671ae295d46e463180bdad2e73e51a7670636feb8fc58e14c426
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:77f6f8dca22790e21b49d4a3e5b2c2d34a14c2511eb61bf406f69b9611e05181
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2666cc7db576490d8041b6acfda665e20af01ea5f7120fdf1d1c04b59674d0ca
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.08,
6
  "eval_steps": 1000,
7
- "global_step": 4000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -324,6 +324,164 @@
324
  "eval_samples_per_second": 307.255,
325
  "eval_steps_per_second": 19.312,
326
  "step": 4000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
327
  }
328
  ],
329
  "logging_steps": 100,
@@ -343,7 +501,7 @@
343
  "attributes": {}
344
  }
345
  },
346
- "total_flos": 1.5679843467264e+17,
347
  "train_batch_size": 120,
348
  "trial_name": null,
349
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.12,
6
  "eval_steps": 1000,
7
+ "global_step": 6000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
324
  "eval_samples_per_second": 307.255,
325
  "eval_steps_per_second": 19.312,
326
  "step": 4000
327
+ },
328
+ {
329
+ "epoch": 0.082,
330
+ "grad_norm": 0.20376908779144287,
331
+ "learning_rate": 0.000984595734445656,
332
+ "loss": 3.8685012817382813,
333
+ "step": 4100
334
+ },
335
+ {
336
+ "epoch": 0.084,
337
+ "grad_norm": 0.18422065675258636,
338
+ "learning_rate": 0.0009838099869394353,
339
+ "loss": 3.853851318359375,
340
+ "step": 4200
341
+ },
342
+ {
343
+ "epoch": 0.086,
344
+ "grad_norm": 0.19115814566612244,
345
+ "learning_rate": 0.0009830050243260172,
346
+ "loss": 3.8678164672851563,
347
+ "step": 4300
348
+ },
349
+ {
350
+ "epoch": 0.088,
351
+ "grad_norm": 0.20425236225128174,
352
+ "learning_rate": 0.0009821808785754791,
353
+ "loss": 3.8287237548828124,
354
+ "step": 4400
355
+ },
356
+ {
357
+ "epoch": 0.09,
358
+ "grad_norm": 0.1944635510444641,
359
+ "learning_rate": 0.0009813375824197811,
360
+ "loss": 3.8451242065429687,
361
+ "step": 4500
362
+ },
363
+ {
364
+ "epoch": 0.092,
365
+ "grad_norm": 0.17947296798229218,
366
+ "learning_rate": 0.0009804751693514645,
367
+ "loss": 3.845400695800781,
368
+ "step": 4600
369
+ },
370
+ {
371
+ "epoch": 0.094,
372
+ "grad_norm": 0.20076608657836914,
373
+ "learning_rate": 0.0009795936736223222,
374
+ "loss": 3.8257626342773436,
375
+ "step": 4700
376
+ },
377
+ {
378
+ "epoch": 0.096,
379
+ "grad_norm": 0.18554596602916718,
380
+ "learning_rate": 0.0009786931302420386,
381
+ "loss": 3.808670959472656,
382
+ "step": 4800
383
+ },
384
+ {
385
+ "epoch": 0.098,
386
+ "grad_norm": 0.23913995921611786,
387
+ "learning_rate": 0.000977773574976799,
388
+ "loss": 3.810527038574219,
389
+ "step": 4900
390
+ },
391
+ {
392
+ "epoch": 0.1,
393
+ "grad_norm": 0.18414103984832764,
394
+ "learning_rate": 0.0009768350443478686,
395
+ "loss": 3.7990948486328127,
396
+ "step": 5000
397
+ },
398
+ {
399
+ "epoch": 0.1,
400
+ "eval_accuracy": 0.33500293894949945,
401
+ "eval_loss": 3.7702300548553467,
402
+ "eval_runtime": 6.4137,
403
+ "eval_samples_per_second": 302.634,
404
+ "eval_steps_per_second": 19.022,
405
+ "step": 5000
406
+ },
407
+ {
408
+ "epoch": 0.102,
409
+ "grad_norm": 0.20944379270076752,
410
+ "learning_rate": 0.0009758775756301432,
411
+ "loss": 3.798116455078125,
412
+ "step": 5100
413
+ },
414
+ {
415
+ "epoch": 0.104,
416
+ "grad_norm": 0.18056762218475342,
417
+ "learning_rate": 0.0009749012068506673,
418
+ "loss": 3.7997445678710937,
419
+ "step": 5200
420
+ },
421
+ {
422
+ "epoch": 0.106,
423
+ "grad_norm": 0.20922255516052246,
424
+ "learning_rate": 0.000973905976787125,
425
+ "loss": 3.7765731811523438,
426
+ "step": 5300
427
+ },
428
+ {
429
+ "epoch": 0.108,
430
+ "grad_norm": 0.19328580796718597,
431
+ "learning_rate": 0.0009728919249662993,
432
+ "loss": 3.729337463378906,
433
+ "step": 5400
434
+ },
435
+ {
436
+ "epoch": 0.11,
437
+ "grad_norm": 0.2031419277191162,
438
+ "learning_rate": 0.0009718590916625022,
439
+ "loss": 3.744057922363281,
440
+ "step": 5500
441
+ },
442
+ {
443
+ "epoch": 0.112,
444
+ "grad_norm": 0.19743800163269043,
445
+ "learning_rate": 0.0009708075178959756,
446
+ "loss": 3.7054071044921875,
447
+ "step": 5600
448
+ },
449
+ {
450
+ "epoch": 0.114,
451
+ "grad_norm": 0.1741236299276352,
452
+ "learning_rate": 0.0009697372454312618,
453
+ "loss": 3.699843444824219,
454
+ "step": 5700
455
+ },
456
+ {
457
+ "epoch": 0.116,
458
+ "grad_norm": 0.1799750030040741,
459
+ "learning_rate": 0.0009686483167755448,
460
+ "loss": 3.7227572631835937,
461
+ "step": 5800
462
+ },
463
+ {
464
+ "epoch": 0.118,
465
+ "grad_norm": 0.2036087065935135,
466
+ "learning_rate": 0.000967540775176962,
467
+ "loss": 3.6742709350585936,
468
+ "step": 5900
469
+ },
470
+ {
471
+ "epoch": 0.12,
472
+ "grad_norm": 0.21463191509246826,
473
+ "learning_rate": 0.0009664146646228871,
474
+ "loss": 3.6995980834960935,
475
+ "step": 6000
476
+ },
477
+ {
478
+ "epoch": 0.12,
479
+ "eval_accuracy": 0.34141821705074654,
480
+ "eval_loss": 3.711214303970337,
481
+ "eval_runtime": 6.3349,
482
+ "eval_samples_per_second": 306.4,
483
+ "eval_steps_per_second": 19.259,
484
+ "step": 6000
485
  }
486
  ],
487
  "logging_steps": 100,
 
501
  "attributes": {}
502
  }
503
  },
504
+ "total_flos": 2.3519765200896e+17,
505
  "train_batch_size": 120,
506
  "trial_name": null,
507
  "trial_params": null