Training in progress, step 40000
Browse files- last-checkpoint/optimizer.pt +1 -1
- last-checkpoint/pytorch_model.bin +1 -1
- last-checkpoint/rng_state_0.pth +1 -1
- last-checkpoint/rng_state_1.pth +1 -1
- last-checkpoint/rng_state_2.pth +1 -1
- last-checkpoint/rng_state_3.pth +1 -1
- last-checkpoint/scaler.pt +1 -1
- last-checkpoint/scheduler.pt +1 -1
- last-checkpoint/trainer_state.json +62 -2
- pytorch_model.bin +1 -1
last-checkpoint/optimizer.pt
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 402587859
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:714bc2836535f20e05f117ef510ba9edbcc38031d847d6dde87e2ad9981ac35e
|
3 |
size 402587859
|
last-checkpoint/pytorch_model.bin
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 201355195
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:85732efbc6097a0b5ae016e5090bf88df65e654b197a3f31c356adab5b7dec57
|
3 |
size 201355195
|
last-checkpoint/rng_state_0.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 14503
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:9b59fc745c089f269945e1b712ad0318c7afe64157e16f9055fcfa94c3d52df7
|
3 |
size 14503
|
last-checkpoint/rng_state_1.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 14503
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:536d3bd2c7b76acdaa9b1920f80987cb6164c9cc562f18aeb6fd3bb4d8827515
|
3 |
size 14503
|
last-checkpoint/rng_state_2.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 14503
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:84fb548de68dfed8cbf315b69645a194052dc91188e17e5094f78bcfe6db676b
|
3 |
size 14503
|
last-checkpoint/rng_state_3.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 14503
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:edf93609c5953d3e51ed27db30ff3c705ffcbd852b78274a073ac9bef5457bca
|
3 |
size 14503
|
last-checkpoint/scaler.pt
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 559
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:776d55c5ff3fd8d529eed1dd2bed058425c39df95f63ea1f8a3fb324f5e2c149
|
3 |
size 559
|
last-checkpoint/scheduler.pt
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 623
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:9a7588470af2a401d553417dd2cec6f6fe08126f49e987c2120567ee0b1332ed
|
3 |
size 623
|
last-checkpoint/trainer_state.json
CHANGED
@@ -1,8 +1,8 @@
|
|
1 |
{
|
2 |
"best_metric": null,
|
3 |
"best_model_checkpoint": null,
|
4 |
-
"epoch": 0.
|
5 |
-
"global_step":
|
6 |
"is_hyper_param_search": false,
|
7 |
"is_local_process_zero": true,
|
8 |
"is_world_process_zero": true,
|
@@ -426,6 +426,66 @@
|
|
426 |
"learning_rate": 0.00014831551629313194,
|
427 |
"loss": 0.3711,
|
428 |
"step": 35000
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
429 |
}
|
430 |
],
|
431 |
"max_steps": 500000,
|
|
|
1 |
{
|
2 |
"best_metric": null,
|
3 |
"best_model_checkpoint": null,
|
4 |
+
"epoch": 0.6808452694019625,
|
5 |
+
"global_step": 40000,
|
6 |
"is_hyper_param_search": false,
|
7 |
"is_local_process_zero": true,
|
8 |
"is_world_process_zero": true,
|
|
|
426 |
"learning_rate": 0.00014831551629313194,
|
427 |
"loss": 0.3711,
|
428 |
"step": 35000
|
429 |
+
},
|
430 |
+
{
|
431 |
+
"epoch": 0.6,
|
432 |
+
"learning_rate": 0.00014826722592314076,
|
433 |
+
"loss": 0.3709,
|
434 |
+
"step": 35500
|
435 |
+
},
|
436 |
+
{
|
437 |
+
"epoch": 0.61,
|
438 |
+
"learning_rate": 0.00014821826178319038,
|
439 |
+
"loss": 0.3709,
|
440 |
+
"step": 36000
|
441 |
+
},
|
442 |
+
{
|
443 |
+
"epoch": 0.62,
|
444 |
+
"learning_rate": 0.0001481687243030059,
|
445 |
+
"loss": 0.3703,
|
446 |
+
"step": 36500
|
447 |
+
},
|
448 |
+
{
|
449 |
+
"epoch": 0.63,
|
450 |
+
"learning_rate": 0.00014811841542465117,
|
451 |
+
"loss": 0.3698,
|
452 |
+
"step": 37000
|
453 |
+
},
|
454 |
+
{
|
455 |
+
"epoch": 0.64,
|
456 |
+
"learning_rate": 0.00014806743424503673,
|
457 |
+
"loss": 0.3697,
|
458 |
+
"step": 37500
|
459 |
+
},
|
460 |
+
{
|
461 |
+
"epoch": 0.65,
|
462 |
+
"learning_rate": 0.0001480157812673262,
|
463 |
+
"loss": 0.3692,
|
464 |
+
"step": 38000
|
465 |
+
},
|
466 |
+
{
|
467 |
+
"epoch": 0.66,
|
468 |
+
"learning_rate": 0.0001479634570013137,
|
469 |
+
"loss": 0.369,
|
470 |
+
"step": 38500
|
471 |
+
},
|
472 |
+
{
|
473 |
+
"epoch": 0.66,
|
474 |
+
"learning_rate": 0.00014791046196341849,
|
475 |
+
"loss": 0.3689,
|
476 |
+
"step": 39000
|
477 |
+
},
|
478 |
+
{
|
479 |
+
"epoch": 0.67,
|
480 |
+
"learning_rate": 0.0001478567966766803,
|
481 |
+
"loss": 0.3685,
|
482 |
+
"step": 39500
|
483 |
+
},
|
484 |
+
{
|
485 |
+
"epoch": 0.68,
|
486 |
+
"learning_rate": 0.0001478024616707538,
|
487 |
+
"loss": 0.3683,
|
488 |
+
"step": 40000
|
489 |
}
|
490 |
],
|
491 |
"max_steps": 500000,
|
pytorch_model.bin
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 201355195
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:85732efbc6097a0b5ae016e5090bf88df65e654b197a3f31c356adab5b7dec57
|
3 |
size 201355195
|