|
{ |
|
"best_metric": NaN, |
|
"best_model_checkpoint": "miner_id_24/checkpoint-50", |
|
"epoch": 0.31689443454149335, |
|
"eval_steps": 25, |
|
"global_step": 50, |
|
"is_hyper_param_search": false, |
|
"is_local_process_zero": true, |
|
"is_world_process_zero": true, |
|
"log_history": [ |
|
{ |
|
"epoch": 0.006337888690829867, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00015, |
|
"loss": 0.0, |
|
"step": 1 |
|
}, |
|
{ |
|
"epoch": 0.006337888690829867, |
|
"eval_loss": NaN, |
|
"eval_runtime": 3.4423, |
|
"eval_samples_per_second": 14.525, |
|
"eval_steps_per_second": 3.776, |
|
"step": 1 |
|
}, |
|
{ |
|
"epoch": 0.012675777381659734, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0003, |
|
"loss": 0.0, |
|
"step": 2 |
|
}, |
|
{ |
|
"epoch": 0.019013666072489603, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.000299878360437632, |
|
"loss": 0.0, |
|
"step": 3 |
|
}, |
|
{ |
|
"epoch": 0.02535155476331947, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00029951366095324104, |
|
"loss": 0.0, |
|
"step": 4 |
|
}, |
|
{ |
|
"epoch": 0.031689443454149334, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00029890655875994835, |
|
"loss": 0.0, |
|
"step": 5 |
|
}, |
|
{ |
|
"epoch": 0.038027332144979206, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0002980581478969406, |
|
"loss": 0.0, |
|
"step": 6 |
|
}, |
|
{ |
|
"epoch": 0.04436522083580907, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00029696995725793764, |
|
"loss": 0.0, |
|
"step": 7 |
|
}, |
|
{ |
|
"epoch": 0.05070310952663894, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00029564394783602234, |
|
"loss": 0.0, |
|
"step": 8 |
|
}, |
|
{ |
|
"epoch": 0.0570409982174688, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0002940825091897988, |
|
"loss": 0.0, |
|
"step": 9 |
|
}, |
|
{ |
|
"epoch": 0.06337888690829867, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00029228845513724634, |
|
"loss": 0.0, |
|
"step": 10 |
|
}, |
|
{ |
|
"epoch": 0.06971677559912855, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00029026501868502873, |
|
"loss": 0.0, |
|
"step": 11 |
|
}, |
|
{ |
|
"epoch": 0.07605466428995841, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0002880158462023983, |
|
"loss": 0.0, |
|
"step": 12 |
|
}, |
|
{ |
|
"epoch": 0.08239255298078828, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0002855449908501917, |
|
"loss": 0.0, |
|
"step": 13 |
|
}, |
|
{ |
|
"epoch": 0.08873044167161814, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00028285690527676035, |
|
"loss": 0.0, |
|
"step": 14 |
|
}, |
|
{ |
|
"epoch": 0.09506833036244801, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.000279956433593997, |
|
"loss": 0.0, |
|
"step": 15 |
|
}, |
|
{ |
|
"epoch": 0.10140621905327787, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00027684880264791867, |
|
"loss": 0.0, |
|
"step": 16 |
|
}, |
|
{ |
|
"epoch": 0.10774410774410774, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00027353961259953696, |
|
"loss": 0.0, |
|
"step": 17 |
|
}, |
|
{ |
|
"epoch": 0.1140819964349376, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00027003482683298933, |
|
"loss": 0.0, |
|
"step": 18 |
|
}, |
|
{ |
|
"epoch": 0.12041988512576748, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00026634076120911777, |
|
"loss": 0.0, |
|
"step": 19 |
|
}, |
|
{ |
|
"epoch": 0.12675777381659734, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0002624640726838608, |
|
"loss": 0.0, |
|
"step": 20 |
|
}, |
|
{ |
|
"epoch": 0.13309566250742721, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00025841174731196877, |
|
"loss": 0.0, |
|
"step": 21 |
|
}, |
|
{ |
|
"epoch": 0.1394335511982571, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.000254191087657661, |
|
"loss": 0.0, |
|
"step": 22 |
|
}, |
|
{ |
|
"epoch": 0.14577143988908695, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0002498096996349117, |
|
"loss": 0.0, |
|
"step": 23 |
|
}, |
|
{ |
|
"epoch": 0.15210932857991682, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0002452754788010787, |
|
"loss": 0.0, |
|
"step": 24 |
|
}, |
|
{ |
|
"epoch": 0.15844721727074668, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00024059659612857536, |
|
"loss": 0.0, |
|
"step": 25 |
|
}, |
|
{ |
|
"epoch": 0.15844721727074668, |
|
"eval_loss": NaN, |
|
"eval_runtime": 2.0847, |
|
"eval_samples_per_second": 23.984, |
|
"eval_steps_per_second": 6.236, |
|
"step": 25 |
|
}, |
|
{ |
|
"epoch": 0.16478510596157656, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00023578148328022626, |
|
"loss": 0.0, |
|
"step": 26 |
|
}, |
|
{ |
|
"epoch": 0.1711229946524064, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00023083881741484068, |
|
"loss": 0.0, |
|
"step": 27 |
|
}, |
|
{ |
|
"epoch": 0.1774608833432363, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00022577750555038587, |
|
"loss": 0.0, |
|
"step": 28 |
|
}, |
|
{ |
|
"epoch": 0.18379877203406614, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.000220606668512939, |
|
"loss": 0.0, |
|
"step": 29 |
|
}, |
|
{ |
|
"epoch": 0.19013666072489602, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00021533562450034164, |
|
"loss": 0.0, |
|
"step": 30 |
|
}, |
|
{ |
|
"epoch": 0.1964745494157259, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00020997387229017774, |
|
"loss": 0.0, |
|
"step": 31 |
|
}, |
|
{ |
|
"epoch": 0.20281243810655575, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00020453107412233428, |
|
"loss": 0.0, |
|
"step": 32 |
|
}, |
|
{ |
|
"epoch": 0.20915032679738563, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0001990170382869919, |
|
"loss": 0.0, |
|
"step": 33 |
|
}, |
|
{ |
|
"epoch": 0.21548821548821548, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00019344170144942302, |
|
"loss": 0.0, |
|
"step": 34 |
|
}, |
|
{ |
|
"epoch": 0.22182610417904536, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00018781511074344962, |
|
"loss": 0.0, |
|
"step": 35 |
|
}, |
|
{ |
|
"epoch": 0.2281639928698752, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0001821474056658286, |
|
"loss": 0.0, |
|
"step": 36 |
|
}, |
|
{ |
|
"epoch": 0.2345018815607051, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00017644879980419374, |
|
"loss": 0.0, |
|
"step": 37 |
|
}, |
|
{ |
|
"epoch": 0.24083977025153497, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00017072956243148002, |
|
"loss": 0.0, |
|
"step": 38 |
|
}, |
|
{ |
|
"epoch": 0.24717765894236482, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.000165, |
|
"loss": 0.0, |
|
"step": 39 |
|
}, |
|
{ |
|
"epoch": 0.25351554763319467, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00015927043756852, |
|
"loss": 0.0, |
|
"step": 40 |
|
}, |
|
{ |
|
"epoch": 0.2598534363240246, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0001535512001958063, |
|
"loss": 0.0, |
|
"step": 41 |
|
}, |
|
{ |
|
"epoch": 0.26619132501485443, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00014785259433417133, |
|
"loss": 0.0, |
|
"step": 42 |
|
}, |
|
{ |
|
"epoch": 0.2725292137056843, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00014218488925655037, |
|
"loss": 0.0, |
|
"step": 43 |
|
}, |
|
{ |
|
"epoch": 0.2788671023965142, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00013655829855057698, |
|
"loss": 0.0, |
|
"step": 44 |
|
}, |
|
{ |
|
"epoch": 0.28520499108734404, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00013098296171300814, |
|
"loss": 0.0, |
|
"step": 45 |
|
}, |
|
{ |
|
"epoch": 0.2915428797781739, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.0001254689258776657, |
|
"loss": 0.0, |
|
"step": 46 |
|
}, |
|
{ |
|
"epoch": 0.29788076846900374, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00012002612770982222, |
|
"loss": 0.0, |
|
"step": 47 |
|
}, |
|
{ |
|
"epoch": 0.30421865715983365, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00011466437549965834, |
|
"loss": 0.0, |
|
"step": 48 |
|
}, |
|
{ |
|
"epoch": 0.3105565458506635, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00010939333148706099, |
|
"loss": 0.0, |
|
"step": 49 |
|
}, |
|
{ |
|
"epoch": 0.31689443454149335, |
|
"grad_norm": NaN, |
|
"learning_rate": 0.00010422249444961407, |
|
"loss": 0.0, |
|
"step": 50 |
|
}, |
|
{ |
|
"epoch": 0.31689443454149335, |
|
"eval_loss": NaN, |
|
"eval_runtime": 2.0827, |
|
"eval_samples_per_second": 24.008, |
|
"eval_steps_per_second": 6.242, |
|
"step": 50 |
|
} |
|
], |
|
"logging_steps": 1, |
|
"max_steps": 76, |
|
"num_input_tokens_seen": 0, |
|
"num_train_epochs": 1, |
|
"save_steps": 50, |
|
"stateful_callbacks": { |
|
"EarlyStoppingCallback": { |
|
"args": { |
|
"early_stopping_patience": 1, |
|
"early_stopping_threshold": 0.0 |
|
}, |
|
"attributes": { |
|
"early_stopping_patience_counter": 0 |
|
} |
|
}, |
|
"TrainerControl": { |
|
"args": { |
|
"should_epoch_stop": false, |
|
"should_evaluate": false, |
|
"should_log": false, |
|
"should_save": true, |
|
"should_training_stop": false |
|
}, |
|
"attributes": {} |
|
} |
|
}, |
|
"total_flos": 7.375761253114839e+17, |
|
"train_batch_size": 1, |
|
"trial_name": null, |
|
"trial_params": null |
|
} |
|
|