|
{ |
|
"best_metric": null, |
|
"best_model_checkpoint": null, |
|
"epoch": 3.0, |
|
"eval_steps": 3, |
|
"global_step": 33, |
|
"is_hyper_param_search": false, |
|
"is_local_process_zero": true, |
|
"is_world_process_zero": true, |
|
"log_history": [ |
|
{ |
|
"epoch": 0.09090909090909091, |
|
"grad_norm": 0.13040611147880554, |
|
"learning_rate": 1e-05, |
|
"loss": 3.3054, |
|
"step": 1 |
|
}, |
|
{ |
|
"epoch": 0.09090909090909091, |
|
"eval_loss": 1.6250120401382446, |
|
"eval_runtime": 5.3483, |
|
"eval_samples_per_second": 1.87, |
|
"eval_steps_per_second": 0.374, |
|
"step": 1 |
|
}, |
|
{ |
|
"epoch": 0.18181818181818182, |
|
"grad_norm": 0.1381409764289856, |
|
"learning_rate": 2e-05, |
|
"loss": 3.2584, |
|
"step": 2 |
|
}, |
|
{ |
|
"epoch": 0.2727272727272727, |
|
"grad_norm": 0.13030081987380981, |
|
"learning_rate": 3e-05, |
|
"loss": 3.2721, |
|
"step": 3 |
|
}, |
|
{ |
|
"epoch": 0.2727272727272727, |
|
"eval_loss": 1.6240813732147217, |
|
"eval_runtime": 5.3485, |
|
"eval_samples_per_second": 1.87, |
|
"eval_steps_per_second": 0.374, |
|
"step": 3 |
|
}, |
|
{ |
|
"epoch": 0.36363636363636365, |
|
"grad_norm": 0.14820316433906555, |
|
"learning_rate": 4e-05, |
|
"loss": 3.2555, |
|
"step": 4 |
|
}, |
|
{ |
|
"epoch": 0.45454545454545453, |
|
"grad_norm": 0.15054553747177124, |
|
"learning_rate": 5e-05, |
|
"loss": 3.3107, |
|
"step": 5 |
|
}, |
|
{ |
|
"epoch": 0.5454545454545454, |
|
"grad_norm": 0.1470218300819397, |
|
"learning_rate": 6e-05, |
|
"loss": 3.2923, |
|
"step": 6 |
|
}, |
|
{ |
|
"epoch": 0.5454545454545454, |
|
"eval_loss": 1.6198304891586304, |
|
"eval_runtime": 5.3572, |
|
"eval_samples_per_second": 1.867, |
|
"eval_steps_per_second": 0.373, |
|
"step": 6 |
|
}, |
|
{ |
|
"epoch": 0.6363636363636364, |
|
"grad_norm": 0.15020014345645905, |
|
"learning_rate": 7e-05, |
|
"loss": 3.1834, |
|
"step": 7 |
|
}, |
|
{ |
|
"epoch": 0.7272727272727273, |
|
"grad_norm": 0.1556359976530075, |
|
"learning_rate": 8e-05, |
|
"loss": 3.223, |
|
"step": 8 |
|
}, |
|
{ |
|
"epoch": 0.8181818181818182, |
|
"grad_norm": 0.1711636781692505, |
|
"learning_rate": 9e-05, |
|
"loss": 3.1513, |
|
"step": 9 |
|
}, |
|
{ |
|
"epoch": 0.8181818181818182, |
|
"eval_loss": 1.6020313501358032, |
|
"eval_runtime": 5.3568, |
|
"eval_samples_per_second": 1.867, |
|
"eval_steps_per_second": 0.373, |
|
"step": 9 |
|
}, |
|
{ |
|
"epoch": 0.9090909090909091, |
|
"grad_norm": 0.23649035394191742, |
|
"learning_rate": 0.0001, |
|
"loss": 3.2135, |
|
"step": 10 |
|
}, |
|
{ |
|
"epoch": 1.0, |
|
"grad_norm": 0.23038285970687866, |
|
"learning_rate": 9.953429730181653e-05, |
|
"loss": 3.138, |
|
"step": 11 |
|
}, |
|
{ |
|
"epoch": 1.0909090909090908, |
|
"grad_norm": 0.2272975891828537, |
|
"learning_rate": 9.814586436738998e-05, |
|
"loss": 3.1555, |
|
"step": 12 |
|
}, |
|
{ |
|
"epoch": 1.0909090909090908, |
|
"eval_loss": 1.5541090965270996, |
|
"eval_runtime": 5.3607, |
|
"eval_samples_per_second": 1.865, |
|
"eval_steps_per_second": 0.373, |
|
"step": 12 |
|
}, |
|
{ |
|
"epoch": 1.1818181818181819, |
|
"grad_norm": 0.24618901312351227, |
|
"learning_rate": 9.586056507527266e-05, |
|
"loss": 3.0046, |
|
"step": 13 |
|
}, |
|
{ |
|
"epoch": 1.2727272727272727, |
|
"grad_norm": 0.23367342352867126, |
|
"learning_rate": 9.272097022732443e-05, |
|
"loss": 3.1042, |
|
"step": 14 |
|
}, |
|
{ |
|
"epoch": 1.3636363636363638, |
|
"grad_norm": 0.20990729331970215, |
|
"learning_rate": 8.8785564535221e-05, |
|
"loss": 3.0543, |
|
"step": 15 |
|
}, |
|
{ |
|
"epoch": 1.3636363636363638, |
|
"eval_loss": 1.5030436515808105, |
|
"eval_runtime": 5.3534, |
|
"eval_samples_per_second": 1.868, |
|
"eval_steps_per_second": 0.374, |
|
"step": 15 |
|
}, |
|
{ |
|
"epoch": 1.4545454545454546, |
|
"grad_norm": 0.29300206899642944, |
|
"learning_rate": 8.412765716093272e-05, |
|
"loss": 3.0298, |
|
"step": 16 |
|
}, |
|
{ |
|
"epoch": 1.5454545454545454, |
|
"grad_norm": 0.2927170991897583, |
|
"learning_rate": 7.883401610574336e-05, |
|
"loss": 3.001, |
|
"step": 17 |
|
}, |
|
{ |
|
"epoch": 1.6363636363636362, |
|
"grad_norm": 0.29279839992523193, |
|
"learning_rate": 7.300325188655761e-05, |
|
"loss": 3.0495, |
|
"step": 18 |
|
}, |
|
{ |
|
"epoch": 1.6363636363636362, |
|
"eval_loss": 1.449854850769043, |
|
"eval_runtime": 5.3549, |
|
"eval_samples_per_second": 1.867, |
|
"eval_steps_per_second": 0.373, |
|
"step": 18 |
|
}, |
|
{ |
|
"epoch": 1.7272727272727273, |
|
"grad_norm": 0.27877599000930786, |
|
"learning_rate": 6.674398060854931e-05, |
|
"loss": 2.8904, |
|
"step": 19 |
|
}, |
|
{ |
|
"epoch": 1.8181818181818183, |
|
"grad_norm": 0.3109539747238159, |
|
"learning_rate": 6.01728006526317e-05, |
|
"loss": 2.9107, |
|
"step": 20 |
|
}, |
|
{ |
|
"epoch": 1.9090909090909092, |
|
"grad_norm": 0.2679404020309448, |
|
"learning_rate": 5.341212066823355e-05, |
|
"loss": 2.9001, |
|
"step": 21 |
|
}, |
|
{ |
|
"epoch": 1.9090909090909092, |
|
"eval_loss": 1.3991481065750122, |
|
"eval_runtime": 5.366, |
|
"eval_samples_per_second": 1.864, |
|
"eval_steps_per_second": 0.373, |
|
"step": 21 |
|
}, |
|
{ |
|
"epoch": 2.0, |
|
"grad_norm": 0.2852272093296051, |
|
"learning_rate": 4.658787933176646e-05, |
|
"loss": 2.7902, |
|
"step": 22 |
|
}, |
|
{ |
|
"epoch": 2.090909090909091, |
|
"grad_norm": 0.3013867437839508, |
|
"learning_rate": 3.982719934736832e-05, |
|
"loss": 2.8728, |
|
"step": 23 |
|
}, |
|
{ |
|
"epoch": 2.1818181818181817, |
|
"grad_norm": 0.3320194184780121, |
|
"learning_rate": 3.325601939145069e-05, |
|
"loss": 2.7332, |
|
"step": 24 |
|
}, |
|
{ |
|
"epoch": 2.1818181818181817, |
|
"eval_loss": 1.3631479740142822, |
|
"eval_runtime": 5.355, |
|
"eval_samples_per_second": 1.867, |
|
"eval_steps_per_second": 0.373, |
|
"step": 24 |
|
}, |
|
{ |
|
"epoch": 2.2727272727272725, |
|
"grad_norm": 0.3708702623844147, |
|
"learning_rate": 2.6996748113442394e-05, |
|
"loss": 2.7049, |
|
"step": 25 |
|
}, |
|
{ |
|
"epoch": 2.3636363636363638, |
|
"grad_norm": 0.3256295323371887, |
|
"learning_rate": 2.1165983894256647e-05, |
|
"loss": 2.9422, |
|
"step": 26 |
|
}, |
|
{ |
|
"epoch": 2.4545454545454546, |
|
"grad_norm": 0.3431509733200073, |
|
"learning_rate": 1.5872342839067306e-05, |
|
"loss": 2.6983, |
|
"step": 27 |
|
}, |
|
{ |
|
"epoch": 2.4545454545454546, |
|
"eval_loss": 1.3413066864013672, |
|
"eval_runtime": 5.3602, |
|
"eval_samples_per_second": 1.866, |
|
"eval_steps_per_second": 0.373, |
|
"step": 27 |
|
}, |
|
{ |
|
"epoch": 2.5454545454545454, |
|
"grad_norm": 0.34741801023483276, |
|
"learning_rate": 1.1214435464779006e-05, |
|
"loss": 2.7043, |
|
"step": 28 |
|
}, |
|
{ |
|
"epoch": 2.6363636363636362, |
|
"grad_norm": 0.370277464389801, |
|
"learning_rate": 7.2790297726755716e-06, |
|
"loss": 2.5969, |
|
"step": 29 |
|
}, |
|
{ |
|
"epoch": 2.7272727272727275, |
|
"grad_norm": 0.39455196261405945, |
|
"learning_rate": 4.139434924727359e-06, |
|
"loss": 2.7283, |
|
"step": 30 |
|
}, |
|
{ |
|
"epoch": 2.7272727272727275, |
|
"eval_loss": 1.3317468166351318, |
|
"eval_runtime": 5.3596, |
|
"eval_samples_per_second": 1.866, |
|
"eval_steps_per_second": 0.373, |
|
"step": 30 |
|
}, |
|
{ |
|
"epoch": 2.8181818181818183, |
|
"grad_norm": 0.4511341452598572, |
|
"learning_rate": 1.8541356326100433e-06, |
|
"loss": 2.5703, |
|
"step": 31 |
|
}, |
|
{ |
|
"epoch": 2.909090909090909, |
|
"grad_norm": 0.3779066801071167, |
|
"learning_rate": 4.6570269818346224e-07, |
|
"loss": 2.6689, |
|
"step": 32 |
|
}, |
|
{ |
|
"epoch": 3.0, |
|
"grad_norm": 0.39674925804138184, |
|
"learning_rate": 0.0, |
|
"loss": 2.6383, |
|
"step": 33 |
|
}, |
|
{ |
|
"epoch": 3.0, |
|
"eval_loss": 1.3296959400177002, |
|
"eval_runtime": 5.3524, |
|
"eval_samples_per_second": 1.868, |
|
"eval_steps_per_second": 0.374, |
|
"step": 33 |
|
} |
|
], |
|
"logging_steps": 1, |
|
"max_steps": 33, |
|
"num_input_tokens_seen": 0, |
|
"num_train_epochs": 3, |
|
"save_steps": 25, |
|
"stateful_callbacks": { |
|
"TrainerControl": { |
|
"args": { |
|
"should_epoch_stop": false, |
|
"should_evaluate": false, |
|
"should_log": false, |
|
"should_save": true, |
|
"should_training_stop": true |
|
}, |
|
"attributes": {} |
|
} |
|
}, |
|
"total_flos": 8.379135550291968e+16, |
|
"train_batch_size": 8, |
|
"trial_name": null, |
|
"trial_params": null |
|
} |
|
|