|
{ |
|
"best_metric": null, |
|
"best_model_checkpoint": null, |
|
"epoch": 1.19019281123542, |
|
"global_step": 5000, |
|
"is_hyper_param_search": false, |
|
"is_local_process_zero": true, |
|
"is_world_process_zero": true, |
|
"log_history": [ |
|
{ |
|
"epoch": 0.02, |
|
"learning_rate": 0.0003, |
|
"loss": 2235.1048, |
|
"step": 100 |
|
}, |
|
{ |
|
"epoch": 0.05, |
|
"learning_rate": 0.0003, |
|
"loss": 1615.3083, |
|
"step": 200 |
|
}, |
|
{ |
|
"epoch": 0.07, |
|
"learning_rate": 0.0003, |
|
"loss": 1223.9066, |
|
"step": 300 |
|
}, |
|
{ |
|
"epoch": 0.1, |
|
"learning_rate": 0.0003, |
|
"loss": 602.0252, |
|
"step": 400 |
|
}, |
|
{ |
|
"epoch": 0.12, |
|
"learning_rate": 0.0003, |
|
"loss": 478.0811, |
|
"step": 500 |
|
}, |
|
{ |
|
"epoch": 0.12, |
|
"eval_cer": 0.2157549269468942, |
|
"eval_loss": 376.9426574707031, |
|
"eval_runtime": 1031.6167, |
|
"eval_samples_per_second": 7.186, |
|
"eval_steps_per_second": 0.719, |
|
"eval_wer": 0.7605279103234114, |
|
"step": 500 |
|
}, |
|
{ |
|
"epoch": 0.14, |
|
"learning_rate": 0.0003, |
|
"loss": 421.4082, |
|
"step": 600 |
|
}, |
|
{ |
|
"epoch": 0.17, |
|
"learning_rate": 0.0003, |
|
"loss": 394.7663, |
|
"step": 700 |
|
}, |
|
{ |
|
"epoch": 0.19, |
|
"learning_rate": 0.0003, |
|
"loss": 379.9422, |
|
"step": 800 |
|
}, |
|
{ |
|
"epoch": 0.21, |
|
"learning_rate": 0.0003, |
|
"loss": 373.7412, |
|
"step": 900 |
|
}, |
|
{ |
|
"epoch": 0.24, |
|
"learning_rate": 0.0003, |
|
"loss": 349.4335, |
|
"step": 1000 |
|
}, |
|
{ |
|
"epoch": 0.24, |
|
"eval_cer": 0.15168241477586192, |
|
"eval_loss": 274.2783508300781, |
|
"eval_runtime": 1026.2842, |
|
"eval_samples_per_second": 7.223, |
|
"eval_steps_per_second": 0.723, |
|
"eval_wer": 0.5870256759827311, |
|
"step": 1000 |
|
}, |
|
{ |
|
"epoch": 0.26, |
|
"learning_rate": 0.0003, |
|
"loss": 346.411, |
|
"step": 1100 |
|
}, |
|
{ |
|
"epoch": 0.29, |
|
"learning_rate": 0.0003, |
|
"loss": 335.0877, |
|
"step": 1200 |
|
}, |
|
{ |
|
"epoch": 0.31, |
|
"learning_rate": 0.0003, |
|
"loss": 330.8003, |
|
"step": 1300 |
|
}, |
|
{ |
|
"epoch": 0.33, |
|
"learning_rate": 0.0003, |
|
"loss": 324.7547, |
|
"step": 1400 |
|
}, |
|
{ |
|
"epoch": 0.36, |
|
"learning_rate": 0.0003, |
|
"loss": 308.4796, |
|
"step": 1500 |
|
}, |
|
{ |
|
"epoch": 0.36, |
|
"eval_cer": 0.13990026347021764, |
|
"eval_loss": 249.27769470214844, |
|
"eval_runtime": 1029.4748, |
|
"eval_samples_per_second": 7.201, |
|
"eval_steps_per_second": 0.721, |
|
"eval_wer": 0.5321707187760357, |
|
"step": 1500 |
|
}, |
|
{ |
|
"epoch": 0.38, |
|
"learning_rate": 0.0003, |
|
"loss": 311.2614, |
|
"step": 1600 |
|
}, |
|
{ |
|
"epoch": 0.4, |
|
"learning_rate": 0.0003, |
|
"loss": 299.9645, |
|
"step": 1700 |
|
}, |
|
{ |
|
"epoch": 0.43, |
|
"learning_rate": 0.0003, |
|
"loss": 299.9626, |
|
"step": 1800 |
|
}, |
|
{ |
|
"epoch": 0.45, |
|
"learning_rate": 0.0003, |
|
"loss": 291.1237, |
|
"step": 1900 |
|
}, |
|
{ |
|
"epoch": 0.48, |
|
"learning_rate": 0.0003, |
|
"loss": 280.2542, |
|
"step": 2000 |
|
}, |
|
{ |
|
"epoch": 0.48, |
|
"eval_cer": 0.12837916028623791, |
|
"eval_loss": 232.43905639648438, |
|
"eval_runtime": 1027.1097, |
|
"eval_samples_per_second": 7.217, |
|
"eval_steps_per_second": 0.722, |
|
"eval_wer": 0.48547678557903506, |
|
"step": 2000 |
|
}, |
|
{ |
|
"epoch": 0.5, |
|
"learning_rate": 0.0003, |
|
"loss": 297.7207, |
|
"step": 2100 |
|
}, |
|
{ |
|
"epoch": 0.52, |
|
"learning_rate": 0.0003, |
|
"loss": 278.3572, |
|
"step": 2200 |
|
}, |
|
{ |
|
"epoch": 0.55, |
|
"learning_rate": 0.0003, |
|
"loss": 282.0629, |
|
"step": 2300 |
|
}, |
|
{ |
|
"epoch": 0.57, |
|
"learning_rate": 0.0003, |
|
"loss": 277.2473, |
|
"step": 2400 |
|
}, |
|
{ |
|
"epoch": 0.6, |
|
"learning_rate": 0.0003, |
|
"loss": 270.2169, |
|
"step": 2500 |
|
}, |
|
{ |
|
"epoch": 0.6, |
|
"eval_cer": 0.12593015213453937, |
|
"eval_loss": 211.3223114013672, |
|
"eval_runtime": 1032.1072, |
|
"eval_samples_per_second": 7.182, |
|
"eval_steps_per_second": 0.719, |
|
"eval_wer": 0.46879497083996063, |
|
"step": 2500 |
|
}, |
|
{ |
|
"epoch": 0.62, |
|
"learning_rate": 0.0003, |
|
"loss": 272.5899, |
|
"step": 2600 |
|
}, |
|
{ |
|
"epoch": 0.64, |
|
"learning_rate": 0.0003, |
|
"loss": 262.8966, |
|
"step": 2700 |
|
}, |
|
{ |
|
"epoch": 0.67, |
|
"learning_rate": 0.0003, |
|
"loss": 267.9274, |
|
"step": 2800 |
|
}, |
|
{ |
|
"epoch": 0.69, |
|
"learning_rate": 0.0003, |
|
"loss": 254.0362, |
|
"step": 2900 |
|
}, |
|
{ |
|
"epoch": 0.71, |
|
"learning_rate": 0.0003, |
|
"loss": 261.8612, |
|
"step": 3000 |
|
}, |
|
{ |
|
"epoch": 0.71, |
|
"eval_cer": 0.11589729236582261, |
|
"eval_loss": 196.6202850341797, |
|
"eval_runtime": 1032.1872, |
|
"eval_samples_per_second": 7.182, |
|
"eval_steps_per_second": 0.719, |
|
"eval_wer": 0.4461864727713398, |
|
"step": 3000 |
|
}, |
|
{ |
|
"epoch": 0.74, |
|
"learning_rate": 0.0003, |
|
"loss": 263.8872, |
|
"step": 3100 |
|
}, |
|
{ |
|
"epoch": 0.76, |
|
"learning_rate": 0.0003, |
|
"loss": 263.278, |
|
"step": 3200 |
|
}, |
|
{ |
|
"epoch": 0.79, |
|
"learning_rate": 0.0003, |
|
"loss": 252.0182, |
|
"step": 3300 |
|
}, |
|
{ |
|
"epoch": 0.81, |
|
"learning_rate": 0.0003, |
|
"loss": 256.6052, |
|
"step": 3400 |
|
}, |
|
{ |
|
"epoch": 0.83, |
|
"learning_rate": 0.0003, |
|
"loss": 251.7534, |
|
"step": 3500 |
|
}, |
|
{ |
|
"epoch": 0.83, |
|
"eval_cer": 0.12125281568657002, |
|
"eval_loss": 186.7011260986328, |
|
"eval_runtime": 1035.9213, |
|
"eval_samples_per_second": 7.156, |
|
"eval_steps_per_second": 0.716, |
|
"eval_wer": 0.4463568885859274, |
|
"step": 3500 |
|
}, |
|
{ |
|
"epoch": 0.86, |
|
"learning_rate": 0.0003, |
|
"loss": 230.9344, |
|
"step": 3600 |
|
}, |
|
{ |
|
"epoch": 0.88, |
|
"learning_rate": 0.0003, |
|
"loss": 246.4894, |
|
"step": 3700 |
|
}, |
|
{ |
|
"epoch": 0.9, |
|
"learning_rate": 0.0003, |
|
"loss": 239.637, |
|
"step": 3800 |
|
}, |
|
{ |
|
"epoch": 0.93, |
|
"learning_rate": 0.0003, |
|
"loss": 251.8844, |
|
"step": 3900 |
|
}, |
|
{ |
|
"epoch": 0.95, |
|
"learning_rate": 0.0003, |
|
"loss": 242.5334, |
|
"step": 4000 |
|
}, |
|
{ |
|
"epoch": 0.95, |
|
"eval_cer": 0.10615777533175987, |
|
"eval_loss": 173.9308319091797, |
|
"eval_runtime": 1037.0086, |
|
"eval_samples_per_second": 7.148, |
|
"eval_steps_per_second": 0.716, |
|
"eval_wer": 0.4061387563432553, |
|
"step": 4000 |
|
}, |
|
{ |
|
"epoch": 0.98, |
|
"learning_rate": 0.0003, |
|
"loss": 237.7298, |
|
"step": 4100 |
|
}, |
|
{ |
|
"epoch": 1.0, |
|
"learning_rate": 0.0003, |
|
"loss": 238.2786, |
|
"step": 4200 |
|
}, |
|
{ |
|
"epoch": 1.02, |
|
"learning_rate": 0.0003, |
|
"loss": 209.7386, |
|
"step": 4300 |
|
}, |
|
{ |
|
"epoch": 1.05, |
|
"learning_rate": 0.0003, |
|
"loss": 201.507, |
|
"step": 4400 |
|
}, |
|
{ |
|
"epoch": 1.07, |
|
"learning_rate": 0.0003, |
|
"loss": 217.7602, |
|
"step": 4500 |
|
}, |
|
{ |
|
"epoch": 1.07, |
|
"eval_cer": 0.1168149976182723, |
|
"eval_loss": 161.6786651611328, |
|
"eval_runtime": 1036.8937, |
|
"eval_samples_per_second": 7.149, |
|
"eval_steps_per_second": 0.716, |
|
"eval_wer": 0.4201885935014769, |
|
"step": 4500 |
|
}, |
|
{ |
|
"epoch": 1.09, |
|
"learning_rate": 0.0003, |
|
"loss": 209.7251, |
|
"step": 4600 |
|
}, |
|
{ |
|
"epoch": 1.12, |
|
"learning_rate": 0.0003, |
|
"loss": 212.4263, |
|
"step": 4700 |
|
}, |
|
{ |
|
"epoch": 1.14, |
|
"learning_rate": 0.0003, |
|
"loss": 220.1039, |
|
"step": 4800 |
|
}, |
|
{ |
|
"epoch": 1.17, |
|
"learning_rate": 0.0003, |
|
"loss": 218.4896, |
|
"step": 4900 |
|
}, |
|
{ |
|
"epoch": 1.19, |
|
"learning_rate": 0.0, |
|
"loss": 217.312, |
|
"step": 5000 |
|
}, |
|
{ |
|
"epoch": 1.19, |
|
"eval_cer": 0.11287236361581252, |
|
"eval_loss": 172.40834045410156, |
|
"eval_runtime": 1040.3699, |
|
"eval_samples_per_second": 7.125, |
|
"eval_steps_per_second": 0.713, |
|
"eval_wer": 0.41411042944785276, |
|
"step": 5000 |
|
}, |
|
{ |
|
"epoch": 1.19, |
|
"step": 5000, |
|
"total_flos": 8.046046597054969e+18, |
|
"train_loss": 372.1765015625, |
|
"train_runtime": 24790.5372, |
|
"train_samples_per_second": 2.017, |
|
"train_steps_per_second": 0.202 |
|
} |
|
], |
|
"max_steps": 5000, |
|
"num_train_epochs": 2, |
|
"total_flos": 8.046046597054969e+18, |
|
"trial_name": null, |
|
"trial_params": null |
|
} |
|
|