{ "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 171, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.01, "learning_rate": 1.1111111111111112e-05, "loss": 1.8643, "step": 1 }, { "epoch": 0.03, "learning_rate": 5.555555555555556e-05, "loss": 1.7663, "step": 5 }, { "epoch": 0.06, "learning_rate": 0.00011111111111111112, "loss": 1.4851, "step": 10 }, { "epoch": 0.09, "learning_rate": 0.0001666666666666667, "loss": 1.3257, "step": 15 }, { "epoch": 0.12, "learning_rate": 0.0001999156886888064, "loss": 1.2837, "step": 20 }, { "epoch": 0.15, "learning_rate": 0.00019896881839082556, "loss": 1.2272, "step": 25 }, { "epoch": 0.18, "learning_rate": 0.00019697969360350098, "loss": 1.2551, "step": 30 }, { "epoch": 0.2, "learning_rate": 0.00019396926207859084, "loss": 1.2263, "step": 35 }, { "epoch": 0.23, "learning_rate": 0.00018996922709216455, "loss": 1.2222, "step": 40 }, { "epoch": 0.26, "learning_rate": 0.00018502171357296144, "loss": 1.242, "step": 45 }, { "epoch": 0.29, "learning_rate": 0.00017917882447886582, "loss": 1.214, "step": 50 }, { "epoch": 0.32, "learning_rate": 0.00017250209209335927, "loss": 1.2186, "step": 55 }, { "epoch": 0.35, "learning_rate": 0.0001650618300204242, "loss": 1.2202, "step": 60 }, { "epoch": 0.38, "learning_rate": 0.00015693639270213136, "loss": 1.1952, "step": 65 }, { "epoch": 0.41, "learning_rate": 0.0001482113502570349, "loss": 1.2162, "step": 70 }, { "epoch": 0.44, "learning_rate": 0.00013897858732926793, "loss": 1.1951, "step": 75 }, { "epoch": 0.47, "learning_rate": 0.00012933533543848461, "loss": 1.1806, "step": 80 }, { "epoch": 0.5, "learning_rate": 0.00011938314902110701, "loss": 1.2158, "step": 85 }, { "epoch": 0.53, "learning_rate": 0.00010922683594633021, "loss": 1.1702, "step": 90 }, { "epoch": 0.56, "learning_rate": 9.897335376977102e-05, "loss": 1.186, "step": 95 }, { "epoch": 0.58, "learning_rate": 8.87306833484679e-05, "loss": 1.1802, "step": 100 }, { "epoch": 0.61, "learning_rate": 7.860669167935028e-05, "loss": 1.1626, "step": 105 }, { "epoch": 0.64, "learning_rate": 6.870799593678459e-05, "loss": 1.177, "step": 110 }, { "epoch": 0.67, "learning_rate": 5.913884067217685e-05, "loss": 1.1495, "step": 115 }, { "epoch": 0.7, "learning_rate": 5.000000000000002e-05, "loss": 1.1842, "step": 120 }, { "epoch": 0.73, "learning_rate": 4.1387716331478565e-05, "loss": 1.1736, "step": 125 }, { "epoch": 0.76, "learning_rate": 3.339268683227499e-05, "loss": 1.1638, "step": 130 }, { "epoch": 0.79, "learning_rate": 2.6099108277934103e-05, "loss": 1.1465, "step": 135 }, { "epoch": 0.82, "learning_rate": 1.9583790365845822e-05, "loss": 1.1706, "step": 140 }, { "epoch": 0.85, "learning_rate": 1.3915346821563235e-05, "loss": 1.152, "step": 145 }, { "epoch": 0.88, "learning_rate": 9.153472818047625e-06, "loss": 1.1454, "step": 150 }, { "epoch": 0.91, "learning_rate": 5.348316317440549e-06, "loss": 1.1639, "step": 155 }, { "epoch": 0.94, "learning_rate": 2.539949955849985e-06, "loss": 1.1614, "step": 160 }, { "epoch": 0.96, "learning_rate": 7.579490328064265e-07, "loss": 1.1605, "step": 165 }, { "epoch": 0.99, "learning_rate": 2.108004964086474e-08, "loss": 1.1641, "step": 170 }, { "epoch": 1.0, "eval_loss": 1.2067391872406006, "eval_runtime": 235.4134, "eval_samples_per_second": 2.68, "eval_steps_per_second": 0.671, "step": 171 }, { "epoch": 1.0, "step": 171, "total_flos": 6.011757435184742e+16, "train_loss": 1.2207316202029848, "train_runtime": 1132.3581, "train_samples_per_second": 0.604, "train_steps_per_second": 0.151 } ], "logging_steps": 5, "max_steps": 171, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 100, "total_flos": 6.011757435184742e+16, "train_batch_size": 2, "trial_name": null, "trial_params": null }