{ "best_metric": null, "best_model_checkpoint": null, "epoch": 25.0, "eval_steps": 500, "global_step": 25, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 1.0, "grad_norm": 0.8025223016738892, "learning_rate": 2.9999999999999997e-05, "loss": 1.4923, "step": 1 }, { "epoch": 2.0, "grad_norm": 0.8024629354476929, "learning_rate": 5.9999999999999995e-05, "loss": 1.4923, "step": 2 }, { "epoch": 3.0, "grad_norm": 0.7746701240539551, "learning_rate": 8.999999999999999e-05, "loss": 1.4838, "step": 3 }, { "epoch": 4.0, "grad_norm": 0.6885385513305664, "learning_rate": 0.00011999999999999999, "loss": 1.4412, "step": 4 }, { "epoch": 5.0, "grad_norm": 0.6546556353569031, "learning_rate": 0.00015, "loss": 1.3718, "step": 5 }, { "epoch": 6.0, "grad_norm": 0.7697901725769043, "learning_rate": 0.00017999999999999998, "loss": 1.2978, "step": 6 }, { "epoch": 7.0, "grad_norm": 0.8412020802497864, "learning_rate": 0.00020999999999999998, "loss": 1.2189, "step": 7 }, { "epoch": 8.0, "grad_norm": 0.7889145016670227, "learning_rate": 0.00023999999999999998, "loss": 1.1297, "step": 8 }, { "epoch": 9.0, "grad_norm": 0.7386413812637329, "learning_rate": 0.00027, "loss": 1.0375, "step": 9 }, { "epoch": 10.0, "grad_norm": 0.7709745168685913, "learning_rate": 0.0003, "loss": 0.9459, "step": 10 }, { "epoch": 11.0, "grad_norm": 0.7179133296012878, "learning_rate": 0.0002967221401100708, "loss": 0.8415, "step": 11 }, { "epoch": 12.0, "grad_norm": 0.696864128112793, "learning_rate": 0.0002870318186463901, "loss": 0.7327, "step": 12 }, { "epoch": 13.0, "grad_norm": 0.6389071345329285, "learning_rate": 0.0002713525491562421, "loss": 0.6332, "step": 13 }, { "epoch": 14.0, "grad_norm": 0.5981886982917786, "learning_rate": 0.0002503695909538287, "loss": 0.5437, "step": 14 }, { "epoch": 15.0, "grad_norm": 0.6274304389953613, "learning_rate": 0.000225, "loss": 0.4655, "step": 15 }, { "epoch": 16.0, "grad_norm": 0.5762436389923096, "learning_rate": 0.0001963525491562421, "loss": 0.394, "step": 16 }, { "epoch": 17.0, "grad_norm": 0.5552861094474792, "learning_rate": 0.000165679269490148, "loss": 0.332, "step": 17 }, { "epoch": 18.0, "grad_norm": 0.536261796951294, "learning_rate": 0.000134320730509852, "loss": 0.2785, "step": 18 }, { "epoch": 19.0, "grad_norm": 0.48586374521255493, "learning_rate": 0.0001036474508437579, "loss": 0.2342, "step": 19 }, { "epoch": 20.0, "grad_norm": 0.45940732955932617, "learning_rate": 7.500000000000002e-05, "loss": 0.1992, "step": 20 }, { "epoch": 21.0, "grad_norm": 0.42822158336639404, "learning_rate": 4.963040904617131e-05, "loss": 0.1728, "step": 21 }, { "epoch": 22.0, "grad_norm": 0.4039425849914551, "learning_rate": 2.8647450843757897e-05, "loss": 0.1533, "step": 22 }, { "epoch": 23.0, "grad_norm": 0.38404396176338196, "learning_rate": 1.2968181353609852e-05, "loss": 0.1403, "step": 23 }, { "epoch": 24.0, "grad_norm": 0.3684459328651428, "learning_rate": 3.2778598899291465e-06, "loss": 0.1325, "step": 24 }, { "epoch": 25.0, "grad_norm": 0.3580010235309601, "learning_rate": 0.0, "loss": 0.129, "step": 25 } ], "logging_steps": 1, "max_steps": 25, "num_input_tokens_seen": 0, "num_train_epochs": 25, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 3628959635865600.0, "train_batch_size": 16, "trial_name": null, "trial_params": null }