{ "best_metric": null, "best_model_checkpoint": null, "epoch": 3.0, "eval_steps": 3, "global_step": 33, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.09090909090909091, "grad_norm": 0.13040611147880554, "learning_rate": 1e-05, "loss": 3.3054, "step": 1 }, { "epoch": 0.09090909090909091, "eval_loss": 1.6250120401382446, "eval_runtime": 5.3483, "eval_samples_per_second": 1.87, "eval_steps_per_second": 0.374, "step": 1 }, { "epoch": 0.18181818181818182, "grad_norm": 0.1381409764289856, "learning_rate": 2e-05, "loss": 3.2584, "step": 2 }, { "epoch": 0.2727272727272727, "grad_norm": 0.13030081987380981, "learning_rate": 3e-05, "loss": 3.2721, "step": 3 }, { "epoch": 0.2727272727272727, "eval_loss": 1.6240813732147217, "eval_runtime": 5.3485, "eval_samples_per_second": 1.87, "eval_steps_per_second": 0.374, "step": 3 }, { "epoch": 0.36363636363636365, "grad_norm": 0.14820316433906555, "learning_rate": 4e-05, "loss": 3.2555, "step": 4 }, { "epoch": 0.45454545454545453, "grad_norm": 0.15054553747177124, "learning_rate": 5e-05, "loss": 3.3107, "step": 5 }, { "epoch": 0.5454545454545454, "grad_norm": 0.1470218300819397, "learning_rate": 6e-05, "loss": 3.2923, "step": 6 }, { "epoch": 0.5454545454545454, "eval_loss": 1.6198304891586304, "eval_runtime": 5.3572, "eval_samples_per_second": 1.867, "eval_steps_per_second": 0.373, "step": 6 }, { "epoch": 0.6363636363636364, "grad_norm": 0.15020014345645905, "learning_rate": 7e-05, "loss": 3.1834, "step": 7 }, { "epoch": 0.7272727272727273, "grad_norm": 0.1556359976530075, "learning_rate": 8e-05, "loss": 3.223, "step": 8 }, { "epoch": 0.8181818181818182, "grad_norm": 0.1711636781692505, "learning_rate": 9e-05, "loss": 3.1513, "step": 9 }, { "epoch": 0.8181818181818182, "eval_loss": 1.6020313501358032, "eval_runtime": 5.3568, "eval_samples_per_second": 1.867, "eval_steps_per_second": 0.373, "step": 9 }, { "epoch": 0.9090909090909091, "grad_norm": 0.23649035394191742, "learning_rate": 0.0001, "loss": 3.2135, "step": 10 }, { "epoch": 1.0, "grad_norm": 0.23038285970687866, "learning_rate": 9.953429730181653e-05, "loss": 3.138, "step": 11 }, { "epoch": 1.0909090909090908, "grad_norm": 0.2272975891828537, "learning_rate": 9.814586436738998e-05, "loss": 3.1555, "step": 12 }, { "epoch": 1.0909090909090908, "eval_loss": 1.5541090965270996, "eval_runtime": 5.3607, "eval_samples_per_second": 1.865, "eval_steps_per_second": 0.373, "step": 12 }, { "epoch": 1.1818181818181819, "grad_norm": 0.24618901312351227, "learning_rate": 9.586056507527266e-05, "loss": 3.0046, "step": 13 }, { "epoch": 1.2727272727272727, "grad_norm": 0.23367342352867126, "learning_rate": 9.272097022732443e-05, "loss": 3.1042, "step": 14 }, { "epoch": 1.3636363636363638, "grad_norm": 0.20990729331970215, "learning_rate": 8.8785564535221e-05, "loss": 3.0543, "step": 15 }, { "epoch": 1.3636363636363638, "eval_loss": 1.5030436515808105, "eval_runtime": 5.3534, "eval_samples_per_second": 1.868, "eval_steps_per_second": 0.374, "step": 15 }, { "epoch": 1.4545454545454546, "grad_norm": 0.29300206899642944, "learning_rate": 8.412765716093272e-05, "loss": 3.0298, "step": 16 }, { "epoch": 1.5454545454545454, "grad_norm": 0.2927170991897583, "learning_rate": 7.883401610574336e-05, "loss": 3.001, "step": 17 }, { "epoch": 1.6363636363636362, "grad_norm": 0.29279839992523193, "learning_rate": 7.300325188655761e-05, "loss": 3.0495, "step": 18 }, { "epoch": 1.6363636363636362, "eval_loss": 1.449854850769043, "eval_runtime": 5.3549, "eval_samples_per_second": 1.867, "eval_steps_per_second": 0.373, "step": 18 }, { "epoch": 1.7272727272727273, "grad_norm": 0.27877599000930786, "learning_rate": 6.674398060854931e-05, "loss": 2.8904, "step": 19 }, { "epoch": 1.8181818181818183, "grad_norm": 0.3109539747238159, "learning_rate": 6.01728006526317e-05, "loss": 2.9107, "step": 20 }, { "epoch": 1.9090909090909092, "grad_norm": 0.2679404020309448, "learning_rate": 5.341212066823355e-05, "loss": 2.9001, "step": 21 }, { "epoch": 1.9090909090909092, "eval_loss": 1.3991481065750122, "eval_runtime": 5.366, "eval_samples_per_second": 1.864, "eval_steps_per_second": 0.373, "step": 21 }, { "epoch": 2.0, "grad_norm": 0.2852272093296051, "learning_rate": 4.658787933176646e-05, "loss": 2.7902, "step": 22 }, { "epoch": 2.090909090909091, "grad_norm": 0.3013867437839508, "learning_rate": 3.982719934736832e-05, "loss": 2.8728, "step": 23 }, { "epoch": 2.1818181818181817, "grad_norm": 0.3320194184780121, "learning_rate": 3.325601939145069e-05, "loss": 2.7332, "step": 24 }, { "epoch": 2.1818181818181817, "eval_loss": 1.3631479740142822, "eval_runtime": 5.355, "eval_samples_per_second": 1.867, "eval_steps_per_second": 0.373, "step": 24 }, { "epoch": 2.2727272727272725, "grad_norm": 0.3708702623844147, "learning_rate": 2.6996748113442394e-05, "loss": 2.7049, "step": 25 }, { "epoch": 2.3636363636363638, "grad_norm": 0.3256295323371887, "learning_rate": 2.1165983894256647e-05, "loss": 2.9422, "step": 26 }, { "epoch": 2.4545454545454546, "grad_norm": 0.3431509733200073, "learning_rate": 1.5872342839067306e-05, "loss": 2.6983, "step": 27 }, { "epoch": 2.4545454545454546, "eval_loss": 1.3413066864013672, "eval_runtime": 5.3602, "eval_samples_per_second": 1.866, "eval_steps_per_second": 0.373, "step": 27 }, { "epoch": 2.5454545454545454, "grad_norm": 0.34741801023483276, "learning_rate": 1.1214435464779006e-05, "loss": 2.7043, "step": 28 }, { "epoch": 2.6363636363636362, "grad_norm": 0.370277464389801, "learning_rate": 7.2790297726755716e-06, "loss": 2.5969, "step": 29 }, { "epoch": 2.7272727272727275, "grad_norm": 0.39455196261405945, "learning_rate": 4.139434924727359e-06, "loss": 2.7283, "step": 30 }, { "epoch": 2.7272727272727275, "eval_loss": 1.3317468166351318, "eval_runtime": 5.3596, "eval_samples_per_second": 1.866, "eval_steps_per_second": 0.373, "step": 30 }, { "epoch": 2.8181818181818183, "grad_norm": 0.4511341452598572, "learning_rate": 1.8541356326100433e-06, "loss": 2.5703, "step": 31 }, { "epoch": 2.909090909090909, "grad_norm": 0.3779066801071167, "learning_rate": 4.6570269818346224e-07, "loss": 2.6689, "step": 32 }, { "epoch": 3.0, "grad_norm": 0.39674925804138184, "learning_rate": 0.0, "loss": 2.6383, "step": 33 }, { "epoch": 3.0, "eval_loss": 1.3296959400177002, "eval_runtime": 5.3524, "eval_samples_per_second": 1.868, "eval_steps_per_second": 0.374, "step": 33 } ], "logging_steps": 1, "max_steps": 33, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 25, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 8.379135550291968e+16, "train_batch_size": 8, "trial_name": null, "trial_params": null }