{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 5.0, "eval_steps": 500, "global_step": 265, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.09603841536614646, "grad_norm": 0.7072356343269348, "learning_rate": 1.6000000000000003e-05, "loss": 3.1026, "step": 5 }, { "epoch": 0.19207683073229292, "grad_norm": 0.7724522948265076, "learning_rate": 3.6e-05, "loss": 3.1732, "step": 10 }, { "epoch": 0.28811524609843936, "grad_norm": 0.9103713035583496, "learning_rate": 5.6000000000000006e-05, "loss": 3.1209, "step": 15 }, { "epoch": 0.38415366146458585, "grad_norm": 0.9525737166404724, "learning_rate": 7.6e-05, "loss": 3.0653, "step": 20 }, { "epoch": 0.4801920768307323, "grad_norm": 0.9877885580062866, "learning_rate": 9.6e-05, "loss": 2.7498, "step": 25 }, { "epoch": 0.5762304921968787, "grad_norm": 0.8707166910171509, "learning_rate": 0.000116, "loss": 2.6275, "step": 30 }, { "epoch": 0.6722689075630253, "grad_norm": 1.3398517370224, "learning_rate": 0.00013600000000000003, "loss": 2.475, "step": 35 }, { "epoch": 0.7683073229291717, "grad_norm": 1.2803456783294678, "learning_rate": 0.00015600000000000002, "loss": 2.207, "step": 40 }, { "epoch": 0.8643457382953181, "grad_norm": 1.3441357612609863, "learning_rate": 0.00017600000000000002, "loss": 1.83, "step": 45 }, { "epoch": 0.9603841536614646, "grad_norm": 0.9121316075325012, "learning_rate": 0.000196, "loss": 1.7398, "step": 50 }, { "epoch": 1.0384153661464586, "grad_norm": 0.7634927034378052, "learning_rate": 0.00019627906976744185, "loss": 1.8408, "step": 55 }, { "epoch": 1.134453781512605, "grad_norm": 0.8097577691078186, "learning_rate": 0.0001916279069767442, "loss": 1.6271, "step": 60 }, { "epoch": 1.2304921968787514, "grad_norm": 0.7660069465637207, "learning_rate": 0.00018697674418604652, "loss": 1.707, "step": 65 }, { "epoch": 1.3265306122448979, "grad_norm": 0.7814516425132751, "learning_rate": 0.00018232558139534886, "loss": 1.6397, "step": 70 }, { "epoch": 1.4225690276110443, "grad_norm": 1.1940962076187134, "learning_rate": 0.00017767441860465117, "loss": 1.6707, "step": 75 }, { "epoch": 1.5186074429771907, "grad_norm": 1.1954811811447144, "learning_rate": 0.00017302325581395348, "loss": 1.5648, "step": 80 }, { "epoch": 1.6146458583433372, "grad_norm": 1.1031644344329834, "learning_rate": 0.00016837209302325584, "loss": 1.5376, "step": 85 }, { "epoch": 1.7106842737094838, "grad_norm": 1.2271337509155273, "learning_rate": 0.00016372093023255815, "loss": 1.5206, "step": 90 }, { "epoch": 1.8067226890756303, "grad_norm": 1.5525450706481934, "learning_rate": 0.00015906976744186046, "loss": 1.4246, "step": 95 }, { "epoch": 1.9027611044417767, "grad_norm": 1.0928661823272705, "learning_rate": 0.0001544186046511628, "loss": 1.4618, "step": 100 }, { "epoch": 1.9987995198079231, "grad_norm": 1.430029273033142, "learning_rate": 0.0001497674418604651, "loss": 1.373, "step": 105 }, { "epoch": 2.076830732292917, "grad_norm": 1.303673505783081, "learning_rate": 0.00014511627906976747, "loss": 1.2444, "step": 110 }, { "epoch": 2.1728691476590636, "grad_norm": 1.9656398296356201, "learning_rate": 0.00014046511627906978, "loss": 1.2816, "step": 115 }, { "epoch": 2.26890756302521, "grad_norm": 1.633468747138977, "learning_rate": 0.0001358139534883721, "loss": 1.2165, "step": 120 }, { "epoch": 2.3649459783913565, "grad_norm": 1.5862339735031128, "learning_rate": 0.00013116279069767442, "loss": 1.2797, "step": 125 }, { "epoch": 2.460984393757503, "grad_norm": 1.2453504800796509, "learning_rate": 0.00012651162790697676, "loss": 1.2173, "step": 130 }, { "epoch": 2.5570228091236493, "grad_norm": 1.7630853652954102, "learning_rate": 0.00012186046511627907, "loss": 1.2752, "step": 135 }, { "epoch": 2.6530612244897958, "grad_norm": 2.465102195739746, "learning_rate": 0.00011720930232558141, "loss": 1.12, "step": 140 }, { "epoch": 2.7490996398559426, "grad_norm": 2.226787805557251, "learning_rate": 0.00011255813953488372, "loss": 1.1776, "step": 145 }, { "epoch": 2.8451380552220886, "grad_norm": 2.306854248046875, "learning_rate": 0.00010790697674418607, "loss": 1.1599, "step": 150 }, { "epoch": 2.9411764705882355, "grad_norm": 1.749025821685791, "learning_rate": 0.00010325581395348838, "loss": 1.089, "step": 155 }, { "epoch": 3.0192076830732293, "grad_norm": 2.295750856399536, "learning_rate": 9.86046511627907e-05, "loss": 1.2437, "step": 160 }, { "epoch": 3.1152460984393757, "grad_norm": 2.4508137702941895, "learning_rate": 9.395348837209302e-05, "loss": 1.097, "step": 165 }, { "epoch": 3.211284513805522, "grad_norm": 2.6268515586853027, "learning_rate": 8.930232558139535e-05, "loss": 1.0338, "step": 170 }, { "epoch": 3.3073229291716686, "grad_norm": 2.7699310779571533, "learning_rate": 8.465116279069768e-05, "loss": 1.0605, "step": 175 }, { "epoch": 3.403361344537815, "grad_norm": 2.2988967895507812, "learning_rate": 8e-05, "loss": 1.0653, "step": 180 }, { "epoch": 3.4993997599039615, "grad_norm": 3.493929386138916, "learning_rate": 7.534883720930233e-05, "loss": 0.9758, "step": 185 }, { "epoch": 3.595438175270108, "grad_norm": 1.8273813724517822, "learning_rate": 7.069767441860465e-05, "loss": 0.9684, "step": 190 }, { "epoch": 3.6914765906362543, "grad_norm": 2.5960426330566406, "learning_rate": 6.604651162790698e-05, "loss": 0.9856, "step": 195 }, { "epoch": 3.787515006002401, "grad_norm": 2.3766703605651855, "learning_rate": 6.139534883720931e-05, "loss": 0.8215, "step": 200 }, { "epoch": 3.883553421368547, "grad_norm": 1.8748396635055542, "learning_rate": 5.674418604651163e-05, "loss": 0.8761, "step": 205 }, { "epoch": 3.979591836734694, "grad_norm": 3.0460057258605957, "learning_rate": 5.209302325581395e-05, "loss": 0.872, "step": 210 }, { "epoch": 4.057623049219688, "grad_norm": 2.0809686183929443, "learning_rate": 4.744186046511628e-05, "loss": 0.7135, "step": 215 }, { "epoch": 4.153661464585834, "grad_norm": 2.5814781188964844, "learning_rate": 4.2790697674418605e-05, "loss": 0.957, "step": 220 }, { "epoch": 4.249699879951981, "grad_norm": 2.4936790466308594, "learning_rate": 3.8139534883720935e-05, "loss": 0.9361, "step": 225 }, { "epoch": 4.345738295318127, "grad_norm": 2.3625705242156982, "learning_rate": 3.348837209302326e-05, "loss": 0.8608, "step": 230 }, { "epoch": 4.441776710684274, "grad_norm": 1.8307788372039795, "learning_rate": 2.8837209302325585e-05, "loss": 0.8447, "step": 235 }, { "epoch": 4.53781512605042, "grad_norm": 2.044945001602173, "learning_rate": 2.4186046511627908e-05, "loss": 0.8737, "step": 240 }, { "epoch": 4.633853541416567, "grad_norm": 2.2939560413360596, "learning_rate": 1.9534883720930235e-05, "loss": 0.9019, "step": 245 }, { "epoch": 4.729891956782713, "grad_norm": 2.2410664558410645, "learning_rate": 1.488372093023256e-05, "loss": 0.7932, "step": 250 }, { "epoch": 4.82593037214886, "grad_norm": 2.9812769889831543, "learning_rate": 1.0232558139534884e-05, "loss": 0.8736, "step": 255 }, { "epoch": 4.921968787515006, "grad_norm": 2.267162799835205, "learning_rate": 5.581395348837209e-06, "loss": 0.7703, "step": 260 }, { "epoch": 5.0, "grad_norm": 10.240612030029297, "learning_rate": 9.30232558139535e-07, "loss": 0.853, "step": 265 } ], "logging_steps": 5, "max_steps": 265, "num_input_tokens_seen": 0, "num_train_epochs": 5, "save_steps": 50, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 9.10237507780608e+16, "train_batch_size": 1, "trial_name": null, "trial_params": null }