| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 4.0, |
| "eval_steps": 500, |
| "global_step": 568, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.035211267605633804, |
| "grad_norm": 6.78125, |
| "learning_rate": 9.23356539870189e-06, |
| "loss": 1.7337, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.07042253521126761, |
| "grad_norm": 5.96875, |
| "learning_rate": 2.077552214707925e-05, |
| "loss": 1.5631, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.1056338028169014, |
| "grad_norm": 4.8125, |
| "learning_rate": 3.231747889545661e-05, |
| "loss": 1.5268, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.14084507042253522, |
| "grad_norm": 4.96875, |
| "learning_rate": 4.3859435643833976e-05, |
| "loss": 1.315, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.176056338028169, |
| "grad_norm": 3.515625, |
| "learning_rate": 5.5401392392211335e-05, |
| "loss": 1.0877, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.2112676056338028, |
| "grad_norm": 3.953125, |
| "learning_rate": 6.69433491405887e-05, |
| "loss": 1.0149, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.24647887323943662, |
| "grad_norm": 3.34375, |
| "learning_rate": 7.848530588896606e-05, |
| "loss": 0.8898, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.28169014084507044, |
| "grad_norm": 2.8125, |
| "learning_rate": 9.002726263734342e-05, |
| "loss": 0.6927, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.31690140845070425, |
| "grad_norm": 2.890625, |
| "learning_rate": 0.00010156921938572079, |
| "loss": 0.6622, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.352112676056338, |
| "grad_norm": 2.71875, |
| "learning_rate": 0.00011311117613409813, |
| "loss": 0.6008, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.3873239436619718, |
| "grad_norm": 2.9375, |
| "learning_rate": 0.0001246531328824755, |
| "loss": 0.5495, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.4225352112676056, |
| "grad_norm": 2.3125, |
| "learning_rate": 0.00013619508963085285, |
| "loss": 0.5099, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.45774647887323944, |
| "grad_norm": 2.859375, |
| "learning_rate": 0.00014773704637923023, |
| "loss": 0.4783, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.49295774647887325, |
| "grad_norm": 1.828125, |
| "learning_rate": 0.0001592790031276076, |
| "loss": 0.4256, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.528169014084507, |
| "grad_norm": 2.671875, |
| "learning_rate": 0.00017082095987598495, |
| "loss": 0.4219, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.5633802816901409, |
| "grad_norm": 2.3125, |
| "learning_rate": 0.0001823629166243623, |
| "loss": 0.3816, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.5985915492957746, |
| "grad_norm": 3.046875, |
| "learning_rate": 0.00019390487337273967, |
| "loss": 0.3935, |
| "step": 85 |
| }, |
| { |
| "epoch": 0.6338028169014085, |
| "grad_norm": 2.828125, |
| "learning_rate": 0.00019620338945371542, |
| "loss": 0.3491, |
| "step": 90 |
| }, |
| { |
| "epoch": 0.6690140845070423, |
| "grad_norm": 2.53125, |
| "learning_rate": 0.0001961632757176071, |
| "loss": 0.3588, |
| "step": 95 |
| }, |
| { |
| "epoch": 0.704225352112676, |
| "grad_norm": 2.4375, |
| "learning_rate": 0.00019609232312076385, |
| "loss": 0.3503, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.7394366197183099, |
| "grad_norm": 2.984375, |
| "learning_rate": 0.00019599056142107619, |
| "loss": 0.3755, |
| "step": 105 |
| }, |
| { |
| "epoch": 0.7746478873239436, |
| "grad_norm": 3.578125, |
| "learning_rate": 0.00019585803329793365, |
| "loss": 0.2912, |
| "step": 110 |
| }, |
| { |
| "epoch": 0.8098591549295775, |
| "grad_norm": 1.8203125, |
| "learning_rate": 0.0001956947943343249, |
| "loss": 0.3251, |
| "step": 115 |
| }, |
| { |
| "epoch": 0.8450704225352113, |
| "grad_norm": 1.6796875, |
| "learning_rate": 0.0001955009129935257, |
| "loss": 0.2976, |
| "step": 120 |
| }, |
| { |
| "epoch": 0.8802816901408451, |
| "grad_norm": 1.96875, |
| "learning_rate": 0.00019527647059038538, |
| "loss": 0.2934, |
| "step": 125 |
| }, |
| { |
| "epoch": 0.9154929577464789, |
| "grad_norm": 1.765625, |
| "learning_rate": 0.00019502156125722272, |
| "loss": 0.2827, |
| "step": 130 |
| }, |
| { |
| "epoch": 0.9507042253521126, |
| "grad_norm": 1.78125, |
| "learning_rate": 0.00019473629190434644, |
| "loss": 0.2146, |
| "step": 135 |
| }, |
| { |
| "epoch": 0.9859154929577465, |
| "grad_norm": 1.4765625, |
| "learning_rate": 0.00019442078217521644, |
| "loss": 0.2386, |
| "step": 140 |
| }, |
| { |
| "epoch": 1.0, |
| "eval_loss": 0.2483566701412201, |
| "eval_runtime": 13.5632, |
| "eval_samples_per_second": 14.746, |
| "eval_steps_per_second": 14.746, |
| "step": 142 |
| }, |
| { |
| "epoch": 1.0211267605633803, |
| "grad_norm": 1.34375, |
| "learning_rate": 0.00019407516439626476, |
| "loss": 0.1576, |
| "step": 145 |
| }, |
| { |
| "epoch": 1.056338028169014, |
| "grad_norm": 1.3125, |
| "learning_rate": 0.00019369958352139717, |
| "loss": 0.1596, |
| "step": 150 |
| }, |
| { |
| "epoch": 1.091549295774648, |
| "grad_norm": 1.140625, |
| "learning_rate": 0.0001932941970711986, |
| "loss": 0.1234, |
| "step": 155 |
| }, |
| { |
| "epoch": 1.1267605633802817, |
| "grad_norm": 2.28125, |
| "learning_rate": 0.00019285917506686845, |
| "loss": 0.1538, |
| "step": 160 |
| }, |
| { |
| "epoch": 1.1619718309859155, |
| "grad_norm": 1.4375, |
| "learning_rate": 0.00019239469995891258, |
| "loss": 0.162, |
| "step": 165 |
| }, |
| { |
| "epoch": 1.1971830985915493, |
| "grad_norm": 1.6171875, |
| "learning_rate": 0.00019190096655062266, |
| "loss": 0.1486, |
| "step": 170 |
| }, |
| { |
| "epoch": 1.232394366197183, |
| "grad_norm": 1.3359375, |
| "learning_rate": 0.0001913781819163747, |
| "loss": 0.1008, |
| "step": 175 |
| }, |
| { |
| "epoch": 1.267605633802817, |
| "grad_norm": 1.90625, |
| "learning_rate": 0.00019082656531478102, |
| "loss": 0.1478, |
| "step": 180 |
| }, |
| { |
| "epoch": 1.3028169014084507, |
| "grad_norm": 1.6484375, |
| "learning_rate": 0.00019024634809673195, |
| "loss": 0.178, |
| "step": 185 |
| }, |
| { |
| "epoch": 1.3380281690140845, |
| "grad_norm": 1.3046875, |
| "learning_rate": 0.00018963777360836602, |
| "loss": 0.1387, |
| "step": 190 |
| }, |
| { |
| "epoch": 1.3732394366197183, |
| "grad_norm": 1.1640625, |
| "learning_rate": 0.0001890010970890094, |
| "loss": 0.1286, |
| "step": 195 |
| }, |
| { |
| "epoch": 1.408450704225352, |
| "grad_norm": 1.234375, |
| "learning_rate": 0.0001883365855641272, |
| "loss": 0.1432, |
| "step": 200 |
| }, |
| { |
| "epoch": 1.443661971830986, |
| "grad_norm": 1.0625, |
| "learning_rate": 0.00018764451773333154, |
| "loss": 0.1314, |
| "step": 205 |
| }, |
| { |
| "epoch": 1.4788732394366197, |
| "grad_norm": 1.6328125, |
| "learning_rate": 0.00018692518385349347, |
| "loss": 0.1008, |
| "step": 210 |
| }, |
| { |
| "epoch": 1.5140845070422535, |
| "grad_norm": 1.71875, |
| "learning_rate": 0.0001861788856170078, |
| "loss": 0.0998, |
| "step": 215 |
| }, |
| { |
| "epoch": 1.5492957746478875, |
| "grad_norm": 2.109375, |
| "learning_rate": 0.00018540593602526156, |
| "loss": 0.1174, |
| "step": 220 |
| }, |
| { |
| "epoch": 1.584507042253521, |
| "grad_norm": 1.5234375, |
| "learning_rate": 0.00018460665925735983, |
| "loss": 0.1313, |
| "step": 225 |
| }, |
| { |
| "epoch": 1.619718309859155, |
| "grad_norm": 1.5390625, |
| "learning_rate": 0.0001837813905341631, |
| "loss": 0.106, |
| "step": 230 |
| }, |
| { |
| "epoch": 1.6549295774647887, |
| "grad_norm": 1.03125, |
| "learning_rate": 0.00018293047597769393, |
| "loss": 0.0816, |
| "step": 235 |
| }, |
| { |
| "epoch": 1.6901408450704225, |
| "grad_norm": 1.421875, |
| "learning_rate": 0.00018205427246597178, |
| "loss": 0.12, |
| "step": 240 |
| }, |
| { |
| "epoch": 1.7253521126760565, |
| "grad_norm": 1.265625, |
| "learning_rate": 0.00018115314748333612, |
| "loss": 0.1031, |
| "step": 245 |
| }, |
| { |
| "epoch": 1.76056338028169, |
| "grad_norm": 0.9921875, |
| "learning_rate": 0.00018022747896632185, |
| "loss": 0.0974, |
| "step": 250 |
| }, |
| { |
| "epoch": 1.795774647887324, |
| "grad_norm": 1.1328125, |
| "learning_rate": 0.0001792776551451508, |
| "loss": 0.083, |
| "step": 255 |
| }, |
| { |
| "epoch": 1.8309859154929577, |
| "grad_norm": 0.6171875, |
| "learning_rate": 0.0001783040743809056, |
| "loss": 0.0782, |
| "step": 260 |
| }, |
| { |
| "epoch": 1.8661971830985915, |
| "grad_norm": 1.375, |
| "learning_rate": 0.00017730714499845533, |
| "loss": 0.0951, |
| "step": 265 |
| }, |
| { |
| "epoch": 1.9014084507042255, |
| "grad_norm": 0.94921875, |
| "learning_rate": 0.00017628728511520167, |
| "loss": 0.093, |
| "step": 270 |
| }, |
| { |
| "epoch": 1.936619718309859, |
| "grad_norm": 0.80859375, |
| "learning_rate": 0.00017524492246571847, |
| "loss": 0.0913, |
| "step": 275 |
| }, |
| { |
| "epoch": 1.971830985915493, |
| "grad_norm": 0.94921875, |
| "learning_rate": 0.00017418049422235742, |
| "loss": 0.081, |
| "step": 280 |
| }, |
| { |
| "epoch": 2.0, |
| "eval_loss": 0.12189739942550659, |
| "eval_runtime": 13.5415, |
| "eval_samples_per_second": 14.769, |
| "eval_steps_per_second": 14.769, |
| "step": 284 |
| }, |
| { |
| "epoch": 2.007042253521127, |
| "grad_norm": 0.70703125, |
| "learning_rate": 0.00017309444681189588, |
| "loss": 0.0673, |
| "step": 285 |
| }, |
| { |
| "epoch": 2.0422535211267605, |
| "grad_norm": 0.6875, |
| "learning_rate": 0.00017198723572830297, |
| "loss": 0.0346, |
| "step": 290 |
| }, |
| { |
| "epoch": 2.0774647887323945, |
| "grad_norm": 0.82421875, |
| "learning_rate": 0.00017085932534170327, |
| "loss": 0.0389, |
| "step": 295 |
| }, |
| { |
| "epoch": 2.112676056338028, |
| "grad_norm": 1.6015625, |
| "learning_rate": 0.00016971118870361723, |
| "loss": 0.0378, |
| "step": 300 |
| }, |
| { |
| "epoch": 2.147887323943662, |
| "grad_norm": 0.6171875, |
| "learning_rate": 0.00016854330734856118, |
| "loss": 0.0377, |
| "step": 305 |
| }, |
| { |
| "epoch": 2.183098591549296, |
| "grad_norm": 0.59765625, |
| "learning_rate": 0.00016735617109208897, |
| "loss": 0.05, |
| "step": 310 |
| }, |
| { |
| "epoch": 2.2183098591549295, |
| "grad_norm": 0.71875, |
| "learning_rate": 0.00016615027782536105, |
| "loss": 0.0418, |
| "step": 315 |
| }, |
| { |
| "epoch": 2.2535211267605635, |
| "grad_norm": 0.8203125, |
| "learning_rate": 0.00016492613330632605, |
| "loss": 0.0279, |
| "step": 320 |
| }, |
| { |
| "epoch": 2.288732394366197, |
| "grad_norm": 1.0546875, |
| "learning_rate": 0.0001636842509476034, |
| "loss": 0.0399, |
| "step": 325 |
| }, |
| { |
| "epoch": 2.323943661971831, |
| "grad_norm": 0.640625, |
| "learning_rate": 0.00016242515160115535, |
| "loss": 0.0466, |
| "step": 330 |
| }, |
| { |
| "epoch": 2.359154929577465, |
| "grad_norm": 0.7734375, |
| "learning_rate": 0.00016114936333983893, |
| "loss": 0.0464, |
| "step": 335 |
| }, |
| { |
| "epoch": 2.3943661971830985, |
| "grad_norm": 1.015625, |
| "learning_rate": 0.00015985742123592942, |
| "loss": 0.0317, |
| "step": 340 |
| }, |
| { |
| "epoch": 2.4295774647887325, |
| "grad_norm": 0.6328125, |
| "learning_rate": 0.00015854986713670827, |
| "loss": 0.0278, |
| "step": 345 |
| }, |
| { |
| "epoch": 2.464788732394366, |
| "grad_norm": 0.50390625, |
| "learning_rate": 0.00015722724943720946, |
| "loss": 0.0337, |
| "step": 350 |
| }, |
| { |
| "epoch": 2.5, |
| "grad_norm": 0.390625, |
| "learning_rate": 0.00015589012285021977, |
| "loss": 0.0394, |
| "step": 355 |
| }, |
| { |
| "epoch": 2.535211267605634, |
| "grad_norm": 0.7265625, |
| "learning_rate": 0.0001545390481736294, |
| "loss": 0.0306, |
| "step": 360 |
| }, |
| { |
| "epoch": 2.5704225352112675, |
| "grad_norm": 1.4296875, |
| "learning_rate": 0.00015317459205523032, |
| "loss": 0.031, |
| "step": 365 |
| }, |
| { |
| "epoch": 2.6056338028169015, |
| "grad_norm": 0.3984375, |
| "learning_rate": 0.00015179732675506114, |
| "loss": 0.0389, |
| "step": 370 |
| }, |
| { |
| "epoch": 2.640845070422535, |
| "grad_norm": 0.7265625, |
| "learning_rate": 0.00015040782990539858, |
| "loss": 0.0292, |
| "step": 375 |
| }, |
| { |
| "epoch": 2.676056338028169, |
| "grad_norm": 0.51953125, |
| "learning_rate": 0.00014900668426849503, |
| "loss": 0.03, |
| "step": 380 |
| }, |
| { |
| "epoch": 2.711267605633803, |
| "grad_norm": 0.478515625, |
| "learning_rate": 0.00014759447749216532, |
| "loss": 0.0317, |
| "step": 385 |
| }, |
| { |
| "epoch": 2.7464788732394365, |
| "grad_norm": 1.9140625, |
| "learning_rate": 0.00014617180186332412, |
| "loss": 0.0266, |
| "step": 390 |
| }, |
| { |
| "epoch": 2.7816901408450705, |
| "grad_norm": 0.494140625, |
| "learning_rate": 0.00014473925405957756, |
| "loss": 0.021, |
| "step": 395 |
| }, |
| { |
| "epoch": 2.816901408450704, |
| "grad_norm": 0.494140625, |
| "learning_rate": 0.00014329743489897354, |
| "loss": 0.025, |
| "step": 400 |
| }, |
| { |
| "epoch": 2.852112676056338, |
| "grad_norm": 0.7265625, |
| "learning_rate": 0.00014184694908801575, |
| "loss": 0.0227, |
| "step": 405 |
| }, |
| { |
| "epoch": 2.887323943661972, |
| "grad_norm": 0.48828125, |
| "learning_rate": 0.00014038840496804626, |
| "loss": 0.0239, |
| "step": 410 |
| }, |
| { |
| "epoch": 2.9225352112676055, |
| "grad_norm": 0.70703125, |
| "learning_rate": 0.00013892241426010418, |
| "loss": 0.0273, |
| "step": 415 |
| }, |
| { |
| "epoch": 2.9577464788732395, |
| "grad_norm": 0.322265625, |
| "learning_rate": 0.00013744959180836652, |
| "loss": 0.0172, |
| "step": 420 |
| }, |
| { |
| "epoch": 2.992957746478873, |
| "grad_norm": 0.52734375, |
| "learning_rate": 0.00013597055532227952, |
| "loss": 0.0275, |
| "step": 425 |
| }, |
| { |
| "epoch": 3.0, |
| "eval_loss": 0.07534319907426834, |
| "eval_runtime": 13.5825, |
| "eval_samples_per_second": 14.725, |
| "eval_steps_per_second": 14.725, |
| "step": 426 |
| }, |
| { |
| "epoch": 3.028169014084507, |
| "grad_norm": 0.439453125, |
| "learning_rate": 0.00013448592511748788, |
| "loss": 0.0107, |
| "step": 430 |
| }, |
| { |
| "epoch": 3.063380281690141, |
| "grad_norm": 0.341796875, |
| "learning_rate": 0.00013299632385567123, |
| "loss": 0.0188, |
| "step": 435 |
| }, |
| { |
| "epoch": 3.0985915492957745, |
| "grad_norm": 0.39453125, |
| "learning_rate": 0.0001315023762833965, |
| "loss": 0.014, |
| "step": 440 |
| }, |
| { |
| "epoch": 3.1338028169014085, |
| "grad_norm": 0.671875, |
| "learning_rate": 0.0001300047089700961, |
| "loss": 0.0215, |
| "step": 445 |
| }, |
| { |
| "epoch": 3.169014084507042, |
| "grad_norm": 0.54296875, |
| "learning_rate": 0.00012850395004528108, |
| "loss": 0.0149, |
| "step": 450 |
| }, |
| { |
| "epoch": 3.204225352112676, |
| "grad_norm": 0.3359375, |
| "learning_rate": 0.00012700072893510078, |
| "loss": 0.014, |
| "step": 455 |
| }, |
| { |
| "epoch": 3.23943661971831, |
| "grad_norm": 0.73046875, |
| "learning_rate": 0.0001254956760983578, |
| "loss": 0.0205, |
| "step": 460 |
| }, |
| { |
| "epoch": 3.2746478873239435, |
| "grad_norm": 0.330078125, |
| "learning_rate": 0.00012398942276209052, |
| "loss": 0.0176, |
| "step": 465 |
| }, |
| { |
| "epoch": 3.3098591549295775, |
| "grad_norm": 0.53125, |
| "learning_rate": 0.00012248260065683312, |
| "loss": 0.014, |
| "step": 470 |
| }, |
| { |
| "epoch": 3.345070422535211, |
| "grad_norm": 0.17578125, |
| "learning_rate": 0.00012097584175166445, |
| "loss": 0.0108, |
| "step": 475 |
| }, |
| { |
| "epoch": 3.380281690140845, |
| "grad_norm": 0.42578125, |
| "learning_rate": 0.00011946977798915689, |
| "loss": 0.0145, |
| "step": 480 |
| }, |
| { |
| "epoch": 3.415492957746479, |
| "grad_norm": 0.1806640625, |
| "learning_rate": 0.00011796504102033635, |
| "loss": 0.0108, |
| "step": 485 |
| }, |
| { |
| "epoch": 3.4507042253521125, |
| "grad_norm": 0.2119140625, |
| "learning_rate": 0.00011646226193976462, |
| "loss": 0.0112, |
| "step": 490 |
| }, |
| { |
| "epoch": 3.4859154929577465, |
| "grad_norm": 0.5703125, |
| "learning_rate": 0.00011496207102085474, |
| "loss": 0.0182, |
| "step": 495 |
| }, |
| { |
| "epoch": 3.52112676056338, |
| "grad_norm": 0.50390625, |
| "learning_rate": 0.00011346509745153135, |
| "loss": 0.0162, |
| "step": 500 |
| }, |
| { |
| "epoch": 3.52112676056338, |
| "eval_loss": 0.07581495493650436, |
| "eval_runtime": 13.6223, |
| "eval_samples_per_second": 14.682, |
| "eval_steps_per_second": 14.682, |
| "step": 500 |
| }, |
| { |
| "epoch": 3.556338028169014, |
| "grad_norm": 0.94140625, |
| "learning_rate": 0.00011197196907034575, |
| "loss": 0.0191, |
| "step": 505 |
| }, |
| { |
| "epoch": 3.591549295774648, |
| "grad_norm": 0.3515625, |
| "learning_rate": 0.00011048331210315723, |
| "loss": 0.0159, |
| "step": 510 |
| }, |
| { |
| "epoch": 3.626760563380282, |
| "grad_norm": 0.20703125, |
| "learning_rate": 0.00010899975090049065, |
| "loss": 0.0136, |
| "step": 515 |
| }, |
| { |
| "epoch": 3.6619718309859155, |
| "grad_norm": 0.18359375, |
| "learning_rate": 0.00010752190767568055, |
| "loss": 0.0126, |
| "step": 520 |
| }, |
| { |
| "epoch": 3.697183098591549, |
| "grad_norm": 0.49609375, |
| "learning_rate": 0.00010605040224391162, |
| "loss": 0.0211, |
| "step": 525 |
| }, |
| { |
| "epoch": 3.732394366197183, |
| "grad_norm": 0.1298828125, |
| "learning_rate": 0.00010458585176226496, |
| "loss": 0.0124, |
| "step": 530 |
| }, |
| { |
| "epoch": 3.767605633802817, |
| "grad_norm": 0.169921875, |
| "learning_rate": 0.00010312887047087915, |
| "loss": 0.0105, |
| "step": 535 |
| }, |
| { |
| "epoch": 3.802816901408451, |
| "grad_norm": 0.32421875, |
| "learning_rate": 0.0001016800694353349, |
| "loss": 0.0149, |
| "step": 540 |
| }, |
| { |
| "epoch": 3.8380281690140845, |
| "grad_norm": 0.1884765625, |
| "learning_rate": 0.00010024005629037063, |
| "loss": 0.0098, |
| "step": 545 |
| }, |
| { |
| "epoch": 3.873239436619718, |
| "grad_norm": 0.443359375, |
| "learning_rate": 9.88094349850375e-05, |
| "loss": 0.0139, |
| "step": 550 |
| }, |
| { |
| "epoch": 3.908450704225352, |
| "grad_norm": 0.1259765625, |
| "learning_rate": 9.738880552939999e-05, |
| "loss": 0.0106, |
| "step": 555 |
| }, |
| { |
| "epoch": 3.943661971830986, |
| "grad_norm": 0.1884765625, |
| "learning_rate": 9.597876374288852e-05, |
| "loss": 0.0109, |
| "step": 560 |
| }, |
| { |
| "epoch": 3.97887323943662, |
| "grad_norm": 0.20703125, |
| "learning_rate": 9.457990100440958e-05, |
| "loss": 0.0133, |
| "step": 565 |
| }, |
| { |
| "epoch": 4.0, |
| "eval_loss": 0.06904773414134979, |
| "eval_runtime": 13.6132, |
| "eval_samples_per_second": 14.692, |
| "eval_steps_per_second": 14.692, |
| "step": 568 |
| } |
| ], |
| "logging_steps": 5, |
| "max_steps": 852, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 6, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 3.130372693992407e+17, |
| "train_batch_size": 14, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|