kvarela's picture
initial
3b64d62 verified
Raw
History Blame Contribute Delete
7.59 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 3.787515006002401,
"eval_steps": 500,
"global_step": 200,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.09603841536614646,
"grad_norm": 0.7072356343269348,
"learning_rate": 1.6000000000000003e-05,
"loss": 3.1026,
"step": 5
},
{
"epoch": 0.19207683073229292,
"grad_norm": 0.7724522948265076,
"learning_rate": 3.6e-05,
"loss": 3.1732,
"step": 10
},
{
"epoch": 0.28811524609843936,
"grad_norm": 0.9103713035583496,
"learning_rate": 5.6000000000000006e-05,
"loss": 3.1209,
"step": 15
},
{
"epoch": 0.38415366146458585,
"grad_norm": 0.9525737166404724,
"learning_rate": 7.6e-05,
"loss": 3.0653,
"step": 20
},
{
"epoch": 0.4801920768307323,
"grad_norm": 0.9877885580062866,
"learning_rate": 9.6e-05,
"loss": 2.7498,
"step": 25
},
{
"epoch": 0.5762304921968787,
"grad_norm": 0.8707166910171509,
"learning_rate": 0.000116,
"loss": 2.6275,
"step": 30
},
{
"epoch": 0.6722689075630253,
"grad_norm": 1.3398517370224,
"learning_rate": 0.00013600000000000003,
"loss": 2.475,
"step": 35
},
{
"epoch": 0.7683073229291717,
"grad_norm": 1.2803456783294678,
"learning_rate": 0.00015600000000000002,
"loss": 2.207,
"step": 40
},
{
"epoch": 0.8643457382953181,
"grad_norm": 1.3441357612609863,
"learning_rate": 0.00017600000000000002,
"loss": 1.83,
"step": 45
},
{
"epoch": 0.9603841536614646,
"grad_norm": 0.9121316075325012,
"learning_rate": 0.000196,
"loss": 1.7398,
"step": 50
},
{
"epoch": 1.0384153661464586,
"grad_norm": 0.7634927034378052,
"learning_rate": 0.00019627906976744185,
"loss": 1.8408,
"step": 55
},
{
"epoch": 1.134453781512605,
"grad_norm": 0.8097577691078186,
"learning_rate": 0.0001916279069767442,
"loss": 1.6271,
"step": 60
},
{
"epoch": 1.2304921968787514,
"grad_norm": 0.7660069465637207,
"learning_rate": 0.00018697674418604652,
"loss": 1.707,
"step": 65
},
{
"epoch": 1.3265306122448979,
"grad_norm": 0.7814516425132751,
"learning_rate": 0.00018232558139534886,
"loss": 1.6397,
"step": 70
},
{
"epoch": 1.4225690276110443,
"grad_norm": 1.1940962076187134,
"learning_rate": 0.00017767441860465117,
"loss": 1.6707,
"step": 75
},
{
"epoch": 1.5186074429771907,
"grad_norm": 1.1954811811447144,
"learning_rate": 0.00017302325581395348,
"loss": 1.5648,
"step": 80
},
{
"epoch": 1.6146458583433372,
"grad_norm": 1.1031644344329834,
"learning_rate": 0.00016837209302325584,
"loss": 1.5376,
"step": 85
},
{
"epoch": 1.7106842737094838,
"grad_norm": 1.2271337509155273,
"learning_rate": 0.00016372093023255815,
"loss": 1.5206,
"step": 90
},
{
"epoch": 1.8067226890756303,
"grad_norm": 1.5525450706481934,
"learning_rate": 0.00015906976744186046,
"loss": 1.4246,
"step": 95
},
{
"epoch": 1.9027611044417767,
"grad_norm": 1.0928661823272705,
"learning_rate": 0.0001544186046511628,
"loss": 1.4618,
"step": 100
},
{
"epoch": 1.9987995198079231,
"grad_norm": 1.430029273033142,
"learning_rate": 0.0001497674418604651,
"loss": 1.373,
"step": 105
},
{
"epoch": 2.076830732292917,
"grad_norm": 1.303673505783081,
"learning_rate": 0.00014511627906976747,
"loss": 1.2444,
"step": 110
},
{
"epoch": 2.1728691476590636,
"grad_norm": 1.9656398296356201,
"learning_rate": 0.00014046511627906978,
"loss": 1.2816,
"step": 115
},
{
"epoch": 2.26890756302521,
"grad_norm": 1.633468747138977,
"learning_rate": 0.0001358139534883721,
"loss": 1.2165,
"step": 120
},
{
"epoch": 2.3649459783913565,
"grad_norm": 1.5862339735031128,
"learning_rate": 0.00013116279069767442,
"loss": 1.2797,
"step": 125
},
{
"epoch": 2.460984393757503,
"grad_norm": 1.2453504800796509,
"learning_rate": 0.00012651162790697676,
"loss": 1.2173,
"step": 130
},
{
"epoch": 2.5570228091236493,
"grad_norm": 1.7630853652954102,
"learning_rate": 0.00012186046511627907,
"loss": 1.2752,
"step": 135
},
{
"epoch": 2.6530612244897958,
"grad_norm": 2.465102195739746,
"learning_rate": 0.00011720930232558141,
"loss": 1.12,
"step": 140
},
{
"epoch": 2.7490996398559426,
"grad_norm": 2.226787805557251,
"learning_rate": 0.00011255813953488372,
"loss": 1.1776,
"step": 145
},
{
"epoch": 2.8451380552220886,
"grad_norm": 2.306854248046875,
"learning_rate": 0.00010790697674418607,
"loss": 1.1599,
"step": 150
},
{
"epoch": 2.9411764705882355,
"grad_norm": 1.749025821685791,
"learning_rate": 0.00010325581395348838,
"loss": 1.089,
"step": 155
},
{
"epoch": 3.0192076830732293,
"grad_norm": 2.295750856399536,
"learning_rate": 9.86046511627907e-05,
"loss": 1.2437,
"step": 160
},
{
"epoch": 3.1152460984393757,
"grad_norm": 2.4508137702941895,
"learning_rate": 9.395348837209302e-05,
"loss": 1.097,
"step": 165
},
{
"epoch": 3.211284513805522,
"grad_norm": 2.6268515586853027,
"learning_rate": 8.930232558139535e-05,
"loss": 1.0338,
"step": 170
},
{
"epoch": 3.3073229291716686,
"grad_norm": 2.7699310779571533,
"learning_rate": 8.465116279069768e-05,
"loss": 1.0605,
"step": 175
},
{
"epoch": 3.403361344537815,
"grad_norm": 2.2988967895507812,
"learning_rate": 8e-05,
"loss": 1.0653,
"step": 180
},
{
"epoch": 3.4993997599039615,
"grad_norm": 3.493929386138916,
"learning_rate": 7.534883720930233e-05,
"loss": 0.9758,
"step": 185
},
{
"epoch": 3.595438175270108,
"grad_norm": 1.8273813724517822,
"learning_rate": 7.069767441860465e-05,
"loss": 0.9684,
"step": 190
},
{
"epoch": 3.6914765906362543,
"grad_norm": 2.5960426330566406,
"learning_rate": 6.604651162790698e-05,
"loss": 0.9856,
"step": 195
},
{
"epoch": 3.787515006002401,
"grad_norm": 2.3766703605651855,
"learning_rate": 6.139534883720931e-05,
"loss": 0.8215,
"step": 200
}
],
"logging_steps": 5,
"max_steps": 265,
"num_input_tokens_seen": 0,
"num_train_epochs": 5,
"save_steps": 50,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 6.89507643949056e+16,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}