abaddon182 commited on
Commit
9f13f98
·
verified ·
1 Parent(s): 996fb2f

Training in progress, step 1500, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f99e35d5fd1075c7f66039ae7553a8b20accd9a21ae8e8d321af76b14a56e319
3
  size 72936
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7dd741a4454ec5b43eb9924c4895e4cdf24e08a48927167a73c24aa37aa1f9ef
3
  size 72936
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:324559ce798f2a672e8e627fedbb8375cfb43d230ec52e61fe3c7362deb5399f
3
  size 154612
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e59b67ef2c123a46b6761940e97c4867d5688bc27b8ee64995ef8543db551f2e
3
  size 154612
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:175868e3df28c2b22cec44c6ea174b07fee7fa72d88e131198f05ca389499a34
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c93bf75027ec35141a6743a1bb774c7870f6cd4648ef4faf5055b346f442127
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6444f03632ac809ddd302ec167645d7a82c68acafc185363d92b0bcd166284dc
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:740f92207430d59e3e915864ad915fc31827de287779f64d8e590410bcf177e5
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
- "best_metric": 12.383695602416992,
3
- "best_model_checkpoint": "miner_id_24/checkpoint-1350",
4
- "epoch": 0.7529280535415505,
5
  "eval_steps": 150,
6
- "global_step": 1350,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -1032,6 +1032,119 @@
1032
  "eval_samples_per_second": 66.201,
1033
  "eval_steps_per_second": 8.286,
1034
  "step": 1350
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1035
  }
1036
  ],
1037
  "logging_steps": 10,
@@ -1055,12 +1168,12 @@
1055
  "should_evaluate": false,
1056
  "should_log": false,
1057
  "should_save": true,
1058
- "should_training_stop": false
1059
  },
1060
  "attributes": {}
1061
  }
1062
  },
1063
- "total_flos": 2603247206400.0,
1064
  "train_batch_size": 8,
1065
  "trial_name": null,
1066
  "trial_params": null
 
1
  {
2
+ "best_metric": 12.383665084838867,
3
+ "best_model_checkpoint": "miner_id_24/checkpoint-1500",
4
+ "epoch": 0.8365867261572784,
5
  "eval_steps": 150,
6
+ "global_step": 1500,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
1032
  "eval_samples_per_second": 66.201,
1033
  "eval_steps_per_second": 8.286,
1034
  "step": 1350
1035
+ },
1036
+ {
1037
+ "epoch": 0.758505298382599,
1038
+ "grad_norm": 0.07976693660020828,
1039
+ "learning_rate": 2.282587464572594e-06,
1040
+ "loss": 12.3927,
1041
+ "step": 1360
1042
+ },
1043
+ {
1044
+ "epoch": 0.7640825432236475,
1045
+ "grad_norm": 0.08914309740066528,
1046
+ "learning_rate": 1.9702322308350674e-06,
1047
+ "loss": 12.3887,
1048
+ "step": 1370
1049
+ },
1050
+ {
1051
+ "epoch": 0.769659788064696,
1052
+ "grad_norm": 0.07854153960943222,
1053
+ "learning_rate": 1.6804223604318825e-06,
1054
+ "loss": 12.3809,
1055
+ "step": 1380
1056
+ },
1057
+ {
1058
+ "epoch": 0.7752370329057445,
1059
+ "grad_norm": 0.11203592270612717,
1060
+ "learning_rate": 1.413293891264722e-06,
1061
+ "loss": 12.3859,
1062
+ "step": 1390
1063
+ },
1064
+ {
1065
+ "epoch": 0.7808142777467931,
1066
+ "grad_norm": 0.15510566532611847,
1067
+ "learning_rate": 1.1689722144956671e-06,
1068
+ "loss": 12.3824,
1069
+ "step": 1400
1070
+ },
1071
+ {
1072
+ "epoch": 0.7863915225878416,
1073
+ "grad_norm": 0.0713285580277443,
1074
+ "learning_rate": 9.475720156880419e-07,
1075
+ "loss": 12.39,
1076
+ "step": 1410
1077
+ },
1078
+ {
1079
+ "epoch": 0.7919687674288901,
1080
+ "grad_norm": 0.0701255202293396,
1081
+ "learning_rate": 7.491972209725806e-07,
1082
+ "loss": 12.3835,
1083
+ "step": 1420
1084
+ },
1085
+ {
1086
+ "epoch": 0.7975460122699386,
1087
+ "grad_norm": 0.08518093079328537,
1088
+ "learning_rate": 5.739409482640956e-07,
1089
+ "loss": 12.3846,
1090
+ "step": 1430
1091
+ },
1092
+ {
1093
+ "epoch": 0.8031232571109872,
1094
+ "grad_norm": 0.0968775674700737,
1095
+ "learning_rate": 4.2188546355153013e-07,
1096
+ "loss": 12.3836,
1097
+ "step": 1440
1098
+ },
1099
+ {
1100
+ "epoch": 0.8087005019520357,
1101
+ "grad_norm": 0.1857132762670517,
1102
+ "learning_rate": 2.9310214228202013e-07,
1103
+ "loss": 12.3894,
1104
+ "step": 1450
1105
+ },
1106
+ {
1107
+ "epoch": 0.8142777467930842,
1108
+ "grad_norm": 0.0695808008313179,
1109
+ "learning_rate": 1.8765143585693922e-07,
1110
+ "loss": 12.3885,
1111
+ "step": 1460
1112
+ },
1113
+ {
1114
+ "epoch": 0.8198549916341328,
1115
+ "grad_norm": 0.09991304576396942,
1116
+ "learning_rate": 1.0558284325578038e-07,
1117
+ "loss": 12.3892,
1118
+ "step": 1470
1119
+ },
1120
+ {
1121
+ "epoch": 0.8254322364751813,
1122
+ "grad_norm": 0.11328619718551636,
1123
+ "learning_rate": 4.6934887801164396e-08,
1124
+ "loss": 12.3822,
1125
+ "step": 1480
1126
+ },
1127
+ {
1128
+ "epoch": 0.8310094813162298,
1129
+ "grad_norm": 0.0934273824095726,
1130
+ "learning_rate": 1.173509907579362e-08,
1131
+ "loss": 12.3941,
1132
+ "step": 1490
1133
+ },
1134
+ {
1135
+ "epoch": 0.8365867261572784,
1136
+ "grad_norm": 0.15032252669334412,
1137
+ "learning_rate": 0.0,
1138
+ "loss": 12.3824,
1139
+ "step": 1500
1140
+ },
1141
+ {
1142
+ "epoch": 0.8365867261572784,
1143
+ "eval_loss": 12.383665084838867,
1144
+ "eval_runtime": 22.815,
1145
+ "eval_samples_per_second": 66.185,
1146
+ "eval_steps_per_second": 8.284,
1147
+ "step": 1500
1148
  }
1149
  ],
1150
  "logging_steps": 10,
 
1168
  "should_evaluate": false,
1169
  "should_log": false,
1170
  "should_save": true,
1171
+ "should_training_stop": true
1172
  },
1173
  "attributes": {}
1174
  }
1175
  },
1176
+ "total_flos": 2892496896000.0,
1177
  "train_batch_size": 8,
1178
  "trial_name": null,
1179
  "trial_params": null