aleegis commited on
Commit
89c6efa
·
verified ·
1 Parent(s): 3f53734

Training in progress, step 1200, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f3932ab744355321baf314cfcf3f680a1a100f6d2cc0cdf70928fca050e52b1d
3
  size 156926880
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d28c490785058699a1e8618cc1052385a14f1c2c1441e89fecafc2205d72d9ca
3
  size 156926880
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dcccf170a4eef4e0c4cdd3ca33e5171ee94ae67610116ac91abe27db39167e8a
3
  size 314002891
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ed23a7b705bd44c54abea95056794b8b513b6c64cbb9ecd27c2015441d4792a2
3
  size 314002891
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6b7a86f1c599f39d8a7fb7c8203b48a6e2a98c7ce96cf4a912160b2dea652b9d
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c195c9fb096cf067be0bf1bfeb335315e6b38d45cb599558e0bfc1c9b068455c
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:30d1a0c1b59fa100b9aabdf7da50a7e8a0b70c1c9662946b0b6100f8b47bc48c
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a717c54c61f00318563a2243900cad87ed16f178d7acf7675538ea64c8f7c0e3
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.026781330436982043,
6
  "eval_steps": 500,
7
- "global_step": 900,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -904,6 +904,307 @@
904
  "learning_rate": 4.1891103844721636e-05,
905
  "loss": 1.8271,
906
  "step": 896
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
907
  }
908
  ],
909
  "logging_steps": 7,
@@ -923,7 +1224,7 @@
923
  "attributes": {}
924
  }
925
  },
926
- "total_flos": 1.788146752684032e+17,
927
  "train_batch_size": 8,
928
  "trial_name": null,
929
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.03570844058264272,
6
  "eval_steps": 500,
7
+ "global_step": 1200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
904
  "learning_rate": 4.1891103844721636e-05,
905
  "loss": 1.8271,
906
  "step": 896
907
+ },
908
+ {
909
+ "epoch": 0.02687060153843865,
910
+ "grad_norm": 2.7906270027160645,
911
+ "learning_rate": 4.108851081114169e-05,
912
+ "loss": 1.7191,
913
+ "step": 903
914
+ },
915
+ {
916
+ "epoch": 0.027078900775170733,
917
+ "grad_norm": 1.597091794013977,
918
+ "learning_rate": 4.028828243900141e-05,
919
+ "loss": 1.7157,
920
+ "step": 910
921
+ },
922
+ {
923
+ "epoch": 0.027287200011902813,
924
+ "grad_norm": 1.7898212671279907,
925
+ "learning_rate": 3.949063106870031e-05,
926
+ "loss": 1.6832,
927
+ "step": 917
928
+ },
929
+ {
930
+ "epoch": 0.027495499248634897,
931
+ "grad_norm": 4.160453796386719,
932
+ "learning_rate": 3.869576835683109e-05,
933
+ "loss": 1.8242,
934
+ "step": 924
935
+ },
936
+ {
937
+ "epoch": 0.027703798485366977,
938
+ "grad_norm": 2.273709535598755,
939
+ "learning_rate": 3.790390522001662e-05,
940
+ "loss": 1.6938,
941
+ "step": 931
942
+ },
943
+ {
944
+ "epoch": 0.02791209772209906,
945
+ "grad_norm": 1.60932195186615,
946
+ "learning_rate": 3.711525177894331e-05,
947
+ "loss": 1.6732,
948
+ "step": 938
949
+ },
950
+ {
951
+ "epoch": 0.028120396958831145,
952
+ "grad_norm": 2.062462091445923,
953
+ "learning_rate": 3.6330017302605576e-05,
954
+ "loss": 1.5316,
955
+ "step": 945
956
+ },
957
+ {
958
+ "epoch": 0.028328696195563226,
959
+ "grad_norm": 1.7259407043457031,
960
+ "learning_rate": 3.554841015277641e-05,
961
+ "loss": 1.5854,
962
+ "step": 952
963
+ },
964
+ {
965
+ "epoch": 0.02853699543229531,
966
+ "grad_norm": 1.953240156173706,
967
+ "learning_rate": 3.477063772871861e-05,
968
+ "loss": 1.4713,
969
+ "step": 959
970
+ },
971
+ {
972
+ "epoch": 0.02874529466902739,
973
+ "grad_norm": 3.1948041915893555,
974
+ "learning_rate": 3.399690641215142e-05,
975
+ "loss": 1.6005,
976
+ "step": 966
977
+ },
978
+ {
979
+ "epoch": 0.028953593905759474,
980
+ "grad_norm": 2.461402177810669,
981
+ "learning_rate": 3.322742151248725e-05,
982
+ "loss": 1.9629,
983
+ "step": 973
984
+ },
985
+ {
986
+ "epoch": 0.029161893142491558,
987
+ "grad_norm": 2.2598836421966553,
988
+ "learning_rate": 3.246238721235283e-05,
989
+ "loss": 1.9584,
990
+ "step": 980
991
+ },
992
+ {
993
+ "epoch": 0.02937019237922364,
994
+ "grad_norm": 1.823764681816101,
995
+ "learning_rate": 3.1702006513409396e-05,
996
+ "loss": 1.4632,
997
+ "step": 987
998
+ },
999
+ {
1000
+ "epoch": 0.029578491615955722,
1001
+ "grad_norm": 1.822013258934021,
1002
+ "learning_rate": 3.09464811824863e-05,
1003
+ "loss": 1.5673,
1004
+ "step": 994
1005
+ },
1006
+ {
1007
+ "epoch": 0.029786790852687803,
1008
+ "grad_norm": 1.8545750379562378,
1009
+ "learning_rate": 3.019601169804216e-05,
1010
+ "loss": 1.5401,
1011
+ "step": 1001
1012
+ },
1013
+ {
1014
+ "epoch": 0.029995090089419887,
1015
+ "grad_norm": 2.142122983932495,
1016
+ "learning_rate": 2.9450797196968023e-05,
1017
+ "loss": 1.541,
1018
+ "step": 1008
1019
+ },
1020
+ {
1021
+ "epoch": 0.03020338932615197,
1022
+ "grad_norm": 2.207737684249878,
1023
+ "learning_rate": 2.8711035421746367e-05,
1024
+ "loss": 1.6909,
1025
+ "step": 1015
1026
+ },
1027
+ {
1028
+ "epoch": 0.03041168856288405,
1029
+ "grad_norm": 1.393803596496582,
1030
+ "learning_rate": 2.7976922667980272e-05,
1031
+ "loss": 1.5975,
1032
+ "step": 1022
1033
+ },
1034
+ {
1035
+ "epoch": 0.030619987799616135,
1036
+ "grad_norm": 2.638187885284424,
1037
+ "learning_rate": 2.7248653732306316e-05,
1038
+ "loss": 1.6129,
1039
+ "step": 1029
1040
+ },
1041
+ {
1042
+ "epoch": 0.030828287036348215,
1043
+ "grad_norm": 1.743645191192627,
1044
+ "learning_rate": 2.6526421860705473e-05,
1045
+ "loss": 1.621,
1046
+ "step": 1036
1047
+ },
1048
+ {
1049
+ "epoch": 0.0310365862730803,
1050
+ "grad_norm": 2.1193864345550537,
1051
+ "learning_rate": 2.581041869722519e-05,
1052
+ "loss": 1.3971,
1053
+ "step": 1043
1054
+ },
1055
+ {
1056
+ "epoch": 0.031244885509812383,
1057
+ "grad_norm": 2.1348867416381836,
1058
+ "learning_rate": 2.5100834233126823e-05,
1059
+ "loss": 1.5024,
1060
+ "step": 1050
1061
+ },
1062
+ {
1063
+ "epoch": 0.03145318474654447,
1064
+ "grad_norm": 1.6680853366851807,
1065
+ "learning_rate": 2.4397856756471432e-05,
1066
+ "loss": 1.3568,
1067
+ "step": 1057
1068
+ },
1069
+ {
1070
+ "epoch": 0.03166148398327655,
1071
+ "grad_norm": 2.6139674186706543,
1072
+ "learning_rate": 2.3701672802157566e-05,
1073
+ "loss": 1.5204,
1074
+ "step": 1064
1075
+ },
1076
+ {
1077
+ "epoch": 0.03186978322000863,
1078
+ "grad_norm": 2.4763057231903076,
1079
+ "learning_rate": 2.3012467102424373e-05,
1080
+ "loss": 1.5295,
1081
+ "step": 1071
1082
+ },
1083
+ {
1084
+ "epoch": 0.032078082456740716,
1085
+ "grad_norm": 1.9144788980484009,
1086
+ "learning_rate": 2.23304225378328e-05,
1087
+ "loss": 1.8166,
1088
+ "step": 1078
1089
+ },
1090
+ {
1091
+ "epoch": 0.032286381693472796,
1092
+ "grad_norm": 1.9162483215332031,
1093
+ "learning_rate": 2.1655720088738453e-05,
1094
+ "loss": 1.6142,
1095
+ "step": 1085
1096
+ },
1097
+ {
1098
+ "epoch": 0.032494680930204876,
1099
+ "grad_norm": 2.354734182357788,
1100
+ "learning_rate": 2.0988538787268374e-05,
1101
+ "loss": 1.7982,
1102
+ "step": 1092
1103
+ },
1104
+ {
1105
+ "epoch": 0.03270298016693696,
1106
+ "grad_norm": 2.0897438526153564,
1107
+ "learning_rate": 2.0329055669814934e-05,
1108
+ "loss": 1.6379,
1109
+ "step": 1099
1110
+ },
1111
+ {
1112
+ "epoch": 0.032911279403669044,
1113
+ "grad_norm": 2.5271339416503906,
1114
+ "learning_rate": 1.9677445730059346e-05,
1115
+ "loss": 1.5626,
1116
+ "step": 1106
1117
+ },
1118
+ {
1119
+ "epoch": 0.033119578640401125,
1120
+ "grad_norm": 1.9526691436767578,
1121
+ "learning_rate": 1.9033881872537006e-05,
1122
+ "loss": 1.5815,
1123
+ "step": 1113
1124
+ },
1125
+ {
1126
+ "epoch": 0.033327877877133205,
1127
+ "grad_norm": 2.1989901065826416,
1128
+ "learning_rate": 1.8398534866757454e-05,
1129
+ "loss": 1.8776,
1130
+ "step": 1120
1131
+ },
1132
+ {
1133
+ "epoch": 0.03353617711386529,
1134
+ "grad_norm": 1.9625588655471802,
1135
+ "learning_rate": 1.7771573301890664e-05,
1136
+ "loss": 1.4475,
1137
+ "step": 1127
1138
+ },
1139
+ {
1140
+ "epoch": 0.03374447635059737,
1141
+ "grad_norm": 2.7089996337890625,
1142
+ "learning_rate": 1.715316354203188e-05,
1143
+ "loss": 1.7354,
1144
+ "step": 1134
1145
+ },
1146
+ {
1147
+ "epoch": 0.03395277558732945,
1148
+ "grad_norm": 2.1591153144836426,
1149
+ "learning_rate": 1.6543469682057106e-05,
1150
+ "loss": 1.6762,
1151
+ "step": 1141
1152
+ },
1153
+ {
1154
+ "epoch": 0.03416107482406154,
1155
+ "grad_norm": 2.157672882080078,
1156
+ "learning_rate": 1.594265350408039e-05,
1157
+ "loss": 1.8244,
1158
+ "step": 1148
1159
+ },
1160
+ {
1161
+ "epoch": 0.03436937406079362,
1162
+ "grad_norm": 2.2080414295196533,
1163
+ "learning_rate": 1.5350874434525142e-05,
1164
+ "loss": 1.6367,
1165
+ "step": 1155
1166
+ },
1167
+ {
1168
+ "epoch": 0.0345776732975257,
1169
+ "grad_norm": 2.246075391769409,
1170
+ "learning_rate": 1.4768289501820265e-05,
1171
+ "loss": 1.6959,
1172
+ "step": 1162
1173
+ },
1174
+ {
1175
+ "epoch": 0.03478597253425778,
1176
+ "grad_norm": 1.6588751077651978,
1177
+ "learning_rate": 1.4195053294732758e-05,
1178
+ "loss": 1.5196,
1179
+ "step": 1169
1180
+ },
1181
+ {
1182
+ "epoch": 0.03499427177098987,
1183
+ "grad_norm": 1.9793294668197632,
1184
+ "learning_rate": 1.3631317921347563e-05,
1185
+ "loss": 1.6595,
1186
+ "step": 1176
1187
+ },
1188
+ {
1189
+ "epoch": 0.03520257100772195,
1190
+ "grad_norm": 1.935530424118042,
1191
+ "learning_rate": 1.3077232968705805e-05,
1192
+ "loss": 1.7511,
1193
+ "step": 1183
1194
+ },
1195
+ {
1196
+ "epoch": 0.03541087024445403,
1197
+ "grad_norm": 1.967477560043335,
1198
+ "learning_rate": 1.2532945463111855e-05,
1199
+ "loss": 1.53,
1200
+ "step": 1190
1201
+ },
1202
+ {
1203
+ "epoch": 0.03561916948118612,
1204
+ "grad_norm": 3.320049524307251,
1205
+ "learning_rate": 1.1998599831119912e-05,
1206
+ "loss": 1.6357,
1207
+ "step": 1197
1208
  }
1209
  ],
1210
  "logging_steps": 7,
 
1224
  "attributes": {}
1225
  }
1226
  },
1227
+ "total_flos": 2.384195670245376e+17,
1228
  "train_batch_size": 8,
1229
  "trial_name": null,
1230
  "trial_params": null