Training in progress, step 690, checkpoint
Browse files
last-checkpoint/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1976163472
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8da4a61f4ad08e274cdbf73ebb57438c58fd588d4c0fa3e41f3415fecff11a2a
|
| 3 |
size 1976163472
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1816765835
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:50794c663f5c018c35812f32b3df9ca6b67d62b3ddbada7d522c08bb8bdf691b
|
| 3 |
size 1816765835
|
last-checkpoint/scaler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1383
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f7d5d33e43d9e54d35ed6ab24ad21ab50d80e1d6b8855d6a0d3c240494595653
|
| 3 |
size 1383
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:94500ce5ab9390767daf1d7b015ce76c5d2ee92a8f74d860763427890a6af7b5
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -2,9 +2,9 @@
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 500,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -470,6 +470,27 @@
|
|
| 470 |
"learning_rate": 9.784499672988883e-06,
|
| 471 |
"loss": 1.041367530822754,
|
| 472 |
"step": 660
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 473 |
}
|
| 474 |
],
|
| 475 |
"logging_steps": 10,
|
|
@@ -489,7 +510,7 @@
|
|
| 489 |
"attributes": {}
|
| 490 |
}
|
| 491 |
},
|
| 492 |
-
"total_flos": 2.
|
| 493 |
"train_batch_size": 1,
|
| 494 |
"trial_name": null,
|
| 495 |
"trial_params": null
|
|
|
|
| 2 |
"best_global_step": null,
|
| 3 |
"best_metric": null,
|
| 4 |
"best_model_checkpoint": null,
|
| 5 |
+
"epoch": 0.022564412814297337,
|
| 6 |
"eval_steps": 500,
|
| 7 |
+
"global_step": 690,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 470 |
"learning_rate": 9.784499672988883e-06,
|
| 471 |
"loss": 1.041367530822754,
|
| 472 |
"step": 660
|
| 473 |
+
},
|
| 474 |
+
{
|
| 475 |
+
"epoch": 0.021910371863158282,
|
| 476 |
+
"grad_norm": 1.944808840751648,
|
| 477 |
+
"learning_rate": 9.781229561805102e-06,
|
| 478 |
+
"loss": 1.0567661285400392,
|
| 479 |
+
"step": 670
|
| 480 |
+
},
|
| 481 |
+
{
|
| 482 |
+
"epoch": 0.022237392338727808,
|
| 483 |
+
"grad_norm": 2.1537973880767822,
|
| 484 |
+
"learning_rate": 9.777959450621321e-06,
|
| 485 |
+
"loss": 1.0511420249938965,
|
| 486 |
+
"step": 680
|
| 487 |
+
},
|
| 488 |
+
{
|
| 489 |
+
"epoch": 0.022564412814297337,
|
| 490 |
+
"grad_norm": 1.7468407154083252,
|
| 491 |
+
"learning_rate": 9.774689339437542e-06,
|
| 492 |
+
"loss": 0.9636780738830566,
|
| 493 |
+
"step": 690
|
| 494 |
}
|
| 495 |
],
|
| 496 |
"logging_steps": 10,
|
|
|
|
| 510 |
"attributes": {}
|
| 511 |
}
|
| 512 |
},
|
| 513 |
+
"total_flos": 2.2311643167830016e+16,
|
| 514 |
"train_batch_size": 1,
|
| 515 |
"trial_name": null,
|
| 516 |
"trial_params": null
|