SAIFIINDUSTRIES commited on
Commit
e2a88fa
·
verified ·
1 Parent(s): c3ca86d

Training in progress, step 690, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:82f67cbe9e89ea2da575b60c5410847310ec4df02b9c6d92a6859e332eacbb1a
3
  size 1976163472
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8da4a61f4ad08e274cdbf73ebb57438c58fd588d4c0fa3e41f3415fecff11a2a
3
  size 1976163472
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d2c52059a0ace4821d5364e966b4df8b16f3e773b35b9b1efdd18136d59f9441
3
  size 1816765835
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:50794c663f5c018c35812f32b3df9ca6b67d62b3ddbada7d522c08bb8bdf691b
3
  size 1816765835
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ca202fc25cecd2e3dfae2a87978621e5f552f2c88fe716b567b699919d821583
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f7d5d33e43d9e54d35ed6ab24ad21ab50d80e1d6b8855d6a0d3c240494595653
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:47597edb455ef686d71b9e2fdfad1e99b404dc1a854723bde359a0fdeff5c2f3
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:94500ce5ab9390767daf1d7b015ce76c5d2ee92a8f74d860763427890a6af7b5
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.021583351387588757,
6
  "eval_steps": 500,
7
- "global_step": 660,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -470,6 +470,27 @@
470
  "learning_rate": 9.784499672988883e-06,
471
  "loss": 1.041367530822754,
472
  "step": 660
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
473
  }
474
  ],
475
  "logging_steps": 10,
@@ -489,7 +510,7 @@
489
  "attributes": {}
490
  }
491
  },
492
- "total_flos": 2.129457117633331e+16,
493
  "train_batch_size": 1,
494
  "trial_name": null,
495
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.022564412814297337,
6
  "eval_steps": 500,
7
+ "global_step": 690,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
470
  "learning_rate": 9.784499672988883e-06,
471
  "loss": 1.041367530822754,
472
  "step": 660
473
+ },
474
+ {
475
+ "epoch": 0.021910371863158282,
476
+ "grad_norm": 1.944808840751648,
477
+ "learning_rate": 9.781229561805102e-06,
478
+ "loss": 1.0567661285400392,
479
+ "step": 670
480
+ },
481
+ {
482
+ "epoch": 0.022237392338727808,
483
+ "grad_norm": 2.1537973880767822,
484
+ "learning_rate": 9.777959450621321e-06,
485
+ "loss": 1.0511420249938965,
486
+ "step": 680
487
+ },
488
+ {
489
+ "epoch": 0.022564412814297337,
490
+ "grad_norm": 1.7468407154083252,
491
+ "learning_rate": 9.774689339437542e-06,
492
+ "loss": 0.9636780738830566,
493
+ "step": 690
494
  }
495
  ],
496
  "logging_steps": 10,
 
510
  "attributes": {}
511
  }
512
  },
513
+ "total_flos": 2.2311643167830016e+16,
514
  "train_batch_size": 1,
515
  "trial_name": null,
516
  "trial_params": null