SAIFIINDUSTRIES commited on
Commit
eba7c18
·
verified ·
1 Parent(s): 9c76e7f

Training in progress, step 660, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:753b312283d73a72ca99b1d1d1ffee6620efa88ab793579d9e15f4ca0ae1af73
3
  size 1976163472
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:82f67cbe9e89ea2da575b60c5410847310ec4df02b9c6d92a6859e332eacbb1a
3
  size 1976163472
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:106ff0ce09a38420db18c8b793f298b4cebda4c3454b295a84c99e6c8c842121
3
  size 1816765835
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d2c52059a0ace4821d5364e966b4df8b16f3e773b35b9b1efdd18136d59f9441
3
  size 1816765835
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8b40356dc87f933072688e24d199ff6f594a63bfd128de071b6824c1e3453a79
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ca202fc25cecd2e3dfae2a87978621e5f552f2c88fe716b567b699919d821583
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8858a295d00c72f24ae28810bb37a9a4de6f9a3e105940ebdbeac616da712648
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:47597edb455ef686d71b9e2fdfad1e99b404dc1a854723bde359a0fdeff5c2f3
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.020602289960880177,
6
  "eval_steps": 500,
7
- "global_step": 630,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -449,6 +449,27 @@
449
  "learning_rate": 9.794310006540224e-06,
450
  "loss": 1.0060985565185547,
451
  "step": 630
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
452
  }
453
  ],
454
  "logging_steps": 10,
@@ -468,7 +489,7 @@
468
  "attributes": {}
469
  }
470
  },
471
- "total_flos": 2.031704549462016e+16,
472
  "train_batch_size": 1,
473
  "trial_name": null,
474
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.021583351387588757,
6
  "eval_steps": 500,
7
+ "global_step": 660,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
449
  "learning_rate": 9.794310006540224e-06,
450
  "loss": 1.0060985565185547,
451
  "step": 630
452
+ },
453
+ {
454
+ "epoch": 0.020929310436449702,
455
+ "grad_norm": 2.1337103843688965,
456
+ "learning_rate": 9.791039895356443e-06,
457
+ "loss": 1.0044135093688964,
458
+ "step": 640
459
+ },
460
+ {
461
+ "epoch": 0.021256330912019228,
462
+ "grad_norm": 2.3290762901306152,
463
+ "learning_rate": 9.787769784172662e-06,
464
+ "loss": 1.0389455795288085,
465
+ "step": 650
466
+ },
467
+ {
468
+ "epoch": 0.021583351387588757,
469
+ "grad_norm": 2.148353338241577,
470
+ "learning_rate": 9.784499672988883e-06,
471
+ "loss": 1.041367530822754,
472
+ "step": 660
473
  }
474
  ],
475
  "logging_steps": 10,
 
489
  "attributes": {}
490
  }
491
  },
492
+ "total_flos": 2.129457117633331e+16,
493
  "train_batch_size": 1,
494
  "trial_name": null,
495
  "trial_params": null