w-ahmad commited on
Commit
2f30f6b
·
verified ·
1 Parent(s): 48e9bac

Delete outio

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. outio/mlp-linear-9L_run/README.md +0 -72
  2. outio/mlp-linear-9L_run/checkpoint-100/config.json +0 -35
  3. outio/mlp-linear-9L_run/checkpoint-100/model.safetensors +0 -3
  4. outio/mlp-linear-9L_run/checkpoint-100/optimizer.pt +0 -3
  5. outio/mlp-linear-9L_run/checkpoint-100/rng_state.pth +0 -3
  6. outio/mlp-linear-9L_run/checkpoint-100/scheduler.pt +0 -3
  7. outio/mlp-linear-9L_run/checkpoint-100/tokenizer.json +0 -0
  8. outio/mlp-linear-9L_run/checkpoint-100/tokenizer_config.json +0 -13
  9. outio/mlp-linear-9L_run/checkpoint-100/trainer_state.json +0 -85
  10. outio/mlp-linear-9L_run/checkpoint-100/training_args.bin +0 -3
  11. outio/mlp-linear-9L_run/checkpoint-200/config.json +0 -35
  12. outio/mlp-linear-9L_run/checkpoint-200/model.safetensors +0 -3
  13. outio/mlp-linear-9L_run/checkpoint-200/optimizer.pt +0 -3
  14. outio/mlp-linear-9L_run/checkpoint-200/rng_state.pth +0 -3
  15. outio/mlp-linear-9L_run/checkpoint-200/scheduler.pt +0 -3
  16. outio/mlp-linear-9L_run/checkpoint-200/tokenizer.json +0 -0
  17. outio/mlp-linear-9L_run/checkpoint-200/tokenizer_config.json +0 -13
  18. outio/mlp-linear-9L_run/checkpoint-200/trainer_state.json +0 -136
  19. outio/mlp-linear-9L_run/checkpoint-200/training_args.bin +0 -3
  20. outio/mlp-linear-9L_run/checkpoint-300/config.json +0 -35
  21. outio/mlp-linear-9L_run/checkpoint-300/model.safetensors +0 -3
  22. outio/mlp-linear-9L_run/checkpoint-300/optimizer.pt +0 -3
  23. outio/mlp-linear-9L_run/checkpoint-300/rng_state.pth +0 -3
  24. outio/mlp-linear-9L_run/checkpoint-300/scheduler.pt +0 -3
  25. outio/mlp-linear-9L_run/checkpoint-300/tokenizer.json +0 -0
  26. outio/mlp-linear-9L_run/checkpoint-300/tokenizer_config.json +0 -13
  27. outio/mlp-linear-9L_run/checkpoint-300/trainer_state.json +0 -187
  28. outio/mlp-linear-9L_run/checkpoint-300/training_args.bin +0 -3
  29. outio/mlp-linear-9L_run/checkpoint-400/config.json +0 -35
  30. outio/mlp-linear-9L_run/checkpoint-400/model.safetensors +0 -3
  31. outio/mlp-linear-9L_run/checkpoint-400/optimizer.pt +0 -3
  32. outio/mlp-linear-9L_run/checkpoint-400/rng_state.pth +0 -3
  33. outio/mlp-linear-9L_run/checkpoint-400/scheduler.pt +0 -3
  34. outio/mlp-linear-9L_run/checkpoint-400/tokenizer.json +0 -0
  35. outio/mlp-linear-9L_run/checkpoint-400/tokenizer_config.json +0 -13
  36. outio/mlp-linear-9L_run/checkpoint-400/trainer_state.json +0 -238
  37. outio/mlp-linear-9L_run/checkpoint-400/training_args.bin +0 -3
  38. outio/mlp-linear-9L_run/checkpoint-500/config.json +0 -35
  39. outio/mlp-linear-9L_run/checkpoint-500/model.safetensors +0 -3
  40. outio/mlp-linear-9L_run/checkpoint-500/optimizer.pt +0 -3
  41. outio/mlp-linear-9L_run/checkpoint-500/rng_state.pth +0 -3
  42. outio/mlp-linear-9L_run/checkpoint-500/scheduler.pt +0 -3
  43. outio/mlp-linear-9L_run/checkpoint-500/tokenizer.json +0 -0
  44. outio/mlp-linear-9L_run/checkpoint-500/tokenizer_config.json +0 -13
  45. outio/mlp-linear-9L_run/checkpoint-500/trainer_state.json +0 -289
  46. outio/mlp-linear-9L_run/checkpoint-500/training_args.bin +0 -3
  47. outio/mlp-linear-9L_run/checkpoint-600/config.json +0 -35
  48. outio/mlp-linear-9L_run/checkpoint-600/model.safetensors +0 -3
  49. outio/mlp-linear-9L_run/checkpoint-600/optimizer.pt +0 -3
  50. outio/mlp-linear-9L_run/checkpoint-600/rng_state.pth +0 -3
outio/mlp-linear-9L_run/README.md DELETED
@@ -1,72 +0,0 @@
1
- ---
2
- library_name: transformers
3
- tags:
4
- - generated_from_trainer
5
- model-index:
6
- - name: finale-mlp-linear-9L
7
- results: []
8
- ---
9
-
10
- <!-- This model card has been generated automatically according to the information the Trainer had access to. You
11
- should probably proofread and complete it, then remove this comment. -->
12
-
13
- # finale-mlp-linear-9L
14
-
15
- This model is a fine-tuned version of [](https://huggingface.co/) on an unknown dataset.
16
- It achieves the following results on the evaluation set:
17
- - Loss: 3.1081
18
-
19
- ## Model description
20
-
21
- More information needed
22
-
23
- ## Intended uses & limitations
24
-
25
- More information needed
26
-
27
- ## Training and evaluation data
28
-
29
- More information needed
30
-
31
- ## Training procedure
32
-
33
- ### Training hyperparameters
34
-
35
- The following hyperparameters were used during training:
36
- - learning_rate: 0.0005
37
- - train_batch_size: 80
38
- - eval_batch_size: 80
39
- - seed: 42
40
- - gradient_accumulation_steps: 16
41
- - total_train_batch_size: 1280
42
- - optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
43
- - lr_scheduler_type: constant
44
- - training_steps: 750
45
-
46
- ### Training results
47
-
48
- | Training Loss | Epoch | Step | Validation Loss |
49
- |:-------------:|:------:|:----:|:---------------:|
50
- | 102.0503 | 0.0674 | 50 | 5.6659 |
51
- | 77.5119 | 0.1348 | 100 | 4.6899 |
52
- | 70.3764 | 0.2022 | 150 | 4.2638 |
53
- | 63.8118 | 0.2696 | 200 | 3.9344 |
54
- | 61.1106 | 0.3370 | 250 | 3.7483 |
55
- | 58.3348 | 0.4044 | 300 | 3.6138 |
56
- | 56.6996 | 0.4719 | 350 | 3.4927 |
57
- | 54.9361 | 0.5393 | 400 | 3.4158 |
58
- | 53.8945 | 0.6067 | 450 | 3.3447 |
59
- | 52.8687 | 0.6741 | 500 | 3.2968 |
60
- | 52.3214 | 0.7415 | 550 | 3.2509 |
61
- | 51.4070 | 0.8089 | 600 | 3.2120 |
62
- | 50.8616 | 0.8763 | 650 | 3.1778 |
63
- | 50.1968 | 0.9437 | 700 | 3.1350 |
64
- | 49.7485 | 1.0108 | 750 | 3.1081 |
65
-
66
-
67
- ### Framework versions
68
-
69
- - Transformers 5.15.0.dev0
70
- - Pytorch 2.6.0+cu124
71
- - Datasets 5.0.1
72
- - Tokenizers 0.22.2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-100/config.json DELETED
@@ -1,35 +0,0 @@
1
- {
2
- "activation": "linear",
3
- "architectures": [
4
- "TinyLlamaForCausalLM"
5
- ],
6
- "attention_bias": false,
7
- "attention_dropout": 0.0,
8
- "bos_token_id": 1,
9
- "dtype": "bfloat16",
10
- "eos_token_id": 2,
11
- "head_dim": 32,
12
- "hidden_act": "silu",
13
- "hidden_size": 128,
14
- "initializer_range": 0.02,
15
- "intermediate_size": 256,
16
- "max_position_embeddings": 512,
17
- "mlp_bias": false,
18
- "mlp_type": "mlp",
19
- "model_type": "tiny_llama",
20
- "num_attention_heads": 4,
21
- "num_hidden_layers": 9,
22
- "num_key_value_heads": 4,
23
- "pad_token_id": 0,
24
- "pretraining_tp": 1,
25
- "rms_norm_eps": 1e-06,
26
- "rope_parameters": {
27
- "rope_theta": 10000.0,
28
- "rope_type": "default"
29
- },
30
- "tie_word_embeddings": true,
31
- "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
- "transformers_version": "5.15.0.dev0",
33
- "use_cache": false,
34
- "vocab_size": 4096
35
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-100/model.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:5949e1407c2bcab53489e29477bcdd35c96dbc74b1d91d17d7c576a8cd928349
3
- size 4010544
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-100/optimizer.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:962c5708b33c4c24a42c3e37b64d1c725a4f52db56e38f5154a0680dbef3091a
3
- size 8068282
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-100/rng_state.pth DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:3cf9097d4513154245c48236b6ec5137b7ee2a21c9f58f2cba798ea275c6026f
3
- size 14244
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-100/scheduler.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:bf737e6c413389387873c1fe7b9305a2aa5499ddcaf3a71dc59c9e0c189342fa
3
- size 1064
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-100/tokenizer.json DELETED
The diff for this file is too large to render. See raw diff
 
outio/mlp-linear-9L_run/checkpoint-100/tokenizer_config.json DELETED
@@ -1,13 +0,0 @@
1
- {
2
- "add_prefix_space": false,
3
- "backend": "tokenizers",
4
- "bos_token": "<|endoftext|>",
5
- "eos_token": "<|endoftext|>",
6
- "errors": "replace",
7
- "is_local": false,
8
- "local_files_only": false,
9
- "model_max_length": 1024,
10
- "pad_token": "<|endoftext|>",
11
- "tokenizer_class": "GPT2Tokenizer",
12
- "unk_token": "<|endoftext|>"
13
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-100/trainer_state.json DELETED
@@ -1,85 +0,0 @@
1
- {
2
- "best_global_step": null,
3
- "best_metric": null,
4
- "best_model_checkpoint": null,
5
- "epoch": 0.13481631277384565,
6
- "eval_steps": 50,
7
- "global_step": 100,
8
- "is_hyper_param_search": false,
9
- "is_local_process_zero": true,
10
- "is_world_process_zero": true,
11
- "log_history": [
12
- {
13
- "epoch": 0.026963262554769128,
14
- "grad_norm": 19.875,
15
- "learning_rate": 0.0005,
16
- "loss": 120.13565673828126,
17
- "step": 20
18
- },
19
- {
20
- "epoch": 0.053926525109538256,
21
- "grad_norm": 24.625,
22
- "learning_rate": 0.0005,
23
- "loss": 102.05034790039062,
24
- "step": 40
25
- },
26
- {
27
- "epoch": 0.06740815638692282,
28
- "eval_loss": 5.665860652923584,
29
- "eval_runtime": 9.5524,
30
- "eval_samples_per_second": 997.346,
31
- "eval_steps_per_second": 12.562,
32
- "step": 50
33
- },
34
- {
35
- "epoch": 0.08088978766430738,
36
- "grad_norm": 18.125,
37
- "learning_rate": 0.0005,
38
- "loss": 90.97202758789062,
39
- "step": 60
40
- },
41
- {
42
- "epoch": 0.10785305021907651,
43
- "grad_norm": 40.75,
44
- "learning_rate": 0.0005,
45
- "loss": 83.10302734375,
46
- "step": 80
47
- },
48
- {
49
- "epoch": 0.13481631277384565,
50
- "grad_norm": 29.625,
51
- "learning_rate": 0.0005,
52
- "loss": 77.51190185546875,
53
- "step": 100
54
- },
55
- {
56
- "epoch": 0.13481631277384565,
57
- "eval_loss": 4.689935684204102,
58
- "eval_runtime": 9.1111,
59
- "eval_samples_per_second": 1045.652,
60
- "eval_steps_per_second": 13.171,
61
- "step": 100
62
- }
63
- ],
64
- "logging_steps": 20,
65
- "max_steps": 750,
66
- "num_input_tokens_seen": 0,
67
- "num_train_epochs": 2,
68
- "save_steps": 100,
69
- "stateful_callbacks": {
70
- "TrainerControl": {
71
- "args": {
72
- "should_epoch_stop": false,
73
- "should_evaluate": false,
74
- "should_log": false,
75
- "should_save": true,
76
- "should_training_stop": false
77
- },
78
- "attributes": {}
79
- }
80
- },
81
- "total_flos": 580776886272000.0,
82
- "train_batch_size": 80,
83
- "trial_name": null,
84
- "trial_params": null
85
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-100/training_args.bin DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:fa9985a7c7f521fa2a6f7d16c7d254b6d5e243707cd393fa6ea6dfe9b9926147
3
- size 4920
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-200/config.json DELETED
@@ -1,35 +0,0 @@
1
- {
2
- "activation": "linear",
3
- "architectures": [
4
- "TinyLlamaForCausalLM"
5
- ],
6
- "attention_bias": false,
7
- "attention_dropout": 0.0,
8
- "bos_token_id": 1,
9
- "dtype": "bfloat16",
10
- "eos_token_id": 2,
11
- "head_dim": 32,
12
- "hidden_act": "silu",
13
- "hidden_size": 128,
14
- "initializer_range": 0.02,
15
- "intermediate_size": 256,
16
- "max_position_embeddings": 512,
17
- "mlp_bias": false,
18
- "mlp_type": "mlp",
19
- "model_type": "tiny_llama",
20
- "num_attention_heads": 4,
21
- "num_hidden_layers": 9,
22
- "num_key_value_heads": 4,
23
- "pad_token_id": 0,
24
- "pretraining_tp": 1,
25
- "rms_norm_eps": 1e-06,
26
- "rope_parameters": {
27
- "rope_theta": 10000.0,
28
- "rope_type": "default"
29
- },
30
- "tie_word_embeddings": true,
31
- "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
- "transformers_version": "5.15.0.dev0",
33
- "use_cache": false,
34
- "vocab_size": 4096
35
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-200/model.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:6f8ae301a0fa375ffb07a40ad452fd3f506fe3c20deb322496b780d268057061
3
- size 4010544
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-200/optimizer.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:14a31c02e4b99252ca6b74d27ddbc1c0ad2ae33db64621841a3d3a1666664502
3
- size 8068282
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-200/rng_state.pth DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:ecefbb3f17bb76b6655eb0157c98b5287c17fa4b4c72a6b9068b0823ce9fd18d
3
- size 14244
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-200/scheduler.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:2a5c0d895a87c087f330b1f8967bf21b2ccdd5750d550001b66990e4ed22f53c
3
- size 1064
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-200/tokenizer.json DELETED
The diff for this file is too large to render. See raw diff
 
outio/mlp-linear-9L_run/checkpoint-200/tokenizer_config.json DELETED
@@ -1,13 +0,0 @@
1
- {
2
- "add_prefix_space": false,
3
- "backend": "tokenizers",
4
- "bos_token": "<|endoftext|>",
5
- "eos_token": "<|endoftext|>",
6
- "errors": "replace",
7
- "is_local": false,
8
- "local_files_only": false,
9
- "model_max_length": 1024,
10
- "pad_token": "<|endoftext|>",
11
- "tokenizer_class": "GPT2Tokenizer",
12
- "unk_token": "<|endoftext|>"
13
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-200/trainer_state.json DELETED
@@ -1,136 +0,0 @@
1
- {
2
- "best_global_step": null,
3
- "best_metric": null,
4
- "best_model_checkpoint": null,
5
- "epoch": 0.2696326255476913,
6
- "eval_steps": 50,
7
- "global_step": 200,
8
- "is_hyper_param_search": false,
9
- "is_local_process_zero": true,
10
- "is_world_process_zero": true,
11
- "log_history": [
12
- {
13
- "epoch": 0.026963262554769128,
14
- "grad_norm": 19.875,
15
- "learning_rate": 0.0005,
16
- "loss": 120.13565673828126,
17
- "step": 20
18
- },
19
- {
20
- "epoch": 0.053926525109538256,
21
- "grad_norm": 24.625,
22
- "learning_rate": 0.0005,
23
- "loss": 102.05034790039062,
24
- "step": 40
25
- },
26
- {
27
- "epoch": 0.06740815638692282,
28
- "eval_loss": 5.665860652923584,
29
- "eval_runtime": 9.5524,
30
- "eval_samples_per_second": 997.346,
31
- "eval_steps_per_second": 12.562,
32
- "step": 50
33
- },
34
- {
35
- "epoch": 0.08088978766430738,
36
- "grad_norm": 18.125,
37
- "learning_rate": 0.0005,
38
- "loss": 90.97202758789062,
39
- "step": 60
40
- },
41
- {
42
- "epoch": 0.10785305021907651,
43
- "grad_norm": 40.75,
44
- "learning_rate": 0.0005,
45
- "loss": 83.10302734375,
46
- "step": 80
47
- },
48
- {
49
- "epoch": 0.13481631277384565,
50
- "grad_norm": 29.625,
51
- "learning_rate": 0.0005,
52
- "loss": 77.51190185546875,
53
- "step": 100
54
- },
55
- {
56
- "epoch": 0.13481631277384565,
57
- "eval_loss": 4.689935684204102,
58
- "eval_runtime": 9.1111,
59
- "eval_samples_per_second": 1045.652,
60
- "eval_steps_per_second": 13.171,
61
- "step": 100
62
- },
63
- {
64
- "epoch": 0.16177957532861476,
65
- "grad_norm": 19.5,
66
- "learning_rate": 0.0005,
67
- "loss": 73.29612426757812,
68
- "step": 120
69
- },
70
- {
71
- "epoch": 0.1887428378833839,
72
- "grad_norm": 25.25,
73
- "learning_rate": 0.0005,
74
- "loss": 70.37640380859375,
75
- "step": 140
76
- },
77
- {
78
- "epoch": 0.20222446916076844,
79
- "eval_loss": 4.263797760009766,
80
- "eval_runtime": 9.4931,
81
- "eval_samples_per_second": 1003.568,
82
- "eval_steps_per_second": 12.641,
83
- "step": 150
84
- },
85
- {
86
- "epoch": 0.21570610043815303,
87
- "grad_norm": 14.625,
88
- "learning_rate": 0.0005,
89
- "loss": 68.2427978515625,
90
- "step": 160
91
- },
92
- {
93
- "epoch": 0.24266936299292213,
94
- "grad_norm": 13.0625,
95
- "learning_rate": 0.0005,
96
- "loss": 66.13766479492188,
97
- "step": 180
98
- },
99
- {
100
- "epoch": 0.2696326255476913,
101
- "grad_norm": 18.5,
102
- "learning_rate": 0.0005,
103
- "loss": 63.81178588867188,
104
- "step": 200
105
- },
106
- {
107
- "epoch": 0.2696326255476913,
108
- "eval_loss": 3.9343974590301514,
109
- "eval_runtime": 9.3868,
110
- "eval_samples_per_second": 1014.939,
111
- "eval_steps_per_second": 12.784,
112
- "step": 200
113
- }
114
- ],
115
- "logging_steps": 20,
116
- "max_steps": 750,
117
- "num_input_tokens_seen": 0,
118
- "num_train_epochs": 2,
119
- "save_steps": 100,
120
- "stateful_callbacks": {
121
- "TrainerControl": {
122
- "args": {
123
- "should_epoch_stop": false,
124
- "should_evaluate": false,
125
- "should_log": false,
126
- "should_save": true,
127
- "should_training_stop": false
128
- },
129
- "attributes": {}
130
- }
131
- },
132
- "total_flos": 1161553772544000.0,
133
- "train_batch_size": 80,
134
- "trial_name": null,
135
- "trial_params": null
136
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-200/training_args.bin DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:fa9985a7c7f521fa2a6f7d16c7d254b6d5e243707cd393fa6ea6dfe9b9926147
3
- size 4920
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-300/config.json DELETED
@@ -1,35 +0,0 @@
1
- {
2
- "activation": "linear",
3
- "architectures": [
4
- "TinyLlamaForCausalLM"
5
- ],
6
- "attention_bias": false,
7
- "attention_dropout": 0.0,
8
- "bos_token_id": 1,
9
- "dtype": "bfloat16",
10
- "eos_token_id": 2,
11
- "head_dim": 32,
12
- "hidden_act": "silu",
13
- "hidden_size": 128,
14
- "initializer_range": 0.02,
15
- "intermediate_size": 256,
16
- "max_position_embeddings": 512,
17
- "mlp_bias": false,
18
- "mlp_type": "mlp",
19
- "model_type": "tiny_llama",
20
- "num_attention_heads": 4,
21
- "num_hidden_layers": 9,
22
- "num_key_value_heads": 4,
23
- "pad_token_id": 0,
24
- "pretraining_tp": 1,
25
- "rms_norm_eps": 1e-06,
26
- "rope_parameters": {
27
- "rope_theta": 10000.0,
28
- "rope_type": "default"
29
- },
30
- "tie_word_embeddings": true,
31
- "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
- "transformers_version": "5.15.0.dev0",
33
- "use_cache": false,
34
- "vocab_size": 4096
35
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-300/model.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:55dbde6acf6dc9d3dba5e676ff9abfd79b896b31e4cb2ca34ead208008c86666
3
- size 4010544
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-300/optimizer.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:4258f81555e36337063a7b1df64de23ad3d4b82ad674d01e7feda8e361a1d2e5
3
- size 8068282
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-300/rng_state.pth DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:068fbd993087219c15b8c0baa13fc39644a4dcdfe92d8be3fa6434deece90371
3
- size 14244
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-300/scheduler.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:ab52dbe1205ced51cf9c1802cf8a6fb381f63517eae73353a367a22100fbc329
3
- size 1064
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-300/tokenizer.json DELETED
The diff for this file is too large to render. See raw diff
 
outio/mlp-linear-9L_run/checkpoint-300/tokenizer_config.json DELETED
@@ -1,13 +0,0 @@
1
- {
2
- "add_prefix_space": false,
3
- "backend": "tokenizers",
4
- "bos_token": "<|endoftext|>",
5
- "eos_token": "<|endoftext|>",
6
- "errors": "replace",
7
- "is_local": false,
8
- "local_files_only": false,
9
- "model_max_length": 1024,
10
- "pad_token": "<|endoftext|>",
11
- "tokenizer_class": "GPT2Tokenizer",
12
- "unk_token": "<|endoftext|>"
13
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-300/trainer_state.json DELETED
@@ -1,187 +0,0 @@
1
- {
2
- "best_global_step": null,
3
- "best_metric": null,
4
- "best_model_checkpoint": null,
5
- "epoch": 0.4044489383215369,
6
- "eval_steps": 50,
7
- "global_step": 300,
8
- "is_hyper_param_search": false,
9
- "is_local_process_zero": true,
10
- "is_world_process_zero": true,
11
- "log_history": [
12
- {
13
- "epoch": 0.026963262554769128,
14
- "grad_norm": 19.875,
15
- "learning_rate": 0.0005,
16
- "loss": 120.13565673828126,
17
- "step": 20
18
- },
19
- {
20
- "epoch": 0.053926525109538256,
21
- "grad_norm": 24.625,
22
- "learning_rate": 0.0005,
23
- "loss": 102.05034790039062,
24
- "step": 40
25
- },
26
- {
27
- "epoch": 0.06740815638692282,
28
- "eval_loss": 5.665860652923584,
29
- "eval_runtime": 9.5524,
30
- "eval_samples_per_second": 997.346,
31
- "eval_steps_per_second": 12.562,
32
- "step": 50
33
- },
34
- {
35
- "epoch": 0.08088978766430738,
36
- "grad_norm": 18.125,
37
- "learning_rate": 0.0005,
38
- "loss": 90.97202758789062,
39
- "step": 60
40
- },
41
- {
42
- "epoch": 0.10785305021907651,
43
- "grad_norm": 40.75,
44
- "learning_rate": 0.0005,
45
- "loss": 83.10302734375,
46
- "step": 80
47
- },
48
- {
49
- "epoch": 0.13481631277384565,
50
- "grad_norm": 29.625,
51
- "learning_rate": 0.0005,
52
- "loss": 77.51190185546875,
53
- "step": 100
54
- },
55
- {
56
- "epoch": 0.13481631277384565,
57
- "eval_loss": 4.689935684204102,
58
- "eval_runtime": 9.1111,
59
- "eval_samples_per_second": 1045.652,
60
- "eval_steps_per_second": 13.171,
61
- "step": 100
62
- },
63
- {
64
- "epoch": 0.16177957532861476,
65
- "grad_norm": 19.5,
66
- "learning_rate": 0.0005,
67
- "loss": 73.29612426757812,
68
- "step": 120
69
- },
70
- {
71
- "epoch": 0.1887428378833839,
72
- "grad_norm": 25.25,
73
- "learning_rate": 0.0005,
74
- "loss": 70.37640380859375,
75
- "step": 140
76
- },
77
- {
78
- "epoch": 0.20222446916076844,
79
- "eval_loss": 4.263797760009766,
80
- "eval_runtime": 9.4931,
81
- "eval_samples_per_second": 1003.568,
82
- "eval_steps_per_second": 12.641,
83
- "step": 150
84
- },
85
- {
86
- "epoch": 0.21570610043815303,
87
- "grad_norm": 14.625,
88
- "learning_rate": 0.0005,
89
- "loss": 68.2427978515625,
90
- "step": 160
91
- },
92
- {
93
- "epoch": 0.24266936299292213,
94
- "grad_norm": 13.0625,
95
- "learning_rate": 0.0005,
96
- "loss": 66.13766479492188,
97
- "step": 180
98
- },
99
- {
100
- "epoch": 0.2696326255476913,
101
- "grad_norm": 18.5,
102
- "learning_rate": 0.0005,
103
- "loss": 63.81178588867188,
104
- "step": 200
105
- },
106
- {
107
- "epoch": 0.2696326255476913,
108
- "eval_loss": 3.9343974590301514,
109
- "eval_runtime": 9.3868,
110
- "eval_samples_per_second": 1014.939,
111
- "eval_steps_per_second": 12.784,
112
- "step": 200
113
- },
114
- {
115
- "epoch": 0.2965958881024604,
116
- "grad_norm": 16.125,
117
- "learning_rate": 0.0005,
118
- "loss": 62.16768188476563,
119
- "step": 220
120
- },
121
- {
122
- "epoch": 0.3235591506572295,
123
- "grad_norm": 21.25,
124
- "learning_rate": 0.0005,
125
- "loss": 61.1105712890625,
126
- "step": 240
127
- },
128
- {
129
- "epoch": 0.33704078193461406,
130
- "eval_loss": 3.748317241668701,
131
- "eval_runtime": 9.4777,
132
- "eval_samples_per_second": 1005.204,
133
- "eval_steps_per_second": 12.661,
134
- "step": 250
135
- },
136
- {
137
- "epoch": 0.3505224132119987,
138
- "grad_norm": 42.25,
139
- "learning_rate": 0.0005,
140
- "loss": 60.05653686523438,
141
- "step": 260
142
- },
143
- {
144
- "epoch": 0.3774856757667678,
145
- "grad_norm": 27.5,
146
- "learning_rate": 0.0005,
147
- "loss": 59.25806884765625,
148
- "step": 280
149
- },
150
- {
151
- "epoch": 0.4044489383215369,
152
- "grad_norm": 14.375,
153
- "learning_rate": 0.0005,
154
- "loss": 58.334844970703124,
155
- "step": 300
156
- },
157
- {
158
- "epoch": 0.4044489383215369,
159
- "eval_loss": 3.6137855052948,
160
- "eval_runtime": 9.6654,
161
- "eval_samples_per_second": 985.685,
162
- "eval_steps_per_second": 12.415,
163
- "step": 300
164
- }
165
- ],
166
- "logging_steps": 20,
167
- "max_steps": 750,
168
- "num_input_tokens_seen": 0,
169
- "num_train_epochs": 2,
170
- "save_steps": 100,
171
- "stateful_callbacks": {
172
- "TrainerControl": {
173
- "args": {
174
- "should_epoch_stop": false,
175
- "should_evaluate": false,
176
- "should_log": false,
177
- "should_save": true,
178
- "should_training_stop": false
179
- },
180
- "attributes": {}
181
- }
182
- },
183
- "total_flos": 1742330658816000.0,
184
- "train_batch_size": 80,
185
- "trial_name": null,
186
- "trial_params": null
187
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-300/training_args.bin DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:fa9985a7c7f521fa2a6f7d16c7d254b6d5e243707cd393fa6ea6dfe9b9926147
3
- size 4920
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-400/config.json DELETED
@@ -1,35 +0,0 @@
1
- {
2
- "activation": "linear",
3
- "architectures": [
4
- "TinyLlamaForCausalLM"
5
- ],
6
- "attention_bias": false,
7
- "attention_dropout": 0.0,
8
- "bos_token_id": 1,
9
- "dtype": "bfloat16",
10
- "eos_token_id": 2,
11
- "head_dim": 32,
12
- "hidden_act": "silu",
13
- "hidden_size": 128,
14
- "initializer_range": 0.02,
15
- "intermediate_size": 256,
16
- "max_position_embeddings": 512,
17
- "mlp_bias": false,
18
- "mlp_type": "mlp",
19
- "model_type": "tiny_llama",
20
- "num_attention_heads": 4,
21
- "num_hidden_layers": 9,
22
- "num_key_value_heads": 4,
23
- "pad_token_id": 0,
24
- "pretraining_tp": 1,
25
- "rms_norm_eps": 1e-06,
26
- "rope_parameters": {
27
- "rope_theta": 10000.0,
28
- "rope_type": "default"
29
- },
30
- "tie_word_embeddings": true,
31
- "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
- "transformers_version": "5.15.0.dev0",
33
- "use_cache": false,
34
- "vocab_size": 4096
35
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-400/model.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:98e1ba1cf80f5d7c5903f158942635434c3a700f5546a2c4b75eb93cd32d7428
3
- size 4010544
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-400/optimizer.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:fde1712f4eb739cc744787146a27178e32e8a2605380ff4eee1f82a493264a7c
3
- size 8068282
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-400/rng_state.pth DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:f9a6944c405a002fce05f295d08ea6650e2e2ad6dbf5d6da1e9053f7bf7f5827
3
- size 14244
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-400/scheduler.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:700fad8c8dc54616d2078075de4d46f34ce253599b374a9daab808be3cf34871
3
- size 1064
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-400/tokenizer.json DELETED
The diff for this file is too large to render. See raw diff
 
outio/mlp-linear-9L_run/checkpoint-400/tokenizer_config.json DELETED
@@ -1,13 +0,0 @@
1
- {
2
- "add_prefix_space": false,
3
- "backend": "tokenizers",
4
- "bos_token": "<|endoftext|>",
5
- "eos_token": "<|endoftext|>",
6
- "errors": "replace",
7
- "is_local": false,
8
- "local_files_only": false,
9
- "model_max_length": 1024,
10
- "pad_token": "<|endoftext|>",
11
- "tokenizer_class": "GPT2Tokenizer",
12
- "unk_token": "<|endoftext|>"
13
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-400/trainer_state.json DELETED
@@ -1,238 +0,0 @@
1
- {
2
- "best_global_step": null,
3
- "best_metric": null,
4
- "best_model_checkpoint": null,
5
- "epoch": 0.5392652510953826,
6
- "eval_steps": 50,
7
- "global_step": 400,
8
- "is_hyper_param_search": false,
9
- "is_local_process_zero": true,
10
- "is_world_process_zero": true,
11
- "log_history": [
12
- {
13
- "epoch": 0.026963262554769128,
14
- "grad_norm": 19.875,
15
- "learning_rate": 0.0005,
16
- "loss": 120.13565673828126,
17
- "step": 20
18
- },
19
- {
20
- "epoch": 0.053926525109538256,
21
- "grad_norm": 24.625,
22
- "learning_rate": 0.0005,
23
- "loss": 102.05034790039062,
24
- "step": 40
25
- },
26
- {
27
- "epoch": 0.06740815638692282,
28
- "eval_loss": 5.665860652923584,
29
- "eval_runtime": 9.5524,
30
- "eval_samples_per_second": 997.346,
31
- "eval_steps_per_second": 12.562,
32
- "step": 50
33
- },
34
- {
35
- "epoch": 0.08088978766430738,
36
- "grad_norm": 18.125,
37
- "learning_rate": 0.0005,
38
- "loss": 90.97202758789062,
39
- "step": 60
40
- },
41
- {
42
- "epoch": 0.10785305021907651,
43
- "grad_norm": 40.75,
44
- "learning_rate": 0.0005,
45
- "loss": 83.10302734375,
46
- "step": 80
47
- },
48
- {
49
- "epoch": 0.13481631277384565,
50
- "grad_norm": 29.625,
51
- "learning_rate": 0.0005,
52
- "loss": 77.51190185546875,
53
- "step": 100
54
- },
55
- {
56
- "epoch": 0.13481631277384565,
57
- "eval_loss": 4.689935684204102,
58
- "eval_runtime": 9.1111,
59
- "eval_samples_per_second": 1045.652,
60
- "eval_steps_per_second": 13.171,
61
- "step": 100
62
- },
63
- {
64
- "epoch": 0.16177957532861476,
65
- "grad_norm": 19.5,
66
- "learning_rate": 0.0005,
67
- "loss": 73.29612426757812,
68
- "step": 120
69
- },
70
- {
71
- "epoch": 0.1887428378833839,
72
- "grad_norm": 25.25,
73
- "learning_rate": 0.0005,
74
- "loss": 70.37640380859375,
75
- "step": 140
76
- },
77
- {
78
- "epoch": 0.20222446916076844,
79
- "eval_loss": 4.263797760009766,
80
- "eval_runtime": 9.4931,
81
- "eval_samples_per_second": 1003.568,
82
- "eval_steps_per_second": 12.641,
83
- "step": 150
84
- },
85
- {
86
- "epoch": 0.21570610043815303,
87
- "grad_norm": 14.625,
88
- "learning_rate": 0.0005,
89
- "loss": 68.2427978515625,
90
- "step": 160
91
- },
92
- {
93
- "epoch": 0.24266936299292213,
94
- "grad_norm": 13.0625,
95
- "learning_rate": 0.0005,
96
- "loss": 66.13766479492188,
97
- "step": 180
98
- },
99
- {
100
- "epoch": 0.2696326255476913,
101
- "grad_norm": 18.5,
102
- "learning_rate": 0.0005,
103
- "loss": 63.81178588867188,
104
- "step": 200
105
- },
106
- {
107
- "epoch": 0.2696326255476913,
108
- "eval_loss": 3.9343974590301514,
109
- "eval_runtime": 9.3868,
110
- "eval_samples_per_second": 1014.939,
111
- "eval_steps_per_second": 12.784,
112
- "step": 200
113
- },
114
- {
115
- "epoch": 0.2965958881024604,
116
- "grad_norm": 16.125,
117
- "learning_rate": 0.0005,
118
- "loss": 62.16768188476563,
119
- "step": 220
120
- },
121
- {
122
- "epoch": 0.3235591506572295,
123
- "grad_norm": 21.25,
124
- "learning_rate": 0.0005,
125
- "loss": 61.1105712890625,
126
- "step": 240
127
- },
128
- {
129
- "epoch": 0.33704078193461406,
130
- "eval_loss": 3.748317241668701,
131
- "eval_runtime": 9.4777,
132
- "eval_samples_per_second": 1005.204,
133
- "eval_steps_per_second": 12.661,
134
- "step": 250
135
- },
136
- {
137
- "epoch": 0.3505224132119987,
138
- "grad_norm": 42.25,
139
- "learning_rate": 0.0005,
140
- "loss": 60.05653686523438,
141
- "step": 260
142
- },
143
- {
144
- "epoch": 0.3774856757667678,
145
- "grad_norm": 27.5,
146
- "learning_rate": 0.0005,
147
- "loss": 59.25806884765625,
148
- "step": 280
149
- },
150
- {
151
- "epoch": 0.4044489383215369,
152
- "grad_norm": 14.375,
153
- "learning_rate": 0.0005,
154
- "loss": 58.334844970703124,
155
- "step": 300
156
- },
157
- {
158
- "epoch": 0.4044489383215369,
159
- "eval_loss": 3.6137855052948,
160
- "eval_runtime": 9.6654,
161
- "eval_samples_per_second": 985.685,
162
- "eval_steps_per_second": 12.415,
163
- "step": 300
164
- },
165
- {
166
- "epoch": 0.43141220087630605,
167
- "grad_norm": 32.75,
168
- "learning_rate": 0.0005,
169
- "loss": 57.341180419921876,
170
- "step": 320
171
- },
172
- {
173
- "epoch": 0.45837546343107516,
174
- "grad_norm": 12.5625,
175
- "learning_rate": 0.0005,
176
- "loss": 56.699591064453124,
177
- "step": 340
178
- },
179
- {
180
- "epoch": 0.4718570947084597,
181
- "eval_loss": 3.492727518081665,
182
- "eval_runtime": 9.4437,
183
- "eval_samples_per_second": 1008.824,
184
- "eval_steps_per_second": 12.707,
185
- "step": 350
186
- },
187
- {
188
- "epoch": 0.48533872598584427,
189
- "grad_norm": 21.5,
190
- "learning_rate": 0.0005,
191
- "loss": 55.840753173828126,
192
- "step": 360
193
- },
194
- {
195
- "epoch": 0.5123019885406134,
196
- "grad_norm": 14.5625,
197
- "learning_rate": 0.0005,
198
- "loss": 55.377203369140624,
199
- "step": 380
200
- },
201
- {
202
- "epoch": 0.5392652510953826,
203
- "grad_norm": 14.6875,
204
- "learning_rate": 0.0005,
205
- "loss": 54.936090087890626,
206
- "step": 400
207
- },
208
- {
209
- "epoch": 0.5392652510953826,
210
- "eval_loss": 3.41583251953125,
211
- "eval_runtime": 9.1837,
212
- "eval_samples_per_second": 1037.382,
213
- "eval_steps_per_second": 13.067,
214
- "step": 400
215
- }
216
- ],
217
- "logging_steps": 20,
218
- "max_steps": 750,
219
- "num_input_tokens_seen": 0,
220
- "num_train_epochs": 2,
221
- "save_steps": 100,
222
- "stateful_callbacks": {
223
- "TrainerControl": {
224
- "args": {
225
- "should_epoch_stop": false,
226
- "should_evaluate": false,
227
- "should_log": false,
228
- "should_save": true,
229
- "should_training_stop": false
230
- },
231
- "attributes": {}
232
- }
233
- },
234
- "total_flos": 2323107545088000.0,
235
- "train_batch_size": 80,
236
- "trial_name": null,
237
- "trial_params": null
238
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-400/training_args.bin DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:fa9985a7c7f521fa2a6f7d16c7d254b6d5e243707cd393fa6ea6dfe9b9926147
3
- size 4920
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-500/config.json DELETED
@@ -1,35 +0,0 @@
1
- {
2
- "activation": "linear",
3
- "architectures": [
4
- "TinyLlamaForCausalLM"
5
- ],
6
- "attention_bias": false,
7
- "attention_dropout": 0.0,
8
- "bos_token_id": 1,
9
- "dtype": "bfloat16",
10
- "eos_token_id": 2,
11
- "head_dim": 32,
12
- "hidden_act": "silu",
13
- "hidden_size": 128,
14
- "initializer_range": 0.02,
15
- "intermediate_size": 256,
16
- "max_position_embeddings": 512,
17
- "mlp_bias": false,
18
- "mlp_type": "mlp",
19
- "model_type": "tiny_llama",
20
- "num_attention_heads": 4,
21
- "num_hidden_layers": 9,
22
- "num_key_value_heads": 4,
23
- "pad_token_id": 0,
24
- "pretraining_tp": 1,
25
- "rms_norm_eps": 1e-06,
26
- "rope_parameters": {
27
- "rope_theta": 10000.0,
28
- "rope_type": "default"
29
- },
30
- "tie_word_embeddings": true,
31
- "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
- "transformers_version": "5.15.0.dev0",
33
- "use_cache": false,
34
- "vocab_size": 4096
35
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-500/model.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:e0590c4a96baf3691f3f9aa38b42f2cd70d0f1175fce48cef948d63347a301fe
3
- size 4010544
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-500/optimizer.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:fe2e654d5f59c15d4638f2b32a6f2a7dc95d3d2decf4808175611c780a6de959
3
- size 8068282
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-500/rng_state.pth DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:b1a97db8e41139aa1239ba7fb79ddeb0af5998c6305a440c1fe182e6ad02f2f5
3
- size 14244
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-500/scheduler.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:52f9e0f38685de0581cdfb11b92c8c460923d0c53aefc354af6070d2fc4ba852
3
- size 1064
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-500/tokenizer.json DELETED
The diff for this file is too large to render. See raw diff
 
outio/mlp-linear-9L_run/checkpoint-500/tokenizer_config.json DELETED
@@ -1,13 +0,0 @@
1
- {
2
- "add_prefix_space": false,
3
- "backend": "tokenizers",
4
- "bos_token": "<|endoftext|>",
5
- "eos_token": "<|endoftext|>",
6
- "errors": "replace",
7
- "is_local": false,
8
- "local_files_only": false,
9
- "model_max_length": 1024,
10
- "pad_token": "<|endoftext|>",
11
- "tokenizer_class": "GPT2Tokenizer",
12
- "unk_token": "<|endoftext|>"
13
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-500/trainer_state.json DELETED
@@ -1,289 +0,0 @@
1
- {
2
- "best_global_step": null,
3
- "best_metric": null,
4
- "best_model_checkpoint": null,
5
- "epoch": 0.6740815638692281,
6
- "eval_steps": 50,
7
- "global_step": 500,
8
- "is_hyper_param_search": false,
9
- "is_local_process_zero": true,
10
- "is_world_process_zero": true,
11
- "log_history": [
12
- {
13
- "epoch": 0.026963262554769128,
14
- "grad_norm": 19.875,
15
- "learning_rate": 0.0005,
16
- "loss": 120.13565673828126,
17
- "step": 20
18
- },
19
- {
20
- "epoch": 0.053926525109538256,
21
- "grad_norm": 24.625,
22
- "learning_rate": 0.0005,
23
- "loss": 102.05034790039062,
24
- "step": 40
25
- },
26
- {
27
- "epoch": 0.06740815638692282,
28
- "eval_loss": 5.665860652923584,
29
- "eval_runtime": 9.5524,
30
- "eval_samples_per_second": 997.346,
31
- "eval_steps_per_second": 12.562,
32
- "step": 50
33
- },
34
- {
35
- "epoch": 0.08088978766430738,
36
- "grad_norm": 18.125,
37
- "learning_rate": 0.0005,
38
- "loss": 90.97202758789062,
39
- "step": 60
40
- },
41
- {
42
- "epoch": 0.10785305021907651,
43
- "grad_norm": 40.75,
44
- "learning_rate": 0.0005,
45
- "loss": 83.10302734375,
46
- "step": 80
47
- },
48
- {
49
- "epoch": 0.13481631277384565,
50
- "grad_norm": 29.625,
51
- "learning_rate": 0.0005,
52
- "loss": 77.51190185546875,
53
- "step": 100
54
- },
55
- {
56
- "epoch": 0.13481631277384565,
57
- "eval_loss": 4.689935684204102,
58
- "eval_runtime": 9.1111,
59
- "eval_samples_per_second": 1045.652,
60
- "eval_steps_per_second": 13.171,
61
- "step": 100
62
- },
63
- {
64
- "epoch": 0.16177957532861476,
65
- "grad_norm": 19.5,
66
- "learning_rate": 0.0005,
67
- "loss": 73.29612426757812,
68
- "step": 120
69
- },
70
- {
71
- "epoch": 0.1887428378833839,
72
- "grad_norm": 25.25,
73
- "learning_rate": 0.0005,
74
- "loss": 70.37640380859375,
75
- "step": 140
76
- },
77
- {
78
- "epoch": 0.20222446916076844,
79
- "eval_loss": 4.263797760009766,
80
- "eval_runtime": 9.4931,
81
- "eval_samples_per_second": 1003.568,
82
- "eval_steps_per_second": 12.641,
83
- "step": 150
84
- },
85
- {
86
- "epoch": 0.21570610043815303,
87
- "grad_norm": 14.625,
88
- "learning_rate": 0.0005,
89
- "loss": 68.2427978515625,
90
- "step": 160
91
- },
92
- {
93
- "epoch": 0.24266936299292213,
94
- "grad_norm": 13.0625,
95
- "learning_rate": 0.0005,
96
- "loss": 66.13766479492188,
97
- "step": 180
98
- },
99
- {
100
- "epoch": 0.2696326255476913,
101
- "grad_norm": 18.5,
102
- "learning_rate": 0.0005,
103
- "loss": 63.81178588867188,
104
- "step": 200
105
- },
106
- {
107
- "epoch": 0.2696326255476913,
108
- "eval_loss": 3.9343974590301514,
109
- "eval_runtime": 9.3868,
110
- "eval_samples_per_second": 1014.939,
111
- "eval_steps_per_second": 12.784,
112
- "step": 200
113
- },
114
- {
115
- "epoch": 0.2965958881024604,
116
- "grad_norm": 16.125,
117
- "learning_rate": 0.0005,
118
- "loss": 62.16768188476563,
119
- "step": 220
120
- },
121
- {
122
- "epoch": 0.3235591506572295,
123
- "grad_norm": 21.25,
124
- "learning_rate": 0.0005,
125
- "loss": 61.1105712890625,
126
- "step": 240
127
- },
128
- {
129
- "epoch": 0.33704078193461406,
130
- "eval_loss": 3.748317241668701,
131
- "eval_runtime": 9.4777,
132
- "eval_samples_per_second": 1005.204,
133
- "eval_steps_per_second": 12.661,
134
- "step": 250
135
- },
136
- {
137
- "epoch": 0.3505224132119987,
138
- "grad_norm": 42.25,
139
- "learning_rate": 0.0005,
140
- "loss": 60.05653686523438,
141
- "step": 260
142
- },
143
- {
144
- "epoch": 0.3774856757667678,
145
- "grad_norm": 27.5,
146
- "learning_rate": 0.0005,
147
- "loss": 59.25806884765625,
148
- "step": 280
149
- },
150
- {
151
- "epoch": 0.4044489383215369,
152
- "grad_norm": 14.375,
153
- "learning_rate": 0.0005,
154
- "loss": 58.334844970703124,
155
- "step": 300
156
- },
157
- {
158
- "epoch": 0.4044489383215369,
159
- "eval_loss": 3.6137855052948,
160
- "eval_runtime": 9.6654,
161
- "eval_samples_per_second": 985.685,
162
- "eval_steps_per_second": 12.415,
163
- "step": 300
164
- },
165
- {
166
- "epoch": 0.43141220087630605,
167
- "grad_norm": 32.75,
168
- "learning_rate": 0.0005,
169
- "loss": 57.341180419921876,
170
- "step": 320
171
- },
172
- {
173
- "epoch": 0.45837546343107516,
174
- "grad_norm": 12.5625,
175
- "learning_rate": 0.0005,
176
- "loss": 56.699591064453124,
177
- "step": 340
178
- },
179
- {
180
- "epoch": 0.4718570947084597,
181
- "eval_loss": 3.492727518081665,
182
- "eval_runtime": 9.4437,
183
- "eval_samples_per_second": 1008.824,
184
- "eval_steps_per_second": 12.707,
185
- "step": 350
186
- },
187
- {
188
- "epoch": 0.48533872598584427,
189
- "grad_norm": 21.5,
190
- "learning_rate": 0.0005,
191
- "loss": 55.840753173828126,
192
- "step": 360
193
- },
194
- {
195
- "epoch": 0.5123019885406134,
196
- "grad_norm": 14.5625,
197
- "learning_rate": 0.0005,
198
- "loss": 55.377203369140624,
199
- "step": 380
200
- },
201
- {
202
- "epoch": 0.5392652510953826,
203
- "grad_norm": 14.6875,
204
- "learning_rate": 0.0005,
205
- "loss": 54.936090087890626,
206
- "step": 400
207
- },
208
- {
209
- "epoch": 0.5392652510953826,
210
- "eval_loss": 3.41583251953125,
211
- "eval_runtime": 9.1837,
212
- "eval_samples_per_second": 1037.382,
213
- "eval_steps_per_second": 13.067,
214
- "step": 400
215
- },
216
- {
217
- "epoch": 0.5662285136501517,
218
- "grad_norm": 22.0,
219
- "learning_rate": 0.0005,
220
- "loss": 54.31259765625,
221
- "step": 420
222
- },
223
- {
224
- "epoch": 0.5931917762049208,
225
- "grad_norm": 22.5,
226
- "learning_rate": 0.0005,
227
- "loss": 53.894476318359374,
228
- "step": 440
229
- },
230
- {
231
- "epoch": 0.6066734074823054,
232
- "eval_loss": 3.3446545600891113,
233
- "eval_runtime": 9.6871,
234
- "eval_samples_per_second": 983.477,
235
- "eval_steps_per_second": 12.388,
236
- "step": 450
237
- },
238
- {
239
- "epoch": 0.6201550387596899,
240
- "grad_norm": 21.875,
241
- "learning_rate": 0.0005,
242
- "loss": 53.53094482421875,
243
- "step": 460
244
- },
245
- {
246
- "epoch": 0.647118301314459,
247
- "grad_norm": 20.625,
248
- "learning_rate": 0.0005,
249
- "loss": 53.14410400390625,
250
- "step": 480
251
- },
252
- {
253
- "epoch": 0.6740815638692281,
254
- "grad_norm": 19.75,
255
- "learning_rate": 0.0005,
256
- "loss": 52.86866455078125,
257
- "step": 500
258
- },
259
- {
260
- "epoch": 0.6740815638692281,
261
- "eval_loss": 3.2967636585235596,
262
- "eval_runtime": 9.5934,
263
- "eval_samples_per_second": 993.08,
264
- "eval_steps_per_second": 12.509,
265
- "step": 500
266
- }
267
- ],
268
- "logging_steps": 20,
269
- "max_steps": 750,
270
- "num_input_tokens_seen": 0,
271
- "num_train_epochs": 2,
272
- "save_steps": 100,
273
- "stateful_callbacks": {
274
- "TrainerControl": {
275
- "args": {
276
- "should_epoch_stop": false,
277
- "should_evaluate": false,
278
- "should_log": false,
279
- "should_save": true,
280
- "should_training_stop": false
281
- },
282
- "attributes": {}
283
- }
284
- },
285
- "total_flos": 2903884431360000.0,
286
- "train_batch_size": 80,
287
- "trial_name": null,
288
- "trial_params": null
289
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-500/training_args.bin DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:fa9985a7c7f521fa2a6f7d16c7d254b6d5e243707cd393fa6ea6dfe9b9926147
3
- size 4920
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-600/config.json DELETED
@@ -1,35 +0,0 @@
1
- {
2
- "activation": "linear",
3
- "architectures": [
4
- "TinyLlamaForCausalLM"
5
- ],
6
- "attention_bias": false,
7
- "attention_dropout": 0.0,
8
- "bos_token_id": 1,
9
- "dtype": "bfloat16",
10
- "eos_token_id": 2,
11
- "head_dim": 32,
12
- "hidden_act": "silu",
13
- "hidden_size": 128,
14
- "initializer_range": 0.02,
15
- "intermediate_size": 256,
16
- "max_position_embeddings": 512,
17
- "mlp_bias": false,
18
- "mlp_type": "mlp",
19
- "model_type": "tiny_llama",
20
- "num_attention_heads": 4,
21
- "num_hidden_layers": 9,
22
- "num_key_value_heads": 4,
23
- "pad_token_id": 0,
24
- "pretraining_tp": 1,
25
- "rms_norm_eps": 1e-06,
26
- "rope_parameters": {
27
- "rope_theta": 10000.0,
28
- "rope_type": "default"
29
- },
30
- "tie_word_embeddings": true,
31
- "tokenizer_name": "w-ahmad/tiny-stories-tokenizer",
32
- "transformers_version": "5.15.0.dev0",
33
- "use_cache": false,
34
- "vocab_size": 4096
35
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-600/model.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:d11eb993a659e855b75dbb8749bafe63201dbfe2165c882e08cae34741acad01
3
- size 4010544
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-600/optimizer.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:922e13a7ae7554e8c783fb3452cea228a7c5214f9e24acd104f734aded187450
3
- size 8068282
 
 
 
 
outio/mlp-linear-9L_run/checkpoint-600/rng_state.pth DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:b0b8a6e3878f9402461193b241d8f1ee8546c6ad03f43e5b9dab8f4fc8c4d065
3
- size 14244