saqidr commited on
Commit
d1f1c47
·
verified ·
1 Parent(s): 9bb79c5

Training in progress, step 500

Browse files
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ab42167ffff4c1e729d547711e458440cd560d09dd08d069574cd46a86c79e0d
3
  size 268290900
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:25f9d6eb5310fb82fd6d0ce02178e219ef280bfac553ca1e5a08cd0dd16f2028
3
  size 268290900
run-2/checkpoint-1500/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fbbec414f173d23818e4f1a4d385e4e82b9e46794cb619ec4280120c1996c010
3
  size 268290900
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cda8d10fe1dbb92958cd584c54810ea7b4ca0cbcd757c868dddab676669a0270
3
  size 268290900
run-2/checkpoint-1500/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:eef8ed9d1c4352515c32c273b7b9868df404b50c997e196c96712a96d78c4d7c
3
  size 536643898
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2c932450d510dd3ab268e57603eab4d94e98de0c9535fc4238caafa908691f20
3
  size 536643898
run-2/checkpoint-1500/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ad4d7d251acf36e559c362893a1fb310c9f46b20e8a330025a14b6829ce4ab07
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eb7b4645e59a5c5cb66f63e8bcb5f006c535fd0ea63db9ae152dd586fd465b28
3
  size 1064
run-2/checkpoint-1500/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:02895758c800be06b344602e8d1e4451336b524bdce9f82588ca589e12cfef2e
3
  size 5176
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:baea954a8311f77b8b2d4c67cfd083cf6d7e3bc8a41fb699bd5b137d572b44d3
3
  size 5176
run-2/checkpoint-2000/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2976c14e25a7a0665a6a80691800ba40b64d3af6e97cd20346e4514e1d87fb3b
3
  size 268290900
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:94d5d1048ef2f321733b12e51f297d497c8b46b210d10f2cee5c81dbf1b4b9a9
3
  size 268290900
run-2/checkpoint-2000/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:19cf8d23e59d1a0daf3401e46b60bdc348544f776be10f26cdd37ae080bdd87d
3
  size 536643898
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:90ae0852d9b5eefebaf5766e1e271e13b5498a0b08d0676e99a72850aeaef01f
3
  size 536643898
run-2/checkpoint-2000/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f38866eaf1d2baeb52a55cb38ece6ee67f3213265b0830dd06c82a9148795bea
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2bf6d4c2c13eda2c2dfacdbfc8684055c5603a92b6c0c386d7c757b108195d12
3
  size 1064
run-2/checkpoint-2000/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:02895758c800be06b344602e8d1e4451336b524bdce9f82588ca589e12cfef2e
3
  size 5176
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:baea954a8311f77b8b2d4c67cfd083cf6d7e3bc8a41fb699bd5b137d572b44d3
3
  size 5176
run-2/checkpoint-2500/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:35182c738d53a59a34932d9c059e629f622646c963c6f0c2ad315e97711fa2d7
3
  size 268290900
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5003f3c41fb67a0c518309f8ac556d952bbf9476e879d6516aecca0c25cd6cbf
3
  size 268290900
run-2/checkpoint-2500/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:07f95b5f5b1df52bfda8be3d8cca7cabf1b38044eb5e2b5b4b5f4d48eff6c437
3
  size 536643898
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:838fd059fc0b12714fd51110f4b9d5be6a75b98b9c6da6e8e3c4c3442a3834aa
3
  size 536643898
run-2/checkpoint-2500/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:983ad82afbb61099978c5b5a095761d0cc8a5b9889d078c9a1fcb9751b84b6be
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2fe590653dc8c7914203f68b195f5118633e21a9fa30a5fcd6f1518bae980ac9
3
  size 14244
run-2/checkpoint-2500/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b7c943a77651a184c6653dd4706c14c9a27d52136e81942a2b361c74acb9f665
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2b997ea9328158f28ff50e49c476931e703b821360bc8325e0c4d100e032c865
3
  size 1064
run-2/checkpoint-2500/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:34a2db7a3504e0c57eb385477ee1398acbe37a72c4a5eee4ba4703a02720d57f
3
- size 4728
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:baea954a8311f77b8b2d4c67cfd083cf6d7e3bc8a41fb699bd5b137d572b44d3
3
+ size 5176
run-3/checkpoint-1000/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:49a37519a7193241e4c7880a2bb352a0a188591edb57992228c960f57306b331
3
  size 268290900
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3c6618d8bf4f85416c87ab2fddb2a143230d9298742e6c942f75b60979d22322
3
  size 268290900
run-3/checkpoint-1000/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b6d0a1d579fe73c539aba11316145fa686b4c462dd2052b4e2433f29b811a7f1
3
  size 536643898
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5be13283d31227bbd5f3cdc798cc4f4fe888d3cfbe127e00277f36968b72760a
3
  size 536643898
run-3/checkpoint-1000/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9f182501c34e4ea3ebc7617d27edab7e1367582b147e518cd90295ec7f2eaa0f
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:045cf4abe065257226fe11ac1a91dfd4c5b953cb2366365ef8bb18b4cf33e734
3
  size 1064
run-3/checkpoint-1000/trainer_state.json CHANGED
@@ -10,50 +10,50 @@
10
  "log_history": [
11
  {
12
  "epoch": 1.0,
13
- "eval_accuracy": 0.5893548387096774,
14
- "eval_loss": 0.2303294986486435,
15
- "eval_runtime": 1.354,
16
- "eval_samples_per_second": 2289.482,
17
- "eval_steps_per_second": 48.005,
18
  "step": 318
19
  },
20
  {
21
  "epoch": 1.5723270440251573,
22
- "grad_norm": 0.5280700325965881,
23
- "learning_rate": 1.371069182389937e-05,
24
- "loss": 0.3599,
25
  "step": 500
26
  },
27
  {
28
  "epoch": 2.0,
29
- "eval_accuracy": 0.8180645161290323,
30
- "eval_loss": 0.11683158576488495,
31
- "eval_runtime": 1.346,
32
- "eval_samples_per_second": 2303.101,
33
- "eval_steps_per_second": 48.291,
34
  "step": 636
35
  },
36
  {
37
  "epoch": 3.0,
38
- "eval_accuracy": 0.8654838709677419,
39
- "eval_loss": 0.08170921355485916,
40
- "eval_runtime": 1.3601,
41
- "eval_samples_per_second": 2279.288,
42
- "eval_steps_per_second": 47.792,
43
  "step": 954
44
  },
45
  {
46
  "epoch": 3.1446540880503147,
47
- "grad_norm": 0.49507445096969604,
48
- "learning_rate": 7.421383647798742e-06,
49
- "loss": 0.1347,
50
  "step": 1000
51
  }
52
  ],
53
  "logging_steps": 500,
54
- "max_steps": 1590,
55
  "num_input_tokens_seen": 0,
56
- "num_train_epochs": 5,
57
  "save_steps": 500,
58
  "stateful_callbacks": {
59
  "TrainerControl": {
@@ -71,8 +71,8 @@
71
  "train_batch_size": 48,
72
  "trial_name": null,
73
  "trial_params": {
74
- "alpha": 0.947951511264653,
75
- "num_train_epochs": 5,
76
- "temperature": 6
77
  }
78
  }
 
10
  "log_history": [
11
  {
12
  "epoch": 1.0,
13
+ "eval_accuracy": 0.6190322580645161,
14
+ "eval_loss": 0.25124841928482056,
15
+ "eval_runtime": 1.3533,
16
+ "eval_samples_per_second": 2290.756,
17
+ "eval_steps_per_second": 48.032,
18
  "step": 318
19
  },
20
  {
21
  "epoch": 1.5723270440251573,
22
+ "grad_norm": 0.6181610226631165,
23
+ "learning_rate": 1.606918238993711e-05,
24
+ "loss": 0.4004,
25
  "step": 500
26
  },
27
  {
28
  "epoch": 2.0,
29
+ "eval_accuracy": 0.8406451612903226,
30
+ "eval_loss": 0.1125619187951088,
31
+ "eval_runtime": 1.3516,
32
+ "eval_samples_per_second": 2293.644,
33
+ "eval_steps_per_second": 48.093,
34
  "step": 636
35
  },
36
  {
37
  "epoch": 3.0,
38
+ "eval_accuracy": 0.885483870967742,
39
+ "eval_loss": 0.06991825252771378,
40
+ "eval_runtime": 1.3619,
41
+ "eval_samples_per_second": 2276.288,
42
+ "eval_steps_per_second": 47.729,
43
  "step": 954
44
  },
45
  {
46
  "epoch": 3.1446540880503147,
47
+ "grad_norm": 0.5585916042327881,
48
+ "learning_rate": 1.2138364779874214e-05,
49
+ "loss": 0.1312,
50
  "step": 1000
51
  }
52
  ],
53
  "logging_steps": 500,
54
+ "max_steps": 2544,
55
  "num_input_tokens_seen": 0,
56
+ "num_train_epochs": 8,
57
  "save_steps": 500,
58
  "stateful_callbacks": {
59
  "TrainerControl": {
 
71
  "train_batch_size": 48,
72
  "trial_name": null,
73
  "trial_params": {
74
+ "alpha": 0.17566767797741356,
75
+ "num_train_epochs": 8,
76
+ "temperature": 4
77
  }
78
  }
run-3/checkpoint-1000/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9e2a1cb9dd55c40858a3ac6c021e84bbe7db57465df0ba06d3a7ad08deec13ce
3
  size 5176
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:699ed76d9af91edc99d12562d64ce4055f71cfe483dfd9ab44c7bfc8626ae66f
3
  size 5176
run-3/checkpoint-1500/trainer_state.json CHANGED
@@ -10,66 +10,66 @@
10
  "log_history": [
11
  {
12
  "epoch": 1.0,
13
- "eval_accuracy": 0.5893548387096774,
14
- "eval_loss": 0.2303294986486435,
15
- "eval_runtime": 1.354,
16
- "eval_samples_per_second": 2289.482,
17
- "eval_steps_per_second": 48.005,
18
  "step": 318
19
  },
20
  {
21
  "epoch": 1.5723270440251573,
22
- "grad_norm": 0.5280700325965881,
23
- "learning_rate": 1.371069182389937e-05,
24
- "loss": 0.3599,
25
  "step": 500
26
  },
27
  {
28
  "epoch": 2.0,
29
- "eval_accuracy": 0.8180645161290323,
30
- "eval_loss": 0.11683158576488495,
31
- "eval_runtime": 1.346,
32
- "eval_samples_per_second": 2303.101,
33
- "eval_steps_per_second": 48.291,
34
  "step": 636
35
  },
36
  {
37
  "epoch": 3.0,
38
- "eval_accuracy": 0.8654838709677419,
39
- "eval_loss": 0.08170921355485916,
40
- "eval_runtime": 1.3601,
41
- "eval_samples_per_second": 2279.288,
42
- "eval_steps_per_second": 47.792,
43
  "step": 954
44
  },
45
  {
46
  "epoch": 3.1446540880503147,
47
- "grad_norm": 0.49507445096969604,
48
- "learning_rate": 7.421383647798742e-06,
49
- "loss": 0.1347,
50
  "step": 1000
51
  },
52
  {
53
  "epoch": 4.0,
54
- "eval_accuracy": 0.88,
55
- "eval_loss": 0.06745168566703796,
56
- "eval_runtime": 1.359,
57
- "eval_samples_per_second": 2281.017,
58
- "eval_steps_per_second": 47.828,
59
  "step": 1272
60
  },
61
  {
62
  "epoch": 4.716981132075472,
63
- "grad_norm": 0.40409624576568604,
64
- "learning_rate": 1.1320754716981133e-06,
65
- "loss": 0.0945,
66
  "step": 1500
67
  }
68
  ],
69
  "logging_steps": 500,
70
- "max_steps": 1590,
71
  "num_input_tokens_seen": 0,
72
- "num_train_epochs": 5,
73
  "save_steps": 500,
74
  "stateful_callbacks": {
75
  "TrainerControl": {
@@ -87,8 +87,8 @@
87
  "train_batch_size": 48,
88
  "trial_name": null,
89
  "trial_params": {
90
- "alpha": 0.947951511264653,
91
- "num_train_epochs": 5,
92
- "temperature": 6
93
  }
94
  }
 
10
  "log_history": [
11
  {
12
  "epoch": 1.0,
13
+ "eval_accuracy": 0.6190322580645161,
14
+ "eval_loss": 0.25124841928482056,
15
+ "eval_runtime": 1.3533,
16
+ "eval_samples_per_second": 2290.756,
17
+ "eval_steps_per_second": 48.032,
18
  "step": 318
19
  },
20
  {
21
  "epoch": 1.5723270440251573,
22
+ "grad_norm": 0.6181610226631165,
23
+ "learning_rate": 1.606918238993711e-05,
24
+ "loss": 0.4004,
25
  "step": 500
26
  },
27
  {
28
  "epoch": 2.0,
29
+ "eval_accuracy": 0.8406451612903226,
30
+ "eval_loss": 0.1125619187951088,
31
+ "eval_runtime": 1.3516,
32
+ "eval_samples_per_second": 2293.644,
33
+ "eval_steps_per_second": 48.093,
34
  "step": 636
35
  },
36
  {
37
  "epoch": 3.0,
38
+ "eval_accuracy": 0.885483870967742,
39
+ "eval_loss": 0.06991825252771378,
40
+ "eval_runtime": 1.3619,
41
+ "eval_samples_per_second": 2276.288,
42
+ "eval_steps_per_second": 47.729,
43
  "step": 954
44
  },
45
  {
46
  "epoch": 3.1446540880503147,
47
+ "grad_norm": 0.5585916042327881,
48
+ "learning_rate": 1.2138364779874214e-05,
49
+ "loss": 0.1312,
50
  "step": 1000
51
  },
52
  {
53
  "epoch": 4.0,
54
+ "eval_accuracy": 0.9006451612903226,
55
+ "eval_loss": 0.05187460780143738,
56
+ "eval_runtime": 1.3534,
57
+ "eval_samples_per_second": 2290.5,
58
+ "eval_steps_per_second": 48.027,
59
  "step": 1272
60
  },
61
  {
62
  "epoch": 4.716981132075472,
63
+ "grad_norm": 0.37042877078056335,
64
+ "learning_rate": 8.207547169811321e-06,
65
+ "loss": 0.0787,
66
  "step": 1500
67
  }
68
  ],
69
  "logging_steps": 500,
70
+ "max_steps": 2544,
71
  "num_input_tokens_seen": 0,
72
+ "num_train_epochs": 8,
73
  "save_steps": 500,
74
  "stateful_callbacks": {
75
  "TrainerControl": {
 
87
  "train_batch_size": 48,
88
  "trial_name": null,
89
  "trial_params": {
90
+ "alpha": 0.17566767797741356,
91
+ "num_train_epochs": 8,
92
+ "temperature": 4
93
  }
94
  }
run-3/checkpoint-2000/config.json CHANGED
@@ -326,6 +326,6 @@
326
  "sinusoidal_pos_embds": false,
327
  "tie_weights_": true,
328
  "torch_dtype": "float32",
329
- "transformers_version": "4.37.2",
330
  "vocab_size": 30522
331
  }
 
326
  "sinusoidal_pos_embds": false,
327
  "tie_weights_": true,
328
  "torch_dtype": "float32",
329
+ "transformers_version": "4.41.1",
330
  "vocab_size": 30522
331
  }
run-3/checkpoint-2000/tokenizer.json CHANGED
@@ -1,11 +1,6 @@
1
  {
2
  "version": "1.0",
3
- "truncation": {
4
- "direction": "Right",
5
- "max_length": 512,
6
- "strategy": "LongestFirst",
7
- "stride": 0
8
- },
9
  "padding": null,
10
  "added_tokens": [
11
  {
 
1
  {
2
  "version": "1.0",
3
+ "truncation": null,
 
 
 
 
 
4
  "padding": null,
5
  "added_tokens": [
6
  {
run-3/checkpoint-2000/trainer_state.json CHANGED
@@ -10,94 +10,110 @@
10
  "log_history": [
11
  {
12
  "epoch": 1.0,
13
- "eval_accuracy": 0.5974193548387097,
14
- "eval_loss": 0.20201846957206726,
15
- "eval_runtime": 1.3839,
16
- "eval_samples_per_second": 2240.12,
17
- "eval_steps_per_second": 46.97,
18
  "step": 318
19
  },
20
  {
21
- "epoch": 1.57,
22
- "learning_rate": 1.550763701707098e-05,
23
- "loss": 0.321,
 
24
  "step": 500
25
  },
26
  {
27
  "epoch": 2.0,
28
- "eval_accuracy": 0.8154838709677419,
29
- "eval_loss": 0.10081231594085693,
30
- "eval_runtime": 1.376,
31
- "eval_samples_per_second": 2252.914,
32
- "eval_steps_per_second": 47.239,
33
  "step": 636
34
  },
35
  {
36
  "epoch": 3.0,
37
- "eval_accuracy": 0.8729032258064516,
38
- "eval_loss": 0.06774353235960007,
39
- "eval_runtime": 1.4034,
40
- "eval_samples_per_second": 2208.994,
41
- "eval_steps_per_second": 46.318,
42
  "step": 954
43
  },
44
  {
45
- "epoch": 3.14,
46
- "learning_rate": 1.101527403414196e-05,
47
- "loss": 0.115,
 
48
  "step": 1000
49
  },
50
  {
51
  "epoch": 4.0,
52
- "eval_accuracy": 0.8932258064516129,
53
- "eval_loss": 0.053254324942827225,
54
- "eval_runtime": 1.3981,
55
- "eval_samples_per_second": 2217.32,
56
- "eval_steps_per_second": 46.492,
57
  "step": 1272
58
  },
59
  {
60
- "epoch": 4.72,
61
- "learning_rate": 6.522911051212939e-06,
62
- "loss": 0.0763,
 
63
  "step": 1500
64
  },
65
  {
66
  "epoch": 5.0,
67
- "eval_accuracy": 0.9019354838709678,
68
- "eval_loss": 0.04572787880897522,
69
- "eval_runtime": 1.3764,
70
- "eval_samples_per_second": 2252.312,
71
- "eval_steps_per_second": 47.226,
72
  "step": 1590
73
  },
74
  {
75
  "epoch": 6.0,
76
- "eval_accuracy": 0.9051612903225806,
77
- "eval_loss": 0.041987188160419464,
78
- "eval_runtime": 1.3865,
79
- "eval_samples_per_second": 2235.912,
80
- "eval_steps_per_second": 46.882,
81
  "step": 1908
82
  },
83
  {
84
- "epoch": 6.29,
85
- "learning_rate": 2.0305480682839176e-06,
86
- "loss": 0.0628,
 
87
  "step": 2000
88
  }
89
  ],
90
  "logging_steps": 500,
91
- "max_steps": 2226,
92
  "num_input_tokens_seen": 0,
93
- "num_train_epochs": 7,
94
  "save_steps": 500,
95
- "total_flos": 519271419317532.0,
 
 
 
 
 
 
 
 
 
 
 
 
96
  "train_batch_size": 48,
97
  "trial_name": null,
98
  "trial_params": {
99
- "alpha": 0.27106164774863717,
100
- "num_train_epochs": 7,
101
- "temperature": 12
102
  }
103
  }
 
10
  "log_history": [
11
  {
12
  "epoch": 1.0,
13
+ "eval_accuracy": 0.6190322580645161,
14
+ "eval_loss": 0.25124841928482056,
15
+ "eval_runtime": 1.3533,
16
+ "eval_samples_per_second": 2290.756,
17
+ "eval_steps_per_second": 48.032,
18
  "step": 318
19
  },
20
  {
21
+ "epoch": 1.5723270440251573,
22
+ "grad_norm": 0.6181610226631165,
23
+ "learning_rate": 1.606918238993711e-05,
24
+ "loss": 0.4004,
25
  "step": 500
26
  },
27
  {
28
  "epoch": 2.0,
29
+ "eval_accuracy": 0.8406451612903226,
30
+ "eval_loss": 0.1125619187951088,
31
+ "eval_runtime": 1.3516,
32
+ "eval_samples_per_second": 2293.644,
33
+ "eval_steps_per_second": 48.093,
34
  "step": 636
35
  },
36
  {
37
  "epoch": 3.0,
38
+ "eval_accuracy": 0.885483870967742,
39
+ "eval_loss": 0.06991825252771378,
40
+ "eval_runtime": 1.3619,
41
+ "eval_samples_per_second": 2276.288,
42
+ "eval_steps_per_second": 47.729,
43
  "step": 954
44
  },
45
  {
46
+ "epoch": 3.1446540880503147,
47
+ "grad_norm": 0.5585916042327881,
48
+ "learning_rate": 1.2138364779874214e-05,
49
+ "loss": 0.1312,
50
  "step": 1000
51
  },
52
  {
53
  "epoch": 4.0,
54
+ "eval_accuracy": 0.9006451612903226,
55
+ "eval_loss": 0.05187460780143738,
56
+ "eval_runtime": 1.3534,
57
+ "eval_samples_per_second": 2290.5,
58
+ "eval_steps_per_second": 48.027,
59
  "step": 1272
60
  },
61
  {
62
+ "epoch": 4.716981132075472,
63
+ "grad_norm": 0.37042877078056335,
64
+ "learning_rate": 8.207547169811321e-06,
65
+ "loss": 0.0787,
66
  "step": 1500
67
  },
68
  {
69
  "epoch": 5.0,
70
+ "eval_accuracy": 0.9141935483870968,
71
+ "eval_loss": 0.04222690686583519,
72
+ "eval_runtime": 1.3549,
73
+ "eval_samples_per_second": 2287.962,
74
+ "eval_steps_per_second": 47.973,
75
  "step": 1590
76
  },
77
  {
78
  "epoch": 6.0,
79
+ "eval_accuracy": 0.9203225806451613,
80
+ "eval_loss": 0.03726611286401749,
81
+ "eval_runtime": 1.3638,
82
+ "eval_samples_per_second": 2273.052,
83
+ "eval_steps_per_second": 47.661,
84
  "step": 1908
85
  },
86
  {
87
+ "epoch": 6.289308176100629,
88
+ "grad_norm": 0.2954886257648468,
89
+ "learning_rate": 4.276729559748428e-06,
90
+ "loss": 0.0616,
91
  "step": 2000
92
  }
93
  ],
94
  "logging_steps": 500,
95
+ "max_steps": 2544,
96
  "num_input_tokens_seen": 0,
97
+ "num_train_epochs": 8,
98
  "save_steps": 500,
99
+ "stateful_callbacks": {
100
+ "TrainerControl": {
101
+ "args": {
102
+ "should_epoch_stop": false,
103
+ "should_evaluate": false,
104
+ "should_log": false,
105
+ "should_save": true,
106
+ "should_training_stop": false
107
+ },
108
+ "attributes": {}
109
+ }
110
+ },
111
+ "total_flos": 520991326672152.0,
112
  "train_batch_size": 48,
113
  "trial_name": null,
114
  "trial_params": {
115
+ "alpha": 0.17566767797741356,
116
+ "num_train_epochs": 8,
117
+ "temperature": 4
118
  }
119
  }
run-3/checkpoint-2500/config.json CHANGED
@@ -326,6 +326,6 @@
326
  "sinusoidal_pos_embds": false,
327
  "tie_weights_": true,
328
  "torch_dtype": "float32",
329
- "transformers_version": "4.37.2",
330
  "vocab_size": 30522
331
  }
 
326
  "sinusoidal_pos_embds": false,
327
  "tie_weights_": true,
328
  "torch_dtype": "float32",
329
+ "transformers_version": "4.41.1",
330
  "vocab_size": 30522
331
  }
run-3/checkpoint-2500/tokenizer.json CHANGED
@@ -1,11 +1,6 @@
1
  {
2
  "version": "1.0",
3
- "truncation": {
4
- "direction": "Right",
5
- "max_length": 512,
6
- "strategy": "LongestFirst",
7
- "stride": 0
8
- },
9
  "padding": null,
10
  "added_tokens": [
11
  {
 
1
  {
2
  "version": "1.0",
3
+ "truncation": null,
 
 
 
 
 
4
  "padding": null,
5
  "added_tokens": [
6
  {
run-3/checkpoint-2500/trainer_state.json CHANGED
@@ -10,109 +10,126 @@
10
  "log_history": [
11
  {
12
  "epoch": 1.0,
13
- "eval_accuracy": 0.6403225806451613,
14
- "eval_loss": 0.23227573931217194,
15
- "eval_runtime": 1.2548,
16
- "eval_samples_per_second": 2470.604,
17
- "eval_steps_per_second": 51.803,
18
  "step": 318
19
  },
20
  {
21
- "epoch": 1.57,
22
- "learning_rate": 1.650593990216632e-05,
23
- "loss": 0.3708,
 
24
  "step": 500
25
  },
26
  {
27
  "epoch": 2.0,
28
- "eval_accuracy": 0.832258064516129,
29
- "eval_loss": 0.10681243985891342,
30
- "eval_runtime": 1.2626,
31
- "eval_samples_per_second": 2455.178,
32
- "eval_steps_per_second": 51.48,
33
  "step": 636
34
  },
35
  {
36
  "epoch": 3.0,
37
- "eval_accuracy": 0.8841935483870967,
38
- "eval_loss": 0.06609988212585449,
39
- "eval_runtime": 1.3493,
40
- "eval_samples_per_second": 2297.478,
41
- "eval_steps_per_second": 48.173,
42
  "step": 954
43
  },
44
  {
45
- "epoch": 3.14,
46
- "learning_rate": 1.3011879804332637e-05,
47
- "loss": 0.1227,
 
48
  "step": 1000
49
  },
50
  {
51
  "epoch": 4.0,
52
- "eval_accuracy": 0.9041935483870968,
53
- "eval_loss": 0.04876786842942238,
54
- "eval_runtime": 1.3474,
55
- "eval_samples_per_second": 2300.769,
56
- "eval_steps_per_second": 48.242,
57
  "step": 1272
58
  },
59
  {
60
- "epoch": 4.72,
61
- "learning_rate": 9.517819706498952e-06,
62
- "loss": 0.0748,
 
63
  "step": 1500
64
  },
65
  {
66
  "epoch": 5.0,
67
- "eval_accuracy": 0.9145161290322581,
68
- "eval_loss": 0.04008982330560684,
69
- "eval_runtime": 1.2747,
70
- "eval_samples_per_second": 2431.964,
71
- "eval_steps_per_second": 50.993,
72
  "step": 1590
73
  },
74
  {
75
  "epoch": 6.0,
76
- "eval_accuracy": 0.9232258064516129,
77
- "eval_loss": 0.035276755690574646,
78
- "eval_runtime": 1.3204,
79
- "eval_samples_per_second": 2347.825,
80
- "eval_steps_per_second": 49.229,
81
  "step": 1908
82
  },
83
  {
84
- "epoch": 6.29,
85
- "learning_rate": 6.02375960866527e-06,
86
- "loss": 0.0578,
 
87
  "step": 2000
88
  },
89
  {
90
  "epoch": 7.0,
91
- "eval_accuracy": 0.9264516129032258,
92
- "eval_loss": 0.03217238560318947,
93
- "eval_runtime": 1.2775,
94
- "eval_samples_per_second": 2426.53,
95
- "eval_steps_per_second": 50.879,
96
  "step": 2226
97
  },
98
  {
99
- "epoch": 7.86,
100
- "learning_rate": 2.5296995108315863e-06,
101
- "loss": 0.0505,
 
102
  "step": 2500
103
  }
104
  ],
105
  "logging_steps": 500,
106
- "max_steps": 2862,
107
  "num_input_tokens_seen": 0,
108
- "num_train_epochs": 9,
109
  "save_steps": 500,
110
- "total_flos": 649131947564688.0,
 
 
 
 
 
 
 
 
 
 
 
 
111
  "train_batch_size": 48,
112
  "trial_name": null,
113
  "trial_params": {
114
- "alpha": 0.6247539052050974,
115
- "num_train_epochs": 9,
116
- "temperature": 5
117
  }
118
  }
 
10
  "log_history": [
11
  {
12
  "epoch": 1.0,
13
+ "eval_accuracy": 0.6190322580645161,
14
+ "eval_loss": 0.25124841928482056,
15
+ "eval_runtime": 1.3533,
16
+ "eval_samples_per_second": 2290.756,
17
+ "eval_steps_per_second": 48.032,
18
  "step": 318
19
  },
20
  {
21
+ "epoch": 1.5723270440251573,
22
+ "grad_norm": 0.6181610226631165,
23
+ "learning_rate": 1.606918238993711e-05,
24
+ "loss": 0.4004,
25
  "step": 500
26
  },
27
  {
28
  "epoch": 2.0,
29
+ "eval_accuracy": 0.8406451612903226,
30
+ "eval_loss": 0.1125619187951088,
31
+ "eval_runtime": 1.3516,
32
+ "eval_samples_per_second": 2293.644,
33
+ "eval_steps_per_second": 48.093,
34
  "step": 636
35
  },
36
  {
37
  "epoch": 3.0,
38
+ "eval_accuracy": 0.885483870967742,
39
+ "eval_loss": 0.06991825252771378,
40
+ "eval_runtime": 1.3619,
41
+ "eval_samples_per_second": 2276.288,
42
+ "eval_steps_per_second": 47.729,
43
  "step": 954
44
  },
45
  {
46
+ "epoch": 3.1446540880503147,
47
+ "grad_norm": 0.5585916042327881,
48
+ "learning_rate": 1.2138364779874214e-05,
49
+ "loss": 0.1312,
50
  "step": 1000
51
  },
52
  {
53
  "epoch": 4.0,
54
+ "eval_accuracy": 0.9006451612903226,
55
+ "eval_loss": 0.05187460780143738,
56
+ "eval_runtime": 1.3534,
57
+ "eval_samples_per_second": 2290.5,
58
+ "eval_steps_per_second": 48.027,
59
  "step": 1272
60
  },
61
  {
62
+ "epoch": 4.716981132075472,
63
+ "grad_norm": 0.37042877078056335,
64
+ "learning_rate": 8.207547169811321e-06,
65
+ "loss": 0.0787,
66
  "step": 1500
67
  },
68
  {
69
  "epoch": 5.0,
70
+ "eval_accuracy": 0.9141935483870968,
71
+ "eval_loss": 0.04222690686583519,
72
+ "eval_runtime": 1.3549,
73
+ "eval_samples_per_second": 2287.962,
74
+ "eval_steps_per_second": 47.973,
75
  "step": 1590
76
  },
77
  {
78
  "epoch": 6.0,
79
+ "eval_accuracy": 0.9203225806451613,
80
+ "eval_loss": 0.03726611286401749,
81
+ "eval_runtime": 1.3638,
82
+ "eval_samples_per_second": 2273.052,
83
+ "eval_steps_per_second": 47.661,
84
  "step": 1908
85
  },
86
  {
87
+ "epoch": 6.289308176100629,
88
+ "grad_norm": 0.2954886257648468,
89
+ "learning_rate": 4.276729559748428e-06,
90
+ "loss": 0.0616,
91
  "step": 2000
92
  },
93
  {
94
  "epoch": 7.0,
95
+ "eval_accuracy": 0.9219354838709677,
96
+ "eval_loss": 0.035025350749492645,
97
+ "eval_runtime": 1.3626,
98
+ "eval_samples_per_second": 2275.027,
99
+ "eval_steps_per_second": 47.702,
100
  "step": 2226
101
  },
102
  {
103
+ "epoch": 7.861635220125786,
104
+ "grad_norm": 0.26834845542907715,
105
+ "learning_rate": 3.459119496855346e-07,
106
+ "loss": 0.0549,
107
  "step": 2500
108
  }
109
  ],
110
  "logging_steps": 500,
111
+ "max_steps": 2544,
112
  "num_input_tokens_seen": 0,
113
+ "num_train_epochs": 8,
114
  "save_steps": 500,
115
+ "stateful_callbacks": {
116
+ "TrainerControl": {
117
+ "args": {
118
+ "should_epoch_stop": false,
119
+ "should_evaluate": false,
120
+ "should_log": false,
121
+ "should_save": true,
122
+ "should_training_stop": false
123
+ },
124
+ "attributes": {}
125
+ }
126
+ },
127
+ "total_flos": 651155886807636.0,
128
  "train_batch_size": 48,
129
  "trial_name": null,
130
  "trial_params": {
131
+ "alpha": 0.17566767797741356,
132
+ "num_train_epochs": 8,
133
+ "temperature": 4
134
  }
135
  }
run-3/checkpoint-500/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:afd2ce840c42e4f81187b486c6afb8eb396a25dd802b57910141a11d216310a4
3
  size 268290900
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:25f9d6eb5310fb82fd6d0ce02178e219ef280bfac553ca1e5a08cd0dd16f2028
3
  size 268290900
run-3/checkpoint-500/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:433eb5c99790bde8ccfa94533a3e385c22d8cc1001199b3afcfbf61d951e87e1
3
  size 536643898
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:61b66909a996ee50dd5b57e0c400b5d2288582050b5f926b1c289f7a221255f6
3
  size 536643898
run-3/checkpoint-500/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9e1264523e958cf7990dc5f42d876cc12129475c4603804cf66868aaf25c2c24
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:90dc4637e972cc69c745eebddd8a7560dca27d2318df3e23f8e145abbf236536
3
  size 1064
run-3/checkpoint-500/trainer_state.json CHANGED
@@ -10,25 +10,25 @@
10
  "log_history": [
11
  {
12
  "epoch": 1.0,
13
- "eval_accuracy": 0.5893548387096774,
14
- "eval_loss": 0.2303294986486435,
15
- "eval_runtime": 1.354,
16
- "eval_samples_per_second": 2289.482,
17
- "eval_steps_per_second": 48.005,
18
  "step": 318
19
  },
20
  {
21
  "epoch": 1.5723270440251573,
22
- "grad_norm": 0.5280700325965881,
23
- "learning_rate": 1.371069182389937e-05,
24
- "loss": 0.3599,
25
  "step": 500
26
  }
27
  ],
28
  "logging_steps": 500,
29
- "max_steps": 1590,
30
  "num_input_tokens_seen": 0,
31
- "num_train_epochs": 5,
32
  "save_steps": 500,
33
  "stateful_callbacks": {
34
  "TrainerControl": {
@@ -46,8 +46,8 @@
46
  "train_batch_size": 48,
47
  "trial_name": null,
48
  "trial_params": {
49
- "alpha": 0.947951511264653,
50
- "num_train_epochs": 5,
51
- "temperature": 6
52
  }
53
  }
 
10
  "log_history": [
11
  {
12
  "epoch": 1.0,
13
+ "eval_accuracy": 0.6190322580645161,
14
+ "eval_loss": 0.25124841928482056,
15
+ "eval_runtime": 1.3533,
16
+ "eval_samples_per_second": 2290.756,
17
+ "eval_steps_per_second": 48.032,
18
  "step": 318
19
  },
20
  {
21
  "epoch": 1.5723270440251573,
22
+ "grad_norm": 0.6181610226631165,
23
+ "learning_rate": 1.606918238993711e-05,
24
+ "loss": 0.4004,
25
  "step": 500
26
  }
27
  ],
28
  "logging_steps": 500,
29
+ "max_steps": 2544,
30
  "num_input_tokens_seen": 0,
31
+ "num_train_epochs": 8,
32
  "save_steps": 500,
33
  "stateful_callbacks": {
34
  "TrainerControl": {
 
46
  "train_batch_size": 48,
47
  "trial_name": null,
48
  "trial_params": {
49
+ "alpha": 0.17566767797741356,
50
+ "num_train_epochs": 8,
51
+ "temperature": 4
52
  }
53
  }
run-3/checkpoint-500/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9e2a1cb9dd55c40858a3ac6c021e84bbe7db57465df0ba06d3a7ad08deec13ce
3
  size 5176
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:699ed76d9af91edc99d12562d64ce4055f71cfe483dfd9ab44c7bfc8626ae66f
3
  size 5176
training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:baea954a8311f77b8b2d4c67cfd083cf6d7e3bc8a41fb699bd5b137d572b44d3
3
  size 5176
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:699ed76d9af91edc99d12562d64ce4055f71cfe483dfd9ab44c7bfc8626ae66f
3
  size 5176