Training in progress, step 300, checkpoint
Browse files- last-checkpoint/model.safetensors +1 -1
- last-checkpoint/optimizer.pt +1 -1
- last-checkpoint/rng_state_0.pth +1 -1
- last-checkpoint/rng_state_1.pth +1 -1
- last-checkpoint/rng_state_2.pth +1 -1
- last-checkpoint/rng_state_3.pth +1 -1
- last-checkpoint/rng_state_4.pth +1 -1
- last-checkpoint/rng_state_5.pth +1 -1
- last-checkpoint/rng_state_6.pth +1 -1
- last-checkpoint/rng_state_7.pth +1 -1
- last-checkpoint/scheduler.pt +1 -1
- last-checkpoint/trainer_state.json +74 -4
last-checkpoint/model.safetensors
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 2066752
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:ae2287b45658d1cae69b4bfe25e777f022036e8d980148fa7635d454487588a9
|
3 |
size 2066752
|
last-checkpoint/optimizer.pt
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 2162798
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:556e1830f3d5b4cc26513099e09450196ba4a64ae97c75af17431421d368e625
|
3 |
size 2162798
|
last-checkpoint/rng_state_0.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 15984
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:4faa065a55913b65f4b0549e4d93d87e8865c0f6ec216f40a3de4d251a15322a
|
3 |
size 15984
|
last-checkpoint/rng_state_1.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 15984
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:10ae8864af9d168bc9a94e5c5625da874d35a133304d7d7414b10c80148467d4
|
3 |
size 15984
|
last-checkpoint/rng_state_2.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 15984
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:ece1ed46b8aa193251efdc1d8393b3bb872b53f6ba93c31cc3efc627b34d74be
|
3 |
size 15984
|
last-checkpoint/rng_state_3.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 15984
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:3018d94a8b9b3b95a3578032d80b8d3f31c01fab9a615c48039128422aba13ef
|
3 |
size 15984
|
last-checkpoint/rng_state_4.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 15984
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:d53200af24bc9fb65fd10ddaa327aeda1804119510e1de3d0cbe9297bfbcade4
|
3 |
size 15984
|
last-checkpoint/rng_state_5.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 15984
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:86c1416ffb6786e15b9550c111cad94cc2bc7bb18dea5ef0cdef3475e2015ab2
|
3 |
size 15984
|
last-checkpoint/rng_state_6.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 15984
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:47857ef45555e1fc8f192b42aa88a26a9f993ff97303b3c590fa1e726221a814
|
3 |
size 15984
|
last-checkpoint/rng_state_7.pth
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 15984
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:1912f7357d97d0c536c21d15f5ae8d134b73c487be1987df761ea730336d8216
|
3 |
size 15984
|
last-checkpoint/scheduler.pt
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 1064
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:74485e67705dc36efbfb69b1e54f842e1ff07894d01bb0e36d6d2526a318b300
|
3 |
size 1064
|
last-checkpoint/trainer_state.json
CHANGED
@@ -1,9 +1,9 @@
|
|
1 |
{
|
2 |
"best_metric": null,
|
3 |
"best_model_checkpoint": null,
|
4 |
-
"epoch":
|
5 |
"eval_steps": 200,
|
6 |
-
"global_step":
|
7 |
"is_hyper_param_search": false,
|
8 |
"is_local_process_zero": true,
|
9 |
"is_world_process_zero": true,
|
@@ -163,6 +163,76 @@
|
|
163 |
"eval_samples_per_second": 1561.466,
|
164 |
"eval_steps_per_second": 6.242,
|
165 |
"step": 200
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
166 |
}
|
167 |
],
|
168 |
"logging_steps": 10,
|
@@ -177,12 +247,12 @@
|
|
177 |
"should_evaluate": false,
|
178 |
"should_log": false,
|
179 |
"should_save": true,
|
180 |
-
"should_training_stop":
|
181 |
},
|
182 |
"attributes": {}
|
183 |
}
|
184 |
},
|
185 |
-
"total_flos":
|
186 |
"train_batch_size": 32,
|
187 |
"trial_name": null,
|
188 |
"trial_params": null
|
|
|
1 |
{
|
2 |
"best_metric": null,
|
3 |
"best_model_checkpoint": null,
|
4 |
+
"epoch": 100.0,
|
5 |
"eval_steps": 200,
|
6 |
+
"global_step": 300,
|
7 |
"is_hyper_param_search": false,
|
8 |
"is_local_process_zero": true,
|
9 |
"is_world_process_zero": true,
|
|
|
163 |
"eval_samples_per_second": 1561.466,
|
164 |
"eval_steps_per_second": 6.242,
|
165 |
"step": 200
|
166 |
+
},
|
167 |
+
{
|
168 |
+
"epoch": 70.0,
|
169 |
+
"grad_norm": 0.36328125,
|
170 |
+
"learning_rate": 4.12214747707527e-05,
|
171 |
+
"loss": 9.6678,
|
172 |
+
"step": 210
|
173 |
+
},
|
174 |
+
{
|
175 |
+
"epoch": 73.33333333333333,
|
176 |
+
"grad_norm": 0.36328125,
|
177 |
+
"learning_rate": 3.308693936411421e-05,
|
178 |
+
"loss": 9.6641,
|
179 |
+
"step": 220
|
180 |
+
},
|
181 |
+
{
|
182 |
+
"epoch": 76.66666666666667,
|
183 |
+
"grad_norm": 0.36328125,
|
184 |
+
"learning_rate": 2.5685517452260567e-05,
|
185 |
+
"loss": 9.6616,
|
186 |
+
"step": 230
|
187 |
+
},
|
188 |
+
{
|
189 |
+
"epoch": 80.0,
|
190 |
+
"grad_norm": 0.36328125,
|
191 |
+
"learning_rate": 1.9098300562505266e-05,
|
192 |
+
"loss": 9.6605,
|
193 |
+
"step": 240
|
194 |
+
},
|
195 |
+
{
|
196 |
+
"epoch": 83.33333333333333,
|
197 |
+
"grad_norm": 0.365234375,
|
198 |
+
"learning_rate": 1.339745962155613e-05,
|
199 |
+
"loss": 9.6596,
|
200 |
+
"step": 250
|
201 |
+
},
|
202 |
+
{
|
203 |
+
"epoch": 86.66666666666667,
|
204 |
+
"grad_norm": 0.36328125,
|
205 |
+
"learning_rate": 8.645454235739903e-06,
|
206 |
+
"loss": 9.6597,
|
207 |
+
"step": 260
|
208 |
+
},
|
209 |
+
{
|
210 |
+
"epoch": 90.0,
|
211 |
+
"grad_norm": 0.36328125,
|
212 |
+
"learning_rate": 4.8943483704846475e-06,
|
213 |
+
"loss": 9.6595,
|
214 |
+
"step": 270
|
215 |
+
},
|
216 |
+
{
|
217 |
+
"epoch": 93.33333333333333,
|
218 |
+
"grad_norm": 0.361328125,
|
219 |
+
"learning_rate": 2.1852399266194314e-06,
|
220 |
+
"loss": 9.6595,
|
221 |
+
"step": 280
|
222 |
+
},
|
223 |
+
{
|
224 |
+
"epoch": 96.66666666666667,
|
225 |
+
"grad_norm": 0.3671875,
|
226 |
+
"learning_rate": 5.478104631726711e-07,
|
227 |
+
"loss": 9.659,
|
228 |
+
"step": 290
|
229 |
+
},
|
230 |
+
{
|
231 |
+
"epoch": 100.0,
|
232 |
+
"grad_norm": 0.36328125,
|
233 |
+
"learning_rate": 0.0,
|
234 |
+
"loss": 9.6596,
|
235 |
+
"step": 300
|
236 |
}
|
237 |
],
|
238 |
"logging_steps": 10,
|
|
|
247 |
"should_evaluate": false,
|
248 |
"should_log": false,
|
249 |
"should_save": true,
|
250 |
+
"should_training_stop": true
|
251 |
},
|
252 |
"attributes": {}
|
253 |
}
|
254 |
},
|
255 |
+
"total_flos": 490990259404800.0,
|
256 |
"train_batch_size": 32,
|
257 |
"trial_name": null,
|
258 |
"trial_params": null
|