yuweiiizz commited on
Commit
9014134
1 Parent(s): 3aca47d

Training in progress, step 1000, checkpoint

Browse files
last-checkpoint/config.json ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "openai/whisper-small",
3
+ "activation_dropout": 0.0,
4
+ "activation_function": "gelu",
5
+ "apply_spec_augment": false,
6
+ "architectures": [
7
+ "WhisperForConditionalGeneration"
8
+ ],
9
+ "attention_dropout": 0.0,
10
+ "begin_suppress_tokens": [
11
+ 220,
12
+ 50257
13
+ ],
14
+ "bos_token_id": 50257,
15
+ "classifier_proj_size": 256,
16
+ "d_model": 768,
17
+ "decoder_attention_heads": 12,
18
+ "decoder_ffn_dim": 3072,
19
+ "decoder_layerdrop": 0.0,
20
+ "decoder_layers": 12,
21
+ "decoder_start_token_id": 50258,
22
+ "dropout": 0.0,
23
+ "encoder_attention_heads": 12,
24
+ "encoder_ffn_dim": 3072,
25
+ "encoder_layerdrop": 0.0,
26
+ "encoder_layers": 12,
27
+ "eos_token_id": 50257,
28
+ "forced_decoder_ids": null,
29
+ "init_std": 0.02,
30
+ "is_encoder_decoder": true,
31
+ "mask_feature_length": 10,
32
+ "mask_feature_min_masks": 0,
33
+ "mask_feature_prob": 0.0,
34
+ "mask_time_length": 10,
35
+ "mask_time_min_masks": 2,
36
+ "mask_time_prob": 0.05,
37
+ "max_length": 448,
38
+ "max_source_positions": 1500,
39
+ "max_target_positions": 448,
40
+ "median_filter_width": 7,
41
+ "model_type": "whisper",
42
+ "num_hidden_layers": 12,
43
+ "num_mel_bins": 80,
44
+ "pad_token_id": 50257,
45
+ "scale_embedding": false,
46
+ "suppress_tokens": [],
47
+ "torch_dtype": "float32",
48
+ "transformers_version": "4.40.2",
49
+ "use_cache": false,
50
+ "use_weighted_layer_sum": false,
51
+ "vocab_size": 51865
52
+ }
last-checkpoint/generation_config.json ADDED
@@ -0,0 +1,266 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alignment_heads": [
3
+ [
4
+ 5,
5
+ 3
6
+ ],
7
+ [
8
+ 5,
9
+ 9
10
+ ],
11
+ [
12
+ 8,
13
+ 0
14
+ ],
15
+ [
16
+ 8,
17
+ 4
18
+ ],
19
+ [
20
+ 8,
21
+ 7
22
+ ],
23
+ [
24
+ 8,
25
+ 8
26
+ ],
27
+ [
28
+ 9,
29
+ 0
30
+ ],
31
+ [
32
+ 9,
33
+ 7
34
+ ],
35
+ [
36
+ 9,
37
+ 9
38
+ ],
39
+ [
40
+ 10,
41
+ 5
42
+ ]
43
+ ],
44
+ "begin_suppress_tokens": [
45
+ 220,
46
+ 50257
47
+ ],
48
+ "bos_token_id": 50257,
49
+ "decoder_start_token_id": 50258,
50
+ "eos_token_id": 50257,
51
+ "forced_decoder_ids": [
52
+ [
53
+ 1,
54
+ null
55
+ ],
56
+ [
57
+ 2,
58
+ 50359
59
+ ]
60
+ ],
61
+ "is_multilingual": true,
62
+ "lang_to_id": {
63
+ "<|af|>": 50327,
64
+ "<|am|>": 50334,
65
+ "<|ar|>": 50272,
66
+ "<|as|>": 50350,
67
+ "<|az|>": 50304,
68
+ "<|ba|>": 50355,
69
+ "<|be|>": 50330,
70
+ "<|bg|>": 50292,
71
+ "<|bn|>": 50302,
72
+ "<|bo|>": 50347,
73
+ "<|br|>": 50309,
74
+ "<|bs|>": 50315,
75
+ "<|ca|>": 50270,
76
+ "<|cs|>": 50283,
77
+ "<|cy|>": 50297,
78
+ "<|da|>": 50285,
79
+ "<|de|>": 50261,
80
+ "<|el|>": 50281,
81
+ "<|en|>": 50259,
82
+ "<|es|>": 50262,
83
+ "<|et|>": 50307,
84
+ "<|eu|>": 50310,
85
+ "<|fa|>": 50300,
86
+ "<|fi|>": 50277,
87
+ "<|fo|>": 50338,
88
+ "<|fr|>": 50265,
89
+ "<|gl|>": 50319,
90
+ "<|gu|>": 50333,
91
+ "<|haw|>": 50352,
92
+ "<|ha|>": 50354,
93
+ "<|he|>": 50279,
94
+ "<|hi|>": 50276,
95
+ "<|hr|>": 50291,
96
+ "<|ht|>": 50339,
97
+ "<|hu|>": 50286,
98
+ "<|hy|>": 50312,
99
+ "<|id|>": 50275,
100
+ "<|is|>": 50311,
101
+ "<|it|>": 50274,
102
+ "<|ja|>": 50266,
103
+ "<|jw|>": 50356,
104
+ "<|ka|>": 50329,
105
+ "<|kk|>": 50316,
106
+ "<|km|>": 50323,
107
+ "<|kn|>": 50306,
108
+ "<|ko|>": 50264,
109
+ "<|la|>": 50294,
110
+ "<|lb|>": 50345,
111
+ "<|ln|>": 50353,
112
+ "<|lo|>": 50336,
113
+ "<|lt|>": 50293,
114
+ "<|lv|>": 50301,
115
+ "<|mg|>": 50349,
116
+ "<|mi|>": 50295,
117
+ "<|mk|>": 50308,
118
+ "<|ml|>": 50296,
119
+ "<|mn|>": 50314,
120
+ "<|mr|>": 50320,
121
+ "<|ms|>": 50282,
122
+ "<|mt|>": 50343,
123
+ "<|my|>": 50346,
124
+ "<|ne|>": 50313,
125
+ "<|nl|>": 50271,
126
+ "<|nn|>": 50342,
127
+ "<|no|>": 50288,
128
+ "<|oc|>": 50328,
129
+ "<|pa|>": 50321,
130
+ "<|pl|>": 50269,
131
+ "<|ps|>": 50340,
132
+ "<|pt|>": 50267,
133
+ "<|ro|>": 50284,
134
+ "<|ru|>": 50263,
135
+ "<|sa|>": 50344,
136
+ "<|sd|>": 50332,
137
+ "<|si|>": 50322,
138
+ "<|sk|>": 50298,
139
+ "<|sl|>": 50305,
140
+ "<|sn|>": 50324,
141
+ "<|so|>": 50326,
142
+ "<|sq|>": 50317,
143
+ "<|sr|>": 50303,
144
+ "<|su|>": 50357,
145
+ "<|sv|>": 50273,
146
+ "<|sw|>": 50318,
147
+ "<|ta|>": 50287,
148
+ "<|te|>": 50299,
149
+ "<|tg|>": 50331,
150
+ "<|th|>": 50289,
151
+ "<|tk|>": 50341,
152
+ "<|tl|>": 50348,
153
+ "<|tr|>": 50268,
154
+ "<|tt|>": 50351,
155
+ "<|uk|>": 50280,
156
+ "<|ur|>": 50290,
157
+ "<|uz|>": 50337,
158
+ "<|vi|>": 50278,
159
+ "<|yi|>": 50335,
160
+ "<|yo|>": 50325,
161
+ "<|zh|>": 50260
162
+ },
163
+ "language": "<|zh|>",
164
+ "max_initial_timestamp_index": 50,
165
+ "max_length": 448,
166
+ "no_timestamps_token_id": 50363,
167
+ "pad_token_id": 50257,
168
+ "prev_sot_token_id": 50361,
169
+ "return_timestamps": false,
170
+ "suppress_tokens": [
171
+ 1,
172
+ 2,
173
+ 7,
174
+ 8,
175
+ 9,
176
+ 10,
177
+ 14,
178
+ 25,
179
+ 26,
180
+ 27,
181
+ 28,
182
+ 29,
183
+ 31,
184
+ 58,
185
+ 59,
186
+ 60,
187
+ 61,
188
+ 62,
189
+ 63,
190
+ 90,
191
+ 91,
192
+ 92,
193
+ 93,
194
+ 359,
195
+ 503,
196
+ 522,
197
+ 542,
198
+ 873,
199
+ 893,
200
+ 902,
201
+ 918,
202
+ 922,
203
+ 931,
204
+ 1350,
205
+ 1853,
206
+ 1982,
207
+ 2460,
208
+ 2627,
209
+ 3246,
210
+ 3253,
211
+ 3268,
212
+ 3536,
213
+ 3846,
214
+ 3961,
215
+ 4183,
216
+ 4667,
217
+ 6585,
218
+ 6647,
219
+ 7273,
220
+ 9061,
221
+ 9383,
222
+ 10428,
223
+ 10929,
224
+ 11938,
225
+ 12033,
226
+ 12331,
227
+ 12562,
228
+ 13793,
229
+ 14157,
230
+ 14635,
231
+ 15265,
232
+ 15618,
233
+ 16553,
234
+ 16604,
235
+ 18362,
236
+ 18956,
237
+ 20075,
238
+ 21675,
239
+ 22520,
240
+ 26130,
241
+ 26161,
242
+ 26435,
243
+ 28279,
244
+ 29464,
245
+ 31650,
246
+ 32302,
247
+ 32470,
248
+ 36865,
249
+ 42863,
250
+ 47425,
251
+ 49870,
252
+ 50254,
253
+ 50258,
254
+ 50358,
255
+ 50359,
256
+ 50360,
257
+ 50361,
258
+ 50362
259
+ ],
260
+ "task": "transcribe",
261
+ "task_to_id": {
262
+ "transcribe": 50359,
263
+ "translate": 50358
264
+ },
265
+ "transformers_version": "4.40.2"
266
+ }
last-checkpoint/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:030a13576845e75e2e63857e192c4030bf4d3836c56fae6ae173563df4028294
3
+ size 966995080
last-checkpoint/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6e2a5e113b303da23dff325add4179a0637b04a5ff3dd27d50eeb16d6ffd8f46
3
+ size 1925064044
last-checkpoint/preprocessor_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chunk_length": 30,
3
+ "feature_extractor_type": "WhisperFeatureExtractor",
4
+ "feature_size": 80,
5
+ "hop_length": 160,
6
+ "n_fft": 400,
7
+ "n_samples": 480000,
8
+ "nb_max_frames": 3000,
9
+ "padding_side": "right",
10
+ "padding_value": 0.0,
11
+ "processor_class": "WhisperProcessor",
12
+ "return_attention_mask": false,
13
+ "sampling_rate": 16000
14
+ }
last-checkpoint/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d1f09b1f1f9b06ad2afb12e89fc8695073b76afcf9ea0b3552c7069932117824
3
+ size 14244
last-checkpoint/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a51738ca0af55e803ac3fdbc0e3b67846eda4ca13018dd2960c216586e47c984
3
+ size 1064
last-checkpoint/trainer_state.json ADDED
@@ -0,0 +1,310 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": 28.522079327423317,
3
+ "best_model_checkpoint": "./whisper-small-taiwanese-hanzi/checkpoint-1000",
4
+ "epoch": 0.4,
5
+ "eval_steps": 1000,
6
+ "global_step": 1000,
7
+ "is_hyper_param_search": false,
8
+ "is_local_process_zero": true,
9
+ "is_world_process_zero": true,
10
+ "log_history": [
11
+ {
12
+ "epoch": 0.01,
13
+ "grad_norm": 191.53482055664062,
14
+ "learning_rate": 5.000000000000001e-07,
15
+ "loss": 7.0515,
16
+ "step": 25
17
+ },
18
+ {
19
+ "epoch": 0.02,
20
+ "grad_norm": 50.858787536621094,
21
+ "learning_rate": 1.0000000000000002e-06,
22
+ "loss": 5.441,
23
+ "step": 50
24
+ },
25
+ {
26
+ "epoch": 0.03,
27
+ "grad_norm": 33.031028747558594,
28
+ "learning_rate": 1.5e-06,
29
+ "loss": 3.8697,
30
+ "step": 75
31
+ },
32
+ {
33
+ "epoch": 0.04,
34
+ "grad_norm": 28.192623138427734,
35
+ "learning_rate": 2.0000000000000003e-06,
36
+ "loss": 2.7348,
37
+ "step": 100
38
+ },
39
+ {
40
+ "epoch": 0.05,
41
+ "grad_norm": 23.84247589111328,
42
+ "learning_rate": 2.5e-06,
43
+ "loss": 2.3454,
44
+ "step": 125
45
+ },
46
+ {
47
+ "epoch": 0.06,
48
+ "grad_norm": 25.436233520507812,
49
+ "learning_rate": 3e-06,
50
+ "loss": 2.1468,
51
+ "step": 150
52
+ },
53
+ {
54
+ "epoch": 0.07,
55
+ "grad_norm": 24.16190528869629,
56
+ "learning_rate": 3.5e-06,
57
+ "loss": 1.8151,
58
+ "step": 175
59
+ },
60
+ {
61
+ "epoch": 0.08,
62
+ "grad_norm": 22.0955810546875,
63
+ "learning_rate": 4.000000000000001e-06,
64
+ "loss": 1.6193,
65
+ "step": 200
66
+ },
67
+ {
68
+ "epoch": 0.09,
69
+ "grad_norm": 16.28850746154785,
70
+ "learning_rate": 4.5e-06,
71
+ "loss": 1.3734,
72
+ "step": 225
73
+ },
74
+ {
75
+ "epoch": 0.1,
76
+ "grad_norm": 13.348727226257324,
77
+ "learning_rate": 5e-06,
78
+ "loss": 1.288,
79
+ "step": 250
80
+ },
81
+ {
82
+ "epoch": 0.11,
83
+ "grad_norm": 12.8897123336792,
84
+ "learning_rate": 5.500000000000001e-06,
85
+ "loss": 1.2431,
86
+ "step": 275
87
+ },
88
+ {
89
+ "epoch": 0.12,
90
+ "grad_norm": 16.30738639831543,
91
+ "learning_rate": 6e-06,
92
+ "loss": 1.1198,
93
+ "step": 300
94
+ },
95
+ {
96
+ "epoch": 0.13,
97
+ "grad_norm": 13.086812973022461,
98
+ "learning_rate": 6.5000000000000004e-06,
99
+ "loss": 1.0947,
100
+ "step": 325
101
+ },
102
+ {
103
+ "epoch": 0.14,
104
+ "grad_norm": 12.245979309082031,
105
+ "learning_rate": 7e-06,
106
+ "loss": 1.0291,
107
+ "step": 350
108
+ },
109
+ {
110
+ "epoch": 0.15,
111
+ "grad_norm": 14.182358741760254,
112
+ "learning_rate": 7.500000000000001e-06,
113
+ "loss": 0.9875,
114
+ "step": 375
115
+ },
116
+ {
117
+ "epoch": 0.16,
118
+ "grad_norm": 19.550535202026367,
119
+ "learning_rate": 8.000000000000001e-06,
120
+ "loss": 0.9415,
121
+ "step": 400
122
+ },
123
+ {
124
+ "epoch": 0.17,
125
+ "grad_norm": 15.594605445861816,
126
+ "learning_rate": 8.5e-06,
127
+ "loss": 0.9123,
128
+ "step": 425
129
+ },
130
+ {
131
+ "epoch": 0.18,
132
+ "grad_norm": 12.243993759155273,
133
+ "learning_rate": 9e-06,
134
+ "loss": 0.8811,
135
+ "step": 450
136
+ },
137
+ {
138
+ "epoch": 0.19,
139
+ "grad_norm": 16.33646583557129,
140
+ "learning_rate": 9.5e-06,
141
+ "loss": 0.8885,
142
+ "step": 475
143
+ },
144
+ {
145
+ "epoch": 0.2,
146
+ "grad_norm": 12.957507133483887,
147
+ "learning_rate": 1e-05,
148
+ "loss": 0.8375,
149
+ "step": 500
150
+ },
151
+ {
152
+ "epoch": 0.21,
153
+ "grad_norm": 15.454781532287598,
154
+ "learning_rate": 9.944444444444445e-06,
155
+ "loss": 0.7655,
156
+ "step": 525
157
+ },
158
+ {
159
+ "epoch": 0.22,
160
+ "grad_norm": 11.995880126953125,
161
+ "learning_rate": 9.88888888888889e-06,
162
+ "loss": 0.7816,
163
+ "step": 550
164
+ },
165
+ {
166
+ "epoch": 0.23,
167
+ "grad_norm": 12.509295463562012,
168
+ "learning_rate": 9.833333333333333e-06,
169
+ "loss": 0.7413,
170
+ "step": 575
171
+ },
172
+ {
173
+ "epoch": 0.24,
174
+ "grad_norm": 12.734782218933105,
175
+ "learning_rate": 9.777777777777779e-06,
176
+ "loss": 0.8081,
177
+ "step": 600
178
+ },
179
+ {
180
+ "epoch": 0.25,
181
+ "grad_norm": 11.664021492004395,
182
+ "learning_rate": 9.722222222222223e-06,
183
+ "loss": 0.7416,
184
+ "step": 625
185
+ },
186
+ {
187
+ "epoch": 0.26,
188
+ "grad_norm": 10.193824768066406,
189
+ "learning_rate": 9.666666666666667e-06,
190
+ "loss": 0.7186,
191
+ "step": 650
192
+ },
193
+ {
194
+ "epoch": 0.27,
195
+ "grad_norm": 11.241517066955566,
196
+ "learning_rate": 9.611111111111112e-06,
197
+ "loss": 0.7178,
198
+ "step": 675
199
+ },
200
+ {
201
+ "epoch": 0.28,
202
+ "grad_norm": 11.208189964294434,
203
+ "learning_rate": 9.555555555555556e-06,
204
+ "loss": 0.6749,
205
+ "step": 700
206
+ },
207
+ {
208
+ "epoch": 0.29,
209
+ "grad_norm": 9.345088005065918,
210
+ "learning_rate": 9.5e-06,
211
+ "loss": 0.6773,
212
+ "step": 725
213
+ },
214
+ {
215
+ "epoch": 0.3,
216
+ "grad_norm": 11.954216003417969,
217
+ "learning_rate": 9.444444444444445e-06,
218
+ "loss": 0.6648,
219
+ "step": 750
220
+ },
221
+ {
222
+ "epoch": 0.31,
223
+ "grad_norm": 10.755352020263672,
224
+ "learning_rate": 9.38888888888889e-06,
225
+ "loss": 0.673,
226
+ "step": 775
227
+ },
228
+ {
229
+ "epoch": 0.32,
230
+ "grad_norm": 12.048798561096191,
231
+ "learning_rate": 9.333333333333334e-06,
232
+ "loss": 0.6619,
233
+ "step": 800
234
+ },
235
+ {
236
+ "epoch": 0.33,
237
+ "grad_norm": 10.677817344665527,
238
+ "learning_rate": 9.277777777777778e-06,
239
+ "loss": 0.6395,
240
+ "step": 825
241
+ },
242
+ {
243
+ "epoch": 0.34,
244
+ "grad_norm": 12.70429515838623,
245
+ "learning_rate": 9.222222222222224e-06,
246
+ "loss": 0.5982,
247
+ "step": 850
248
+ },
249
+ {
250
+ "epoch": 0.35,
251
+ "grad_norm": 11.555309295654297,
252
+ "learning_rate": 9.166666666666666e-06,
253
+ "loss": 0.6028,
254
+ "step": 875
255
+ },
256
+ {
257
+ "epoch": 0.36,
258
+ "grad_norm": 20.559389114379883,
259
+ "learning_rate": 9.111111111111112e-06,
260
+ "loss": 0.6572,
261
+ "step": 900
262
+ },
263
+ {
264
+ "epoch": 0.37,
265
+ "grad_norm": 8.118528366088867,
266
+ "learning_rate": 9.055555555555556e-06,
267
+ "loss": 0.595,
268
+ "step": 925
269
+ },
270
+ {
271
+ "epoch": 0.38,
272
+ "grad_norm": 11.243496894836426,
273
+ "learning_rate": 9e-06,
274
+ "loss": 0.6036,
275
+ "step": 950
276
+ },
277
+ {
278
+ "epoch": 0.39,
279
+ "grad_norm": 10.028908729553223,
280
+ "learning_rate": 8.944444444444446e-06,
281
+ "loss": 0.581,
282
+ "step": 975
283
+ },
284
+ {
285
+ "epoch": 0.4,
286
+ "grad_norm": 13.079019546508789,
287
+ "learning_rate": 8.888888888888888e-06,
288
+ "loss": 0.5463,
289
+ "step": 1000
290
+ },
291
+ {
292
+ "epoch": 0.4,
293
+ "eval_cer": 28.522079327423317,
294
+ "eval_loss": 0.558542788028717,
295
+ "eval_runtime": 1740.114,
296
+ "eval_samples_per_second": 2.262,
297
+ "eval_steps_per_second": 0.283,
298
+ "step": 1000
299
+ }
300
+ ],
301
+ "logging_steps": 25,
302
+ "max_steps": 5000,
303
+ "num_input_tokens_seen": 0,
304
+ "num_train_epochs": 2,
305
+ "save_steps": 1000,
306
+ "total_flos": 4.61736640512e+18,
307
+ "train_batch_size": 8,
308
+ "trial_name": null,
309
+ "trial_params": null
310
+ }
last-checkpoint/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5db4033f6b868aaeb993204292a7f37e97e009a90ed66631bd9548f433d7f150
3
+ size 5176