yuweiiizz commited on
Commit
50d3007
·
verified ·
1 Parent(s): e9ced29

Training in progress, step 1000, checkpoint

Browse files
last-checkpoint/config.json ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "openai/whisper-small",
3
+ "activation_dropout": 0.0,
4
+ "activation_function": "gelu",
5
+ "apply_spec_augment": false,
6
+ "architectures": [
7
+ "WhisperForConditionalGeneration"
8
+ ],
9
+ "attention_dropout": 0.0,
10
+ "begin_suppress_tokens": [
11
+ 220,
12
+ 50257
13
+ ],
14
+ "bos_token_id": 50257,
15
+ "classifier_proj_size": 256,
16
+ "d_model": 768,
17
+ "decoder_attention_heads": 12,
18
+ "decoder_ffn_dim": 3072,
19
+ "decoder_layerdrop": 0.0,
20
+ "decoder_layers": 12,
21
+ "decoder_start_token_id": 50258,
22
+ "dropout": 0.0,
23
+ "encoder_attention_heads": 12,
24
+ "encoder_ffn_dim": 3072,
25
+ "encoder_layerdrop": 0.0,
26
+ "encoder_layers": 12,
27
+ "eos_token_id": 50257,
28
+ "forced_decoder_ids": null,
29
+ "init_std": 0.02,
30
+ "is_encoder_decoder": true,
31
+ "mask_feature_length": 10,
32
+ "mask_feature_min_masks": 0,
33
+ "mask_feature_prob": 0.0,
34
+ "mask_time_length": 10,
35
+ "mask_time_min_masks": 2,
36
+ "mask_time_prob": 0.05,
37
+ "max_length": 448,
38
+ "max_source_positions": 1500,
39
+ "max_target_positions": 448,
40
+ "median_filter_width": 7,
41
+ "model_type": "whisper",
42
+ "num_hidden_layers": 12,
43
+ "num_mel_bins": 80,
44
+ "pad_token_id": 50257,
45
+ "scale_embedding": false,
46
+ "suppress_tokens": [],
47
+ "torch_dtype": "float32",
48
+ "transformers_version": "4.40.0",
49
+ "use_cache": false,
50
+ "use_weighted_layer_sum": false,
51
+ "vocab_size": 51865
52
+ }
last-checkpoint/generation_config.json ADDED
@@ -0,0 +1,266 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alignment_heads": [
3
+ [
4
+ 5,
5
+ 3
6
+ ],
7
+ [
8
+ 5,
9
+ 9
10
+ ],
11
+ [
12
+ 8,
13
+ 0
14
+ ],
15
+ [
16
+ 8,
17
+ 4
18
+ ],
19
+ [
20
+ 8,
21
+ 7
22
+ ],
23
+ [
24
+ 8,
25
+ 8
26
+ ],
27
+ [
28
+ 9,
29
+ 0
30
+ ],
31
+ [
32
+ 9,
33
+ 7
34
+ ],
35
+ [
36
+ 9,
37
+ 9
38
+ ],
39
+ [
40
+ 10,
41
+ 5
42
+ ]
43
+ ],
44
+ "begin_suppress_tokens": [
45
+ 220,
46
+ 50257
47
+ ],
48
+ "bos_token_id": 50257,
49
+ "decoder_start_token_id": 50258,
50
+ "eos_token_id": 50257,
51
+ "forced_decoder_ids": [
52
+ [
53
+ 1,
54
+ null
55
+ ],
56
+ [
57
+ 2,
58
+ 50359
59
+ ]
60
+ ],
61
+ "is_multilingual": true,
62
+ "lang_to_id": {
63
+ "<|af|>": 50327,
64
+ "<|am|>": 50334,
65
+ "<|ar|>": 50272,
66
+ "<|as|>": 50350,
67
+ "<|az|>": 50304,
68
+ "<|ba|>": 50355,
69
+ "<|be|>": 50330,
70
+ "<|bg|>": 50292,
71
+ "<|bn|>": 50302,
72
+ "<|bo|>": 50347,
73
+ "<|br|>": 50309,
74
+ "<|bs|>": 50315,
75
+ "<|ca|>": 50270,
76
+ "<|cs|>": 50283,
77
+ "<|cy|>": 50297,
78
+ "<|da|>": 50285,
79
+ "<|de|>": 50261,
80
+ "<|el|>": 50281,
81
+ "<|en|>": 50259,
82
+ "<|es|>": 50262,
83
+ "<|et|>": 50307,
84
+ "<|eu|>": 50310,
85
+ "<|fa|>": 50300,
86
+ "<|fi|>": 50277,
87
+ "<|fo|>": 50338,
88
+ "<|fr|>": 50265,
89
+ "<|gl|>": 50319,
90
+ "<|gu|>": 50333,
91
+ "<|haw|>": 50352,
92
+ "<|ha|>": 50354,
93
+ "<|he|>": 50279,
94
+ "<|hi|>": 50276,
95
+ "<|hr|>": 50291,
96
+ "<|ht|>": 50339,
97
+ "<|hu|>": 50286,
98
+ "<|hy|>": 50312,
99
+ "<|id|>": 50275,
100
+ "<|is|>": 50311,
101
+ "<|it|>": 50274,
102
+ "<|ja|>": 50266,
103
+ "<|jw|>": 50356,
104
+ "<|ka|>": 50329,
105
+ "<|kk|>": 50316,
106
+ "<|km|>": 50323,
107
+ "<|kn|>": 50306,
108
+ "<|ko|>": 50264,
109
+ "<|la|>": 50294,
110
+ "<|lb|>": 50345,
111
+ "<|ln|>": 50353,
112
+ "<|lo|>": 50336,
113
+ "<|lt|>": 50293,
114
+ "<|lv|>": 50301,
115
+ "<|mg|>": 50349,
116
+ "<|mi|>": 50295,
117
+ "<|mk|>": 50308,
118
+ "<|ml|>": 50296,
119
+ "<|mn|>": 50314,
120
+ "<|mr|>": 50320,
121
+ "<|ms|>": 50282,
122
+ "<|mt|>": 50343,
123
+ "<|my|>": 50346,
124
+ "<|ne|>": 50313,
125
+ "<|nl|>": 50271,
126
+ "<|nn|>": 50342,
127
+ "<|no|>": 50288,
128
+ "<|oc|>": 50328,
129
+ "<|pa|>": 50321,
130
+ "<|pl|>": 50269,
131
+ "<|ps|>": 50340,
132
+ "<|pt|>": 50267,
133
+ "<|ro|>": 50284,
134
+ "<|ru|>": 50263,
135
+ "<|sa|>": 50344,
136
+ "<|sd|>": 50332,
137
+ "<|si|>": 50322,
138
+ "<|sk|>": 50298,
139
+ "<|sl|>": 50305,
140
+ "<|sn|>": 50324,
141
+ "<|so|>": 50326,
142
+ "<|sq|>": 50317,
143
+ "<|sr|>": 50303,
144
+ "<|su|>": 50357,
145
+ "<|sv|>": 50273,
146
+ "<|sw|>": 50318,
147
+ "<|ta|>": 50287,
148
+ "<|te|>": 50299,
149
+ "<|tg|>": 50331,
150
+ "<|th|>": 50289,
151
+ "<|tk|>": 50341,
152
+ "<|tl|>": 50348,
153
+ "<|tr|>": 50268,
154
+ "<|tt|>": 50351,
155
+ "<|uk|>": 50280,
156
+ "<|ur|>": 50290,
157
+ "<|uz|>": 50337,
158
+ "<|vi|>": 50278,
159
+ "<|yi|>": 50335,
160
+ "<|yo|>": 50325,
161
+ "<|zh|>": 50260
162
+ },
163
+ "language": "<|zh|>",
164
+ "max_initial_timestamp_index": 50,
165
+ "max_length": 448,
166
+ "no_timestamps_token_id": 50363,
167
+ "pad_token_id": 50257,
168
+ "prev_sot_token_id": 50361,
169
+ "return_timestamps": false,
170
+ "suppress_tokens": [
171
+ 1,
172
+ 2,
173
+ 7,
174
+ 8,
175
+ 9,
176
+ 10,
177
+ 14,
178
+ 25,
179
+ 26,
180
+ 27,
181
+ 28,
182
+ 29,
183
+ 31,
184
+ 58,
185
+ 59,
186
+ 60,
187
+ 61,
188
+ 62,
189
+ 63,
190
+ 90,
191
+ 91,
192
+ 92,
193
+ 93,
194
+ 359,
195
+ 503,
196
+ 522,
197
+ 542,
198
+ 873,
199
+ 893,
200
+ 902,
201
+ 918,
202
+ 922,
203
+ 931,
204
+ 1350,
205
+ 1853,
206
+ 1982,
207
+ 2460,
208
+ 2627,
209
+ 3246,
210
+ 3253,
211
+ 3268,
212
+ 3536,
213
+ 3846,
214
+ 3961,
215
+ 4183,
216
+ 4667,
217
+ 6585,
218
+ 6647,
219
+ 7273,
220
+ 9061,
221
+ 9383,
222
+ 10428,
223
+ 10929,
224
+ 11938,
225
+ 12033,
226
+ 12331,
227
+ 12562,
228
+ 13793,
229
+ 14157,
230
+ 14635,
231
+ 15265,
232
+ 15618,
233
+ 16553,
234
+ 16604,
235
+ 18362,
236
+ 18956,
237
+ 20075,
238
+ 21675,
239
+ 22520,
240
+ 26130,
241
+ 26161,
242
+ 26435,
243
+ 28279,
244
+ 29464,
245
+ 31650,
246
+ 32302,
247
+ 32470,
248
+ 36865,
249
+ 42863,
250
+ 47425,
251
+ 49870,
252
+ 50254,
253
+ 50258,
254
+ 50358,
255
+ 50359,
256
+ 50360,
257
+ 50361,
258
+ 50362
259
+ ],
260
+ "task": "transcribe",
261
+ "task_to_id": {
262
+ "transcribe": 50359,
263
+ "translate": 50358
264
+ },
265
+ "transformers_version": "4.40.0"
266
+ }
last-checkpoint/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d58a30e201bef181ee7edcccc40b40dbf6259019eafbb9948724df310fb0dadf
3
+ size 966995080
last-checkpoint/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:989686289c4be0b33c86225518ba4685fdb5b737c5a29b3a5fd4191ec82b15c5
3
+ size 1925064044
last-checkpoint/preprocessor_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chunk_length": 30,
3
+ "feature_extractor_type": "WhisperFeatureExtractor",
4
+ "feature_size": 80,
5
+ "hop_length": 160,
6
+ "n_fft": 400,
7
+ "n_samples": 480000,
8
+ "nb_max_frames": 3000,
9
+ "padding_side": "right",
10
+ "padding_value": 0.0,
11
+ "processor_class": "WhisperProcessor",
12
+ "return_attention_mask": false,
13
+ "sampling_rate": 16000
14
+ }
last-checkpoint/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51bea2a28f129bf069e5a02ae44edfec13f51109373355626e9228154b0d41f5
3
+ size 14244
last-checkpoint/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d2d378b4bd36bf44babbc26f567786bedc31fd4875330753b97c0f677a367397
3
+ size 1064
last-checkpoint/trainer_state.json ADDED
@@ -0,0 +1,310 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": 43.3160621761658,
3
+ "best_model_checkpoint": "./whisper-small-taiwanese/checkpoint-1000",
4
+ "epoch": 0.6451612903225806,
5
+ "eval_steps": 1000,
6
+ "global_step": 1000,
7
+ "is_hyper_param_search": false,
8
+ "is_local_process_zero": true,
9
+ "is_world_process_zero": true,
10
+ "log_history": [
11
+ {
12
+ "epoch": 0.016129032258064516,
13
+ "grad_norm": 229.72373962402344,
14
+ "learning_rate": 5.376344086021506e-07,
15
+ "loss": 7.9674,
16
+ "step": 25
17
+ },
18
+ {
19
+ "epoch": 0.03225806451612903,
20
+ "grad_norm": 50.686302185058594,
21
+ "learning_rate": 1.0752688172043011e-06,
22
+ "loss": 5.7026,
23
+ "step": 50
24
+ },
25
+ {
26
+ "epoch": 0.04838709677419355,
27
+ "grad_norm": 32.474510192871094,
28
+ "learning_rate": 1.6129032258064516e-06,
29
+ "loss": 3.7065,
30
+ "step": 75
31
+ },
32
+ {
33
+ "epoch": 0.06451612903225806,
34
+ "grad_norm": 30.973085403442383,
35
+ "learning_rate": 2.1505376344086023e-06,
36
+ "loss": 2.6906,
37
+ "step": 100
38
+ },
39
+ {
40
+ "epoch": 0.08064516129032258,
41
+ "grad_norm": 28.370464324951172,
42
+ "learning_rate": 2.688172043010753e-06,
43
+ "loss": 2.3087,
44
+ "step": 125
45
+ },
46
+ {
47
+ "epoch": 0.0967741935483871,
48
+ "grad_norm": 29.259729385375977,
49
+ "learning_rate": 3.225806451612903e-06,
50
+ "loss": 2.0589,
51
+ "step": 150
52
+ },
53
+ {
54
+ "epoch": 0.11290322580645161,
55
+ "grad_norm": 29.08380699157715,
56
+ "learning_rate": 3.763440860215054e-06,
57
+ "loss": 1.8731,
58
+ "step": 175
59
+ },
60
+ {
61
+ "epoch": 0.12903225806451613,
62
+ "grad_norm": 22.745624542236328,
63
+ "learning_rate": 4.3010752688172045e-06,
64
+ "loss": 1.5257,
65
+ "step": 200
66
+ },
67
+ {
68
+ "epoch": 0.14516129032258066,
69
+ "grad_norm": 16.694580078125,
70
+ "learning_rate": 4.838709677419355e-06,
71
+ "loss": 1.4005,
72
+ "step": 225
73
+ },
74
+ {
75
+ "epoch": 0.16129032258064516,
76
+ "grad_norm": 18.02663803100586,
77
+ "learning_rate": 5.376344086021506e-06,
78
+ "loss": 1.3308,
79
+ "step": 250
80
+ },
81
+ {
82
+ "epoch": 0.1774193548387097,
83
+ "grad_norm": 14.609949111938477,
84
+ "learning_rate": 5.9139784946236566e-06,
85
+ "loss": 1.2143,
86
+ "step": 275
87
+ },
88
+ {
89
+ "epoch": 0.1935483870967742,
90
+ "grad_norm": 16.727527618408203,
91
+ "learning_rate": 6.451612903225806e-06,
92
+ "loss": 1.1925,
93
+ "step": 300
94
+ },
95
+ {
96
+ "epoch": 0.20967741935483872,
97
+ "grad_norm": 15.254867553710938,
98
+ "learning_rate": 6.989247311827958e-06,
99
+ "loss": 1.1482,
100
+ "step": 325
101
+ },
102
+ {
103
+ "epoch": 0.22580645161290322,
104
+ "grad_norm": 16.119234085083008,
105
+ "learning_rate": 7.526881720430108e-06,
106
+ "loss": 1.0825,
107
+ "step": 350
108
+ },
109
+ {
110
+ "epoch": 0.24193548387096775,
111
+ "grad_norm": 13.577301025390625,
112
+ "learning_rate": 8.064516129032258e-06,
113
+ "loss": 1.099,
114
+ "step": 375
115
+ },
116
+ {
117
+ "epoch": 0.25806451612903225,
118
+ "grad_norm": 15.483856201171875,
119
+ "learning_rate": 8.602150537634409e-06,
120
+ "loss": 1.0654,
121
+ "step": 400
122
+ },
123
+ {
124
+ "epoch": 0.27419354838709675,
125
+ "grad_norm": 15.842108726501465,
126
+ "learning_rate": 9.13978494623656e-06,
127
+ "loss": 0.9747,
128
+ "step": 425
129
+ },
130
+ {
131
+ "epoch": 0.2903225806451613,
132
+ "grad_norm": 13.010821342468262,
133
+ "learning_rate": 9.67741935483871e-06,
134
+ "loss": 0.9679,
135
+ "step": 450
136
+ },
137
+ {
138
+ "epoch": 0.3064516129032258,
139
+ "grad_norm": 15.315924644470215,
140
+ "learning_rate": 9.97610513739546e-06,
141
+ "loss": 0.9001,
142
+ "step": 475
143
+ },
144
+ {
145
+ "epoch": 0.3225806451612903,
146
+ "grad_norm": 15.252881050109863,
147
+ "learning_rate": 9.916367980884111e-06,
148
+ "loss": 0.9019,
149
+ "step": 500
150
+ },
151
+ {
152
+ "epoch": 0.3387096774193548,
153
+ "grad_norm": 15.013239860534668,
154
+ "learning_rate": 9.856630824372761e-06,
155
+ "loss": 0.9167,
156
+ "step": 525
157
+ },
158
+ {
159
+ "epoch": 0.3548387096774194,
160
+ "grad_norm": 12.44570255279541,
161
+ "learning_rate": 9.79689366786141e-06,
162
+ "loss": 0.8644,
163
+ "step": 550
164
+ },
165
+ {
166
+ "epoch": 0.3709677419354839,
167
+ "grad_norm": 13.266128540039062,
168
+ "learning_rate": 9.737156511350062e-06,
169
+ "loss": 0.8954,
170
+ "step": 575
171
+ },
172
+ {
173
+ "epoch": 0.3870967741935484,
174
+ "grad_norm": 13.153059005737305,
175
+ "learning_rate": 9.67741935483871e-06,
176
+ "loss": 0.8364,
177
+ "step": 600
178
+ },
179
+ {
180
+ "epoch": 0.4032258064516129,
181
+ "grad_norm": 15.848042488098145,
182
+ "learning_rate": 9.61768219832736e-06,
183
+ "loss": 0.8667,
184
+ "step": 625
185
+ },
186
+ {
187
+ "epoch": 0.41935483870967744,
188
+ "grad_norm": 13.445392608642578,
189
+ "learning_rate": 9.557945041816011e-06,
190
+ "loss": 0.8155,
191
+ "step": 650
192
+ },
193
+ {
194
+ "epoch": 0.43548387096774194,
195
+ "grad_norm": 13.883005142211914,
196
+ "learning_rate": 9.49820788530466e-06,
197
+ "loss": 0.8446,
198
+ "step": 675
199
+ },
200
+ {
201
+ "epoch": 0.45161290322580644,
202
+ "grad_norm": 13.22021198272705,
203
+ "learning_rate": 9.43847072879331e-06,
204
+ "loss": 0.8255,
205
+ "step": 700
206
+ },
207
+ {
208
+ "epoch": 0.46774193548387094,
209
+ "grad_norm": 14.165966987609863,
210
+ "learning_rate": 9.37873357228196e-06,
211
+ "loss": 0.8034,
212
+ "step": 725
213
+ },
214
+ {
215
+ "epoch": 0.4838709677419355,
216
+ "grad_norm": 12.320103645324707,
217
+ "learning_rate": 9.31899641577061e-06,
218
+ "loss": 0.7439,
219
+ "step": 750
220
+ },
221
+ {
222
+ "epoch": 0.5,
223
+ "grad_norm": 13.079719543457031,
224
+ "learning_rate": 9.25925925925926e-06,
225
+ "loss": 0.7574,
226
+ "step": 775
227
+ },
228
+ {
229
+ "epoch": 0.5161290322580645,
230
+ "grad_norm": 12.108668327331543,
231
+ "learning_rate": 9.19952210274791e-06,
232
+ "loss": 0.7844,
233
+ "step": 800
234
+ },
235
+ {
236
+ "epoch": 0.532258064516129,
237
+ "grad_norm": 12.974024772644043,
238
+ "learning_rate": 9.13978494623656e-06,
239
+ "loss": 0.783,
240
+ "step": 825
241
+ },
242
+ {
243
+ "epoch": 0.5483870967741935,
244
+ "grad_norm": 14.670340538024902,
245
+ "learning_rate": 9.08004778972521e-06,
246
+ "loss": 0.7084,
247
+ "step": 850
248
+ },
249
+ {
250
+ "epoch": 0.5645161290322581,
251
+ "grad_norm": 15.380485534667969,
252
+ "learning_rate": 9.02031063321386e-06,
253
+ "loss": 0.7624,
254
+ "step": 875
255
+ },
256
+ {
257
+ "epoch": 0.5806451612903226,
258
+ "grad_norm": 14.00020694732666,
259
+ "learning_rate": 8.96057347670251e-06,
260
+ "loss": 0.7031,
261
+ "step": 900
262
+ },
263
+ {
264
+ "epoch": 0.5967741935483871,
265
+ "grad_norm": 11.307880401611328,
266
+ "learning_rate": 8.90083632019116e-06,
267
+ "loss": 0.6797,
268
+ "step": 925
269
+ },
270
+ {
271
+ "epoch": 0.6129032258064516,
272
+ "grad_norm": 14.682994842529297,
273
+ "learning_rate": 8.84109916367981e-06,
274
+ "loss": 0.6679,
275
+ "step": 950
276
+ },
277
+ {
278
+ "epoch": 0.6290322580645161,
279
+ "grad_norm": 14.844277381896973,
280
+ "learning_rate": 8.78136200716846e-06,
281
+ "loss": 0.7079,
282
+ "step": 975
283
+ },
284
+ {
285
+ "epoch": 0.6451612903225806,
286
+ "grad_norm": 14.752099990844727,
287
+ "learning_rate": 8.72162485065711e-06,
288
+ "loss": 0.6757,
289
+ "step": 1000
290
+ },
291
+ {
292
+ "epoch": 0.6451612903225806,
293
+ "eval_cer": 43.3160621761658,
294
+ "eval_loss": 0.6164932250976562,
295
+ "eval_runtime": 945.7136,
296
+ "eval_samples_per_second": 2.412,
297
+ "eval_steps_per_second": 0.302,
298
+ "step": 1000
299
+ }
300
+ ],
301
+ "logging_steps": 25,
302
+ "max_steps": 4650,
303
+ "num_input_tokens_seen": 0,
304
+ "num_train_epochs": 3,
305
+ "save_steps": 1000,
306
+ "total_flos": 4.61736640512e+18,
307
+ "train_batch_size": 16,
308
+ "trial_name": null,
309
+ "trial_params": null
310
+ }
last-checkpoint/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fdbd7ffd023398f8cec6e5726c887d0bce38c6797a0f638b634302be6e3c8ab1
3
+ size 5176