TracNetwork commited on
Commit
a83f71f
·
verified ·
1 Parent(s): 02c7dfa

Mayhem admin artifact nvidia/parakeet-tdt-0.6b-v3 safetensors

Browse files
.gitattributes CHANGED
@@ -33,3 +33,6 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ parakeet-tdt-0.6b-v3.nemo filter=lfs diff=lfs merge=lfs -text
37
+ plots/*.png filter=lfs diff=lfs merge=lfs -text
38
+ plots/asr.png filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,1317 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: cc-by-4.0
3
+ track_downloads: true
4
+ language:
5
+ - en
6
+ - es
7
+ - fr
8
+ - de
9
+ - bg
10
+ - hr
11
+ - cs
12
+ - da
13
+ - nl
14
+ - et
15
+ - fi
16
+ - el
17
+ - hu
18
+ - it
19
+ - lv
20
+ - lt
21
+ - mt
22
+ - pl
23
+ - pt
24
+ - ro
25
+ - sk
26
+ - sl
27
+ - sv
28
+ - ru
29
+ - uk
30
+ pipeline_tag: automatic-speech-recognition
31
+ library_name: transformers
32
+ datasets:
33
+ - nvidia/Granary
34
+ - nemo/asr-set-3.0
35
+ tags:
36
+ - automatic-speech-recognition
37
+ - speech
38
+ - audio
39
+ - Transducer
40
+ - Transformer
41
+ - TDT
42
+ - FastConformer
43
+ - Conformer
44
+ - pytorch
45
+ - NeMo
46
+ - hf-asr-leaderboard
47
+ - Transformers
48
+ widget:
49
+ - example_title: Librispeech sample 1
50
+ src: https://cdn-media.huggingface.co/speech_samples/sample1.flac
51
+ - example_title: Librispeech sample 2
52
+ src: https://cdn-media.huggingface.co/speech_samples/sample2.flac
53
+ metrics:
54
+ - wer
55
+ model-index:
56
+ - name: parakeet-tdt-0.6b-v3
57
+ results:
58
+ - task:
59
+ type: automatic-speech-recognition
60
+ name: Automatic Speech Recognition
61
+ dataset:
62
+ name: AMI (Meetings test)
63
+ type: edinburghcstr/ami
64
+ config: ihm
65
+ split: test
66
+ args:
67
+ language: en
68
+ metrics:
69
+ - type: wer
70
+ value: 11.31
71
+ name: Test WER
72
+ - task:
73
+ type: automatic-speech-recognition
74
+ name: Automatic Speech Recognition
75
+ dataset:
76
+ name: Earnings-22
77
+ type: revdotcom/earnings22
78
+ split: test
79
+ args:
80
+ language: en
81
+ metrics:
82
+ - type: wer
83
+ value: 11.42
84
+ name: Test WER
85
+ - task:
86
+ type: automatic-speech-recognition
87
+ name: Automatic Speech Recognition
88
+ dataset:
89
+ name: GigaSpeech
90
+ type: speechcolab/gigaspeech
91
+ split: test
92
+ args:
93
+ language: en
94
+ metrics:
95
+ - type: wer
96
+ value: 9.59
97
+ name: Test WER
98
+ - task:
99
+ type: automatic-speech-recognition
100
+ name: Automatic Speech Recognition
101
+ dataset:
102
+ name: LibriSpeech (clean)
103
+ type: librispeech_asr
104
+ config: other
105
+ split: test
106
+ args:
107
+ language: en
108
+ metrics:
109
+ - type: wer
110
+ value: 1.93
111
+ name: Test WER
112
+ - type: wer
113
+ value: 3.59
114
+ name: Test WER
115
+ - task:
116
+ type: Automatic Speech Recognition
117
+ name: automatic-speech-recognition
118
+ dataset:
119
+ name: SPGI Speech
120
+ type: kensho/spgispeech
121
+ config: test
122
+ split: test
123
+ args:
124
+ language: en
125
+ metrics:
126
+ - type: wer
127
+ value: 3.97
128
+ name: Test WER
129
+ - task:
130
+ type: Automatic Speech Recognition
131
+ name: automatic-speech-recognition
132
+ dataset:
133
+ name: tedlium-v3
134
+ type: LIUM/tedlium
135
+ config: release1
136
+ split: test
137
+ args:
138
+ language: en
139
+ metrics:
140
+ - type: wer
141
+ value: 2.75
142
+ name: Test WER
143
+ - task:
144
+ type: automatic-speech-recognition
145
+ name: Automatic Speech Recognition
146
+ dataset:
147
+ name: Vox Populi
148
+ type: facebook/voxpopuli
149
+ config: en
150
+ split: test
151
+ args:
152
+ language: en
153
+ metrics:
154
+ - type: wer
155
+ value: 6.14
156
+ name: Test WER
157
+ - task:
158
+ type: Automatic Speech Recognition
159
+ name: automatic-speech-recognition
160
+ dataset:
161
+ name: FLEURS
162
+ type: google/fleurs
163
+ config: bg_bg
164
+ split: test
165
+ args:
166
+ language: bg
167
+ metrics:
168
+ - type: wer
169
+ value: 12.64
170
+ name: Test WER (Bg)
171
+ - task:
172
+ type: Automatic Speech Recognition
173
+ name: automatic-speech-recognition
174
+ dataset:
175
+ name: FLEURS
176
+ type: google/fleurs
177
+ config: cs_cz
178
+ split: test
179
+ args:
180
+ language: cs
181
+ metrics:
182
+ - type: wer
183
+ value: 11.01
184
+ name: Test WER (Cs)
185
+ - task:
186
+ type: Automatic Speech Recognition
187
+ name: automatic-speech-recognition
188
+ dataset:
189
+ name: FLEURS
190
+ type: google/fleurs
191
+ config: da_dk
192
+ split: test
193
+ args:
194
+ language: da
195
+ metrics:
196
+ - type: wer
197
+ value: 18.41
198
+ name: Test WER (Da)
199
+ - task:
200
+ type: Automatic Speech Recognition
201
+ name: automatic-speech-recognition
202
+ dataset:
203
+ name: FLEURS
204
+ type: google/fleurs
205
+ config: de_de
206
+ split: test
207
+ args:
208
+ language: de
209
+ metrics:
210
+ - type: wer
211
+ value: 5.04
212
+ name: Test WER (De)
213
+ - task:
214
+ type: Automatic Speech Recognition
215
+ name: automatic-speech-recognition
216
+ dataset:
217
+ name: FLEURS
218
+ type: google/fleurs
219
+ config: el_gr
220
+ split: test
221
+ args:
222
+ language: el
223
+ metrics:
224
+ - type: wer
225
+ value: 20.7
226
+ name: Test WER (El)
227
+ - task:
228
+ type: Automatic Speech Recognition
229
+ name: automatic-speech-recognition
230
+ dataset:
231
+ name: FLEURS
232
+ type: google/fleurs
233
+ config: en_us
234
+ split: test
235
+ args:
236
+ language: en
237
+ metrics:
238
+ - type: wer
239
+ value: 4.85
240
+ name: Test WER (En)
241
+ - task:
242
+ type: Automatic Speech Recognition
243
+ name: automatic-speech-recognition
244
+ dataset:
245
+ name: FLEURS
246
+ type: google/fleurs
247
+ config: es_419
248
+ split: test
249
+ args:
250
+ language: es
251
+ metrics:
252
+ - type: wer
253
+ value: 3.45
254
+ name: Test WER (Es)
255
+ - task:
256
+ type: Automatic Speech Recognition
257
+ name: automatic-speech-recognition
258
+ dataset:
259
+ name: FLEURS
260
+ type: google/fleurs
261
+ config: et_ee
262
+ split: test
263
+ args:
264
+ language: et
265
+ metrics:
266
+ - type: wer
267
+ value: 17.73
268
+ name: Test WER (Et)
269
+ - task:
270
+ type: Automatic Speech Recognition
271
+ name: automatic-speech-recognition
272
+ dataset:
273
+ name: FLEURS
274
+ type: google/fleurs
275
+ config: fi_fi
276
+ split: test
277
+ args:
278
+ language: fi
279
+ metrics:
280
+ - type: wer
281
+ value: 13.21
282
+ name: Test WER (Fi)
283
+ - task:
284
+ type: Automatic Speech Recognition
285
+ name: automatic-speech-recognition
286
+ dataset:
287
+ name: FLEURS
288
+ type: google/fleurs
289
+ config: fr_fr
290
+ split: test
291
+ args:
292
+ language: fr
293
+ metrics:
294
+ - type: wer
295
+ value: 5.15
296
+ name: Test WER (Fr)
297
+ - task:
298
+ type: Automatic Speech Recognition
299
+ name: automatic-speech-recognition
300
+ dataset:
301
+ name: FLEURS
302
+ type: google/fleurs
303
+ config: hr_hr
304
+ split: test
305
+ args:
306
+ language: hr
307
+ metrics:
308
+ - type: wer
309
+ value: 12.46
310
+ name: Test WER (Hr)
311
+ - task:
312
+ type: Automatic Speech Recognition
313
+ name: automatic-speech-recognition
314
+ dataset:
315
+ name: FLEURS
316
+ type: google/fleurs
317
+ config: hu_hu
318
+ split: test
319
+ args:
320
+ language: hu
321
+ metrics:
322
+ - type: wer
323
+ value: 15.72
324
+ name: Test WER (Hu)
325
+ - task:
326
+ type: Automatic Speech Recognition
327
+ name: automatic-speech-recognition
328
+ dataset:
329
+ name: FLEURS
330
+ type: google/fleurs
331
+ config: it_it
332
+ split: test
333
+ args:
334
+ language: it
335
+ metrics:
336
+ - type: wer
337
+ value: 3
338
+ name: Test WER (It)
339
+ - task:
340
+ type: Automatic Speech Recognition
341
+ name: automatic-speech-recognition
342
+ dataset:
343
+ name: FLEURS
344
+ type: google/fleurs
345
+ config: lt_lt
346
+ split: test
347
+ args:
348
+ language: lt
349
+ metrics:
350
+ - type: wer
351
+ value: 20.35
352
+ name: Test WER (Lt)
353
+ - task:
354
+ type: Automatic Speech Recognition
355
+ name: automatic-speech-recognition
356
+ dataset:
357
+ name: FLEURS
358
+ type: google/fleurs
359
+ config: lv_lv
360
+ split: test
361
+ args:
362
+ language: lv
363
+ metrics:
364
+ - type: wer
365
+ value: 22.84
366
+ name: Test WER (Lv)
367
+ - task:
368
+ type: Automatic Speech Recognition
369
+ name: automatic-speech-recognition
370
+ dataset:
371
+ name: FLEURS
372
+ type: google/fleurs
373
+ config: mt_mt
374
+ split: test
375
+ args:
376
+ language: mt
377
+ metrics:
378
+ - type: wer
379
+ value: 20.46
380
+ name: Test WER (Mt)
381
+ - task:
382
+ type: Automatic Speech Recognition
383
+ name: automatic-speech-recognition
384
+ dataset:
385
+ name: FLEURS
386
+ type: google/fleurs
387
+ config: nl_nl
388
+ split: test
389
+ args:
390
+ language: nl
391
+ metrics:
392
+ - type: wer
393
+ value: 7.48
394
+ name: Test WER (Nl)
395
+ - task:
396
+ type: Automatic Speech Recognition
397
+ name: automatic-speech-recognition
398
+ dataset:
399
+ name: FLEURS
400
+ type: google/fleurs
401
+ config: pl_pl
402
+ split: test
403
+ args:
404
+ language: pl
405
+ metrics:
406
+ - type: wer
407
+ value: 7.31
408
+ name: Test WER (Pl)
409
+ - task:
410
+ type: Automatic Speech Recognition
411
+ name: automatic-speech-recognition
412
+ dataset:
413
+ name: FLEURS
414
+ type: google/fleurs
415
+ config: pt_br
416
+ split: test
417
+ args:
418
+ language: pt
419
+ metrics:
420
+ - type: wer
421
+ value: 4.76
422
+ name: Test WER (Pt)
423
+ - task:
424
+ type: Automatic Speech Recognition
425
+ name: automatic-speech-recognition
426
+ dataset:
427
+ name: FLEURS
428
+ type: google/fleurs
429
+ config: ro_ro
430
+ split: test
431
+ args:
432
+ language: ro
433
+ metrics:
434
+ - type: wer
435
+ value: 12.44
436
+ name: Test WER (Ro)
437
+ - task:
438
+ type: Automatic Speech Recognition
439
+ name: automatic-speech-recognition
440
+ dataset:
441
+ name: FLEURS
442
+ type: google/fleurs
443
+ config: ru_ru
444
+ split: test
445
+ args:
446
+ language: ru
447
+ metrics:
448
+ - type: wer
449
+ value: 5.51
450
+ name: Test WER (Ru)
451
+ - task:
452
+ type: Automatic Speech Recognition
453
+ name: automatic-speech-recognition
454
+ dataset:
455
+ name: FLEURS
456
+ type: google/fleurs
457
+ config: sk_sk
458
+ split: test
459
+ args:
460
+ language: sk
461
+ metrics:
462
+ - type: wer
463
+ value: 8.82
464
+ name: Test WER (Sk)
465
+ - task:
466
+ type: Automatic Speech Recognition
467
+ name: automatic-speech-recognition
468
+ dataset:
469
+ name: FLEURS
470
+ type: google/fleurs
471
+ config: sl_si
472
+ split: test
473
+ args:
474
+ language: sl
475
+ metrics:
476
+ - type: wer
477
+ value: 24.03
478
+ name: Test WER (Sl)
479
+ - task:
480
+ type: Automatic Speech Recognition
481
+ name: automatic-speech-recognition
482
+ dataset:
483
+ name: FLEURS
484
+ type: google/fleurs
485
+ config: sv_se
486
+ split: test
487
+ args:
488
+ language: sv
489
+ metrics:
490
+ - type: wer
491
+ value: 15.08
492
+ name: Test WER (Sv)
493
+ - task:
494
+ type: Automatic Speech Recognition
495
+ name: automatic-speech-recognition
496
+ dataset:
497
+ name: FLEURS
498
+ type: google/fleurs
499
+ config: uk_ua
500
+ split: test
501
+ args:
502
+ language: uk
503
+ metrics:
504
+ - type: wer
505
+ value: 6.79
506
+ name: Test WER (Uk)
507
+ - task:
508
+ type: Automatic Speech Recognition
509
+ name: automatic-speech-recognition
510
+ dataset:
511
+ name: Multilingual LibriSpeech
512
+ type: facebook/multilingual_librispeech
513
+ config: spanish
514
+ split: test
515
+ args:
516
+ language: es
517
+ metrics:
518
+ - type: wer
519
+ value: 4.39
520
+ name: Test WER (Es)
521
+ - task:
522
+ type: Automatic Speech Recognition
523
+ name: automatic-speech-recognition
524
+ dataset:
525
+ name: Multilingual LibriSpeech
526
+ type: facebook/multilingual_librispeech
527
+ config: french
528
+ split: test
529
+ args:
530
+ language: fr
531
+ metrics:
532
+ - type: wer
533
+ value: 4.97
534
+ name: Test WER (Fr)
535
+ - task:
536
+ type: Automatic Speech Recognition
537
+ name: automatic-speech-recognition
538
+ dataset:
539
+ name: Multilingual LibriSpeech
540
+ type: facebook/multilingual_librispeech
541
+ config: italian
542
+ split: test
543
+ args:
544
+ language: it
545
+ metrics:
546
+ - type: wer
547
+ value: 10.08
548
+ name: Test WER (It)
549
+ - task:
550
+ type: Automatic Speech Recognition
551
+ name: automatic-speech-recognition
552
+ dataset:
553
+ name: Multilingual LibriSpeech
554
+ type: facebook/multilingual_librispeech
555
+ config: dutch
556
+ split: test
557
+ args:
558
+ language: nl
559
+ metrics:
560
+ - type: wer
561
+ value: 12.78
562
+ name: Test WER (Nl)
563
+ - task:
564
+ type: Automatic Speech Recognition
565
+ name: automatic-speech-recognition
566
+ dataset:
567
+ name: Multilingual LibriSpeech
568
+ type: facebook/multilingual_librispeech
569
+ config: polish
570
+ split: test
571
+ args:
572
+ language: pl
573
+ metrics:
574
+ - type: wer
575
+ value: 7.28
576
+ name: Test WER (Pl)
577
+ - task:
578
+ type: Automatic Speech Recognition
579
+ name: automatic-speech-recognition
580
+ dataset:
581
+ name: Multilingual LibriSpeech
582
+ type: facebook/multilingual_librispeech
583
+ config: portuguese
584
+ split: test
585
+ args:
586
+ language: pt
587
+ metrics:
588
+ - type: wer
589
+ value: 7.5
590
+ name: Test WER (Pt)
591
+ - task:
592
+ type: Automatic Speech Recognition
593
+ name: automatic-speech-recognition
594
+ dataset:
595
+ name: CoVoST2
596
+ type: covost2
597
+ config: de
598
+ split: test
599
+ args:
600
+ language: de
601
+ metrics:
602
+ - type: wer
603
+ value: 4.84
604
+ name: Test WER (De)
605
+ - task:
606
+ type: Automatic Speech Recognition
607
+ name: automatic-speech-recognition
608
+ dataset:
609
+ name: CoVoST2
610
+ type: covost2
611
+ config: en
612
+ split: test
613
+ args:
614
+ language: en
615
+ metrics:
616
+ - type: wer
617
+ value: 6.8
618
+ name: Test WER (En)
619
+ - task:
620
+ type: Automatic Speech Recognition
621
+ name: automatic-speech-recognition
622
+ dataset:
623
+ name: CoVoST2
624
+ type: covost2
625
+ config: es
626
+ split: test
627
+ args:
628
+ language: es
629
+ metrics:
630
+ - type: wer
631
+ value: 3.41
632
+ name: Test WER (Es)
633
+ - task:
634
+ type: Automatic Speech Recognition
635
+ name: automatic-speech-recognition
636
+ dataset:
637
+ name: CoVoST2
638
+ type: covost2
639
+ config: et
640
+ split: test
641
+ args:
642
+ language: et
643
+ metrics:
644
+ - type: wer
645
+ value: 22.04
646
+ name: Test WER (Et)
647
+ - task:
648
+ type: Automatic Speech Recognition
649
+ name: automatic-speech-recognition
650
+ dataset:
651
+ name: CoVoST2
652
+ type: covost2
653
+ config: fr
654
+ split: test
655
+ args:
656
+ language: fr
657
+ metrics:
658
+ - type: wer
659
+ value: 6.05
660
+ name: Test WER (Fr)
661
+ - task:
662
+ type: Automatic Speech Recognition
663
+ name: automatic-speech-recognition
664
+ dataset:
665
+ name: CoVoST2
666
+ type: covost2
667
+ config: it
668
+ split: test
669
+ args:
670
+ language: it
671
+ metrics:
672
+ - type: wer
673
+ value: 3.69
674
+ name: Test WER (It)
675
+ - task:
676
+ type: Automatic Speech Recognition
677
+ name: automatic-speech-recognition
678
+ dataset:
679
+ name: CoVoST2
680
+ type: covost2
681
+ config: lv
682
+ split: test
683
+ args:
684
+ language: lv
685
+ metrics:
686
+ - type: wer
687
+ value: 38.36
688
+ name: Test WER (Lv)
689
+ - task:
690
+ type: Automatic Speech Recognition
691
+ name: automatic-speech-recognition
692
+ dataset:
693
+ name: CoVoST2
694
+ type: covost2
695
+ config: nl
696
+ split: test
697
+ args:
698
+ language: nl
699
+ metrics:
700
+ - type: wer
701
+ value: 6.5
702
+ name: Test WER (Nl)
703
+ - task:
704
+ type: Automatic Speech Recognition
705
+ name: automatic-speech-recognition
706
+ dataset:
707
+ name: CoVoST2
708
+ type: covost2
709
+ config: pt
710
+ split: test
711
+ args:
712
+ language: pt
713
+ metrics:
714
+ - type: wer
715
+ value: 3.96
716
+ name: Test WER (Pt)
717
+ - task:
718
+ type: Automatic Speech Recognition
719
+ name: automatic-speech-recognition
720
+ dataset:
721
+ name: CoVoST2
722
+ type: covost2
723
+ config: ru
724
+ split: test
725
+ args:
726
+ language: ru
727
+ metrics:
728
+ - type: wer
729
+ value: 3
730
+ name: Test WER (Ru)
731
+ - task:
732
+ type: Automatic Speech Recognition
733
+ name: automatic-speech-recognition
734
+ dataset:
735
+ name: CoVoST2
736
+ type: covost2
737
+ config: sl
738
+ split: test
739
+ args:
740
+ language: sl
741
+ metrics:
742
+ - type: wer
743
+ value: 31.8
744
+ name: Test WER (Sl)
745
+ - task:
746
+ type: Automatic Speech Recognition
747
+ name: automatic-speech-recognition
748
+ dataset:
749
+ name: CoVoST2
750
+ type: covost2
751
+ config: sv
752
+ split: test
753
+ args:
754
+ language: sv
755
+ metrics:
756
+ - type: wer
757
+ value: 20.16
758
+ name: Test WER (Sv)
759
+ - task:
760
+ type: Automatic Speech Recognition
761
+ name: automatic-speech-recognition
762
+ dataset:
763
+ name: CoVoST2
764
+ type: covost2
765
+ config: uk
766
+ split: test
767
+ args:
768
+ language: uk
769
+ metrics:
770
+ - type: wer
771
+ value: 5.1
772
+ name: Test WER (Uk)
773
+ ---
774
+
775
+ # **<span style="color:#76b900;">🦜 parakeet-tdt-0.6b-v3: Multilingual Speech-to-Text Model</span>**
776
+
777
+ <style>
778
+ img {
779
+ display: inline;
780
+ }
781
+ </style>
782
+
783
+ [![Model architecture](https://img.shields.io/badge/Model_Arch-FastConformer--TDT-blue#model-badge)](#model-architecture)
784
+ | [![Model size](https://img.shields.io/badge/Params-0.6B-green#model-badge)](#model-architecture)
785
+ | [![Language](https://img.shields.io/badge/Language-EU_Languages-blue#model-badge)](#datasets)
786
+
787
+ ## <span style="color:#466f00;">Description:</span>
788
+
789
+ `parakeet-tdt-0.6b-v3` is a 600-million-parameter multilingual automatic speech recognition (ASR) model designed for high-throughput speech-to-text transcription. It extends the [parakeet-tdt-0.6b-v2](https://huggingface.co/nvidia/parakeet-tdt-0.6b-v2) model by expanding language support from English to 25 European languages. The model automatically detects the language of the audio and transcribes it without requiring additional prompting. It is part of a series of models that leverage the [Granary](https://huggingface.co/datasets/nvidia/Granary) [1, 2] multilingual corpus as their primary training dataset.
790
+
791
+ 🗣️ Try Demo here: https://huggingface.co/spaces/nvidia/parakeet-tdt-0.6b-v3
792
+
793
+ **Supported Languages:**
794
+ Bulgarian (**bg**), Croatian (**hr**), Czech (**cs**), Danish (**da**), Dutch (**nl**), English (**en**), Estonian (**et**), Finnish (**fi**), French (**fr**), German (**de**), Greek (**el**), Hungarian (**hu**), Italian (**it**), Latvian (**lv**), Lithuanian (**lt**), Maltese (**mt**), Polish (**pl**), Portuguese (**pt**), Romanian (**ro**), Slovak (**sk**), Slovenian (**sl**), Spanish (**es**), Swedish (**sv**), Russian (**ru**), Ukrainian (**uk**)
795
+
796
+ This model is ready for commercial/non-commercial use.
797
+
798
+ ## <span style="color:#466f00;">Key Features:</span>
799
+
800
+ `parakeet-tdt-0.6b-v3`'s key features are built on the foundation of its predecessor, [parakeet-tdt-0.6b-v2](https://huggingface.co/nvidia/parakeet-tdt-0.6b-v2), and include:
801
+
802
+ * Automatic **punctuation** and **capitalization**
803
+ * Accurate **word-level** and **segment-level** timestamps
804
+ * **Long audio** transcription, supporting audio **up to 24 minutes** long with full attention (on A100 80GB) or up to 3 hours with local attention.
805
+ * Released under a **permissive CC BY 4.0 license**
806
+
807
+ For full details on the model architecture, training methodology, datasets, and evaluation results, check out the **[Technical Report](https://arxiv.org/abs/2509.14128)**.
808
+
809
+
810
+ ## <span style="color:#466f00;">License/Terms of Use:</span>
811
+
812
+ GOVERNING TERMS: Use of this model is governed by the [CC-BY-4.0](https://creativecommons.org/licenses/by/4.0/legalcode.en) license.
813
+
814
+ ### <span style="color:#466f00;">Discover more from NVIDIA:</span>
815
+ For documentation, deployment guides, enterprise-ready APIs, and the latest open models—including Nemotron and other cutting-edge speech, translation, and generative AI—visit the NVIDIA Developer Portal at developer.nvidia.com.
816
+ Join the community to access tools, support, and resources to accelerate your development with NVIDIA’s NeMo, Riva, NIM, and foundation models.<br>
817
+
818
+ #### <span style="color:#466f00;">Explore more from NVIDIA:</span> <br>
819
+ What is [Nemotron](https://www.nvidia.com/en-us/ai-data-science/foundation-models/nemotron/)?<br>
820
+ NVIDIA Developer [Nemotron](https://developer.nvidia.com/nemotron)<br>
821
+ [NVIDIA Riva Speech](https://developer.nvidia.com/riva?sortBy=developer_learning_library%2Fsort%2Ffeatured_in.riva%3Adesc%2Ctitle%3Aasc#demos)<br>
822
+ [NeMo Documentation](https://docs.nvidia.com/nemo-framework/user-guide/latest/nemotoolkit/asr/models.html)<br>
823
+
824
+ ## Automatic Speech Recognition (ASR) Performance
825
+
826
+ ![ASR WER Comparison](plots/asr.png)
827
+
828
+ *Figure 1: ASR WER comparison across different models. This does not include Punctuation and Capitalisation errors.*
829
+
830
+ ---
831
+
832
+ ### Evaluation Notes
833
+
834
+ **Note 1:** The above evaluations are conducted for 24 supported languages, excluding Latvian since `seamless-m4t-v2-large` and `seamless-m4t-medium` do not support it.
835
+
836
+ **Note 2:** Performance differences may be partly attributed to Portuguese variant differences - our training data uses European Portuguese while most benchmarks use Brazilian Portuguese.
837
+
838
+ ### <span style="color:#466f00;">Deployment Geography:</span>
839
+ Global
840
+
841
+
842
+ ### <span style="color:#466f00;">Use Case:</span>
843
+
844
+ This model serves developers, researchers, academics, and industries building applications that require speech-to-text capabilities, including but not limited to: conversational AI, voice assistants, transcription services, subtitle generation, and voice analytics platforms.
845
+
846
+
847
+ ### <span style="color:#466f00;">Release Date:</span>
848
+
849
+ Huggingface [08/14/2025](https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3)
850
+
851
+
852
+ ### <span style="color:#466f00;">Model Architecture:</span>
853
+
854
+ **Architecture Type**:
855
+
856
+ FastConformer-TDT
857
+
858
+ **Network Architecture**:
859
+
860
+ * This model was developed based on [FastConformer encoder](https://docs.nvidia.com/deeplearning/nemo/user-guide/docs/en/main/asr/models.html#fast-conformer) architecture[3] and TDT decoder[4]
861
+ * This model has 600 million model parameters.
862
+
863
+ ### <span style="color:#466f00;">Input:</span>
864
+ **Input Type(s):** 16kHz Audio
865
+ **Input Format(s):** `.wav` and `.flac` audio formats
866
+ **Input Parameters:** 1D (audio signal)
867
+ **Other Properties Related to Input:** Monochannel audio
868
+
869
+ ### <span style="color:#466f00;">Output:</span>
870
+ **Output Type(s):** Text
871
+ **Output Format:** String
872
+ **Output Parameters:** 1D (text)
873
+ **Other Properties Related to Output:** Punctuations and Capitalizations included.
874
+
875
+ Our AI models are designed and/or optimized to run on NVIDIA GPU-accelerated systems. By leveraging NVIDIA's hardware (e.g. GPU cores) and software frameworks (e.g., CUDA libraries), the model achieves faster training and inference times compared to CPU-only solutions.
876
+
877
+ For more information, refer to the [NeMo documentation](https://docs.nvidia.com/deeplearning/nemo/user-guide/docs/en/main/asr/models.html#fast-conformer).
878
+
879
+ ## <span style="color:#466f00;">How to Use this Model:</span>
880
+
881
+ To train, fine-tune or play with the model you will need to install [NVIDIA NeMo](https://github.com/NVIDIA/NeMo). We recommend you install it after you've installed latest PyTorch version.
882
+ ```bash
883
+ pip install -U nemo_toolkit['asr']
884
+ ```
885
+ The model is available for use in the NeMo toolkit [5], and can be used as a pre-trained checkpoint for inference or for fine-tuning on another dataset.
886
+
887
+ You can also run Parakeet TDT with [Transformers](https://github.com/huggingface/transformers) 🤗 (more below).
888
+
889
+ ### 1) NeMo usage
890
+
891
+ #### Automatically instantiate the model
892
+
893
+ ```python
894
+ import nemo.collections.asr as nemo_asr
895
+ asr_model = nemo_asr.models.ASRModel.from_pretrained(model_name="nvidia/parakeet-tdt-0.6b-v3")
896
+ ```
897
+
898
+ #### Transcribing using Python
899
+ First, let's get a sample
900
+ ```bash
901
+ wget https://dldata-public.s3.us-east-2.amazonaws.com/2086-149220-0033.wav
902
+ ```
903
+ Then simply do:
904
+ ```python
905
+ output = asr_model.transcribe(['2086-149220-0033.wav'])
906
+ print(output[0].text)
907
+ ```
908
+
909
+ #### Transcribing with timestamps
910
+
911
+ To transcribe with timestamps:
912
+ ```python
913
+ output = asr_model.transcribe(['2086-149220-0033.wav'], timestamps=True)
914
+ # by default, timestamps are enabled for char, word and segment level
915
+ word_timestamps = output[0].timestamp['word'] # word level timestamps for first sample
916
+ segment_timestamps = output[0].timestamp['segment'] # segment level timestamps
917
+ char_timestamps = output[0].timestamp['char'] # char level timestamps
918
+
919
+ for stamp in segment_timestamps:
920
+ print(f"{stamp['start']}s - {stamp['end']}s : {stamp['segment']}")
921
+ ```
922
+
923
+ #### Transcribing long-form audio
924
+
925
+ ```python
926
+ #updating self-attention model of fast-conformer encoder
927
+ #setting attention left and right context sizes to 256
928
+ asr_model.change_attention_model(self_attention_model="rel_pos_local_attn", att_context_size=[256, 256])
929
+
930
+ output = asr_model.transcribe(['2086-149220-0033.wav'])
931
+
932
+ print(output[0].text)
933
+ ```
934
+
935
+ #### Streaming with Parakeet models
936
+
937
+ To use parakeet models in streaming mode use this [script](https://github.com/NVIDIA/NeMo/blob/main/examples/asr/asr_chunked_inference/rnnt/speech_to_text_streaming_infer_rnnt.py) as shown below:
938
+
939
+ ```bash
940
+ python NeMo/main/examples/asr/asr_chunked_inference/rnnt/speech_to_text_streaming_infer_rnnt.py \
941
+ pretrained_name="nvidia/parakeet-tdt-0.6b-v3" \
942
+ model_path=null \
943
+ audio_dir="<optional path to folder of audio files>" \
944
+ dataset_manifest="<optional path to manifest>" \
945
+ output_filename="<optional output filename>" \
946
+ right_context_secs=2.0 \
947
+ chunk_secs=2 \
948
+ left_context_secs=10.0 \
949
+ batch_size=32 \
950
+ clean_groundtruth_text=False
951
+ ```
952
+
953
+ NVIDIA NIM for v2 parakeet model is available at [https://build.nvidia.com/nvidia/parakeet-tdt-0_6b-v2](https://build.nvidia.com/nvidia/parakeet-tdt-0_6b-v2).
954
+
955
+
956
+ ### 2) [Transformers](https://github.com/huggingface/transformers) 🤗 usage
957
+
958
+
959
+ Until Parakeet TDT is part of an official Transformers release, you can use it by installing from source.
960
+
961
+ ```bash
962
+ pip install git+https://github.com/huggingface/transformers
963
+ ```
964
+
965
+ <details>
966
+ <summary>➡️ Pipeline usage</summary>
967
+
968
+ ```python
969
+ from transformers import pipeline
970
+
971
+ pipe = pipeline("automatic-speech-recognition", model="nvidia/parakeet-tdt-0.6b-v3")
972
+ out = pipe("https://huggingface.co/datasets/hf-internal-testing/dummy-audio-samples/resolve/main/bcn_weather.mp3")
973
+ print(out)
974
+ ```
975
+ </details>
976
+
977
+ <details>
978
+ <summary>➡️ AutoModel</summary>
979
+
980
+ ```python
981
+ from transformers import AutoModelForTDT, AutoProcessor
982
+ from datasets import load_dataset, Audio
983
+ import torch
984
+
985
+ device = "cuda" if torch.cuda.is_available() else "cpu"
986
+ num_samples = 3
987
+
988
+ model_id = "nvidia/parakeet-tdt-0.6b-v3"
989
+ processor = AutoProcessor.from_pretrained(model_id)
990
+ model = AutoModelForTDT.from_pretrained(model_id, dtype="auto", device_map=device)
991
+
992
+ ds = load_dataset("hf-internal-testing/librispeech_asr_dummy", "clean", split="validation")
993
+ ds = ds.cast_column("audio", Audio(sampling_rate=processor.feature_extractor.sampling_rate))
994
+ speech_samples = [el["array"] for el in ds["audio"][:num_samples]]
995
+
996
+ inputs = processor(speech_samples, sampling_rate=processor.feature_extractor.sampling_rate)
997
+ inputs.to(model.device, dtype=model.dtype)
998
+ output = model.generate(**inputs, return_dict_in_generate=True)
999
+ print(processor.decode(output.sequences, skip_special_tokens=True))
1000
+ ```
1001
+ </details>
1002
+
1003
+
1004
+ <details>
1005
+ <summary>➡️ Timestamping</summary>
1006
+
1007
+ ```python
1008
+ from datasets import Audio, load_dataset
1009
+ from transformers import AutoModelForTDT, AutoProcessor
1010
+ import torch
1011
+
1012
+ device = "cuda" if torch.cuda.is_available() else "cpu"
1013
+ num_samples = 3
1014
+
1015
+ model_id = "nvidia/parakeet-tdt-0.6b-v3"
1016
+ processor = AutoProcessor.from_pretrained(model_id)
1017
+ model = AutoModelForTDT.from_pretrained(model_id, dtype="auto", device_map=device)
1018
+
1019
+ ds = load_dataset("hf-internal-testing/librispeech_asr_dummy", "clean", split="validation")
1020
+ ds = ds.cast_column("audio", Audio(sampling_rate=processor.feature_extractor.sampling_rate))
1021
+ speech_samples = [el["array"] for el in ds["audio"][:num_samples]]
1022
+
1023
+ inputs = processor(speech_samples, sampling_rate=processor.feature_extractor.sampling_rate)
1024
+ inputs.to(model.device, dtype=model.dtype)
1025
+ output = model.generate(**inputs, return_dict_in_generate=True)
1026
+ decoded_output, decoded_timestamps = processor.decode(
1027
+ output.sequences,
1028
+ durations=output.durations,
1029
+ skip_special_tokens=True,
1030
+ )
1031
+ print("Transcription:", decoded_output)
1032
+ print("Timestamped tokens:", decoded_timestamps)
1033
+ ```
1034
+ </details>
1035
+
1036
+ <details>
1037
+ <summary>➡️ Training</summary>
1038
+
1039
+ ```python
1040
+ from transformers import AutoModelForTDT, AutoProcessor
1041
+ from datasets import load_dataset, Audio
1042
+ import torch
1043
+
1044
+ device = "cuda" if torch.cuda.is_available() else "cpu"
1045
+
1046
+ model_id = "nvidia/parakeet-tdt-0.6b-v3"
1047
+ NUM_SAMPLES = 4
1048
+
1049
+ processor = AutoProcessor.from_pretrained(model_id)
1050
+ model = AutoModelForTDT.from_pretrained(model_id, dtype=torch.bfloat16, device_map=device)
1051
+ model.train()
1052
+
1053
+ ds = load_dataset("hf-internal-testing/librispeech_asr_dummy", "clean", split="validation")
1054
+ ds = ds.cast_column("audio", Audio(sampling_rate=processor.feature_extractor.sampling_rate))
1055
+ speech_samples = [el["array"] for el in ds["audio"][:NUM_SAMPLES]]
1056
+ text_samples = ds["text"][:NUM_SAMPLES]
1057
+
1058
+ # passing `text` to the processor will prepare inputs' `labels` key
1059
+ inputs = processor(audio=speech_samples, text=text_samples, sampling_rate=processor.feature_extractor.sampling_rate)
1060
+ inputs.to(device=model.device, dtype=model.dtype)
1061
+
1062
+ outputs = model(**inputs)
1063
+ print("Loss:", outputs.loss.item())
1064
+ outputs.loss.backward()
1065
+ ```
1066
+ </details>
1067
+
1068
+ For more details about usage, please refer to the [Transformers' documentation](https://huggingface.co/docs/transformers/en/model_doc/parakeet).
1069
+
1070
+
1071
+ ## <span style="color:#466f00;">Software Integration:</span>
1072
+
1073
+ **Runtime Engine(s):**
1074
+ * NeMo 2.4
1075
+
1076
+
1077
+ **Supported Hardware Microarchitecture Compatibility:**
1078
+ * NVIDIA Ampere
1079
+ * NVIDIA Blackwell
1080
+ * NVIDIA Hopper
1081
+ * NVIDIA Volta
1082
+
1083
+ **[Preferred/Supported] Operating System(s):**
1084
+
1085
+ - Linux
1086
+
1087
+ **Hardware Specific Requirements:**
1088
+
1089
+ At least 2GB RAM for model to load. The bigger the RAM, the larger audio input it supports.
1090
+
1091
+ #### Model Version
1092
+
1093
+ Current version: `parakeet-tdt-0.6b-v3`. Previous versions can be [accessed](https://huggingface.co/collections/nvidia/parakeet-659711f49d1469e51546e021) here.
1094
+
1095
+ ## <span style="color:#466f00;">Training and Evaluation Datasets:</span>
1096
+
1097
+ ### <span style="color:#466f00;">Training</span>
1098
+
1099
+ This model was trained using the NeMo toolkit [5], following the strategies below:
1100
+
1101
+ - Initialized from a CTC multilingual checkpoint pretrained on the Granary dataset \[1] \[2].
1102
+ - Trained for 150,000 steps on 128 A100 GPUs.
1103
+ - Dataset corpora and languages were balanced using a temperature sampling value of 0.5.
1104
+ - Stage 2 fine-tuning was performed for 5,000 steps on 4 A100 GPUs using approximately 7,500 hours of high-quality, human-transcribed data of NeMo ASR Set 3.0.
1105
+
1106
+ Training was conducted using this [example script](https://github.com/NVIDIA/NeMo/blob/main/examples/asr/asr_transducer/speech_to_text_rnnt_bpe.py) and [TDT configuration](https://github.com/NVIDIA/NeMo/blob/main/examples/asr/conf/fastconformer/hybrid_transducer_ctc/fastconformer_hybrid_tdt_ctc_bpe.yaml).
1107
+
1108
+ During the training, a unified SentencePiece Tokenizer \[6] with a vocabulary of **8,192 tokens** was used. The unified tokenizer was constructed from the training set transcripts using this [script](https://github.com/NVIDIA/NeMo/blob/main/scripts/tokenizers/process_asr_text_tokenizer.py) and was optimized across all 25 supported languages.
1109
+
1110
+ ### <span style="color:#466f00;">Training Dataset</span>
1111
+ The model was trained on the combination of [Granary dataset's ASR subset](https://huggingface.co/datasets/nvidia/Granary) and in-house dataset NeMo ASR Set 3.0:
1112
+
1113
+ - 10,000 hours from human-transcribed NeMo ASR Set 3.0, including:
1114
+ - LibriSpeech (960 hours)
1115
+ - Fisher Corpus
1116
+ - National Speech Corpus Part 1
1117
+ - VCTK
1118
+ - Europarl-ASR
1119
+ - Multilingual LibriSpeech
1120
+ - Mozilla Common Voice (v7.0)
1121
+ - AMI
1122
+
1123
+ - 660,000 hours of pseudo-labeled data from Granary \[1] \[2], including:
1124
+ - [YTC](https://huggingface.co/datasets/FBK-MT/mosel) \[7]
1125
+ - [MOSEL](https://huggingface.co/datasets/FBK-MT/mosel) \[8]
1126
+ - [YODAS](https://huggingface.co/datasets/espnet/yodas-granary) \[9]
1127
+
1128
+ All transcriptions preserve punctuation and capitalization. The Granary dataset will be made publicly available after presentation at Interspeech 2025.
1129
+
1130
+ **Data Collection Method by dataset**
1131
+
1132
+ * Hybrid: Automated, Human
1133
+
1134
+ **Labeling Method by dataset**
1135
+
1136
+ * Hybrid: Synthetic, Human
1137
+
1138
+ **Properties:**
1139
+
1140
+ * Noise robust data from various sources
1141
+ * Single channel, 16kHz sampled data
1142
+
1143
+ #### Evaluation Datasets
1144
+
1145
+ For multilingual ASR performance evaluation:
1146
+ - Fleurs [10]
1147
+ - MLS [11]
1148
+ - CoVoST [12]
1149
+
1150
+ For English ASR performance evaluation:
1151
+ - Hugging Face Open ASR Leaderboard [13] datasets
1152
+
1153
+ **Data Collection Method by dataset**
1154
+ * Human
1155
+
1156
+ **Labeling Method by dataset**
1157
+ * Human
1158
+
1159
+ **Properties:**
1160
+
1161
+ * All are commonly used for benchmarking English ASR systems.
1162
+ * Audio data is typically processed into a 16kHz mono channel format for ASR evaluation, consistent with benchmarks like the [Open ASR Leaderboard](https://huggingface.co/spaces/hf-audio/open_asr_leaderboard).
1163
+
1164
+ ## <span style="color:#466f00;">Performance</span>
1165
+
1166
+ #### Multilingual ASR
1167
+
1168
+ The tables below summarizes the WER (%) using a Transducer decoder with greedy decoding (without an external language model):
1169
+
1170
+
1171
+ | Language | Fleurs | MLS | CoVoST |
1172
+ |----------|--------|-----|--------|
1173
+ | **Average WER ↓** | *11.97%* | *7.83%* | *11.98%* |
1174
+ | **bg** | 12.64% | - | - |
1175
+ | **cs** | 11.01% | - | - |
1176
+ | **da** | 18.41% | - | - |
1177
+ | **de** | 5.04% | - | 4.84% |
1178
+ | **el** | 20.70% | - | - |
1179
+ | **en** | 4.85% | - | 6.80% |
1180
+ | **es** | 3.45% | 4.39% | 3.41% |
1181
+ | **et** | 17.73% | - | 22.04% |
1182
+ | **fi** | 13.21% | - | - |
1183
+ | **fr** | 5.15% | 4.97% | 6.05% |
1184
+ | **hr** | 12.46% | - | - |
1185
+ | **hu** | 15.72% | - | - |
1186
+ | **it** | 3.00% | 10.08% | 3.69% |
1187
+ | **lt** | 20.35% | - | - |
1188
+ | **lv** | 22.84% | - | 38.36% |
1189
+ | **mt** | 20.46% | - | - |
1190
+ | **nl** | 7.48% | 12.78% | 6.50% |
1191
+ | **pl** | 7.31% | 7.28% | - |
1192
+ | **pt** | 4.76% | 7.50% | 3.96% |
1193
+ | **ro** | 12.44% | - | - |
1194
+ | **ru** | 5.51% | - | 3.00% |
1195
+ | **sk** | 8.82% | - | - |
1196
+ | **sl** | 24.03% | - | 31.80% |
1197
+ | **sv** | 15.08% | - | 20.16% |
1198
+ | **uk** | 6.79% | - | 5.10% |
1199
+
1200
+ **Note:** WERs are calculated after removing Punctuation and Capitalization from reference and predicted text.
1201
+
1202
+
1203
+ #### Huggingface Open-ASR-Leaderboard
1204
+
1205
+ | **Model** | **Avg WER** | **AMI** | **Earnings-22** | **GigaSpeech** | **LS test-clean** | **LS test-other** | **SPGI Speech** | **TEDLIUM-v3** | **VoxPopuli** |
1206
+ |:-------------|:-------------:|:---------:|:------------------:|:----------------:|:-----------------:|:-----------------:|:------------------:|:----------------:|:---------------:|
1207
+ | `parakeet-tdt-0.6b-v3` | 6.34% | 11.31% | 11.42% | 9.59% | 1.93% | 3.59% | 3.97% | 2.75% | 6.14% |
1208
+
1209
+ Additional evaluation details are available on the [Hugging Face ASR Leaderboard](https://huggingface.co/spaces/hf-audio/open_asr_leaderboard).[13]
1210
+
1211
+ ### Noise Robustness
1212
+ Performance across different Signal-to-Noise Ratios (SNR) using MUSAN music and noise samples [14]:
1213
+
1214
+ | **SNR Level** | **Avg WER** | **AMI** | **Earnings** | **GigaSpeech** | **LS test-clean** | **LS test-other** | **SPGI** | **Tedlium** | **VoxPopuli** | **Relative Change** |
1215
+ |:---------------|:-------------:|:----------:|:------------:|:----------------:|:-----------------:|:-----------------:|:-----------:|:-------------:|:---------------:|:-----------------:|
1216
+ | Clean | 6.34% | 11.31% | 11.42% | 9.59% | 1.93% | 3.59% | 3.97% | 2.75% | 6.14% | - |
1217
+ | SNR 10 | 7.12% | 13.99% | 11.79% | 9.96% | 2.15% | 4.55% | 4.45% | 3.05% | 6.99% | -12.28% |
1218
+ | SNR 5 | 8.23% | 17.59% | 13.01% | 10.69% | 2.62% | 6.05% | 5.23% | 3.33% | 7.31% | -29.81% |
1219
+ | SNR 0 | 11.66% | 24.44% | 17.34% | 13.60% | 4.82% | 10.38% | 8.41% | 5.39% | 8.91% | -83.97% |
1220
+ | SNR -5 | 19.88% | 34.91% | 26.92% | 21.41% | 12.21% | 19.98% | 16.96% | 11.36% | 15.30% | -213.64% |
1221
+
1222
+
1223
+
1224
+ ## <span style="color:#466f00;">References</span>
1225
+
1226
+ [1] [Granary: Speech Recognition and Translation Dataset in 25 European Languages](https://arxiv.org/abs/2505.13404)
1227
+
1228
+ [2] [NVIDIA Granary Dataset Card](https://huggingface.co/datasets/nvidia/Granary)
1229
+
1230
+ [3] [Fast Conformer with Linearly Scalable Attention for Efficient Speech Recognition](https://arxiv.org/abs/2305.05084)
1231
+
1232
+ [4] [Efficient Sequence Transduction by Jointly Predicting Tokens and Durations](https://arxiv.org/abs/2304.06795)
1233
+
1234
+ [5] [NVIDIA NeMo Toolkit](https://github.com/NVIDIA/NeMo)
1235
+
1236
+ [6] [Google Sentencepiece Tokenizer](https://github.com/google/sentencepiece)
1237
+
1238
+ [7] [Youtube-Commons](https://huggingface.co/datasets/PleIAs/YouTube-Commons)
1239
+
1240
+ [8] [MOSEL: 950,000 Hours of Speech Data for Open-Source Speech Foundation Model Training on EU Languages](https://arxiv.org/abs/2410.01036)
1241
+
1242
+ [9] [YODAS: Youtube-Oriented Dataset for Audio and Speech](https://arxiv.org/pdf/2406.00899)
1243
+
1244
+ [10] [FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech](https://arxiv.org/abs/2205.12446)
1245
+
1246
+ [11] [MLS: A Large-Scale Multilingual Dataset for Speech Research](https://arxiv.org/abs/2012.03411)
1247
+
1248
+ [12] [CoVoST 2 and Massively Multilingual Speech-to-Text Translation](https://arxiv.org/abs/2007.10310)
1249
+
1250
+ [13] [HuggingFace ASR Leaderboard](https://huggingface.co/spaces/hf-audio/open_asr_leaderboard)
1251
+
1252
+ [14] [MUSAN: A Music, Speech, and Noise Corpus](https://arxiv.org/abs/1510.08484)
1253
+
1254
+ ## <span style="color:#466f00;">Inference:</span>
1255
+
1256
+ **Engine**:
1257
+ * NVIDIA NeMo
1258
+
1259
+ **Test Hardware**:
1260
+ * NVIDIA A10
1261
+ * NVIDIA A100
1262
+ * NVIDIA A30
1263
+ * NVIDIA H100
1264
+ * NVIDIA L4
1265
+ * NVIDIA L40
1266
+ * NVIDIA Turing T4
1267
+ * NVIDIA Volta V100
1268
+
1269
+ ## <span style="color:#466f00;">Ethical Considerations:</span>
1270
+ NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their supporting model team to ensure this model meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
1271
+
1272
+ For more detailed information on ethical considerations for this model, please see the Model Card++ Explainability, Bias, Safety & Security, and Privacy Subcards [here](https://developer.nvidia.com/blog/enhancing-ai-transparency-and-ethical-considerations-with-model-card/).
1273
+
1274
+ Please report security vulnerabilities or NVIDIA AI Concerns [here](https://www.nvidia.com/en-us/support/submit-security-vulnerability/).
1275
+
1276
+ ## <span style="color:#466f00;">Bias:</span>
1277
+
1278
+ Field | Response
1279
+ ---------------------------------------------------------------------------------------------------|---------------
1280
+ Participation considerations from adversely impacted groups [protected classes](https://www.senate.ca.gov/content/protected-classes) in model design and testing | None
1281
+ Measures taken to mitigate against unwanted bias | None
1282
+
1283
+ ## <span style="color:#466f00;">Explainability:</span>
1284
+
1285
+ Field | Response
1286
+ ------------------------------------------------------------------------------------------------------|---------------------------------------------------------------------------------
1287
+ Intended Domain | Speech to Text Transcription
1288
+ Model Type | FastConformer
1289
+ Intended Users | This model is intended for developers, researchers, academics, and industries building conversational based applications.
1290
+ Output | Text
1291
+ Describe how the model works | Speech input is encoded into embeddings and passed into conformer-based model and output a text response.
1292
+ Name the adversely impacted groups this has been tested to deliver comparable outcomes regardless of | Not Applicable
1293
+ Technical Limitations & Mitigation | Transcripts may be not 100% accurate. Accuracy varies based on language and characteristics of input audio (Domain, Use Case, Accent, Noise, Speech Type, Context of speech, etc.)
1294
+ Verified to have met prescribed NVIDIA quality standards | Yes
1295
+ Performance Metrics | Word Error Rate
1296
+ Potential Known Risks | If a word is not trained in the language model and not presented in vocabulary, the word is not likely to be recognized. Not recommended for word-for-word/incomplete sentences as accuracy varies based on the context of input text
1297
+ Licensing | GOVERNING TERMS: Use of this model is governed by the [CC-BY-4.0](https://creativecommons.org/licenses/by/4.0/legalcode.en) license.
1298
+
1299
+ ## <span style="color:#466f00;">Privacy:</span>
1300
+
1301
+ Field | Response
1302
+ ----------------------------------------------------------------------------------------------------------------------------------|-----------------------------------------------
1303
+ Generatable or reverse engineerable personal data? | None
1304
+ Personal data used to create this model? | None
1305
+ Is there provenance for all datasets used in training? | Yes
1306
+ Does data labeling (annotation, metadata) comply with privacy laws? | Yes
1307
+ Is data compliant with data subject requests for data correction or removal, if such a request was made? | No, not possible with externally-sourced data.
1308
+ Applicable Privacy Policy | https://www.nvidia.com/en-us/about-nvidia/privacy-policy/
1309
+
1310
+ ## <span style="color:#466f00;">Safety:</span>
1311
+
1312
+ Field | Response
1313
+ ---------------------------------------------------|----------------------------------
1314
+ Model Application(s) | Speech to Text Transcription
1315
+ Describe the life critical impact | None
1316
+ Use Case Restrictions | Abide by [CC-BY-4.0](https://creativecommons.org/licenses/by/4.0/legalcode.en) License
1317
+ Model and dataset restrictions | The Principle of least privilege (PoLP) is applied limiting access for dataset generation and model development. Restrictions enforce dataset access during training, and dataset license constraints adhered to.
config.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "ParakeetForTDT"
4
+ ],
5
+ "blank_token_id": 8192,
6
+ "decoder_hidden_size": 640,
7
+ "dtype": "float32",
8
+ "durations": [
9
+ 0,
10
+ 1,
11
+ 2,
12
+ 3,
13
+ 4
14
+ ],
15
+ "encoder_config": {
16
+ "activation_dropout": 0.1,
17
+ "attention_bias": false,
18
+ "attention_dropout": 0.1,
19
+ "conv_kernel_size": 9,
20
+ "convolution_bias": false,
21
+ "dropout": 0.1,
22
+ "dropout_positions": 0.0,
23
+ "hidden_act": "silu",
24
+ "hidden_size": 1024,
25
+ "initializer_range": 0.02,
26
+ "intermediate_size": 4096,
27
+ "layerdrop": 0.1,
28
+ "max_position_embeddings": 5000,
29
+ "model_type": "parakeet_encoder",
30
+ "num_attention_heads": 8,
31
+ "num_hidden_layers": 24,
32
+ "num_key_value_heads": 8,
33
+ "num_mel_bins": 128,
34
+ "scale_input": false,
35
+ "subsampling_conv_channels": 256,
36
+ "subsampling_conv_kernel_size": 3,
37
+ "subsampling_conv_stride": 2,
38
+ "subsampling_factor": 8
39
+ },
40
+ "hidden_act": "relu",
41
+ "initializer_range": 0.02,
42
+ "is_encoder_decoder": true,
43
+ "max_symbols_per_step": 10,
44
+ "model_type": "parakeet_tdt",
45
+ "num_decoder_layers": 2,
46
+ "pad_token_id": 2,
47
+ "transformers_version": "5.6.0.dev0",
48
+ "vocab_size": 8193
49
+ }
generation_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "decoder_start_token_id": 8192,
4
+ "eos_token_id": 3,
5
+ "output_attentions": false,
6
+ "output_hidden_states": false,
7
+ "pad_token_id": 2,
8
+ "suppress_tokens": [
9
+ 8193,
10
+ 8194,
11
+ 8195,
12
+ 8196,
13
+ 8197
14
+ ],
15
+ "transformers_version": "5.6.0.dev0"
16
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3a2026366188c8c68598edbbff92f8d11590a08e0ae2e6775544e7b07d6a5e11
3
+ size 2508311120
processor_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "blank_token": "<blank>",
3
+ "feature_extractor": {
4
+ "feature_extractor_type": "ParakeetFeatureExtractor",
5
+ "feature_size": 128,
6
+ "hop_length": 160,
7
+ "n_fft": 512,
8
+ "padding_side": "right",
9
+ "padding_value": 0.0,
10
+ "preemphasis": 0.97,
11
+ "return_attention_mask": true,
12
+ "sampling_rate": 16000,
13
+ "win_length": 400
14
+ },
15
+ "processor_class": "ParakeetProcessor"
16
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "backend": "tokenizers",
3
+ "clean_up_tokenization_spaces": false,
4
+ "eos_token": "<|endoftext|>",
5
+ "model_max_length": 1000000000000000019884624838656,
6
+ "pad_token": "<pad>",
7
+ "processor_class": "ParakeetProcessor",
8
+ "tokenizer_class": "ParakeetTokenizer",
9
+ "unk_token": "<unk>"
10
+ }