refs.bib 16 KB
Newer Older
moto's avatar
moto committed
1
2
3
4
5
6
7
@misc{RESAMPLE,
	author = {Julius O. Smith},
	title = {Digital Audio Resampling Home Page "Theory of Ideal Bandlimited Interpolation" section},
	url = {https://ccrma.stanford.edu/~jos/resample/Theory_Ideal_Bandlimited_Interpolation.html},
	month = {September},
	year = {2020}
}
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
@article{voxpopuli,
  author    = {Changhan Wang and
               Morgane Rivi{\`{e}}re and
               Ann Lee and
               Anne Wu and
               Chaitanya Talnikar and
               Daniel Haziza and
               Mary Williamson and
               Juan Miguel Pino and
               Emmanuel Dupoux},
  title     = {VoxPopuli: {A} Large-Scale Multilingual Speech Corpus for Representation
               Learning, Semi-Supervised Learning and Interpretation},
  journal   = {CoRR},
  volume    = {abs/2101.00390},
  year      = {2021},
  url       = {https://arxiv.org/abs/2101.00390},
  eprinttype = {arXiv},
  eprint    = {2101.00390},
  timestamp = {Thu, 12 Aug 2021 15:37:06 +0200},
  biburl    = {https://dblp.org/rec/journals/corr/abs-2101-00390.bib},
  bibsource = {dblp computer science bibliography, https://dblp.org}
}
moto's avatar
moto committed
30
31
32
33
34
35
36
37
38
39
@article{specaugment,
   title={SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition},
   url={http://dx.doi.org/10.21437/Interspeech.2019-2680},
   DOI={10.21437/interspeech.2019-2680},
   journal={Interspeech 2019},
   publisher={ISCA},
   author={Park, Daniel S. and Chan, William and Zhang, Yu and Chiu, Chung-Cheng and Zoph, Barret and Cubuk, Ekin D. and Le, Quoc V.},
   year={2019},
   month={Sep}
}
moto's avatar
moto committed
40
41
42
43
44
45
@misc{ljspeech17,
  author       = {Keith Ito and Linda Johnson},
  title        = {The LJ Speech Dataset},
  howpublished = {\url{https://keithito.com/LJ-Speech-Dataset/}},
  year         = {2017}
}
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
@misc{conneau2020unsupervised,
      title={Unsupervised Cross-lingual Representation Learning for Speech Recognition}, 
      author={Alexis Conneau and Alexei Baevski and Ronan Collobert and Abdelrahman Mohamed and Michael Auli},
      year={2020},
      eprint={2006.13979},
      archivePrefix={arXiv},
      primaryClass={cs.CL}
}
@inproceedings{Gales2014SpeechRA,
  title={Speech recognition and keyword spotting for low-resource languages: Babel project research at CUED},
  author={Mark John Francis Gales and Kate Knill and Anton Ragni and Shakti Prasad Rath},
  booktitle={SLTU},
  year={2014}
}
@misc{ardila2020common,
      title={Common Voice: A Massively-Multilingual Speech Corpus}, 
      author={Rosana Ardila and Megan Branson and Kelly Davis and Michael Henretty and Michael Kohler and Josh Meyer and Reuben Morais and Lindsay Saunders and Francis M. Tyers and Gregor Weber},
      year={2020},
      eprint={1912.06670},
      archivePrefix={arXiv},
      primaryClass={cs.CL}
}
@article{Pratap_2020,
   title={MLS: A Large-Scale Multilingual Dataset for Speech Research},
   url={http://dx.doi.org/10.21437/Interspeech.2020-2826},
   DOI={10.21437/interspeech.2020-2826},
   journal={Interspeech 2020},
   publisher={ISCA},
   author={Pratap, Vineel and Xu, Qiantong and Sriram, Anuroop and Synnaeve, Gabriel and Collobert, Ronan},
   year={2020},
   month={Oct}
}
@INPROCEEDINGS{librilight,
  author={J. {Kahn} and M. {Rivière} and W. {Zheng} and E. {Kharitonov} and Q. {Xu} and P. E. {Mazaré} and J. {Karadayi} and V. {Liptchinsky} and R. {Collobert} and C. {Fuegen} and T. {Likhomanenko} and G. {Synnaeve} and A. {Joulin} and A. {Mohamed} and E. {Dupoux}},
  booktitle={ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, 
  title={Libri-Light: A Benchmark for ASR with Limited or No Supervision}, 
  year={2020},
  pages={7669-7673},
  note = {\url{https://github.com/facebookresearch/libri-light}},
}
@INPROCEEDINGS{7178964,
  author={Panayotov, Vassil and Chen, Guoguo and Povey, Daniel and Khudanpur, Sanjeev},
  booktitle={2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, 
  title={Librispeech: An ASR corpus based on public domain audio books}, 
  year={2015},
  volume={},
  number={},
  pages={5206-5210},
  doi={10.1109/ICASSP.2015.7178964}
}
@inproceedings{ott2019fairseq,
  title = {fairseq: A Fast, Extensible Toolkit for Sequence Modeling},
  author = {Myle Ott and Sergey Edunov and Alexei Baevski and Angela Fan and Sam Gross and Nathan Ng and David Grangier and Michael Auli},
  booktitle = {Proceedings of NAACL-HLT 2019: Demonstrations},
  year = {2019},
}
moto's avatar
moto committed
102
103
104
105
106
107
108
109
@misc{baevski2020wav2vec,
      title={wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations}, 
      author={Alexei Baevski and Henry Zhou and Abdelrahman Mohamed and Michael Auli},
      year={2020},
      eprint={2006.11477},
      archivePrefix={arXiv},
      primaryClass={cs.CL}
}
moto's avatar
moto committed
110
111
112
113
114
115
116
117
@misc{hsu2021hubert,
      title={HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units}, 
      author={Wei-Ning Hsu and Benjamin Bolte and Yao-Hung Hubert Tsai and Kushal Lakhotia and Ruslan Salakhutdinov and Abdelrahman Mohamed},
      year={2021},
      eprint={2106.07447},
      archivePrefix={arXiv},
      primaryClass={cs.CL}
}
moto's avatar
moto committed
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
@misc{hannun2014deep,
      title={Deep Speech: Scaling up end-to-end speech recognition}, 
      author={Awni Hannun and Carl Case and Jared Casper and Bryan Catanzaro and Greg Diamos and Erich Elsen and Ryan Prenger and Sanjeev Satheesh and Shubho Sengupta and Adam Coates and Andrew Y. Ng},
      year={2014},
      eprint={1412.5567},
      archivePrefix={arXiv},
      primaryClass={cs.CL}
}
@misc{graves2012sequence,
      title={Sequence Transduction with Recurrent Neural Networks}, 
      author={Alex Graves},
      year={2012},
      eprint={1211.3711},
      archivePrefix={arXiv},
      primaryClass={cs.NE}
}
@misc{collobert2016wav2letter,
      title={Wav2Letter: an End-to-End ConvNet-based Speech Recognition System}, 
      author={Ronan Collobert and Christian Puhrsch and Gabriel Synnaeve},
      year={2016},
      eprint={1609.03193},
      archivePrefix={arXiv},
      primaryClass={cs.LG}
}
@misc{kalchbrenner2018efficient,
      title={Efficient Neural Audio Synthesis}, 
      author={Nal Kalchbrenner and Erich Elsen and Karen Simonyan and Seb Noury and Norman Casagrande and Edward Lockhart and Florian Stimberg and Aaron van den Oord and Sander Dieleman and Koray Kavukcuoglu},
      year={2018},
      eprint={1802.08435},
      archivePrefix={arXiv},
      primaryClass={cs.SD}
}
hwangjeff's avatar
hwangjeff committed
150
151
152
153
154
155
156
157
@misc{gulati2020conformer,
      title={Conformer: Convolution-augmented Transformer for Speech Recognition},
      author={Anmol Gulati and James Qin and Chung-Cheng Chiu and Niki Parmar and Yu Zhang and Jiahui Yu and Wei Han and Shibo Wang and Zhengdong Zhang and Yonghui Wu and Ruoming Pang},
      year={2020},
      eprint={2005.08100},
      archivePrefix={arXiv},
      primaryClass={eess.AS}
}
moto's avatar
moto committed
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
@article{Luo_2019,
   title={Conv-TasNet: Surpassing Ideal Time–Frequency Magnitude Masking for Speech Separation},
   volume={27},
   ISSN={2329-9304},
   url={http://dx.doi.org/10.1109/TASLP.2019.2915167},
   DOI={10.1109/taslp.2019.2915167},
   number={8},
   journal={IEEE/ACM Transactions on Audio, Speech, and Language Processing},
   publisher={Institute of Electrical and Electronics Engineers (IEEE)},
   author={Luo, Yi and Mesgarani, Nima},
   year={2019},
   month={Aug},
   pages={1256–1266}
}
@InProceedings{ brian_mcfee-proc-scipy-2015,
  author    = { {B}rian {M}c{F}ee and {C}olin {R}affel and {D}awen {L}iang and {D}aniel {P}.{W}. {E}llis and {M}att {M}c{V}icar and {E}ric {B}attenberg and {O}riol {N}ieto },
  title     = { librosa: {A}udio and {M}usic {S}ignal {A}nalysis in {P}ython },
  booktitle = { {P}roceedings of the 14th {P}ython in {S}cience {C}onference },
  pages     = { 18 - 24 },
  year      = { 2015 },
  editor    = { {K}athryn {H}uff and {J}ames {B}ergstra },
  doi       = { 10.25080/Majora-7b98e3ed-003 }
}
@INPROCEEDINGS{6701851,
  author={Perraudin, Nathanaël and Balazs, Peter and Søndergaard, Peter L.},
  booktitle={2013 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics}, 
  title={A fast Griffin-Lim algorithm}, 
  year={2013},
  volume={},
  number={},
  pages={1-4},
  doi={10.1109/WASPAA.2013.6701851}}
@INPROCEEDINGS{1172092,
  author={Griffin, D. and Jae Lim},
  booktitle={ICASSP '83. IEEE International Conference on Acoustics, Speech, and Signal Processing}, 
  title={Signal estimation from modified short-time Fourier transform}, 
  year={1983},
  volume={8},
  number={},
  pages={804-807},
  doi={10.1109/ICASSP.1983.1172092}}
@INPROCEEDINGS{6854049,
  author={Ghahremani, Pegah and BabaAli, Bagher and Povey, Daniel and Riedhammer, Korbinian and Trmal, Jan and Khudanpur, Sanjeev},
  booktitle={2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, 
  title={A pitch extraction algorithm tuned for automatic speech recognition}, 
  year={2014},
  volume={},
  number={},
  pages={2494-2498},
  doi={10.1109/ICASSP.2014.6854049}}
208
209
210
211
212
213
214
@inproceedings{shen2018natural,
  title={Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram Predictions},
  author={Shen, Jonathan and Pang, Ruoming and Weiss, Ron J and Schuster, Mike and Jaitly, Navdeep and Yang, Zongheng and Chen, Zhifeng and Zhang, Yu and Wang, Yuxuan and Skerrv-Ryan, Rj and others},
  booktitle={2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)},
  pages={4779--4783},
  year={2018},
  organization={IEEE}
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
}
@inproceedings{souden2009optimal,
  title={On optimal frequency-domain multichannel linear filtering for noise reduction},
  author={Souden, Mehrez and Benesty, Jacob and Affes, Sofiene},
  booktitle={IEEE Transactions on audio, speech, and language processing},
  volume={18},
  number={2},
  pages={260--276},
  year={2009},
  publisher={IEEE}
}
@inproceedings{higuchi2016robust,
  title={Robust MVDR beamforming using time-frequency masks for online/offline ASR in noise},
  author={Higuchi, Takuya and Ito, Nobutaka and Yoshioka, Takuya and Nakatani, Tomohiro},
  booktitle={2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)},
  pages={5210--5214},
  year={2016},
  organization={IEEE}
}
hwangjeff's avatar
hwangjeff committed
234
235
236
237
238
239
240
@inproceedings{shi2021emformer,
  title={Emformer: Efficient Memory Transformer Based Acoustic Model for Low Latency Streaming Speech Recognition}, 
  author={Shi, Yangyang and Wang, Yongqiang and Wu, Chunyang and Yeh, Ching-Feng and Chan, Julian and Zhang, Frank and Le, Duc and Seltzer, Mike},
  booktitle={ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, 
  pages={6783-6787},
  year={2021}
}
hwangjeff's avatar
hwangjeff committed
241
242
243
244
245
246
247
248
249
@inproceedings{9747706,
  author={Shi, Yangyang and Wu, Chunyang and Wang, Dilin and Xiao, Alex and Mahadeokar, Jay and Zhang, Xiaohui and Liu, Chunxi and Li, Ke and Shangguan, Yuan and Nagaraja, Varun and Kalinli, Ozlem and Seltzer, Mike},
  booktitle={ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, 
  title={Streaming Transformer Transducer based Speech Recognition Using Non-Causal Convolution}, 
  year={2022},
  volume={},
  number={},
  pages={8277-8281},
  doi={10.1109/ICASSP43922.2022.9747706}}
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
@article{mises1929praktische,
  title={Praktische Verfahren der Gleichungsaufl{\"o}sung.},
  author={Mises, RV and Pollaczek-Geiringer, Hilda},
  journal={ZAMM-Journal of Applied Mathematics and Mechanics/Zeitschrift f{\"u}r Angewandte Mathematik und Mechanik},
  volume={9},
  number={1},
  pages={58--77},
  year={1929},
  publisher={Wiley Online Library}
}
@article{higuchi2017online,
  title={Online MVDR beamformer based on complex Gaussian mixture model with spatial prior for noise robust ASR},
  author={Higuchi, Takuya and Ito, Nobutaka and Araki, Shoko and Yoshioka, Takuya and Delcroix, Marc and Nakatani, Tomohiro},
  journal={IEEE/ACM Transactions on Audio, Speech, and Language Processing},
  volume={25},
  number={4},
  pages={780--793},
  year={2017},
  publisher={IEEE}
}
270
271
272
273
274
275
276
277
278
279
@article{capon1969high,
  title={High-resolution frequency-wavenumber spectrum analysis},
  author={Capon, Jack},
  journal={Proceedings of the IEEE},
  volume={57},
  number={8},
  pages={1408--1418},
  year={1969},
  publisher={IEEE}
}
280
281
282
283
284
285
@article{kahn2022flashlight,
  title={Flashlight: Enabling Innovation in Tools for Machine Learning},
  author={Kahn, Jacob and Pratap, Vineel and Likhomanenko, Tatiana and Xu, Qiantong and Hannun, Awni and Cai, Jeff and Tomasello, Paden and Lee, Ann and Grave, Edouard and Avidov, Gilad and others},
  journal={arXiv preprint arXiv:2201.12465},
  year={2022}
}
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
@TECHREPORT{Kominek03cmuarctic,
  author = {John Kominek and Alan W Black and Ver Ver},
  title = {CMU Arctic Databases for Speech Synthesis},
  institution = {},
  year = {2003}
}
@misc{cosentino2020librimix,
  title={LibriMix: An Open-Source Dataset for Generalizable Speech Separation},
  author={Joris Cosentino and Manuel Pariente and Samuele Cornell and Antoine Deleforge and Emmanuel Vincent},
  year={2020},
  eprint={2005.11262},
  archivePrefix={arXiv},
  primaryClass={eess.AS}
}
@article{Zen2019LibriTTSAC,
  title={LibriTTS: A Corpus Derived from LibriSpeech for Text-to-Speech},
  author={Heiga Zen and Viet-Trung Dang and Robert A. J. Clark and Yu Zhang and Ron J. Weiss and Ye Jia and Z. Chen and Yonghui Wu},
  journal={ArXiv},
  year={2019},
  volume={abs/1904.02882}
}
@article{speechcommandsv2,
  author = { {Warden}, P.},
  title = "{Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition}",
  journal = {ArXiv e-prints},
  archivePrefix = "arXiv",
  eprint = {1804.03209},
  primaryClass = "cs.CL",
  keywords = {Computer Science - Computation and Language, Computer Science - Human-Computer Interaction},
  year = 2018,
  month = apr,
  url = {https://arxiv.org/abs/1804.03209},
}
@inproceedings{rousseau2012tedlium,
  title={TED-LIUM: an Automatic Speech Recognition dedicated corpus},
  author={Rousseau, Anthony and Del{\'e}glise, Paul and Est{\`e}ve, Yannick},
  booktitle={Conference on Language Resources and Evaluation (LREC)},
  pages={125--129},
  year={2012}
}
@misc{yamagishi2019vctk,
  author={Yamagishi, Junichi and Veaux, Christophe and MacDonald, Kirsten},
  title={ {CSTR VCTK Corpus}: English Multi-speaker Corpus for {CSTR} Voice Cloning Toolkit (version 0.92)},
  publisher={University of Edinburgh. The Centre for Speech Technology Research (CSTR)},
  year=2019,
  doi={10.7488/ds/2645},
}
@misc{Sarfjoo2018DeviceRV,
  title={Device Recorded VCTK (Small subset version)},
  author={Seyyed Saeed Sarfjoo and Junichi Yamagishi},
  year={2018}
}
@misc{tzanetakis_essl_cook_2001,
  author    = "Tzanetakis, George and Essl, Georg and Cook, Perry",
  title     = "Automatic Musical Genre Classification Of Audio Signals",
  url       = "http://ismir2001.ismir.net/pdf/tzanetakis.pdf",
  publisher = "The International Society for Music Information Retrieval",
  year      = "2001"
}
@article{Mir2015QUESST2014EQ,
  title={QUESST2014: Evaluating Query-by-Example Speech Search in a zero-resource setting with real-life queries},
  author={Xavier Anguera Miro and Luis Javier Rodriguez-Fuentes and Andi Buzo and Florian Metze and Igor Szoke and Mikel Pe{\~n}agarikano},
  journal={2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)},
  year={2015},
  pages={5833-5837}
}
@misc{cmudict,
  title={The Carnegie Mellon pronuncing dictionary},
  author={Weide, R.L.},
  year={1998},
  url={http://www.speech.cs.cmu.edu/cgi-bin/cmudict},
}
@misc{YesNo,
  title="YesNo",
  url="http://www.openslr.org/1/"
}
362
363
364
365
366
367
368
369
@misc{MUSDB18HQ,
  author       = {Rafii, Zafar and Liutkus, Antoine and Fabian-Robert St{\"o}ter and Mimilakis, Stylianos Ioannis and
                  Bittner, Rachel},
  title        = {{MUSDB18-HQ} - an uncompressed version of MUSDB18},
  month        = dec,
  year         = 2019,
  doi          = {10.5281/zenodo.3338373},
  url          = {https://doi.org/10.5281/zenodo.3338373}
370
371
372
373
374
375
376
377
@inproceedings{fluent,
  author    = {Loren Lugosch and Mirco Ravanelli and Patrick Ignoto and Vikrant Singh Tomar and Yoshua Bengio},
  editor    = {Gernot Kubin and Zdravko Kacic},
  title     = {Speech Model Pre-Training for End-to-End Spoken Language Understanding},
  booktitle = {Proc. of Interspeech},
  pages     = {814--818},
  year      = {2019},
}