Physical papers, 26/03/2026
Physical papers sitting on my desk
- Flow Matching Guide and Code
- Topics in North Saami Phonology
- On The Landscape of Spoken Language Models: A Comprehensive Survey
- T5G2P: Using Text-to-Text Transfer Transformer for Grapheme-to-Phoneme Conversion
- The Swedish Parliament Corpus 1867 – 2022
- Google’s Multilingual Neural Machine Translation System: Enabling Zero-Shot Translation
- Codec-ASR: Training Performant Automatic Speech Recognition Systems with Discrete Speech Representations
- SPACER: A Parallel Dataset of Speech Production And Comprehension of Error Repairs
- Pronunciation adaptation at the lexical level
- Unifying Diarization, Separation, and ASR with Multi-Speaker Encoder
- Hierarchical Phrase-based Translation Representations
- ByT5 model for massively multilingual grapheme-to-phoneme conversion
- WavLLM: Towards Robust and Adaptive Speech Large Language Model
- The taste of IPA: Towards open-vocabulary keyword spotting and forced alignment in any language
- LiteASR: Efficient Automatic Speech Recognition with Low-Rank Approximation
- Gaussian Distributions are Soap Bubbles
- mixup: Data-Dependent Data Augmentation
- Recent Advances in Discrete Speech Tokens: A Review
- How Neural Networks Learn: A Probabilistic Viewpoint
- Dynamic pronunciation models for automatic speech recognition
Flow Matching Guide and Code
Topics in North Saami Phonology
@inproceedings{Bals2005TopicsIN,
title={Topics in North Saami Phonology},
author={Berit Anne Bals and David Odden and K. Gjerum Nielsen},
year={2005},
url={https://api.semanticscholar.org/CorpusID:15957850}
}
On The Landscape of Spoken Language Models: A Comprehensive Survey
- arXiv
- TMLR
@article{
arora2025on,
title={On The Landscape of Spoken Language Models: A Comprehensive Survey},
author={Siddhant Arora and Kai-Wei Chang and Chung-Ming Chien and Yifan Peng and Haibin Wu and Yossi Adi and Emmanuel Dupoux and Hung-yi Lee and Karen Livescu and Shinji Watanabe},
journal={Transactions on Machine Learning Research},
issn={2835-8856},
year={2025},
url={https://openreview.net/forum?id=BvxaP3sVbA},
note={}
}
T5G2P: Using Text-to-Text Transfer Transformer for Grapheme-to-Phoneme Conversion
@inproceedings{rezackova21_interspeech,
title = ,
author = {Markéta Řezáčková and Jan Švec and Daniel Tihelka},
year = {2021},
booktitle = ,
pages = {6--10},
doi = {10.21437/Interspeech.2021-546},
issn = {2958-1796},
}
The Swedish Parliament Corpus 1867 – 2022
@inproceedings{yrjanainen-etal-2024-swedish,
title = "The {S}wedish Parliament Corpus 1867 {--} 2022",
author = {Yrj{\"a}n{\"a}inen, V{\"a}in{\"o} Aleksi and
Mohammadi Nor{\'e}n, Fredrik and
Borges, Robert and
Jarlbrink, Johan and
{\r{A}}berg Brorsson, Lotta and
Olsson, Anders P. and
Snickars, Pelle and
Magnusson, M{\r{a}}ns},
editor = "Calzolari, Nicoletta and
Kan, Min-Yen and
Hoste, Veronique and
Lenci, Alessandro and
Sakti, Sakriani and
Xue, Nianwen",
booktitle = "Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)",
month = may,
year = "2024",
address = "Torino, Italia",
publisher = "ELRA and ICCL",
url = "https://aclanthology.org/2024.lrec-main.1400/",
pages = "16100--16112",
}
Google’s Multilingual Neural Machine Translation System: Enabling Zero-Shot Translation
@article{johnson-etal-2017-googles,
title = "{G}oogle{'}s Multilingual Neural Machine Translation System: Enabling Zero-Shot Translation",
author = "Johnson, Melvin and
Schuster, Mike and
Le, Quoc V. and
Krikun, Maxim and
Wu, Yonghui and
Chen, Zhifeng and
Thorat, Nikhil and
Vi{\'e}gas, Fernanda and
Wattenberg, Martin and
Corrado, Greg and
Hughes, Macduff and
Dean, Jeffrey",
editor = "Lee, Lillian and
Johnson, Mark and
Toutanova, Kristina",
journal = "Transactions of the Association for Computational Linguistics",
volume = "5",
year = "2017",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/Q17-1024/",
doi = "10.1162/tacl_a_00065",
pages = "339--351",
}
Codec-ASR: Training Performant Automatic Speech Recognition Systems with Discrete Speech Representations
@inproceedings{dhawan24_interspeech,
title = ,
author = {Kunal Dhawan and Nithin Rao Koluguri and Ante Jukić and Ryan Langman and Jagadeesh Balam and Boris Ginsburg},
year = {2024},
booktitle = ,
pages = {2574--2578},
doi = {10.21437/Interspeech.2024-330},
issn = {2958-1796},
}
SPACER: A Parallel Dataset of Speech Production And Comprehension of Error Repairs
@inproceedings{upadhye-etal-2025-spacer,
title = "{SPACER}: A Parallel Dataset of Speech Production And Comprehension of Error Repairs",
author = "Upadhye, Shiva and
Li, Jiaxuan and
Futrell, Richard",
editor = "Kuribayashi, Tatsuki and
Rambelli, Giulia and
Takmaz, Ece and
Wicke, Philipp and
Li, Jixing and
Oh, Byung-Doh",
booktitle = "Proceedings of the Workshop on Cognitive Modeling and Computational Linguistics",
month = may,
year = "2025",
address = "Albuquerque, New Mexico, USA",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2025.cmcl-1.19/",
doi = "10.18653/v1/2025.cmcl-1.19",
pages = "144--154",
ISBN = "979-8-89176-227-5",
}
Pronunciation adaptation at the lexical level
@inproceedings{strik01_adaptation,
title = ,
author = {Helmer Strik},
year = {2001},
booktitle = ,
pages = {123--130},
}
Unifying Diarization, Separation, and ASR with Multi-Speaker Encoder
- arXiv
- ICLR 2025
@misc{shakeel2025unifying,
title={Unifying Diarization, Separation, and {ASR} with Multi-Speaker Encoder},
author={Muhammad Shakeel and Yui Sudo and Yifan Peng and Chyi-Jiunn Lin and Shinji Watanabe},
year={2025},
url={https://openreview.net/forum?id=5oaUMZEjWe}
}
Hierarchical Phrase-based Translation Representations
@inproceedings{iglesias-etal-2011-hierarchical,
title = "Hierarchical Phrase-based Translation Representations",
author = "Iglesias, Gonzalo and
Allauzen, Cyril and
Byrne, William and
de Gispert, Adri{\`a} and
Riley, Michael",
editor = "Barzilay, Regina and
Johnson, Mark",
booktitle = "Proceedings of the 2011 Conference on Empirical Methods in Natural Language Processing",
month = jul,
year = "2011",
address = "Edinburgh, Scotland, UK.",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/D11-1127/",
pages = "1373--1383"
}
ByT5 model for massively multilingual grapheme-to-phoneme conversion
@inproceedings{zhu22_interspeech,
title = ,
author = {Jian Zhu and Cong Zhang and David Jurgens},
year = {2022},
booktitle = ,
pages = {446--450},
doi = {10.21437/Interspeech.2022-538},
issn = {2958-1796},
}
WavLLM: Towards Robust and Adaptive Speech Large Language Model
@inproceedings{hu-etal-2024-wavllm,
title = "{W}av{LLM}: Towards Robust and Adaptive Speech Large Language Model",
author = "Hu, Shujie and
Zhou, Long and
Liu, Shujie and
Chen, Sanyuan and
Meng, Lingwei and
Hao, Hongkun and
Pan, Jing and
Liu, Xunying and
Li, Jinyu and
Sivasankaran, Sunit and
Liu, Linquan and
Wei, Furu",
editor = "Al-Onaizan, Yaser and
Bansal, Mohit and
Chen, Yun-Nung",
booktitle = "Findings of the Association for Computational Linguistics: EMNLP 2024",
month = nov,
year = "2024",
address = "Miami, Florida, USA",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2024.findings-emnlp.263/",
doi = "10.18653/v1/2024.findings-emnlp.263",
pages = "4552--4572",
}
The taste of IPA: Towards open-vocabulary keyword spotting and forced alignment in any language
@inproceedings{zhu-etal-2024-taste,
title = "The taste of {IPA}: Towards open-vocabulary keyword spotting and forced alignment in any language",
author = "Zhu, Jian and
Yang, Changbing and
Samir, Farhan and
Islam, Jahurul",
editor = "Duh, Kevin and
Gomez, Helena and
Bethard, Steven",
booktitle = "Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)",
month = jun,
year = "2024",
address = "Mexico City, Mexico",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2024.naacl-long.43/",
doi = "10.18653/v1/2024.naacl-long.43",
pages = "750--772",
}
LiteASR: Efficient Automatic Speech Recognition with Low-Rank Approximation
@inproceedings{kamahori-etal-2025-liteasr,
title = "{L}ite{ASR}: Efficient Automatic Speech Recognition with Low-Rank Approximation",
author = "Kamahori, Keisuke and
Kasai, Jungo and
Kojima, Noriyuki and
Kasikci, Baris",
editor = "Christodoulopoulos, Christos and
Chakraborty, Tanmoy and
Rose, Carolyn and
Peng, Violet",
booktitle = "Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing",
month = nov,
year = "2025",
address = "Suzhou, China",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2025.emnlp-main.169/",
doi = "10.18653/v1/2025.emnlp-main.169",
pages = "3430--3442",
ISBN = "979-8-89176-332-6",
}
Gaussian Distributions are Soap Bubbles
mixup: Data-Dependent Data Augmentation
Recent Advances in Discrete Speech Tokens: A Review
@ARTICLE{11298521,
author={Guo, Yiwei and Li, Zhihan and Wang, Hankun and Li, Bohan and Shao, Chongtian and Zhang, Hanglei and Du, Chenpeng and Chen, Xie and Liu, Shujie and Yu, Kai},
journal={ IEEE Transactions on Pattern Analysis \& Machine Intelligence },
title=,
year={2026},
volume={48},
number={04},
ISSN={1939-3539},
pages={4184-4204},
doi={10.1109/TPAMI.2025.3643619},
url = {https://doi.ieeecomputersociety.org/10.1109/TPAMI.2025.3643619},
publisher={IEEE Computer Society},
address={Los Alamitos, CA, USA},
month=apr}
How Neural Networks Learn: A Probabilistic Viewpoint
Dynamic pronunciation models for automatic speech recognition
- Chapters 3 & 4
@phdthesis{10.5555/931366,
author = {Fosler-Lussier, John Eric and Morgan, Nelson},
title = {Dynamic pronunciation models for automatic speech recognition},
year = {1999},
isbn = {059971154X},
publisher = {University of California, Berkeley},
}