Flow Matching Guide and Code

Topics in North Saami Phonology

@inproceedings{Bals2005TopicsIN,
  title={Topics in North Saami Phonology},
  author={Berit Anne Bals and David Odden and K. Gjerum Nielsen},
  year={2005},
  url={https://api.semanticscholar.org/CorpusID:15957850}
}

On The Landscape of Spoken Language Models: A Comprehensive Survey

@article{
arora2025on,
title={On The Landscape of Spoken Language Models: A Comprehensive Survey},
author={Siddhant Arora and Kai-Wei Chang and Chung-Ming Chien and Yifan Peng and Haibin Wu and Yossi Adi and Emmanuel Dupoux and Hung-yi Lee and Karen Livescu and Shinji Watanabe},
journal={Transactions on Machine Learning Research},
issn={2835-8856},
year={2025},
url={https://openreview.net/forum?id=BvxaP3sVbA},
note={}
}

T5G2P: Using Text-to-Text Transfer Transformer for Grapheme-to-Phoneme Conversion

@inproceedings{rezackova21_interspeech,
  title     = ,
  author    = {Markéta Řezáčková and Jan Švec and Daniel Tihelka},
  year      = {2021},
  booktitle = ,
  pages     = {6--10},
  doi       = {10.21437/Interspeech.2021-546},
  issn      = {2958-1796},
}

The Swedish Parliament Corpus 1867 – 2022

@inproceedings{yrjanainen-etal-2024-swedish,
    title = "The {S}wedish Parliament Corpus 1867 {--} 2022",
    author = {Yrj{\"a}n{\"a}inen, V{\"a}in{\"o} Aleksi  and
      Mohammadi Nor{\'e}n, Fredrik  and
      Borges, Robert  and
      Jarlbrink, Johan  and
      {\r{A}}berg Brorsson, Lotta  and
      Olsson, Anders P.  and
      Snickars, Pelle  and
      Magnusson, M{\r{a}}ns},
    editor = "Calzolari, Nicoletta  and
      Kan, Min-Yen  and
      Hoste, Veronique  and
      Lenci, Alessandro  and
      Sakti, Sakriani  and
      Xue, Nianwen",
    booktitle = "Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)",
    month = may,
    year = "2024",
    address = "Torino, Italia",
    publisher = "ELRA and ICCL",
    url = "https://aclanthology.org/2024.lrec-main.1400/",
    pages = "16100--16112",
}

Google’s Multilingual Neural Machine Translation System: Enabling Zero-Shot Translation

@article{johnson-etal-2017-googles,
    title = "{G}oogle{'}s Multilingual Neural Machine Translation System: Enabling Zero-Shot Translation",
    author = "Johnson, Melvin  and
      Schuster, Mike  and
      Le, Quoc V.  and
      Krikun, Maxim  and
      Wu, Yonghui  and
      Chen, Zhifeng  and
      Thorat, Nikhil  and
      Vi{\'e}gas, Fernanda  and
      Wattenberg, Martin  and
      Corrado, Greg  and
      Hughes, Macduff  and
      Dean, Jeffrey",
    editor = "Lee, Lillian  and
      Johnson, Mark  and
      Toutanova, Kristina",
    journal = "Transactions of the Association for Computational Linguistics",
    volume = "5",
    year = "2017",
    address = "Cambridge, MA",
    publisher = "MIT Press",
    url = "https://aclanthology.org/Q17-1024/",
    doi = "10.1162/tacl_a_00065",
    pages = "339--351",
}

Codec-ASR: Training Performant Automatic Speech Recognition Systems with Discrete Speech Representations

@inproceedings{dhawan24_interspeech,
  title     = ,
  author    = {Kunal Dhawan and Nithin Rao Koluguri and Ante Jukić and Ryan Langman and Jagadeesh Balam and Boris Ginsburg},
  year      = {2024},
  booktitle = ,
  pages     = {2574--2578},
  doi       = {10.21437/Interspeech.2024-330},
  issn      = {2958-1796},
}

SPACER: A Parallel Dataset of Speech Production And Comprehension of Error Repairs

@inproceedings{upadhye-etal-2025-spacer,
    title = "{SPACER}: A Parallel Dataset of Speech Production And Comprehension of Error Repairs",
    author = "Upadhye, Shiva  and
      Li, Jiaxuan  and
      Futrell, Richard",
    editor = "Kuribayashi, Tatsuki  and
      Rambelli, Giulia  and
      Takmaz, Ece  and
      Wicke, Philipp  and
      Li, Jixing  and
      Oh, Byung-Doh",
    booktitle = "Proceedings of the Workshop on Cognitive Modeling and Computational Linguistics",
    month = may,
    year = "2025",
    address = "Albuquerque, New Mexico, USA",
    publisher = "Association for Computational Linguistics",
    url = "https://aclanthology.org/2025.cmcl-1.19/",
    doi = "10.18653/v1/2025.cmcl-1.19",
    pages = "144--154",
    ISBN = "979-8-89176-227-5",
}

Pronunciation adaptation at the lexical level

@inproceedings{strik01_adaptation,
  title     = ,
  author    = {Helmer Strik},
  year      = {2001},
  booktitle = ,
  pages     = {123--130},
}

Unifying Diarization, Separation, and ASR with Multi-Speaker Encoder

@misc{shakeel2025unifying,
title={Unifying Diarization, Separation, and {ASR} with Multi-Speaker Encoder},
author={Muhammad Shakeel and Yui Sudo and Yifan Peng and Chyi-Jiunn Lin and Shinji Watanabe},
year={2025},
url={https://openreview.net/forum?id=5oaUMZEjWe}
}

Hierarchical Phrase-based Translation Representations

@inproceedings{iglesias-etal-2011-hierarchical,
    title = "Hierarchical Phrase-based Translation Representations",
    author = "Iglesias, Gonzalo  and
      Allauzen, Cyril  and
      Byrne, William  and
      de Gispert, Adri{\`a}  and
      Riley, Michael",
    editor = "Barzilay, Regina  and
      Johnson, Mark",
    booktitle = "Proceedings of the 2011 Conference on Empirical Methods in Natural Language Processing",
    month = jul,
    year = "2011",
    address = "Edinburgh, Scotland, UK.",
    publisher = "Association for Computational Linguistics",
    url = "https://aclanthology.org/D11-1127/",
    pages = "1373--1383"
}

ByT5 model for massively multilingual grapheme-to-phoneme conversion

@inproceedings{zhu22_interspeech,
  title     = ,
  author    = {Jian Zhu and Cong Zhang and David Jurgens},
  year      = {2022},
  booktitle = ,
  pages     = {446--450},
  doi       = {10.21437/Interspeech.2022-538},
  issn      = {2958-1796},
}

WavLLM: Towards Robust and Adaptive Speech Large Language Model

@inproceedings{hu-etal-2024-wavllm,
    title = "{W}av{LLM}: Towards Robust and Adaptive Speech Large Language Model",
    author = "Hu, Shujie  and
      Zhou, Long  and
      Liu, Shujie  and
      Chen, Sanyuan  and
      Meng, Lingwei  and
      Hao, Hongkun  and
      Pan, Jing  and
      Liu, Xunying  and
      Li, Jinyu  and
      Sivasankaran, Sunit  and
      Liu, Linquan  and
      Wei, Furu",
    editor = "Al-Onaizan, Yaser  and
      Bansal, Mohit  and
      Chen, Yun-Nung",
    booktitle = "Findings of the Association for Computational Linguistics: EMNLP 2024",
    month = nov,
    year = "2024",
    address = "Miami, Florida, USA",
    publisher = "Association for Computational Linguistics",
    url = "https://aclanthology.org/2024.findings-emnlp.263/",
    doi = "10.18653/v1/2024.findings-emnlp.263",
    pages = "4552--4572",
}

The taste of IPA: Towards open-vocabulary keyword spotting and forced alignment in any language

@inproceedings{zhu-etal-2024-taste,
    title = "The taste of {IPA}: Towards open-vocabulary keyword spotting and forced alignment in any language",
    author = "Zhu, Jian  and
      Yang, Changbing  and
      Samir, Farhan  and
      Islam, Jahurul",
    editor = "Duh, Kevin  and
      Gomez, Helena  and
      Bethard, Steven",
    booktitle = "Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)",
    month = jun,
    year = "2024",
    address = "Mexico City, Mexico",
    publisher = "Association for Computational Linguistics",
    url = "https://aclanthology.org/2024.naacl-long.43/",
    doi = "10.18653/v1/2024.naacl-long.43",
    pages = "750--772",
}

LiteASR: Efficient Automatic Speech Recognition with Low-Rank Approximation

@inproceedings{kamahori-etal-2025-liteasr,
    title = "{L}ite{ASR}: Efficient Automatic Speech Recognition with Low-Rank Approximation",
    author = "Kamahori, Keisuke  and
      Kasai, Jungo  and
      Kojima, Noriyuki  and
      Kasikci, Baris",
    editor = "Christodoulopoulos, Christos  and
      Chakraborty, Tanmoy  and
      Rose, Carolyn  and
      Peng, Violet",
    booktitle = "Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing",
    month = nov,
    year = "2025",
    address = "Suzhou, China",
    publisher = "Association for Computational Linguistics",
    url = "https://aclanthology.org/2025.emnlp-main.169/",
    doi = "10.18653/v1/2025.emnlp-main.169",
    pages = "3430--3442",
    ISBN = "979-8-89176-332-6",
}

Gaussian Distributions are Soap Bubbles

mixup: Data-Dependent Data Augmentation

Recent Advances in Discrete Speech Tokens: A Review

@ARTICLE{11298521,
author={Guo, Yiwei and Li, Zhihan and Wang, Hankun and Li, Bohan and Shao, Chongtian and Zhang, Hanglei and Du, Chenpeng and Chen, Xie and Liu, Shujie and Yu, Kai},
journal={ IEEE Transactions on Pattern Analysis \& Machine Intelligence },
title=,
year={2026},
volume={48},
number={04},
ISSN={1939-3539},
pages={4184-4204},
doi={10.1109/TPAMI.2025.3643619},
url = {https://doi.ieeecomputersociety.org/10.1109/TPAMI.2025.3643619},
publisher={IEEE Computer Society},
address={Los Alamitos, CA, USA},
month=apr}

How Neural Networks Learn: A Probabilistic Viewpoint

Dynamic pronunciation models for automatic speech recognition

  • Chapters 3 & 4
@phdthesis{10.5555/931366,
author = {Fosler-Lussier, John Eric and Morgan, Nelson},
title = {Dynamic pronunciation models for automatic speech recognition},
year = {1999},
isbn = {059971154X},
publisher = {University of California, Berkeley},
}