{"schema_version":1,"site_url":"https://interspeech-2026-wiki.vercel.app","count":1379,"papers":[{"id":"a26_interspeech","title":"MTC-AVSR: Compressed-Token-based Audio-Visual Speech Recognition and Translation with Contrastive Language Alignment","authors":["Lusi A","Zhiyong Duan","Jiang Li","Feilong Bao"],"year":2026,"doi":"10.21437/Interspeech.2026-266","isca_url":"https://www.isca-archive.org/interspeech_2026/a26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/a26_interspeech.pdf","session":"Robust Audio-Visual Speech Recognition","topics":["asr","speech-translation","multilingual"],"category":"asr","labels":["multilingual","efficient-on-device"],"institutions":["Inner Mongolia University","Inner Mongolia Arts University"],"funding":["National Natural Science Foundation","Inner Mongolia Autonomous Region Major Special Science and Technology Project","Inner Mongolia Autonomous Region Natural Science Foundation Project","Inner Mongolia Autonomous Region Science and Technology Plan Project","Inner Mongolia Autonomous Region First-Class Discipline Scientific Research Special Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"a26_interspeech","category":"asr","labels":["multilingual","efficient-on-device"],"institutions":["Inner Mongolia University","Inner Mongolia Arts University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-266","pdf":"https://www.isca-archive.org/interspeech_2026/a26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/a26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/a26_interspeech/markdown.md"},{"id":"adebara26_interspeech","title":"WazobiaSpeech: A Large-Scale Multilingual Speech Corpus for Robust and Fair ASR in Four Nigerian Languages","authors":["Ife Adebara","Oluwaseun Nifemi","Olubayo Adekanmbi","Rashidat Sikiru","Ololade Anjuwon","Ronke Akinmosin","Faiza Sani"],"year":2026,"doi":"10.21437/Interspeech.2026-3519","isca_url":"https://www.isca-archive.org/interspeech_2026/adebara26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/adebara26_interspeech.pdf","session":"Multilingual Speech 1","topics":["asr","multilingual","dataset"],"category":"resources-evaluation","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["University of Alberta","Data Science Nigeria","EqualyzAI","Alberta Machine Intelligence Institute","CIFAR"],"funding":["Gates Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"adebara26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["University of Alberta","Data Science Nigeria","EqualyzAI","Alberta Machine Intelligence Institute","CIFAR"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3519","pdf":"https://www.isca-archive.org/interspeech_2026/adebara26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/adebara26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/adebara26_interspeech/markdown.md"},{"id":"adelson26_interspeech","title":"Beyond Deep Learning: Speech Segmentation and Phone Classification with Neural Assemblies","authors":["Trevor Adelson","Vidhyasaharan Sethu","Ting Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-2041","isca_url":"https://www.isca-archive.org/interspeech_2026/adelson26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/adelson26_interspeech.pdf","session":"Audio Coding and Signal Analysis","topics":["asr","self-supervised","on-device"],"category":"asr","institutions":["University of Melbourne","University of New South Wales"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"adelson26_interspeech","category":"asr","institutions":["University of Melbourne","University of New South Wales"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2041","pdf":"https://www.isca-archive.org/interspeech_2026/adelson26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/adelson26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/adelson26_interspeech/markdown.md"},{"id":"agarwal26_interspeech","title":"Grounding Whisper: An Audio Anchor-Based Approach for Hallucination Mitigation and Throughput-Efficient ASR","authors":["Prateek Agarwal","Saurabh Kumar","Priyanka Bhatt"],"year":2026,"doi":"10.21437/Interspeech.2026-1314","isca_url":"https://www.isca-archive.org/interspeech_2026/agarwal26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/agarwal26_interspeech.pdf","session":"Robust ASR: Hallucinations and Biases","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["self-supervised"],"institutions":["Walmart Global Tech"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"agarwal26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Walmart Global Tech"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1314","pdf":"https://www.isca-archive.org/interspeech_2026/agarwal26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/agarwal26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/agarwal26_interspeech/markdown.md"},{"id":"aghniya26_interspeech","title":"GRATS : A Natural Multi-Speed Mandarin Dataset for Speech Time-Scale Modification Benchmarking","authors":["Ghaida Fathin Aghniya","Dyah A. M. G. Wisnu","Stefano Rini","Yu Tsao"],"year":2026,"doi":"10.21437/Interspeech.2026-1842","isca_url":"https://www.isca-archive.org/interspeech_2026/aghniya26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/aghniya26_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["speech-enhancement","evaluation","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["National Yang Ming Chiao Tung University","Academia Sinica"],"funding":["Bio-ASP Lab at Academia Sinica","National Science and Technology Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"aghniya26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["National Yang Ming Chiao Tung University","Academia Sinica"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1842","pdf":"https://www.isca-archive.org/interspeech_2026/aghniya26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/aghniya26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/aghniya26_interspeech/markdown.md"},{"id":"agrawal26_interspeech","title":"Amadea: An AI Companion for Pitch-Aware Spoken Language Practice","authors":["Mrigendra Agrawal","Ryan Tsui","Kyaw Maung Maung Tet Toe"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/agrawal26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/agrawal26_interspeech.pdf","session":"Speech and Language Learning Technologies","topics":["prosody","speech-feedback","conversational-ai"],"category":"applications-other","labels":["streaming-real-time"],"institutions":["Amadea","University of Queensland","University of Edinburgh"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"agrawal26_interspeech","category":"applications-other","labels":["streaming-real-time"],"institutions":["Amadea","University of Queensland","University of Edinburgh"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/agrawal26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/agrawal26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/agrawal26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/agrawal26_interspeech/markdown.md"},{"id":"aheibam26_interspeech","title":"From Rhythm Metrics to Latent Embeddings: Categorising English and Hindi Varieties in Northeast India","authors":["John Aheibam","Joyshree Chakraborty","Priyankoo Sarmah","Rohit Sinha"],"year":2026,"doi":"10.21437/Interspeech.2026-3462","isca_url":"https://www.isca-archive.org/interspeech_2026/aheibam26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/aheibam26_interspeech.pdf","session":"Cross-Linguistic and L2 Phonetic Studies","topics":["speech-translation","self-supervised","multilingual"],"category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["Indian Institute of Technology Guwahati"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"aheibam26_interspeech","category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["Indian Institute of Technology Guwahati"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3462","pdf":"https://www.isca-archive.org/interspeech_2026/aheibam26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/aheibam26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/aheibam26_interspeech/markdown.md"},{"id":"ahn26_interspeech","title":"Context-Adaptive Automated Audio Captioning with Symmetric Dual-MoE and Dynamic Reward Routing","authors":["Seyun Ahn","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-2897","isca_url":"https://www.isca-archive.org/interspeech_2026/ahn26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ahn26_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["audio-captioning","self-supervised","speech-llm"],"category":"audio-understanding","labels":["self-supervised","generative-model"],"institutions":["Hanyang University"],"funding":["National Research Foundation of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ahn26_interspeech","category":"audio-understanding","labels":["self-supervised","generative-model"],"institutions":["Hanyang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2897","pdf":"https://www.isca-archive.org/interspeech_2026/ahn26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ahn26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ahn26_interspeech/markdown.md"},{"id":"ahn26b_interspeech","title":"Whisper-CD: Accurate Long-Form Speech Recognition using Multi-Negative Contrastive Decoding","authors":["Hoseong Ahn","Jeongyun Chae","Yoonji Park","Kyuhong Shim"],"year":2026,"doi":"10.21437/Interspeech.2026-3058","isca_url":"https://www.isca-archive.org/interspeech_2026/ahn26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ahn26b_interspeech.pdf","session":"Search Methods and Inference Algorithms","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["self-supervised"],"institutions":["Sungkyunkwan University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ahn26b_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Sungkyunkwan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3058","pdf":"https://www.isca-archive.org/interspeech_2026/ahn26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ahn26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ahn26b_interspeech/markdown.md"},{"id":"ai26_interspeech","title":"Stabilizing Short Duration Speaker Verification through Neural Re-scoring with Hybrid Enrollment","authors":["Zhiqi Ai","Cheng Han","Shiyi Mu","Zhiyong Chen","Yongjin Zhou","Shugong Xu"],"year":2026,"doi":"10.21437/Interspeech.2026-228","isca_url":"https://www.isca-archive.org/interspeech_2026/ai26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ai26_interspeech.pdf","session":"Speaker Recognition and Verification","topics":["speaker-verification","keyword-spotting","dataset"],"category":"speaker","labels":["dataset-or-benchmark-release"],"institutions":["Shanghai University","Xi'an Jiaotong-Liverpool University","Hithink RoyalFlush AI Research Institute"],"funding":["Shanghai Municipal Science and Technology Commission","National High Quality Program","Xi'an Jiaotong-Liverpool University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ai26_interspeech","category":"speaker","labels":["dataset-or-benchmark-release"],"institutions":["Shanghai University","Xi'an Jiaotong-Liverpool University","Hithink RoyalFlush AI Research Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-228","pdf":"https://www.isca-archive.org/interspeech_2026/ai26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ai26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ai26_interspeech/markdown.md"},{"id":"ai26b_interspeech","title":"Beyond Two-stage Diffusion TTS: Joint Structure and Content Refinement via Jump Diffusion","authors":["Jiabao Ai","Minghui Zhao","Anton Ragni"],"year":2026,"doi":"10.21437/Interspeech.2026-2875","isca_url":"https://www.isca-archive.org/interspeech_2026/ai26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ai26b_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["tts","self-supervised","prosody"],"category":"tts","labels":["generative-model"],"institutions":["University of Sheffield"],"funding":["UK Research and Innovation","UKRI AI Centre for Doctoral Training in Speech and Language Technologies (SLT) and their Applications"],"code":{"url":"https://anonymousinterpseech.github.io/TTS_Demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ai26b_interspeech","category":"tts","labels":["generative-model"],"institutions":["University of Sheffield"],"code":"https://anonymousinterpseech.github.io/TTS_Demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2875","pdf":"https://www.isca-archive.org/interspeech_2026/ai26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ai26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ai26b_interspeech/markdown.md"},{"id":"akhtar26_interspeech","title":"From Signals to Patterns: Non-Invasive Tuberculosis Detection from Cough Audio using Bandit Weighted Hyperbolic Prototypes","authors":["Mohd Mujtaba Akhtar","Girish","Sanjam Wadhwa","Muskaan Singh","Ning Ma"],"year":2026,"doi":"10.21437/Interspeech.2026-2704","isca_url":"https://www.isca-archive.org/interspeech_2026/akhtar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/akhtar26_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 1","topics":["speech-llm","self-supervised","health"],"category":"health-clinical","labels":["self-supervised"],"institutions":["Ulster University","Thapar Institute of Engineering and Technology","University of Sheffield"],"funding":["United States–Ireland–Northern Ireland R&D Partnership Programme","Engineering and Physical Sciences Research Council"],"code":{"url":"https://github.com/Helixometry/COBALT.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"akhtar26_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["Ulster University","Thapar Institute of Engineering and Technology","University of Sheffield"],"code":"https://github.com/Helixometry/COBALT.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2704","pdf":"https://www.isca-archive.org/interspeech_2026/akhtar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/akhtar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/akhtar26_interspeech/markdown.md"},{"id":"akti26_interspeech","title":"Synthesizing the Lombard Effect: Multi-Level Control of Speech Clarity and Vocal Effort in TTS","authors":["Seymanur Akti","Alexander Waibel"],"year":2026,"doi":"10.21437/Interspeech.2026-1159","isca_url":"https://www.isca-archive.org/interspeech_2026/akti26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/akti26_interspeech.pdf","session":"Controllable and Expressive Speech Synthesis","topics":["tts","self-supervised","multilingual"],"category":"tts","labels":["generative-model","robustness-noise"],"institutions":["Karlsruhe Institute of Technology","Carnegie Mellon University","KIT Campus Transfer"],"funding":["European Union Horizon Europe programme","KIT Campus Transfer GmbH"],"code":{"url":"https://seymanurakti.github.io/synthesizing-lombard-effect/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"akti26_interspeech","category":"tts","labels":["generative-model","robustness-noise"],"institutions":["Karlsruhe Institute of Technology","Carnegie Mellon University","KIT Campus Transfer"],"code":"https://seymanurakti.github.io/synthesizing-lombard-effect/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1159","pdf":"https://www.isca-archive.org/interspeech_2026/akti26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/akti26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/akti26_interspeech/markdown.md"},{"id":"alali26_interspeech","title":"Personal Attribute Leakage in Federated Speech Models","authors":["Hamdan Al-Ali","Ali Reza Ghavamipour","Tommaso Caselli","Fatih Turkmen","Zeerak Talat","Hanan Aldarmaki"],"year":2026,"doi":"10.21437/Interspeech.2026-327","isca_url":"https://www.isca-archive.org/interspeech_2026/alali26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/alali26_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["asr","self-supervised","evaluation"],"category":"deepfake-security","institutions":["Mohamed bin Zayed University of Artificial Intelligence","Maastricht University","University of Groningen","University of Edinburgh"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"alali26_interspeech","category":"deepfake-security","institutions":["Mohamed bin Zayed University of Artificial Intelligence","Maastricht University","University of Groningen","University of Edinburgh"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-327","pdf":"https://www.isca-archive.org/interspeech_2026/alali26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/alali26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/alali26_interspeech/markdown.md"},{"id":"aldeneh26_interspeech","title":"Which Data Matter? Embedding-Based Data Selection for Speech Recognition","authors":["Zakaria Aldeneh","Skyler Seto","Maureen de Seyssel","Jie Chi","Zijin Gu","Takuya Higuchi","Jee-weon Jung","Shinji Watanabe","David Grangier","Barry-John Theobald","Tatiana Likhomanenko"],"year":2026,"doi":"10.21437/Interspeech.2026-3073","isca_url":"https://www.isca-archive.org/interspeech_2026/aldeneh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/aldeneh26_interspeech.pdf","session":"Robust and Efficient ASR","topics":["asr","self-supervised","dataset"],"category":"asr","labels":["self-supervised"],"institutions":["Apple","Carnegie Mellon University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"aldeneh26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Apple","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3073","pdf":"https://www.isca-archive.org/interspeech_2026/aldeneh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/aldeneh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/aldeneh26_interspeech/markdown.md"},{"id":"alhabshi26_interspeech","title":"Multilingual and Cross-lingual Lexical Stress Detection Using SSL Feature Vectors","authors":["Abdulrahman Alhabshi","Beena Ahmed","Mostafa Shahin"],"year":2026,"doi":"10.21437/Interspeech.2026-2914","isca_url":"https://www.isca-archive.org/interspeech_2026/alhabshi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/alhabshi26_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["phonetics","prosody","self-supervised"],"category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["University of New South Wales"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"alhabshi26_interspeech","category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["University of New South Wales"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2914","pdf":"https://www.isca-archive.org/interspeech_2026/alhabshi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/alhabshi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/alhabshi26_interspeech/markdown.md"},{"id":"alhammad26_interspeech","title":"Interpretable Frequency-Band Attention with Gated SSL Fusion for Audio Deepfake Detection","authors":["Abeer Alhammad","Abdullah Aldahlawi"],"year":2026,"doi":"10.21437/Interspeech.2026-2250","isca_url":"https://www.isca-archive.org/interspeech_2026/alhammad26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/alhammad26_interspeech.pdf","session":"Spoofing and Deepfake Detection 3","topics":["audio-deepfake","self-supervised","speaker-verification"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["Thaka"],"funding":["Thaka"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"alhammad26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["Thaka"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2250","pdf":"https://www.isca-archive.org/interspeech_2026/alhammad26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/alhammad26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/alhammad26_interspeech/markdown.md"},{"id":"alharthi26_interspeech","title":"RIVET: Robust Idempotent Voice Attribute Editing","authors":["Dareen Alharthi","Bhuvan Koduru","Rita Singh","Bhiksha Ramakrishnan"],"year":2026,"doi":"10.21437/Interspeech.2026-395","isca_url":"https://www.isca-archive.org/interspeech_2026/alharthi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/alharthi26_interspeech.pdf","session":"Voice Editing","topics":["voice-conversion","self-supervised","speaker-verification"],"category":"tts","labels":["generative-model"],"institutions":["Carnegie Mellon University"],"code":{"url":"https://github.com/DareenHarthi/rivet","stars":6,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"alharthi26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Carnegie Mellon University"],"code":"https://github.com/DareenHarthi/rivet","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-395","pdf":"https://www.isca-archive.org/interspeech_2026/alharthi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/alharthi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/alharthi26_interspeech/markdown.md"},{"id":"ali26_interspeech","title":"Fed-SpeechLLM: Federated Learning Speech Language Models for Multilingual ASR","authors":["Mohamed Nabih Ali","Daniele Falavigna","Alessio Brutti"],"year":2026,"doi":"10.21437/Interspeech.2026-689","isca_url":"https://www.isca-archive.org/interspeech_2026/ali26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ali26_interspeech.pdf","session":"Robust and Real-World ASR Systems","topics":["asr","speech-llm","multilingual"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["Fondazione Bruno Kessler"],"funding":["Ministero delle Imprese e del Made in Italy","European Union"],"code":{"url":"https://github.com/mnabihali/Fed-SpeechLLM","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ali26_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["Fondazione Bruno Kessler"],"code":"https://github.com/mnabihali/Fed-SpeechLLM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-689","pdf":"https://www.isca-archive.org/interspeech_2026/ali26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ali26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ali26_interspeech/markdown.md"},{"id":"ali26b_interspeech","title":"MambAdapter: Lightweight Mamba-Based Adapters for Parameter-Efficient Transfer Learning in Speech and Audio","authors":["Salman Hussain Ali","Umberto Cappellazzo","Mirco Ravanelli"],"year":2026,"doi":"10.21437/Interspeech.2026-1522","isca_url":"https://www.isca-archive.org/interspeech_2026/ali26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ali26b_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 1","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["low-resource","multilingual","efficient-on-device","self-supervised"],"institutions":["Universite de Montreal","Imperial College London","Concordia University","Mila – Quebec AI Institute"],"funding":["NSERC","Digital Research Alliance of Canada","Translated","Apple"],"code":{"url":"https://github.com/salman-ha/MambAdapter","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ali26b_interspeech","category":"asr","labels":["low-resource","multilingual","efficient-on-device","self-supervised"],"institutions":["Universite de Montreal","Imperial College London","Concordia University","Mila – Quebec AI Institute"],"code":"https://github.com/salman-ha/MambAdapter","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1522","pdf":"https://www.isca-archive.org/interspeech_2026/ali26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ali26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ali26b_interspeech/markdown.md"},{"id":"ali26c_interspeech","title":"WASIL: In-the-Wild Arabic Spoken Interactions with LLMs","authors":["Zien Sheikh Ali","Hamdy Mubarak","Soon-Gyo Jung","Hunzalah Hassan Bhatti","Firoj Alam","Shammur Absar Chowdhury"],"year":2026,"doi":"10.21437/Interspeech.2026-2694","isca_url":"https://www.isca-archive.org/interspeech_2026/ali26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ali26c_interspeech.pdf","session":"LLMs and Conversational Interaction","topics":["speech-llm","multilingual","evaluation"],"category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Qatar Computing Research Institute"],"code":{"url":"https://huggingface.co/datasets/QCRI/WASIL","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ali26c_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Qatar Computing Research Institute"],"code":"https://huggingface.co/datasets/QCRI/WASIL","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2694","pdf":"https://www.isca-archive.org/interspeech_2026/ali26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ali26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ali26c_interspeech/markdown.md"},{"id":"alizadeh26_interspeech","title":"The Impact of Informal Persian Speech on Low-Resource ASR and Speech Translation","authors":["Hadi Alizadeh","Mohammad Asgari","Mohammad Sadegh Mehrabikia","Amir Koohnavard"],"year":2026,"doi":"10.21437/Interspeech.2026-2454","isca_url":"https://www.isca-archive.org/interspeech_2026/alizadeh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/alizadeh26_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["asr","speech-translation","low-resource"],"category":"asr","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Toorintan","Islamic Republic of Iran Broadcasting University"],"funding":["Toorintan Company"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"alizadeh26_interspeech","category":"asr","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Toorintan","Islamic Republic of Iran Broadcasting University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2454","pdf":"https://www.isca-archive.org/interspeech_2026/alizadeh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/alizadeh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/alizadeh26_interspeech/markdown.md"},{"id":"allen26_interspeech","title":"Bilingual Speaker Phonetic Alignment to Voice Assistants","authors":["Alyssa Allen","Kathryn Campbell-Kibler"],"year":2026,"doi":"10.21437/Interspeech.2026-2391","isca_url":"https://www.isca-archive.org/interspeech_2026/allen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/allen26_interspeech.pdf","session":"Pronunciation Diversity","topics":["multilingual","self-supervised","evaluation"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Ohio State University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"allen26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Ohio State University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2391","pdf":"https://www.isca-archive.org/interspeech_2026/allen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/allen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/allen26_interspeech/markdown.md"},{"id":"alshubaily26_interspeech","title":"The SSPNet Speaker Personality Corpus Version 2: Investigating the Role of Language Understanding in Automatic Personality Perception","authors":["Nisreen Alshubaily","Emily O'Hara","Tanaya Guha","Alessandro Vinciarelli"],"year":2026,"doi":"10.21437/Interspeech.2026-383","isca_url":"https://www.isca-archive.org/interspeech_2026/alshubaily26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/alshubaily26_interspeech.pdf","session":"Speaker Identity, States, and Traits in Paralinguistics","topics":["paralinguistics","speech-llm","dataset"],"category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["University of Glasgow","Imam Mohammad Ibn Saud Islamic University"],"funding":["UKRI Centre for Doctoral Training in Socially Intelligent Artificial Agents"],"code":{"url":"https://github.com/SocialAI-Glasgow/SSPNet_SPC2.0","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"alshubaily26_interspeech","category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["University of Glasgow","Imam Mohammad Ibn Saud Islamic University"],"code":"https://github.com/SocialAI-Glasgow/SSPNet_SPC2.0","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-383","pdf":"https://www.isca-archive.org/interspeech_2026/alshubaily26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/alshubaily26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/alshubaily26_interspeech/markdown.md"},{"id":"altwlkany26_interspeech","title":"Leveraging Discriminative Capabilities of Self-Supervised Neural Audio Fingerprinting for Efficient Speech Data Annotation","authors":["Kemal Altwlkany","Elmedin Selmanovic"],"year":2026,"doi":"10.21437/Interspeech.2026-1436","isca_url":"https://www.isca-archive.org/interspeech_2026/altwlkany26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/altwlkany26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["self-supervised","speaker-verification","dataset"],"category":"resources-evaluation","labels":["self-supervised"],"institutions":["Infobip","University of Sarajevo"],"funding":["Infobip Global Communication Platform","Important Project of Common European Interest on Next Generation Cloud Infrastructure and Services"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"altwlkany26_interspeech","category":"resources-evaluation","labels":["self-supervised"],"institutions":["Infobip","University of Sarajevo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1436","pdf":"https://www.isca-archive.org/interspeech_2026/altwlkany26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/altwlkany26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/altwlkany26_interspeech/markdown.md"},{"id":"alyafeai26_interspeech","title":"Hamsa: A Manually Annotated Emirati Arabic Corpus for Speech and Language Technologies","authors":["Mohammed Alyafeai","Hamza Alobeidli","Omar Alkaabi","Shaikha Alsuwaidi","Leen AlQadi","Ahmed Alzubaidi","Maitha Alhammadi","Basma El Amel Boussaha","Hakim Hacid"],"year":2026,"doi":"10.21437/Interspeech.2026-1049","isca_url":"https://www.isca-archive.org/interspeech_2026/alyafeai26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/alyafeai26_interspeech.pdf","session":"Speech and Language Representation","topics":["asr","low-resource","multilingual"],"category":"resources-evaluation","labels":["low-resource","self-supervised","dataset-or-benchmark-release"],"institutions":["Technology Innovation Institute"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"alyafeai26_interspeech","category":"resources-evaluation","labels":["low-resource","self-supervised","dataset-or-benchmark-release"],"institutions":["Technology Innovation Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1049","pdf":"https://www.isca-archive.org/interspeech_2026/alyafeai26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/alyafeai26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/alyafeai26_interspeech/markdown.md"},{"id":"anand26_interspeech","title":"ParA-LLM: A Unified Approach to Paralinguistic and Acoustic Speech Understanding","authors":["Nishit Anand","Jiaqi Su","Ke Chen","Yunyun Wang","Dinesh Manocha","Ramani Duraiswami","Rithesh Kumar","Zeyu Jin"],"year":2026,"doi":"10.21437/Interspeech.2026-3015","isca_url":"https://www.isca-archive.org/interspeech_2026/anand26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/anand26_interspeech.pdf","session":"Paralinguistics","topics":["paralinguistics","speech-llm","self-supervised"],"category":"paralinguistics-emotion","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["Adobe Research","University of Maryland, College Park","OpenAI"],"code":{"url":"https://nishitanand.github.io/paralinguistic-understanding-llm","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"anand26_interspeech","category":"paralinguistics-emotion","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["Adobe Research","University of Maryland, College Park","OpenAI"],"code":"https://nishitanand.github.io/paralinguistic-understanding-llm","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3015","pdf":"https://www.isca-archive.org/interspeech_2026/anand26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/anand26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/anand26_interspeech/markdown.md"},{"id":"anand26b_interspeech","title":"Preferences of a Voice-First Nation: Large-Scale Pairwise Evaluation and Preference Analysis for TTS in Indian Languages","authors":["Srija Anand","Ashwin Sankar","Ishvinder Sethi","Aaditya Pareek","Kartik Rajput","Gaurav Yadav","Nikhil Narasimhan","Adish Pandya","Deepon Halder","Mohammed Safi Ur Rahman Khan","Praveen Srinivasa Varadhan","Shobhit Banga","Mitesh M Khapra"],"year":2026,"doi":"10.21437/Interspeech.2026-3357","isca_url":"https://www.isca-archive.org/interspeech_2026/anand26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/anand26b_interspeech.pdf","session":"Speech Synthesis Evaluation 1","topics":["tts","multilingual","evaluation"],"category":"tts","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Indian Institute of Technology Madras","AI4Bharat","Josh Talks"],"funding":["EkStep Foundation","Nilekani Philanthropies"],"code":{"url":"https://huggingface.co/datasets/ai4bharat/SpeechArenaBench/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"anand26b_interspeech","category":"tts","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Indian Institute of Technology Madras","AI4Bharat","Josh Talks"],"code":"https://huggingface.co/datasets/ai4bharat/SpeechArenaBench/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3357","pdf":"https://www.isca-archive.org/interspeech_2026/anand26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/anand26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/anand26b_interspeech/markdown.md"},{"id":"andrusenko26_interspeech","title":"Reducing the Offline-Streaming Gap for Unified ASR Transducer with Consistency Regularization","authors":["Andrei Andrusenko","Vladimir Bataev","Lilit Grigoryan","Nune Tadevosyan","Vitaly Lavrukhin","Boris Ginsburg"],"year":2026,"doi":"10.21437/Interspeech.2026-1195","isca_url":"https://www.isca-archive.org/interspeech_2026/andrusenko26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/andrusenko26_interspeech.pdf","session":"New Training Methods for ASR","topics":["asr","self-supervised","streaming-inference"],"category":"asr","labels":["streaming-real-time"],"institutions":["NVIDIA"],"code":{"url":"https://huggingface.co/nvidia/parakeet-unified-en-0.6b","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"andrusenko26_interspeech","category":"asr","labels":["streaming-real-time"],"institutions":["NVIDIA"],"code":"https://huggingface.co/nvidia/parakeet-unified-en-0.6b","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1195","pdf":"https://www.isca-archive.org/interspeech_2026/andrusenko26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/andrusenko26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/andrusenko26_interspeech/markdown.md"},{"id":"angus26_interspeech","title":"Argmax Pro: Frontier-level Real-time Speech-to-text with Speakers and Custom Vocabulary on Mobile Devices","authors":["Dylan Angus","Chen Cen","Berkin Durmus","Arda Ibis","Brian Keene","Andrey Leonov","Blaise Munyampirwa","Zach Nagengast","Arda Okan","Atila Orhon","Eduardo Pacheco"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/angus26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/angus26_interspeech.pdf","session":"Speech Recognition, Enhancement and Real-Time Systems","topics":["asr","speaker-diarization","on-device"],"category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Argmax"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"angus26_interspeech","category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Argmax"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/angus26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/angus26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/angus26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/angus26_interspeech/markdown.md"},{"id":"annamdevula26_interspeech","title":"CrossAccent-TTS: Cross-Lingual Accent-Intensity Controllable Text-to-Speech via Disentangled Speaker and Accent Representations","authors":["Ram Annamdevula","Ankit Tatawat","Ashishkumar P. Gudmalwar","Nirmesh J. Shah","Pankaj Wasnik"],"year":2026,"doi":"10.21437/Interspeech.2026-1744","isca_url":"https://www.isca-archive.org/interspeech_2026/annamdevula26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/annamdevula26_interspeech.pdf","session":"Controllable and Expressive Speech Synthesis","topics":["tts","speech-llm","multilingual"],"category":"tts","labels":["multilingual","generative-model"],"institutions":["Sony"],"code":{"url":"https://research.sri-media-analysis.com/interspeech26-cross-accent-tts/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"annamdevula26_interspeech","category":"tts","labels":["multilingual","generative-model"],"institutions":["Sony"],"code":"https://research.sri-media-analysis.com/interspeech26-cross-accent-tts/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1744","pdf":"https://www.isca-archive.org/interspeech_2026/annamdevula26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/annamdevula26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/annamdevula26_interspeech/markdown.md"},{"id":"aparin26_interspeech","title":"Whisper Hallucination Detection and Mitigation via Hidden Representation Steering and Sparse AutoEncoders","authors":["Georgii Aparin","Vadim Popov","Tasnima Sadekova","Assel Yermekova"],"year":2026,"doi":"10.21437/Interspeech.2026-1989","isca_url":"https://www.isca-archive.org/interspeech_2026/aparin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/aparin26_interspeech.pdf","session":"Robust and Efficient ASR","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["self-supervised"],"institutions":["National University of Science and Technology MISIS","Higher School of Economics"],"funding":["Basic Research Program at HSE University"],"code":{"url":"https://github.com/audiosae/audio-sae","stars":16,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"aparin26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["National University of Science and Technology MISIS","Higher School of Economics"],"code":"https://github.com/audiosae/audio-sae","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1989","pdf":"https://www.isca-archive.org/interspeech_2026/aparin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/aparin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/aparin26_interspeech/markdown.md"},{"id":"arai26_interspeech","title":"Articulatory Dynamics using Physical Vocal-tract Models","authors":["Takayuki Arai"],"year":2026,"doi":"10.21437/Interspeech.2026-840","isca_url":"https://www.isca-archive.org/interspeech_2026/arai26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/arai26_interspeech.pdf","session":"Speech Production and Perception 2","topics":["phonetics","speech-production","tts"],"category":"phonetics-linguistics","institutions":["Sophia University"],"funding":["JSPS KAKENHI","Sophia University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"arai26_interspeech","category":"phonetics-linguistics","institutions":["Sophia University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-840","pdf":"https://www.isca-archive.org/interspeech_2026/arai26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/arai26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/arai26_interspeech/markdown.md"},{"id":"arai26b_interspeech","title":"Programmable Speech Synthesis without Computers","authors":["Takayuki Arai"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/arai26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/arai26b_interspeech.pdf","session":"Speech Synthesis, Voice Conversion and Audio Generation","topics":["speech-production","phonetics","prosody"],"category":"tts","institutions":["Sophia University"],"funding":["JSPS KAKENHI","Sophia University Special Grant for Academic Research"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"arai26b_interspeech","category":"tts","institutions":["Sophia University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/arai26b_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/arai26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/arai26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/arai26b_interspeech/markdown.md"},{"id":"arcosholzinger26_interspeech","title":"GRIDS: Dimensionality-Aware Anomaly Detection in Learned Representations of Self-Supervised Speech Models","authors":["Sandra Arcos-Holzinger","Sarah M. Erfani","James Bailey","Sanjeev Khudanpur"],"year":2026,"doi":"10.21437/Interspeech.2026-2719","isca_url":"https://www.isca-archive.org/interspeech_2026/arcosholzinger26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/arcosholzinger26_interspeech.pdf","session":"Speech Representations and Alignment","topics":["self-supervised","evaluation","asr"],"category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["University of Melbourne","Monash University","Johns Hopkins University"],"funding":["Australian Government Research Training Program Scholarship"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"arcosholzinger26_interspeech","category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["University of Melbourne","Monash University","Johns Hopkins University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2719","pdf":"https://www.isca-archive.org/interspeech_2026/arcosholzinger26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/arcosholzinger26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/arcosholzinger26_interspeech/markdown.md"},{"id":"arefeen26_interspeech","title":"DAST: A Dual-Stream Voice Anonymization Attacker with Staged Training","authors":["Ridwan Arefeen","Xiaoxiao Miao","Rong Tong","Timothy Liu","Aik Beng Ng","Simon See"],"year":2026,"doi":"10.21437/Interspeech.2026-3094","isca_url":"https://www.isca-archive.org/interspeech_2026/arefeen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/arefeen26_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["speaker-verification","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["Singapore Institute of Technology","Duke Kunshan University","NVIDIA"],"funding":["Singapore Ministry of Education Academic Research Fund Tier 1"],"code":{"url":"https://github.com/monkeyDarefeen/DAST","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"arefeen26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["Singapore Institute of Technology","Duke Kunshan University","NVIDIA"],"code":"https://github.com/monkeyDarefeen/DAST","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3094","pdf":"https://www.isca-archive.org/interspeech_2026/arefeen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/arefeen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/arefeen26_interspeech/markdown.md"},{"id":"arora26_interspeech","title":"Negation in Audio Generation Models","authors":["Arjun Arora","Anshul Jain","Gubbala Mohith Nukesh","Bikash Dutta","Richa Singh","Mayank Vatsa"],"year":2026,"doi":"10.21437/Interspeech.2026-1756","isca_url":"https://www.isca-archive.org/interspeech_2026/arora26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/arora26_interspeech.pdf","session":"Generative Audio and Music","topics":["tts","evaluation","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Indian Institute of Technology Jodhpur"],"funding":["IndiaAI Mission","Meta"],"code":{"url":"https://iab-rubric.org/resources/other-databases/audio-negation-benchmark","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"arora26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Indian Institute of Technology Jodhpur"],"code":"https://iab-rubric.org/resources/other-databases/audio-negation-benchmark","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1756","pdf":"https://www.isca-archive.org/interspeech_2026/arora26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/arora26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/arora26_interspeech/markdown.md"},{"id":"arora26b_interspeech","title":"VIB-AVSR: Variational Information Bottleneck for Noise-Robust LLM-Based Audio-Visual Speech Recognition","authors":["Piyush Arora","Navlika Singh","Umberto Cappellazzo","Stavros Petridis","Maja Pantic"],"year":2026,"doi":"10.21437/Interspeech.2026-1903","isca_url":"https://www.isca-archive.org/interspeech_2026/arora26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/arora26b_interspeech.pdf","session":"Robust Audio-Visual Speech Recognition","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["robustness-noise"],"institutions":["Imperial College London","NatWest AI Research"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"arora26b_interspeech","category":"asr","labels":["robustness-noise"],"institutions":["Imperial College London","NatWest AI Research"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1903","pdf":"https://www.isca-archive.org/interspeech_2026/arora26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/arora26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/arora26b_interspeech/markdown.md"},{"id":"arumugam26_interspeech","title":"A Human-in-the-Loop Multi-Agent Companion for Real-Time Entity Extraction and SLU-Driven ASR Error Correction","authors":["Shiva Shankar Arumugam","Ameyaditya Achar J","Aashraya Sachdeva"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/arumugam26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/arumugam26_interspeech.pdf","session":"Speech Recognition, Enhancement and Real-Time Systems","topics":["asr","speech-llm","evaluation"],"category":"asr","labels":["self-supervised","streaming-real-time"],"institutions":["Observe.AI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"arumugam26_interspeech","category":"asr","labels":["self-supervised","streaming-real-time"],"institutions":["Observe.AI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/arumugam26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/arumugam26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/arumugam26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/arumugam26_interspeech/markdown.md"},{"id":"asaka26_interspeech","title":"Two-Level Uncertainty Suppression for Robust Meeting Diarization","authors":["Shuhei Asaka","Muhammad Shakeel","Chikara Maeda","Benjamin Yen","Takeshi Ashizawa","Naoaki Sumida","Kazuhiro Nakadai"],"year":2026,"doi":"10.21437/Interspeech.2026-1956","isca_url":"https://www.isca-archive.org/interspeech_2026/asaka26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/asaka26_interspeech.pdf","session":"Speaker Diarization 2","topics":["speaker-diarization","self-supervised","multilingual"],"category":"speaker","labels":["self-supervised"],"institutions":["Institute of Science Tokyo","Honda Research Institute Japan"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"asaka26_interspeech","category":"speaker","labels":["self-supervised"],"institutions":["Institute of Science Tokyo","Honda Research Institute Japan"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1956","pdf":"https://www.isca-archive.org/interspeech_2026/asaka26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/asaka26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/asaka26_interspeech/markdown.md"},{"id":"ashikawa26_interspeech","title":"Audio-KWS-Gated Error Memory Retrieval for Incremental ASR Post-Correction","authors":["Taira Ashikawa","Daichi Hayakawa","Takehiko Kagoshima","Tomoki Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-363","isca_url":"https://www.isca-archive.org/interspeech_2026/ashikawa26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ashikawa26_interspeech.pdf","session":"Robust and Real-World ASR Systems","topics":["asr","speech-llm","keyword-spotting"],"category":"asr","institutions":["Toshiba"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ashikawa26_interspeech","category":"asr","institutions":["Toshiba"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-363","pdf":"https://www.isca-archive.org/interspeech_2026/ashikawa26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ashikawa26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ashikawa26_interspeech/markdown.md"},{"id":"awobade26_interspeech","title":"AfriVox-v2: A Domain-Verticalized Benchmark for In-the-Wild African Speech Recognition","authors":["Busayo Awobade","Gabrial Ashungafac","Oluwatoni Otokiti","Tobi Olatunji"],"year":2026,"doi":"10.21437/Interspeech.2026-3140","isca_url":"https://www.isca-archive.org/interspeech_2026/awobade26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/awobade26_interspeech.pdf","session":"Translation","topics":["asr","multilingual","low-resource"],"category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Intron"],"code":{"url":"https://huggingface.co/datasets/intronhealth/afrivox-v2","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"awobade26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Intron"],"code":"https://huggingface.co/datasets/intronhealth/afrivox-v2","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3140","pdf":"https://www.isca-archive.org/interspeech_2026/awobade26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/awobade26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/awobade26_interspeech/markdown.md"},{"id":"azad26_interspeech","title":"Harf-Speech: A Clinically Aligned Framework for Arabic Phoneme-Level Speech Assessment","authors":["Asif Azad","MD Sadik Hossain Shanto","Mohammad Sadat Hossain","Bdour Alwuqaysi","Sabri Boughorbel","Yahya Bokhari","Abdulrhman Aljouie","Ayah Othman Sindi","Ehsan Hoque"],"year":2026,"doi":"10.21437/Interspeech.2026-3472","isca_url":"https://www.isca-archive.org/interspeech_2026/azad26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/azad26_interspeech.pdf","session":"Pathological Speech Assessment 4","topics":["asr","speech-language-therapy","evaluation"],"category":"health-clinical","institutions":["Ministry of Defense","Ability Center","University of Rochester"],"code":{"url":"https://github.com/Iqra-Eval/MSA_phonetiser","stars":5,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"azad26_interspeech","category":"health-clinical","institutions":["Ministry of Defense","Ability Center","University of Rochester"],"code":"https://github.com/Iqra-Eval/MSA_phonetiser","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3472","pdf":"https://www.isca-archive.org/interspeech_2026/azad26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/azad26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/azad26_interspeech/markdown.md"},{"id":"azarski26_interspeech","title":"Temporal Partitioning of Vocal Activity for Detecting Vocal Hyperfunction from Neck-Surface Accelerometer Data","authors":["Łukasz Łazarski","Bartłomiej Eljasiak","Szymon Szmajdziński","Teresa Makuch","Iwan Ryżenkow","Anna Plęs","Władysław Średniawa"],"year":2026,"doi":"10.21437/Interspeech.2026-1435","isca_url":"https://www.isca-archive.org/interspeech_2026/azarski26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/azarski26_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["paralinguistics","health","evaluation"],"category":"health-clinical","institutions":["Samsung"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"azarski26_interspeech","category":"health-clinical","institutions":["Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1435","pdf":"https://www.isca-archive.org/interspeech_2026/azarski26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/azarski26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/azarski26_interspeech/markdown.md"},{"id":"azeemi26_interspeech","title":"Dissecting ASR Failures in Low-Resource South Asian Languages","authors":["Abdul Hameed Azeemi","Ihsan Ayyub Qazi","Maryam Mustafa","Agha Ali Raza"],"year":2026,"doi":"10.21437/Interspeech.2026-1382","isca_url":"https://www.isca-archive.org/interspeech_2026/azeemi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/azeemi26_interspeech.pdf","session":"Multilingual & Low-Resource ASR","topics":["asr","low-resource","multilingual"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Lahore University of Management Sciences"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"azeemi26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Lahore University of Management Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1382","pdf":"https://www.isca-archive.org/interspeech_2026/azeemi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/azeemi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/azeemi26_interspeech/markdown.md"},{"id":"azzouz26_interspeech","title":"Acoustic-to-Articulatory Inversion of Clean Speech Using an MRI-Trained Model","authors":["Sofiane Azzouz"],"year":2026,"doi":"10.21437/Interspeech.2026-734","isca_url":"https://www.isca-archive.org/interspeech_2026/azzouz26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/azzouz26_interspeech.pdf","session":"Speech Production and Perception 2","topics":["phonetics","speech-enhancement","dataset"],"category":"phonetics-linguistics","institutions":["Universite de Lorraine","CNRS","Inria"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"azzouz26_interspeech","category":"phonetics-linguistics","institutions":["Universite de Lorraine","CNRS","Inria"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-734","pdf":"https://www.isca-archive.org/interspeech_2026/azzouz26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/azzouz26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/azzouz26_interspeech/markdown.md"},{"id":"bae26_interspeech","title":"Something from Nothing: Data Augmentation for Robust Severity Level Estimation of Dysarthric Speech","authors":["Jaesung Bae","Xiuwen Zheng","Minje Kim","Chang D. Yoo","Mark Hasegawa-Johnson"],"year":2026,"doi":"10.21437/Interspeech.2026-1390","isca_url":"https://www.isca-archive.org/interspeech_2026/bae26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bae26_interspeech.pdf","session":"Clinical and Inclusive Speech Technology","topics":["paralinguistics","speech-enhancement","self-supervised"],"category":"health-clinical","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["University of Illinois Urbana-Champaign","Korea Advanced Institute of Science & Technology"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","National Science Foundation"],"code":{"url":"https://github.com/JaesungBae/DA-DSQA","stars":9,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bae26_interspeech","category":"health-clinical","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["University of Illinois Urbana-Champaign","Korea Advanced Institute of Science & Technology"],"code":"https://github.com/JaesungBae/DA-DSQA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1390","pdf":"https://www.isca-archive.org/interspeech_2026/bae26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bae26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bae26_interspeech/markdown.md"},{"id":"baek26_interspeech","title":"SPARK: Efficient Audio-Text Matching for User-Defined Keyword Spotting via Spiking Neural Networks","authors":["Seung-Yeop Baek","Sangho Han","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-3336","isca_url":"https://www.isca-archive.org/interspeech_2026/baek26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/baek26_interspeech.pdf","session":"Information Extraction and Retrieval","topics":["keyword-spotting","self-supervised","on-device"],"category":"asr","labels":["efficient-on-device"],"institutions":["Hanyang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea government"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"baek26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["Hanyang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3336","pdf":"https://www.isca-archive.org/interspeech_2026/baek26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/baek26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/baek26_interspeech/markdown.md"},{"id":"bagat26_interspeech","title":"Synthetic Audio Generation Framework for Air Traffic Control Speech Recognition","authors":["Raphaël Bagat","Zhe Zhang","Junichi Yamagishi","Irina Illina","Emmanuel Vincent"],"year":2026,"doi":"10.21437/Interspeech.2026-2422","isca_url":"https://www.isca-archive.org/interspeech_2026/bagat26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bagat26_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["asr","low-resource","self-supervised"],"category":"asr","labels":["low-resource","self-supervised","generative-model"],"institutions":["Universite de Lorraine","CNRS","Inria","National Institute of Informatics"],"funding":["DeepMAUVES project","DGA of french MoD","CNRS","Inria-NII TrustedSpeech Associate Team","MEXT KAKENHI"],"code":{"url":"https://gitlab.inria.fr/rbagat/atc_generation","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bagat26_interspeech","category":"asr","labels":["low-resource","self-supervised","generative-model"],"institutions":["Universite de Lorraine","CNRS","Inria","National Institute of Informatics"],"code":"https://gitlab.inria.fr/rbagat/atc_generation","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2422","pdf":"https://www.isca-archive.org/interspeech_2026/bagat26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bagat26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bagat26_interspeech/markdown.md"},{"id":"bai26_interspeech","title":"Towards Chinese Yue Opera Singing Voice Synthesis: A Benchmark with Dataset, Data Augmentation and Baseline Model","authors":["Peng Bai","Chenyang Lyu","Yue Zhou","Wujin Sun","Longyue Wang","Weihua Luo","Xiaodong Shi"],"year":2026,"doi":"10.21437/Interspeech.2026-131","isca_url":"https://www.isca-archive.org/interspeech_2026/bai26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bai26_interspeech.pdf","session":"Singing Voice and Music Generation","topics":["tts","low-resource","self-supervised"],"category":"tts","labels":["low-resource","dataset-or-benchmark-release","generative-model"],"institutions":["Xiamen University","Alibaba Group"],"funding":["Major Scientific Research Project of the State Language Commission in the 13th Five-Year Plan"],"code":{"url":"https://anonymous.4open.science/api/repo/YueOpera_Benchmark-6914/file/index.html","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bai26_interspeech","category":"tts","labels":["low-resource","dataset-or-benchmark-release","generative-model"],"institutions":["Xiamen University","Alibaba Group"],"code":"https://anonymous.4open.science/api/repo/YueOpera_Benchmark-6914/file/index.html","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-131","pdf":"https://www.isca-archive.org/interspeech_2026/bai26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bai26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bai26_interspeech/markdown.md"},{"id":"bai26b_interspeech","title":"Controllable Accent Normalization via Discrete Diffusion","authors":["Qibing Bai","Yuhan Du","Tom Ko","Shuai Wang","Yannan Wang","Haizhou Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1056","isca_url":"https://www.isca-archive.org/interspeech_2026/bai26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bai26b_interspeech.pdf","session":"Pronunciation Diversity","topics":["speech-translation","self-supervised","voice-conversion"],"category":"tts","labels":["self-supervised","generative-model"],"institutions":["Chinese University of Hong Kong","Nanjing University","Tencent","Shenzhen Loop Area Institute"],"funding":["National Natural Science Foundation of China","Program for Guangdong Introducing Innovative and Entrepreneurial Teams","Yangtze River Delta Science and Technology Innovation Community Joint Research Project","Shenzhen Science and Technology Program","Shenzhen Stability Science Program","Shenzhen Key Lab of MultiModal Cognitive Computing","Guangdong Provincial Key Laboratory of Big Data Computing","The Chinese University of Hong Kong, Shenzhen"],"code":{"url":"https://P1ping.github.io/dlman-demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bai26b_interspeech","category":"tts","labels":["self-supervised","generative-model"],"institutions":["Chinese University of Hong Kong","Nanjing University","Tencent","Shenzhen Loop Area Institute"],"code":"https://P1ping.github.io/dlman-demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1056","pdf":"https://www.isca-archive.org/interspeech_2026/bai26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bai26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bai26b_interspeech/markdown.md"},{"id":"baik26_interspeech","title":"DASH: Dual-View Self-Distillation with Multi-Layer Hidden Representations for Robust Speech Recognition","authors":["Jaeeun Baik","Ui-Hyeop Shin","Jiwon Lee","Woocheol Jeong","Hyung-Min Park"],"year":2026,"doi":"10.21437/Interspeech.2026-3232","isca_url":"https://www.isca-archive.org/interspeech_2026/baik26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/baik26_interspeech.pdf","session":"Robust ASR: Uncertainty and Confidence","topics":["asr","self-supervised","speech-enhancement"],"category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["Sogang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","National Research Foundation of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"baik26_interspeech","category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["Sogang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3232","pdf":"https://www.isca-archive.org/interspeech_2026/baik26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/baik26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/baik26_interspeech/markdown.md"},{"id":"baligar26_interspeech","title":"Toward an Articulatory Weakness Index for Speech Kinematics in Parkinson’s Disease","authors":["Shrishail Baligar","Ahmed Yousef","Daryush Mehta"],"year":2026,"doi":"10.21437/Interspeech.2026-159","isca_url":"https://www.isca-archive.org/interspeech_2026/baligar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/baligar26_interspeech.pdf","session":"Multimodal and Non-Speech Healthcare Applications","topics":["paralinguistics","speech-enhancement","evaluation"],"category":"health-clinical","institutions":["Massachusetts General Hospital"],"funding":["Voice Health Institute","National Institutes of Health"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"baligar26_interspeech","category":"health-clinical","institutions":["Massachusetts General Hospital"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-159","pdf":"https://www.isca-archive.org/interspeech_2026/baligar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/baligar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/baligar26_interspeech/markdown.md"},{"id":"banerasroux26_interspeech","title":"Closing the Speech-Text Gap with Limited Audio for Effective Domain Adaptation in LLM-Based ASR","authors":["Thibault Bañeras-Roux","Sergio Burdisso","Esaú Villatoro-Tello","Dairazalia Sánchez-Cortés","Shiran Liu","Severin Baroudi","Shashi Kumar","Hasindri Watawana","Manjunath K E","Kadri Hacioglu","Petr Motlicek","Andreas Stolcke"],"year":2026,"doi":"10.21437/Interspeech.2026-3383","isca_url":"https://www.isca-archive.org/interspeech_2026/banerasroux26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/banerasroux26_interspeech.pdf","session":"Multimodal Speech Processing and Speech LLM Systems","topics":["asr","self-supervised","speech-llm"],"category":"asr","labels":["low-resource"],"institutions":["Idiap Research Institute","Laboratoire d’Informatique et des Systèmes","EPFL","Uniphore","Brno University of Technology"],"funding":["Idiap Research Institute","Uniphore","EU Horizon 2020"],"code":{"url":"https://github.com/idiap/llm-asr-text-only-adaptation","stars":5,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"banerasroux26_interspeech","category":"asr","labels":["low-resource"],"institutions":["Idiap Research Institute","Laboratoire d’Informatique et des Systèmes","EPFL","Uniphore","Brno University of Technology"],"code":"https://github.com/idiap/llm-asr-text-only-adaptation","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3383","pdf":"https://www.isca-archive.org/interspeech_2026/banerasroux26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/banerasroux26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/banerasroux26_interspeech/markdown.md"},{"id":"banerjee26_interspeech","title":"wav2tok 2.0: Scalable Audio Tokenization Maintaining Explicit Pairwise Token Alignment for Efficient Audio Retrieval","authors":["Adhiraj Banerjee","Vipul Arora"],"year":2026,"doi":"10.21437/Interspeech.2026-141","isca_url":"https://www.isca-archive.org/interspeech_2026/banerjee26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/banerjee26_interspeech.pdf","session":"Information Extraction and Retrieval / Survey Talk","topics":["speech-tokenization","audio-retrieval","spoken-language-understanding"],"category":"asr","institutions":["Indian Institute of Technology Kanpur","KU Leuven"],"code":{"url":"https://github.com/adhiraj69/wav2tok2","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"banerjee26_interspeech","category":"asr","institutions":["Indian Institute of Technology Kanpur","KU Leuven"],"code":"https://github.com/adhiraj69/wav2tok2","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-141","pdf":"https://www.isca-archive.org/interspeech_2026/banerjee26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/banerjee26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/banerjee26_interspeech/markdown.md"},{"id":"baranski26_interspeech","title":"HALAS: A Human-Annotated Dataset of Hallucinations of Modern ASR Systems","authors":["Mateusz Barański","Jan Jasiński","Julitta Bartolewska","Marcin Witkowski","Konrad Kowalczyk"],"year":2026,"doi":"10.21437/Interspeech.2026-337","isca_url":"https://www.isca-archive.org/interspeech_2026/baranski26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/baranski26_interspeech.pdf","session":"Speech Benchmarks, Evaluation, and Resources","topics":["asr","evaluation","self-supervised"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["AGH University of Krakow"],"funding":["National Science Centre, Poland","National Centre for Research and Development, Poland","Excellence initiative – research university"],"code":{"url":"https://github.com/DSP-AGH/HALAS/tree/main","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"baranski26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["AGH University of Krakow"],"code":"https://github.com/DSP-AGH/HALAS/tree/main","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-337","pdf":"https://www.isca-archive.org/interspeech_2026/baranski26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/baranski26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/baranski26_interspeech/markdown.md"},{"id":"bargum26_interspeech","title":"Improving Model Expressivity and Speaker Matching in Low-Latency Voice Conversion","authors":["Anders R. Bargum","Simon Lajboschitz","Stefania Serafin","Cumhur Erkut"],"year":2026,"doi":"10.21437/Interspeech.2026-796","isca_url":"https://www.isca-archive.org/interspeech_2026/bargum26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bargum26_interspeech.pdf","session":"Voice Conversion","topics":["voice-conversion","self-supervised","on-device"],"category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Aalborg University","Khora & Heka VR","Technical University of Denmark"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bargum26_interspeech","category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Aalborg University","Khora & Heka VR","Technical University of Denmark"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-796","pdf":"https://www.isca-archive.org/interspeech_2026/bargum26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bargum26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bargum26_interspeech/markdown.md"},{"id":"barreiros26_interspeech","title":"Massive Open-Vocabulary Keyword Spotting","authors":["Leonor Barreiros","Raul Monteiro","Afonso Mendes","Gonçalo M. Correia"],"year":2026,"doi":"10.21437/Interspeech.2026-1444","isca_url":"https://www.isca-archive.org/interspeech_2026/barreiros26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/barreiros26_interspeech.pdf","topics":["asr","keyword-spotting","self-supervised"],"category":"asr","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Priberam Labs","Instituto Superior Tecnico","Instituto de Telecomunicacoes"],"funding":["Portuguese Recovery and Resilience Plan"],"arxiv":"","code":{"url":"https://github.com/Priberam/Enhance-CB-Whisper","stars":17,"license":""},"contact":"","lab":"","open_to_collaboration":false,"wiki_frontmatter":{"id":"barreiros26_interspeech","category":"asr","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Priberam Labs","Instituto Superior Tecnico","Instituto de Telecomunicacoes"],"code":"https://github.com/Priberam/Enhance-CB-Whisper","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1444","pdf":"https://www.isca-archive.org/interspeech_2026/barreiros26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/barreiros26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/barreiros26_interspeech/markdown.md"},{"id":"bartley26_interspeech","title":"Bootstrapping Endangered Language ASR with Short-Form Corpora","authors":["Christopher Bartley","Anton Ragni"],"year":2026,"doi":"10.21437/Interspeech.2026-1302","isca_url":"https://www.isca-archive.org/interspeech_2026/bartley26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bartley26_interspeech.pdf","session":"Low-Resource & Endangered Language Speech Processing","topics":["asr","low-resource","multilingual"],"category":"asr","labels":["low-resource","efficient-on-device","self-supervised"],"institutions":["University of Sheffield"],"code":{"url":"https://github.com/c-bartley/el-asr","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bartley26_interspeech","category":"asr","labels":["low-resource","efficient-on-device","self-supervised"],"institutions":["University of Sheffield"],"code":"https://github.com/c-bartley/el-asr","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1302","pdf":"https://www.isca-archive.org/interspeech_2026/bartley26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bartley26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bartley26_interspeech/markdown.md"},{"id":"bauer26_interspeech","title":"VAD to the Bone: Ultra-Tiny Speech Activity Detection for Edge Deployment","authors":["Stephen Bauer","Sheila Seidel","Shanza Iftikhar","Scott Veidenheimer","Gorkem Ulkar"],"year":2026,"doi":"10.21437/Interspeech.2026-2523","isca_url":"https://www.isca-archive.org/interspeech_2026/bauer26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bauer26_interspeech.pdf","session":"Audio segmentation","topics":["asr","speech-enhancement","self-supervised"],"category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Analog Devices","University of California, Los Angeles"],"code":{"url":"https://huggingface.co/spaces/kiloVAD-demo","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bauer26_interspeech","category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Analog Devices","University of California, Los Angeles"],"code":"https://huggingface.co/spaces/kiloVAD-demo","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2523","pdf":"https://www.isca-archive.org/interspeech_2026/bauer26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bauer26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bauer26_interspeech/markdown.md"},{"id":"baumann26_interspeech","title":"PhonLLM: Joint Phone Recognition and Phonological Process Inference for Child Speech","authors":["Ilja Baumann","Korbinian Riedhammer","Tobias Bocklet"],"year":2026,"doi":"10.21437/Interspeech.2026-3378","isca_url":"https://www.isca-archive.org/interspeech_2026/baumann26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/baumann26_interspeech.pdf","session":"Prosody, Pronunciation and Specialized Speech Processing","topics":["speech-llm","low-resource","evaluation"],"category":"asr","institutions":["Technische Hochschule Nurnberg"],"funding":["Bavarian Ministry of Health, Care and Prevention","European Union"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"baumann26_interspeech","category":"asr","institutions":["Technische Hochschule Nurnberg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3378","pdf":"https://www.isca-archive.org/interspeech_2026/baumann26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/baumann26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/baumann26_interspeech/markdown.md"},{"id":"beck26_interspeech","title":"AppTek Call-Center Dialogues: A Multi-Accent Long-Form Benchmark for English ASR","authors":["Eugen Beck","Sarah Beranek","Uma Moothiringote","Daniel Mann","Wilfried Michel","Katie Nguyen","Taylor Tragemann"],"year":2026,"doi":"10.21437/Interspeech.2026-2047","isca_url":"https://www.isca-archive.org/interspeech_2026/beck26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/beck26_interspeech.pdf","session":"Long-form Audio & New Attention Approaches","topics":["asr","evaluation","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["AppTek"],"code":{"url":"https://huggingface.co/datasets/apptek-com/apptek_callcenter_dialogues","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"beck26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["AppTek"],"code":"https://huggingface.co/datasets/apptek-com/apptek_callcenter_dialogues","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2047","pdf":"https://www.isca-archive.org/interspeech_2026/beck26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/beck26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/beck26_interspeech/markdown.md"},{"id":"behringer26_interspeech","title":"Assessing the Impact of Noise and Speech Enhancement on the Intelligibility of Speech Codecs","authors":["Lyonel Behringer","Anna Leschanowsky","Anjana Rajasekhar","Emily Kratsch","Guillaume Fuchs"],"year":2026,"doi":"10.21437/Interspeech.2026-1459","isca_url":"https://www.isca-archive.org/interspeech_2026/behringer26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/behringer26_interspeech.pdf","session":"Quality, Intelligibility and Evaluation of Speech and Codecs","topics":["speech-coding","evaluation","speech-enhancement"],"category":"speech-coding","labels":["robustness-noise"],"institutions":["Fraunhofer Institute for Integrated Circuits"],"funding":["Free State of Bavaria"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"behringer26_interspeech","category":"speech-coding","labels":["robustness-noise"],"institutions":["Fraunhofer Institute for Integrated Circuits"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1459","pdf":"https://www.isca-archive.org/interspeech_2026/behringer26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/behringer26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/behringer26_interspeech/markdown.md"},{"id":"bejugam26_interspeech","title":"UFL-GAN: A Multi-Discriminator GAN for Unsupervised Speech Enhancement","authors":["Satvik Bejugam","Venkatesh Parvathala","Sri Rama Murty Kodukula"],"year":2026,"doi":"10.21437/Interspeech.2026-2565","isca_url":"https://www.isca-archive.org/interspeech_2026/bejugam26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bejugam26_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","self-supervised","unsupervised"],"category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Indian Institute of Technology Hyderabad","Eternal Limited"],"code":{"url":"https://siplab-iith.github.io/UFLGAN/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bejugam26_interspeech","category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Indian Institute of Technology Hyderabad","Eternal Limited"],"code":"https://siplab-iith.github.io/UFLGAN/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2565","pdf":"https://www.isca-archive.org/interspeech_2026/bejugam26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bejugam26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bejugam26_interspeech/markdown.md"},{"id":"benslimane26_interspeech","title":"RT-Tango: Real-Time Distributed Binaural Speech Enhancement for Low-Power Hearing Aid Devices","authors":["Zahra Benslimane","Pierre Chouteau","Martyna Poreba","Fabrice Auzanneau","Michal Szczepanski","Fabian Chersi","Romain Serizel"],"year":2026,"doi":"10.21437/Interspeech.2026-3301","isca_url":"https://www.isca-archive.org/interspeech_2026/benslimane26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/benslimane26_interspeech.pdf","session":"Real-Time, Low-Latency and Edge Speech Enhancement","topics":["speech-enhancement","low-resource","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time","robustness-noise"],"institutions":["Universite Paris-Saclay","CEA","Universite de Lorraine","CNRS","Inria"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"benslimane26_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time","robustness-noise"],"institutions":["Universite Paris-Saclay","CEA","Universite de Lorraine","CNRS","Inria"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3301","pdf":"https://www.isca-archive.org/interspeech_2026/benslimane26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/benslimane26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/benslimane26_interspeech/markdown.md"},{"id":"benway26_interspeech","title":"VoxKit: Desktop Phone Alignment and Goodness of Pronunciation Analysis","authors":["Nina R Benway","Beckett Frey","Tristan Mahr","Michael McAuliffe","Prad Kadambi","Visar Berisha","Katherine Hustad"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/benway26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/benway26_interspeech.pdf","session":"Speech and Language Learning Technologies","topics":["evaluation","speech-science","low-resource"],"category":"applications-other","institutions":["University of Maryland, College Park","University of Wisconsin - Madison","Arizona State University"],"funding":["National Institute on Deafness and Other Communication Disorders"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"benway26_interspeech","category":"applications-other","institutions":["University of Maryland, College Park","University of Wisconsin - Madison","Arizona State University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/benway26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/benway26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/benway26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/benway26_interspeech/markdown.md"},{"id":"berdo26_interspeech","title":"Post-Training Speech Enhancement Language Models with Perceptual Rewards","authors":["Frédéric Berdoȥ","Luca A. Lanzendöerfer","Antonis Asonitis","Roger Wattenhofer"],"year":2026,"doi":"10.21437/Interspeech.2026-3405","isca_url":"https://www.isca-archive.org/interspeech_2026/berdo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/berdo26_interspeech.pdf","session":"Language-Model and Codec-Token Speech Enhancement","topics":["speech-enhancement","self-supervised","speech-llm"],"category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["ETH Zurich"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"berdo26_interspeech","category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["ETH Zurich"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3405","pdf":"https://www.isca-archive.org/interspeech_2026/berdo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/berdo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/berdo26_interspeech/markdown.md"},{"id":"beyers26_interspeech","title":"Scaling few-shot spoken word classification with generative meta-continual learning","authors":["Louise Beyers","Batsirayi Mupamhi Ziki","Ruan van der Merwe"],"year":2026,"doi":"10.21437/Interspeech.2026-408","isca_url":"https://www.isca-archive.org/interspeech_2026/beyers26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/beyers26_interspeech.pdf","session":"Efficient Inference for ASR and Speech LMs","topics":["keyword-spotting","few-shot","self-supervised"],"category":"asr","labels":["low-resource","efficient-on-device","self-supervised"],"institutions":["Bytefuse"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"beyers26_interspeech","category":"asr","labels":["low-resource","efficient-on-device","self-supervised"],"institutions":["Bytefuse"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-408","pdf":"https://www.isca-archive.org/interspeech_2026/beyers26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/beyers26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/beyers26_interspeech/markdown.md"},{"id":"bhagtani26_interspeech","title":"Speak or Stay Silent: Context-Aware Turn-Taking in Multi-Party Dialogue","authors":["Kratika Bhagtani","Mrinal Anand","Yu Chen Xu","Amit Kumar Singh Yadav"],"year":2026,"doi":"10.21437/Interspeech.2026-3083","isca_url":"https://www.isca-archive.org/interspeech_2026/bhagtani26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bhagtani26_interspeech.pdf","session":"Entrainment and Dialogue Coordination","topics":["spoken-language-understanding","self-supervised","evaluation"],"category":"speech-llm-dialogue","institutions":["Purdue University","Ishiki Labs"],"code":{"url":"https://github.com/ishikilabsinc/context_aware_modeling","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bhagtani26_interspeech","category":"speech-llm-dialogue","institutions":["Purdue University","Ishiki Labs"],"code":"https://github.com/ishikilabsinc/context_aware_modeling","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3083","pdf":"https://www.isca-archive.org/interspeech_2026/bhagtani26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bhagtani26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bhagtani26_interspeech/markdown.md"},{"id":"bharadwaj26_interspeech","title":"An Empirical Recipe for Universal Phone Recognition","authors":["Shikhar Bharadwaj","Chin-Jou Li","Kwanghee Choi","Eunjung Yeo","William Chen","Shinji Watanabe","David R. Mortensen"],"year":2026,"doi":"10.21437/Interspeech.2026-1462","isca_url":"https://www.isca-archive.org/interspeech_2026/bharadwaj26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bharadwaj26_interspeech.pdf","session":"Low-Resource & Endangered Language Speech Processing","topics":["asr","phonetics","multilingual"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["Carnegie Mellon University","University of Texas at Austin"],"code":{"url":"https://github.com/changelinglab/PhoneticXeus","stars":45,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bharadwaj26_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["Carnegie Mellon University","University of Texas at Austin"],"code":"https://github.com/changelinglab/PhoneticXeus","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1462","pdf":"https://www.isca-archive.org/interspeech_2026/bharadwaj26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bharadwaj26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bharadwaj26_interspeech/markdown.md"},{"id":"bhat26_interspeech","title":"A Gated Multi-Task Whisper Framework for Speech, Emotion, and Scene Understanding","authors":["Manjiri Bhat","Ravindra B. Keskar"],"year":2026,"doi":"10.21437/Interspeech.2026-135","isca_url":"https://www.isca-archive.org/interspeech_2026/bhat26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bhat26_interspeech.pdf","session":"Audio segmentation","topics":["speech-llm","self-supervised","paralinguistics"],"category":"asr","labels":["self-supervised"],"institutions":["Visvesvaraya National Institute of Technology Nagpur"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bhat26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Visvesvaraya National Institute of Technology Nagpur"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-135","pdf":"https://www.isca-archive.org/interspeech_2026/bhat26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bhat26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bhat26_interspeech/markdown.md"},{"id":"bhattacharya26_interspeech","title":"The Sound of Code-Switching: Prosodic Profiles of Spontaneous Spanish-English Speech","authors":["Debasmita Bhattacharya","Michela Marchini","Julia Hirschberg"],"year":2026,"doi":"10.21437/Interspeech.2026-249","isca_url":"https://www.isca-archive.org/interspeech_2026/bhattacharya26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bhattacharya26_interspeech.pdf","session":"Modeling L1 Acquisition","topics":["multilingual","prosody","dataset"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Columbia University","University of Michigan"],"funding":["National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bhattacharya26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Columbia University","University of Michigan"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-249","pdf":"https://www.isca-archive.org/interspeech_2026/bhattacharya26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bhattacharya26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bhattacharya26_interspeech/markdown.md"},{"id":"bhattacharya26b_interspeech","title":"Exploiting Neural Audio Codec Latents for Adversarial Audio Attacks","authors":["Sameek Bhattacharya","Bharath Krishnamurthy","Ajita Rattani"],"year":2026,"doi":"10.21437/Interspeech.2026-3055","isca_url":"https://www.isca-archive.org/interspeech_2026/bhattacharya26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bhattacharya26b_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["speech-enhancement","speaker-verification","self-supervised"],"category":"deepfake-security","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["University of North Texas"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bhattacharya26b_interspeech","category":"deepfake-security","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["University of North Texas"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3055","pdf":"https://www.isca-archive.org/interspeech_2026/bhattacharya26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bhattacharya26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bhattacharya26b_interspeech/markdown.md"},{"id":"bhogale26_interspeech","title":"Voice of India: A Large-Scale Benchmark for Real-World Speech Recognition in India","authors":["Kaushal Bhogale","Manas Dhir","Amritansh Walecha","Manmeet Kaur","Vanshika Chhabra","Aaditya Pareek","Hanuman Sidh","Sagar Jain","Bhaskar Singh","Utkarsh Singh","Tahir Javed","Shobhit Banga","Mitesh M Khapra"],"year":2026,"doi":"10.21437/Interspeech.2026-3189","isca_url":"https://www.isca-archive.org/interspeech_2026/bhogale26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bhogale26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["asr","multilingual","dataset"],"category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Indian Institute of Technology Madras","Josh Talks"],"code":{"url":"https://github.com/JoshTalks/voice-of-india","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bhogale26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Indian Institute of Technology Madras","Josh Talks"],"code":"https://github.com/JoshTalks/voice-of-india","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3189","pdf":"https://www.isca-archive.org/interspeech_2026/bhogale26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bhogale26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bhogale26_interspeech/markdown.md"},{"id":"bhogale26b_interspeech","title":"Vimarsha: Faithful ASR Evaluation for Indian Languages with Demographic Diversity, In-the-Wild Audio and Spelling Variations","authors":["Kaushal Bhogale","Srija Anand","Sadakopa Ramakrishnan Thothathiri","Tahir Javed","Sshubam Verma","Mitesh M Khapra"],"year":2026,"doi":"10.21437/Interspeech.2026-3348","isca_url":"https://www.isca-archive.org/interspeech_2026/bhogale26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bhogale26b_interspeech.pdf","session":"Multilingual & Low-Resource ASR","topics":["asr","multilingual","evaluation"],"category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["Indian Institute of Technology Madras","Sarvam AI"],"funding":["Digital India Bhashini","MeitY","EkStep Foundation","Nilekani Philanthropies"],"code":{"url":"https://github.com/AI4Bharat/Vimarsha","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bhogale26b_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["Indian Institute of Technology Madras","Sarvam AI"],"code":"https://github.com/AI4Bharat/Vimarsha","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3348","pdf":"https://www.isca-archive.org/interspeech_2026/bhogale26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bhogale26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bhogale26b_interspeech/markdown.md"},{"id":"bhooi26_interspeech","title":"Refining the Latent Bridge: Superior ASR Performance via Adapter-Only Alignment with Diffusion LLMs","authors":["Puneet Singh Bhooi","Vinayak Abrol"],"year":2026,"doi":"10.21437/Interspeech.2026-2229","isca_url":"https://www.isca-archive.org/interspeech_2026/bhooi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bhooi26_interspeech.pdf","session":"Robust ASR: Uncertainty and Confidence","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["self-supervised","generative-model"],"institutions":["Indraprastha Institute of Information Technology Delhi"],"funding":["ANRF Core Research Grant","Government of India","Nebius Research Grant","Infosys Foundation","Infosys Centre for AI, IIITD"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bhooi26_interspeech","category":"asr","labels":["self-supervised","generative-model"],"institutions":["Indraprastha Institute of Information Technology Delhi"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2229","pdf":"https://www.isca-archive.org/interspeech_2026/bhooi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bhooi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bhooi26_interspeech/markdown.md"},{"id":"bhosale26_interspeech","title":"Echoes after Edits: Room Impulse Response Estimation for Geometry Update","authors":["Swapnil Bhosale","Yoshiki Masuyama","Moitreya Chatterjee","Christoph Boeddeker","Julius Richter","Gordon Wichern","Jonathan Le Roux"],"year":2026,"doi":"10.21437/Interspeech.2026-2514","isca_url":"https://www.isca-archive.org/interspeech_2026/bhosale26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bhosale26_interspeech.pdf","session":"Spatial Audio 2","topics":["speech-enhancement","dataset"],"category":"enhancement-separation","institutions":["Mitsubishi Electric Research Laboratories","University of Surrey"],"code":{"url":"https://github.com/merlresearch/geometry-edit-rir","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bhosale26_interspeech","category":"enhancement-separation","institutions":["Mitsubishi Electric Research Laboratories","University of Surrey"],"code":"https://github.com/merlresearch/geometry-edit-rir","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2514","pdf":"https://www.isca-archive.org/interspeech_2026/bhosale26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bhosale26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bhosale26_interspeech/markdown.md"},{"id":"bhosale26b_interspeech","title":"Dual-Geometry Manifolds for Few-shot RIR Prediction","authors":["Swapnil Bhosale","Gordon Wichern","Yoshiki Masuyama","Moitreya Chatterjee","Christoph Boeddeker","Julius Richter","Xiatian Zhu","Jonathan Le Roux"],"year":2026,"doi":"10.21437/Interspeech.2026-2630","isca_url":"https://www.isca-archive.org/interspeech_2026/bhosale26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bhosale26b_interspeech.pdf","session":"Spatial Audio 2","topics":["self-supervised","dataset","evaluation"],"category":"enhancement-separation","labels":["low-resource"],"institutions":["Mitsubishi Electric Research Laboratories","University of Surrey"],"code":{"url":"https://github.com/merlresearch/janus-rir","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bhosale26b_interspeech","category":"enhancement-separation","labels":["low-resource"],"institutions":["Mitsubishi Electric Research Laboratories","University of Surrey"],"code":"https://github.com/merlresearch/janus-rir","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2630","pdf":"https://www.isca-archive.org/interspeech_2026/bhosale26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bhosale26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bhosale26b_interspeech/markdown.md"},{"id":"bijoy26_interspeech","title":"Mixture-of-Accent-Adapters for Robust ASR: Injecting Accent Cues into Pretrained Whisper","authors":["Mehedi Hasan Bijoy","Yaroslav Getman","Tamás Grósz","Mikko Kurimo"],"year":2026,"doi":"10.21437/Interspeech.2026-1373","isca_url":"https://www.isca-archive.org/interspeech_2026/bijoy26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bijoy26_interspeech.pdf","session":"Domain Adaptation & Accented ASR","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["Aalto University","South East Technological University"],"funding":["Business Finland","Finnish Cultural Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bijoy26_interspeech","category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["Aalto University","South East Technological University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1373","pdf":"https://www.isca-archive.org/interspeech_2026/bijoy26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bijoy26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bijoy26_interspeech/markdown.md"},{"id":"bird26_interspeech","title":"Speech Technology and Linguistic Diversity","authors":["Steven Bird"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/bird26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bird26_interspeech.pdf","session":"Keynote2 - Steven Bird: Speech Technology and Linguistic Diversity","topics":["low-resource","multilingual"],"category":"asr","labels":["low-resource","multilingual"],"institutions":["Charles Darwin University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bird26_interspeech","category":"asr","labels":["low-resource","multilingual"],"institutions":["Charles Darwin University"],"updated":"2026-09-28","confidence":"abstract-only","source":"https://www.isca-archive.org/interspeech_2026/bird26_interspeech.html"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bird26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bird26_interspeech/markdown.md"},{"id":"boeddeker26_interspeech","title":"Speaker Identity as Sole Supervision for Speech Separation","authors":["Christoph Boeddeker","Yoshiki Masuyama","Julius Richter","Takahiro Edo","Gordon Wichern","Jonathan Le Roux"],"year":2026,"doi":"10.21437/Interspeech.2026-2620","isca_url":"https://www.isca-archive.org/interspeech_2026/boeddeker26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/boeddeker26_interspeech.pdf","session":"Source Separation 2","topics":["speech-separation","self-supervised","speaker-verification"],"category":"enhancement-separation","institutions":["Mitsubishi Electric Research Laboratories"],"code":{"url":"https://github.com/merlresearch/sis_sep","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"boeddeker26_interspeech","category":"enhancement-separation","institutions":["Mitsubishi Electric Research Laboratories"],"code":"https://github.com/merlresearch/sis_sep","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2620","pdf":"https://www.isca-archive.org/interspeech_2026/boeddeker26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/boeddeker26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/boeddeker26_interspeech/markdown.md"},{"id":"bokkahallisatish26_interspeech","title":"The Voice Behind the Words: Quantifying Intersectional Bias in SpeechLLMs","authors":["Shree Harsha Bokkahalli Satish","Christoph Minixhofer","Maria Teleki","James Caverlee","Ondřej Klejch","Peter Bell","Gustav Eje Henter","Éva Székely"],"year":2026,"doi":"10.21437/Interspeech.2026-1918","isca_url":"https://www.isca-archive.org/interspeech_2026/bokkahallisatish26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bokkahallisatish26_interspeech.pdf","session":"Spoken Language Processing: Evaluation and Metrics","topics":["speech-llm","evaluation","multilingual"],"category":"speech-llm-dialogue","institutions":["KTH Royal Institute of Technology","University of Edinburgh","Texas A&M University"],"funding":["Wallenberg AI, Autonomous Systems and Software Program","Knut and Alice Wallenberg Foundation"],"code":{"url":"https://shreeharsha-bs.github.io/interspeech-voice-behind-words-website/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bokkahallisatish26_interspeech","category":"speech-llm-dialogue","institutions":["KTH Royal Institute of Technology","University of Edinburgh","Texas A&M University"],"code":"https://shreeharsha-bs.github.io/interspeech-voice-behind-words-website/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1918","pdf":"https://www.isca-archive.org/interspeech_2026/bokkahallisatish26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bokkahallisatish26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bokkahallisatish26_interspeech/markdown.md"},{"id":"bonzi26_interspeech","title":"Enhancing Audio Reasoning via Semantic Summary Prediction","authors":["Francesco Bonzi","Pooneh Mousavi","Cem Subakan","Mirco Ravanelli"],"year":2026,"doi":"10.21437/Interspeech.2026-1504","isca_url":"https://www.isca-archive.org/interspeech_2026/bonzi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bonzi26_interspeech.pdf","session":"Spoken Language Understanding","topics":["speech-llm","self-supervised","spoken-language-understanding"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Concordia University","Mila - Quebec AI Institute","Universite Laval"],"funding":["NSERC","Digital Research Alliance of Canada","Translated Imminent Program","Apple"],"code":{"url":"https://github.com/FrancescoBonzi/SPARE","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bonzi26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Concordia University","Mila - Quebec AI Institute","Universite Laval"],"code":"https://github.com/FrancescoBonzi/SPARE","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1504","pdf":"https://www.isca-archive.org/interspeech_2026/bonzi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bonzi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bonzi26_interspeech/markdown.md"},{"id":"boo26_interspeech","title":"Referee: Reference-aware Audiovisual Deepfake Detection","authors":["Hyemin Boo","Eunsang Lee","Jiyoung Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1246","isca_url":"https://www.isca-archive.org/interspeech_2026/boo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/boo26_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["deepfake-detection","self-supervised","speaker-verification"],"category":"deepfake-security","labels":["multilingual"],"institutions":["Ewha Womans University"],"funding":["National Research Foundation of Korea","Ministry of Education"],"code":{"url":"https://github.com/ewha-mmai/referee","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"boo26_interspeech","category":"deepfake-security","labels":["multilingual"],"institutions":["Ewha Womans University"],"code":"https://github.com/ewha-mmai/referee","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1246","pdf":"https://www.isca-archive.org/interspeech_2026/boo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/boo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/boo26_interspeech/markdown.md"},{"id":"borodin26_interspeech","title":"Balalaika: Data-Centric, Prosody-Aware Annotation Pipeline for Russian Speech","authors":["Kirill Borodin","Nikita Vasiliev","Vasiliy Kudryavtsev","Maxim Maslov","Mikhail Gorodnichev","Grach Mkrtchian"],"year":2026,"doi":"10.21437/Interspeech.2026-83","isca_url":"https://www.isca-archive.org/interspeech_2026/borodin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/borodin26_interspeech.pdf","session":"Speech Benchmarks, Evaluation, and Resources","topics":["tts","speech-enhancement","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Moscow Technical University of Communications and Informatics","BitmanagerAI"],"code":{"url":"https://github.com/lab260ru/balalaika","stars":21,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"borodin26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Moscow Technical University of Communications and Informatics","BitmanagerAI"],"code":"https://github.com/lab260ru/balalaika","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-83","pdf":"https://www.isca-archive.org/interspeech_2026/borodin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/borodin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/borodin26_interspeech/markdown.md"},{"id":"borodin26b_interspeech","title":"When Spoof Detectors Travel: Evaluation Across 66 Languages in the Low-Resource Language Spoofing Corpus","authors":["Kirill Borodin","Vasiliy Kudryavtsev","Maxim Maslov","Mikhail Gorodnichev","Grach Mkrtchian"],"year":2026,"doi":"10.21437/Interspeech.2026-345","isca_url":"https://www.isca-archive.org/interspeech_2026/borodin26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/borodin26b_interspeech.pdf","session":"Spoofing and Deepfake Detection 1","topics":["audio-deepfake","evaluation","low-resource"],"category":"deepfake-security","labels":["low-resource","multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["Moscow Technical University of Communications and Informatics","BitmanagerAI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"borodin26b_interspeech","category":"deepfake-security","labels":["low-resource","multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["Moscow Technical University of Communications and Informatics","BitmanagerAI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-345","pdf":"https://www.isca-archive.org/interspeech_2026/borodin26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/borodin26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/borodin26b_interspeech/markdown.md"},{"id":"bouziane26_interspeech","title":"Learning Multiple Utterance-Level Attribute Representations with a Unified Speech Encoder","authors":["Maryem Bouziane","Salima Mdhaffar","Yannick Estève"],"year":2026,"doi":"10.21437/Interspeech.2026-3350","isca_url":"https://www.isca-archive.org/interspeech_2026/bouziane26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bouziane26_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["self-supervised","multilingual","speaker-verification"],"category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Avignon Universite"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bouziane26_interspeech","category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Avignon Universite"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3350","pdf":"https://www.isca-archive.org/interspeech_2026/bouziane26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bouziane26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bouziane26_interspeech/markdown.md"},{"id":"bralios26_interspeech","title":"Elastic Time: Dynamic Frame Rate Bottlenecks for Neural Audio Coding","authors":["Dimitrios Bralios","Paris Smaragdis","Minje Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-3031","isca_url":"https://www.isca-archive.org/interspeech_2026/bralios26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bralios26_interspeech.pdf","session":"Neural Audio Codec Architectures","topics":["speech-coding","self-supervised","dataset"],"category":"speech-coding","labels":["efficient-on-device","streaming-real-time"],"institutions":["University of Illinois Urbana-Champaign","Massachusetts Institute of Technology"],"funding":["Electronics and Telecommunications Research Institute"],"code":{"url":"https://github.com/dbralios/elastic-time","stars":6,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bralios26_interspeech","category":"speech-coding","labels":["efficient-on-device","streaming-real-time"],"institutions":["University of Illinois Urbana-Champaign","Massachusetts Institute of Technology"],"code":"https://github.com/dbralios/elastic-time","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3031","pdf":"https://www.isca-archive.org/interspeech_2026/bralios26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bralios26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bralios26_interspeech/markdown.md"},{"id":"braun26_interspeech","title":"Mitigating Scoring Errors and Compensating for Nonverbal Subtests in Speech-Based Dementia Assessment","authors":["Franziska Braun","Christopher Witzl","Andreas Erzigkeit","Hartmut Lehfeld","Thomas Hillemacher","Tobias Bocklet","Korbinian Riedhammer"],"year":2026,"doi":"10.21437/Interspeech.2026-2806","isca_url":"https://www.isca-archive.org/interspeech_2026/braun26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/braun26_interspeech.pdf","session":"Pathological Speech Assessment 2","topics":["health","speech-llm","evaluation"],"category":"health-clinical","labels":["self-supervised"],"institutions":["Technische Hochschule Nurnberg","Geromed GmbH","PMU Klinikum Nurnberg"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"braun26_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["Technische Hochschule Nurnberg","Geromed GmbH","PMU Klinikum Nurnberg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2806","pdf":"https://www.isca-archive.org/interspeech_2026/braun26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/braun26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/braun26_interspeech/markdown.md"},{"id":"bui26_interspeech","title":"CSER: Semantic Evaluation of LLM Auto-Repair for Code-Switching ASR","authors":["Tien Dat Bui","Duy Le-Tuan Nguyen","Nhat Minh Le","Van Hai Do"],"year":2026,"doi":"10.21437/Interspeech.2026-3390","isca_url":"https://www.isca-archive.org/interspeech_2026/bui26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/bui26_interspeech.pdf","session":"Speech Benchmarks, Evaluation, and Resources","topics":["code-switching","asr","evaluation"],"category":"resources-evaluation","labels":["multilingual"],"institutions":["Viettel Group","Thuyloi University"],"code":{"url":"https://github.com/vincentbui-ai/CSER-Benchmark/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"bui26_interspeech","category":"resources-evaluation","labels":["multilingual"],"institutions":["Viettel Group","Thuyloi University"],"code":"https://github.com/vincentbui-ai/CSER-Benchmark/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3390","pdf":"https://www.isca-archive.org/interspeech_2026/bui26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/bui26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/bui26_interspeech/markdown.md"},{"id":"buitrago26_interspeech","title":"Quantifying Cross-Lingual Transfer in Paralinguistic Speech Tasks","authors":["Pol Buitrago","Oriol Pareras","Federico Costa","Javier Hernando"],"year":2026,"doi":"10.21437/Interspeech.2026-2745","isca_url":"https://www.isca-archive.org/interspeech_2026/buitrago26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/buitrago26_interspeech.pdf","session":"Multilingual and Cross-Lingual Paralinguistic Analysis and Processing","topics":["paralinguistics","speaker-verification","self-supervised"],"category":"paralinguistics-emotion","labels":["low-resource","multilingual","self-supervised"],"institutions":["Barcelona Supercomputing Center","Universitat Politecnica de Catalunya"],"funding":["MICIU","AEI","NextGenerationEU","PRTR","Red.es","Ministerio para la Transformacion Digital y de la Funcion Publica"],"code":{"url":"https://github.com/Pol-Buitrago/cltm-framework","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"buitrago26_interspeech","category":"paralinguistics-emotion","labels":["low-resource","multilingual","self-supervised"],"institutions":["Barcelona Supercomputing Center","Universitat Politecnica de Catalunya"],"code":"https://github.com/Pol-Buitrago/cltm-framework","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2745","pdf":"https://www.isca-archive.org/interspeech_2026/buitrago26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/buitrago26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/buitrago26_interspeech/markdown.md"},{"id":"buragohain26_interspeech","title":"Exploiting EEG-based Gamma-Band Time Frequency Feature in WaveNet Decoder Framework for High-Fidelity Speech Reconstruction","authors":["Rantu Buragohain","Saket Maheshwari","Karan Nathwani"],"year":2026,"doi":"10.21437/Interspeech.2026-377","isca_url":"https://www.isca-archive.org/interspeech_2026/buragohain26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/buragohain26_interspeech.pdf","session":"Neurophysiology of Speech","topics":["speech-synthesis","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["Indian Institute of Technology Jammu","GLA University"],"funding":["TIH IIT Guwahati"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"buragohain26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Indian Institute of Technology Jammu","GLA University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-377","pdf":"https://www.isca-archive.org/interspeech_2026/buragohain26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/buragohain26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/buragohain26_interspeech/markdown.md"},{"id":"burdisso26_interspeech","title":"Avoiding Catastrophic Forgetting in Text-Only Adaptation of LLM-based ASR via Multi-View Text Denoising","authors":["Sergio Burdisso","Esaú Villatoro-Tello","Thibault Bañeras-Roux","Shashi Kumar","Srikanth Madikeri","Pradeep Rangappa","Manjunath K E","Petr Motlicek","Andreas Stolcke"],"year":2026,"doi":"10.21437/Interspeech.2026-3422","isca_url":"https://www.isca-archive.org/interspeech_2026/burdisso26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/burdisso26_interspeech.pdf","session":"Multi-Speaker Processing, Personalization, and Adaptation","topics":["asr","speech-llm","self-supervised"],"category":"asr","institutions":["Idiap Research Institute","EPFL","University of Zurich","Uniphore","Brno University of Technology"],"funding":["Idiap Research Institute and Uniphore collaboration project","EU Horizon 2020 project ELOQUENCE"],"code":{"url":"https://github.com/idiap/llm-asr-text-only-adaptation","stars":5,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"burdisso26_interspeech","category":"asr","institutions":["Idiap Research Institute","EPFL","University of Zurich","Uniphore","Brno University of Technology"],"code":"https://github.com/idiap/llm-asr-text-only-adaptation","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3422","pdf":"https://www.isca-archive.org/interspeech_2026/burdisso26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/burdisso26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/burdisso26_interspeech/markdown.md"},{"id":"cai26_interspeech","title":"Relative Importance of Formants to the Intelligibility of Vocoded Speech in Cochlear Implant Simulation","authors":["Ying Cai","Yuting Ding","Xuefei Wang","Fei Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-143","isca_url":"https://www.isca-archive.org/interspeech_2026/cai26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cai26_interspeech.pdf","session":"Assistive Technologies 2","topics":["speech-enhancement","evaluation","phonetics"],"category":"health-clinical","institutions":["Southern University of Science and Technology"],"funding":["National Key Research and Development Program of China","National Natural Science Foundation of China"],"code":{"url":"https://www.haskinslaboratories.org/sws","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cai26_interspeech","category":"health-clinical","institutions":["Southern University of Science and Technology"],"code":"https://www.haskinslaboratories.org/sws","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-143","pdf":"https://www.isca-archive.org/interspeech_2026/cai26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cai26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cai26_interspeech/markdown.md"},{"id":"cai26b_interspeech","title":"Enhancing Temporal Prediction Consistency for Short-Duration Acoustic Scene Classification via Semantic Adversarial Training","authors":["Yiqiang Cai","Yizhou Tan","Peihong Zhang","Yuxuan Liu","Shengchen Li","Xi Shao"],"year":2026,"doi":"10.21437/Interspeech.2026-1955","isca_url":"https://www.isca-archive.org/interspeech_2026/cai26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cai26b_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["acoustic-scene-classification","self-supervised","evaluation"],"category":"audio-understanding","institutions":["Xi'an Jiaotong-Liverpool University","Nanjing University of Posts and Telecommunications"],"funding":["Jiangsu Provincial Major Science and Technology Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cai26b_interspeech","category":"audio-understanding","institutions":["Xi'an Jiaotong-Liverpool University","Nanjing University of Posts and Telecommunications"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1955","pdf":"https://www.isca-archive.org/interspeech_2026/cai26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cai26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cai26b_interspeech/markdown.md"},{"id":"cai26c_interspeech","title":"Beyond Mimicry: Constrained Exploration with GRPO for Joint Multi-Talker ASR and Diarization under Unknown Speaker Counts","authors":["Yunrui Cai","Dingdong Wang","Lingwei Meng","Xixin Wu","Zhiyong Wu","Helen Meng"],"year":2026,"doi":"10.21437/Interspeech.2026-2297","isca_url":"https://www.isca-archive.org/interspeech_2026/cai26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cai26c_interspeech.pdf","session":"Multi-Speaker Processing, Personalization, and Adaptation","topics":["asr","speech-llm","self-supervised"],"category":"asr","institutions":["Chinese University of Hong Kong","Tsinghua University"],"funding":["Centre for Perceptual and Interactive Intelligence","Innovation and Technology Commission of the Hong Kong Special Administrative Region Government"],"code":{"url":"https://github.com/caiyunrui/MT-GRPO","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cai26c_interspeech","category":"asr","institutions":["Chinese University of Hong Kong","Tsinghua University"],"code":"https://github.com/caiyunrui/MT-GRPO","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2297","pdf":"https://www.isca-archive.org/interspeech_2026/cai26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cai26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cai26c_interspeech/markdown.md"},{"id":"cai26d_interspeech","title":"Towards Event-Robust Acoustic Scene Classification","authors":["Yiqiang Cai","Bohan Hu","Yu Yang","Pengwei Lu","Shengchen Li","Xi Shao"],"year":2026,"doi":"10.21437/Interspeech.2026-2350","isca_url":"https://www.isca-archive.org/interspeech_2026/cai26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cai26d_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["asr","self-supervised","evaluation"],"category":"audio-understanding","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["Xi'an Jiaotong-Liverpool University","Zhongdian Zhiheng Information Technology Service Co., Ltd","China Telecom Jiangsu Branch","Nanjing University of Posts and Telecommunications"],"funding":["Jiangsu Provincial Major Science and Technology Project"],"code":{"url":"https://github.com/bohanhu118/Interspeech2026_ESAS","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cai26d_interspeech","category":"audio-understanding","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["Xi'an Jiaotong-Liverpool University","Zhongdian Zhiheng Information Technology Service Co., Ltd","China Telecom Jiangsu Branch","Nanjing University of Posts and Telecommunications"],"code":"https://github.com/bohanhu118/Interspeech2026_ESAS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2350","pdf":"https://www.isca-archive.org/interspeech_2026/cai26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cai26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cai26d_interspeech/markdown.md"},{"id":"calhoun26_interspeech","title":"Listeners' gendered experiences and beliefs affect iconic pitch associations","authors":["Sasha Calhoun","Paul Warren","Sara Gilbert"],"year":2026,"doi":"10.21437/Interspeech.2026-1642","isca_url":"https://www.isca-archive.org/interspeech_2026/calhoun26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/calhoun26_interspeech.pdf","session":"Gender- and Age-Related Speech Characteristics","topics":["paralinguistics","prosody","evaluation"],"category":"phonetics-linguistics","institutions":["Victoria University of Wellington"],"funding":["Faculty Strategic Research Grant from Te Herenga Waka – Victoria University of Wellington"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"calhoun26_interspeech","category":"phonetics-linguistics","institutions":["Victoria University of Wellington"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1642","pdf":"https://www.isca-archive.org/interspeech_2026/calhoun26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/calhoun26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/calhoun26_interspeech/markdown.md"},{"id":"callejas26_interspeech","title":"MultiLinguahah : A New Unsupervised Multilingual Acoustic Laughter Segmentation Method","authors":["Sofia Callejas","Nahuel Gomez","Catherine Pelachaud","Brian Ravenet","Valentin Barriere"],"year":2026,"doi":"10.21437/Interspeech.2026-2352","isca_url":"https://www.isca-archive.org/interspeech_2026/callejas26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/callejas26_interspeech.pdf","session":"Audio signal analysis","topics":["paralinguistics","self-supervised","multilingual"],"category":"paralinguistics-emotion","labels":["low-resource","multilingual","self-supervised"],"institutions":["Universite Paris-Saclay","Universidad de Chile","Sorbonne University"],"code":{"url":"https://github.com/sofia-callejas/Multilinguahah","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"callejas26_interspeech","category":"paralinguistics-emotion","labels":["low-resource","multilingual","self-supervised"],"institutions":["Universite Paris-Saclay","Universidad de Chile","Sorbonne University"],"code":"https://github.com/sofia-callejas/Multilinguahah","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2352","pdf":"https://www.isca-archive.org/interspeech_2026/callejas26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/callejas26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/callejas26_interspeech/markdown.md"},{"id":"camara26_interspeech","title":"An Acoustic Landmark Database of the English Lexicon via Articulatory Synthesis","authors":["Mateo Cámara","José Luis Blanco","Juan Ignacio Godino-Llorente","Jeung-Yoon Choi","Stefanie Shattuck-Hufnagel"],"year":2026,"doi":"10.21437/Interspeech.2026-1374","isca_url":"https://www.isca-archive.org/interspeech_2026/camara26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/camara26_interspeech.pdf","session":"Emotion, Prosody, and Articulation","topics":["speech-enhancement","phonetics","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Universidad Politecnica de Madrid","Massachusetts Institute of Technology"],"funding":["Ministry of Economy and Competitiveness of Spain","Fundacion Santander","MISTI MIT Global Experiences"],"code":{"url":"https://huggingface.co/datasets/mcamara/all-words-in-english-with-pink-trombone","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"camara26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Universidad Politecnica de Madrid","Massachusetts Institute of Technology"],"code":"https://huggingface.co/datasets/mcamara/all-words-in-english-with-pink-trombone","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1374","pdf":"https://www.isca-archive.org/interspeech_2026/camara26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/camara26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/camara26_interspeech/markdown.md"},{"id":"camara26b_interspeech","title":"Word Lengthening as a Function of Utterance Position: A Multi-Corpus Study","authors":["Mateo Cámara","José Luis Blanco","Juan Ignacio Godino-Llorente","Jeung-Yoon Choi","Stefanie Shattuck-Hufnagel"],"year":2026,"doi":"10.21437/Interspeech.2026-1379","isca_url":"https://www.isca-archive.org/interspeech_2026/camara26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/camara26b_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["prosody","phonetics","dataset"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Universidad Politecnica de Madrid","Massachusetts Institute of Technology"],"funding":["Ministry of Economy and Competitiveness of Spain","Fundacion Santander","MISTI MIT Global Experiences program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"camara26b_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Universidad Politecnica de Madrid","Massachusetts Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1379","pdf":"https://www.isca-archive.org/interspeech_2026/camara26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/camara26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/camara26b_interspeech/markdown.md"},{"id":"camara26c_interspeech","title":"Acoustic Landmark Detector based on Conformer and HuBERT","authors":["Mateo Cámara","José Luis Blanco","Juan Ignacio Godino-Llorente","Jeung-Yoon Choi","Stefanie Shattuck-Hufnagel"],"year":2026,"doi":"10.21437/Interspeech.2026-1386","isca_url":"https://www.isca-archive.org/interspeech_2026/camara26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/camara26c_interspeech.pdf","session":"Acoustic Event Detection 1","topics":["phonetics","self-supervised","evaluation"],"category":"asr","labels":["self-supervised"],"institutions":["Universidad Politecnica de Madrid","Massachusetts Institute of Technology"],"funding":["Ministry of Economy and Competitiveness of Spain","Fundacion Santander","MISTI MIT Global Experiences"],"code":{"url":"https://mateocamara.github.io/acoustic-landmarks/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"camara26c_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Universidad Politecnica de Madrid","Massachusetts Institute of Technology"],"code":"https://mateocamara.github.io/acoustic-landmarks/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1386","pdf":"https://www.isca-archive.org/interspeech_2026/camara26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/camara26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/camara26c_interspeech/markdown.md"},{"id":"cao26_interspeech","title":"X-OPD: Cross-Modal On-Policy Distillation for Capability Alignment in Speech LLMs","authors":["Di Cao","Dongjie Fu","Hai Yu","Siqi Zheng","Xu Tan","Tao Jin"],"year":2026,"doi":"10.21437/Interspeech.2026-861","isca_url":"https://www.isca-archive.org/interspeech_2026/cao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cao26_interspeech.pdf","session":"Multimodal Spoken Dialogue Systems","topics":["speech-llm","self-supervised","multilingual"],"category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["Tencent","Zhejiang University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cao26_interspeech","category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["Tencent","Zhejiang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-861","pdf":"https://www.isca-archive.org/interspeech_2026/cao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cao26_interspeech/markdown.md"},{"id":"cao26b_interspeech","title":"Audio-NSP: Data-Centric Semi-Autoregressive Generation for Large Audio-Language Models","authors":["Liang Cao","Xize Cheng","Dongjie Fu","Weihao Wu","Fuming You","Zhiyong Wu","Haifeng Hu"],"year":2026,"doi":"10.21437/Interspeech.2026-1737","isca_url":"https://www.isca-archive.org/interspeech_2026/cao26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cao26b_interspeech.pdf","session":"Efficient Inference for ASR and Speech LMs","topics":["speech-llm","asr","tts"],"category":"speech-llm-dialogue","labels":["efficient-on-device","generative-model"],"institutions":["Tsinghua University","Zhejiang University","Tencent"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cao26b_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","generative-model"],"institutions":["Tsinghua University","Zhejiang University","Tencent"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1737","pdf":"https://www.isca-archive.org/interspeech_2026/cao26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cao26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cao26b_interspeech/markdown.md"},{"id":"cappellazzo26_interspeech","title":"Dr. SHAP-AV: Decoding Relative Modality Contributions via Shapley Attribution in Audio-Visual Speech Recognition","authors":["Umberto Cappellazzo","Stavros Petridis","Maja Pantic"],"year":2026,"doi":"10.21437/Interspeech.2026-417","isca_url":"https://www.isca-archive.org/interspeech_2026/cappellazzo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cappellazzo26_interspeech.pdf","session":"Audio-Visual and Multimodal Perception","topics":["speech-llm","evaluation","self-supervised"],"category":"asr","labels":["robustness-noise"],"institutions":["Imperial College London","NatWest AI Research"],"code":{"url":"https://umbertocappellazzo.github.io/Dr-SHAP-AV","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cappellazzo26_interspeech","category":"asr","labels":["robustness-noise"],"institutions":["Imperial College London","NatWest AI Research"],"code":"https://umbertocappellazzo.github.io/Dr-SHAP-AV","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-417","pdf":"https://www.isca-archive.org/interspeech_2026/cappellazzo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cappellazzo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cappellazzo26_interspeech/markdown.md"},{"id":"carne26_interspeech","title":"Function words: a topic independent approach to word n-gram selection for forensic speaker comparison","authors":["Michael Carne"],"year":2026,"doi":"10.21437/Interspeech.2026-3166","isca_url":"https://www.isca-archive.org/interspeech_2026/carne26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/carne26_interspeech.pdf","session":"Speaker Recognition and Verification","topics":["speaker-verification","evaluation","paralinguistics"],"category":"speaker","institutions":["Australian National University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"carne26_interspeech","category":"speaker","institutions":["Australian National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3166","pdf":"https://www.isca-archive.org/interspeech_2026/carne26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/carne26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/carne26_interspeech/markdown.md"},{"id":"carson26_interspeech","title":"Balancing Speech Reconstruction and Noise Suppression Using Dual-Asymmetric Loss","authors":["Merlin Carson","Suyash Dandekar"],"year":2026,"doi":"10.21437/Interspeech.2026-794","isca_url":"https://www.isca-archive.org/interspeech_2026/carson26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/carson26_interspeech.pdf","session":"SE Architectures, Adaptation and Audio Front-Ends","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Skyworks Solutions","Georgia Institute of Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"carson26_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Skyworks Solutions","Georgia Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-794","pdf":"https://www.isca-archive.org/interspeech_2026/carson26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/carson26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/carson26_interspeech/markdown.md"},{"id":"carvalho26_interspeech","title":"Exploring the potential and limitations of Model Merging for Multi-Domain Adaptation in ASR","authors":["Carlos Carvalho","Francisco Teixeira","Thomas Rolland","Alberto Abad"],"year":2026,"doi":"10.21437/Interspeech.2026-1969","isca_url":"https://www.isca-archive.org/interspeech_2026/carvalho26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/carvalho26_interspeech.pdf","session":"Domain Adaptation & Accented ASR","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["INESC-ID","Universidade de Lisboa"],"funding":["Fundacao para a Ciencia e a Tecnologia","Portuguese Recovery and Resilience Plan"],"code":{"url":"https://github.com/Miamoto/mergewhisper","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"carvalho26_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["INESC-ID","Universidade de Lisboa"],"code":"https://github.com/Miamoto/mergewhisper","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1969","pdf":"https://www.isca-archive.org/interspeech_2026/carvalho26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/carvalho26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/carvalho26_interspeech/markdown.md"},{"id":"casalssalvador26_interspeech","title":"How Attention Shapes Emotion: A Comparative Study of Attention Mechanisms for Speech Emotion Recognition","authors":["Marc Casals-Salvador","Federico Costa","Rodolfo Zevallos","Javier Hernando"],"year":2026,"doi":"10.21437/Interspeech.2026-1907","isca_url":"https://www.isca-archive.org/interspeech_2026/casalssalvador26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/casalssalvador26_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["speech-emotion-recognition","self-supervised","evaluation"],"category":"paralinguistics-emotion","labels":["efficient-on-device"],"institutions":["Barcelona Supercomputing Center","Universitat Politecnica de Catalunya"],"funding":["MICIU/AEI","Red.es, Ministerio para la Transformacion Digital y de la Funcion Publica","NextGenerationEU"],"code":{"url":"https://github.com/marccasals98/AttentionAlternatives","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"casalssalvador26_interspeech","category":"paralinguistics-emotion","labels":["efficient-on-device"],"institutions":["Barcelona Supercomputing Center","Universitat Politecnica de Catalunya"],"code":"https://github.com/marccasals98/AttentionAlternatives","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1907","pdf":"https://www.isca-archive.org/interspeech_2026/casalssalvador26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/casalssalvador26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/casalssalvador26_interspeech/markdown.md"},{"id":"chan26_interspeech","title":"Privacy vs. Performance: Assessing Communication Utility of Anonymized Voice Features","authors":["Amanda Chan","Chee Wee Leong","Candy Olivia Mawalim","Shogo Okada"],"year":2026,"doi":"10.21437/Interspeech.2026-751","isca_url":"https://www.isca-archive.org/interspeech_2026/chan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chan26_interspeech.pdf","session":"Speaker Privacy and Anonymization","topics":["speech-enhancement","evaluation","paralinguistics"],"category":"deepfake-security","institutions":["Educational Testing Service","Japan Advanced Institute of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chan26_interspeech","category":"deepfake-security","institutions":["Educational Testing Service","Japan Advanced Institute of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-751","pdf":"https://www.isca-archive.org/interspeech_2026/chan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chan26_interspeech/markdown.md"},{"id":"chang26_interspeech","title":"TAD: Token-Adaptive Contrastive Decoding with Confidence-Guided Gating for Hallucination Mitigation in Large Audio-Language Models","authors":["Heyu Chang","Nianwen Si","Hao Zhang","Wenlin Zhang","Dan Qu"],"year":2026,"doi":"10.21437/Interspeech.2026-637","isca_url":"https://www.isca-archive.org/interspeech_2026/chang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chang26_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","evaluation","self-supervised"],"category":"audio-understanding","institutions":["Information Engineering University"],"funding":["Henan Province Major Industrial “Challenge-Based Innovation”","Natural Science Foundation of Henan","Science and Technology Key Project Plan of Henan"],"code":{"url":"https://github.com/Changhy26/TAD","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chang26_interspeech","category":"audio-understanding","institutions":["Information Engineering University"],"code":"https://github.com/Changhy26/TAD","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-637","pdf":"https://www.isca-archive.org/interspeech_2026/chang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26_interspeech/markdown.md"},{"id":"chang26b_interspeech","title":"A Two-Stage Defence for Robust Federated Speech Emotion Recognition","authors":["Yi Chang","Sofiane Laridi","Zhao Ren","Gregory Palmer","Björn W. Schuller","Marco Fisichella"],"year":2026,"doi":"10.21437/Interspeech.2026-1125","isca_url":"https://www.isca-archive.org/interspeech_2026/chang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chang26b_interspeech.pdf","session":"Emotion, Prosody, and Articulation","topics":["speech-emotion-recognition","self-supervised","evaluation"],"category":"paralinguistics-emotion","institutions":["Imperial College London","Leibniz University Hannover","University of Bremen"],"funding":["DFG, German Research Foundation","BMBF","KI-Servicezentrum für sensible und kritische Infrastrukturen (KISSKI)"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chang26b_interspeech","category":"paralinguistics-emotion","institutions":["Imperial College London","Leibniz University Hannover","University of Bremen"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1125","pdf":"https://www.isca-archive.org/interspeech_2026/chang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26b_interspeech/markdown.md"},{"id":"chang26c_interspeech","title":"USAD 2.0: Scaling Representation Distillation for Universal Audio Understanding","authors":["Heng-Jui Chang","Alexander Liu","Saurabhchand Bhati","Mrudula Athi","Anton Ratnarajah","Amit Chhetri","James Glass"],"year":2026,"doi":"10.21437/Interspeech.2026-1296","isca_url":"https://www.isca-archive.org/interspeech_2026/chang26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chang26c_interspeech.pdf","session":"Target Speaker Extraction, Speech Separation and Audio Understanding","topics":["self-supervised","multilingual","speech-llm"],"category":"audio-understanding","labels":["multilingual","self-supervised"],"institutions":["Massachusetts Institute of Technology","Amazon"],"code":{"url":"https://hf.co/collections/MIT-SLS/usad2","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chang26c_interspeech","category":"audio-understanding","labels":["multilingual","self-supervised"],"institutions":["Massachusetts Institute of Technology","Amazon"],"code":"https://hf.co/collections/MIT-SLS/usad2","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1296","pdf":"https://www.isca-archive.org/interspeech_2026/chang26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26c_interspeech/markdown.md"},{"id":"chang26d_interspeech","title":"TaigiSpeech: A Low-Resource Real-World Speech Intent Dataset with Scalable Data Mining In-the-Wild","authors":["Kai-Wei Chang","Yi-Cheng Lin","Huang-Cheng Chou","Wenze Ren","Yu-Han Huang","Yun-Shao Tsai","Chien-Cheng Chen","Yu Tsao","Yuan-Fu Liao","Shrikanth Narayanan","James Glass","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1511","isca_url":"https://www.isca-archive.org/interspeech_2026/chang26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chang26d_interspeech.pdf","session":"Benchmarking Foundation Models","topics":["spoken-language-understanding","low-resource","self-supervised"],"category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Massachusetts Institute of Technology","National Taiwan University","National Taiwan University Artificial Intelligence Center of Research Excellence","Academia Sinica","National Yang Ming Chiao Tung University","University of Southern California"],"funding":["National Science and Technology Council","Ministry of Education","National Science Foundation","Intelligence Advanced Research Projects Activity"],"code":{"url":"https://kwchang.org/taigispeech","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chang26d_interspeech","category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Massachusetts Institute of Technology","National Taiwan University","National Taiwan University Artificial Intelligence Center of Research Excellence","Academia Sinica","National Yang Ming Chiao Tung University","University of Southern California"],"code":"https://kwchang.org/taigispeech","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1511","pdf":"https://www.isca-archive.org/interspeech_2026/chang26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26d_interspeech/markdown.md"},{"id":"chang26e_interspeech","title":"Personalized Electrolaryngeal Voice Conversion with a Single Pre-operative Utterance","authors":["Devin Chang","Hsin-Te Hwang","Ming-Chi Yen","Shu-Wei Tsai","Yu Tsao","Hsin-Min Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1942","isca_url":"https://www.isca-archive.org/interspeech_2026/chang26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chang26e_interspeech.pdf","session":"Beyond Speech Technologies in Healthcare","topics":["voice-conversion","speech-enhancement","low-resource"],"category":"tts","labels":["low-resource","generative-model"],"institutions":["Academia Sinica","National Cheng Kung University Hospital"],"funding":["National Science and Technology Council"],"code":{"url":"https://devchang918.github.io/IS2026_one_shot_personalized_ELVC/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chang26e_interspeech","category":"tts","labels":["low-resource","generative-model"],"institutions":["Academia Sinica","National Cheng Kung University Hospital"],"code":"https://devchang918.github.io/IS2026_one_shot_personalized_ELVC/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1942","pdf":"https://www.isca-archive.org/interspeech_2026/chang26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26e_interspeech/markdown.md"},{"id":"chang26f_interspeech","title":"VIP-MINGLE: A Corpus for Videoconference and In-Person Multimodal Interaction in Group Language Engagement","authors":["Andrew Chang","Abhinay K Bodi","Wenxin Deng","Junrui Huang","Venu G Kadamba","Sumanth B H Karanam","Dhiwahar A Kennady","David Poeppel","Dustin Freeman"],"year":2026,"doi":"10.21437/Interspeech.2026-2367","isca_url":"https://www.isca-archive.org/interspeech_2026/chang26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chang26f_interspeech.pdf","session":"Datasets","topics":["multilingual","self-supervised","dataset"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["New York University"],"funding":["NYU Discovery Research Fund for Human Health","Leon Levy Foundation","New York Academy of Sciences"],"code":{"url":"https://doi.org/10.5281/zenodo.20670131","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chang26f_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["New York University"],"code":"https://doi.org/10.5281/zenodo.20670131","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2367","pdf":"https://www.isca-archive.org/interspeech_2026/chang26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26f_interspeech/markdown.md"},{"id":"chang26g_interspeech","title":"From Words to Sentences: Contextual Predictability Overrides Phonetic Ambiguity in Lexical Competition","authors":["Will Chih-Chao Chang","Jiaxuan Li","Xin Xie"],"year":2026,"doi":"10.21437/Interspeech.2026-3106","isca_url":"https://www.isca-archive.org/interspeech_2026/chang26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chang26g_interspeech.pdf","session":"Modeling L1 Acquisition","topics":["phonetics","speech-perception","spoken-language-understanding"],"category":"phonetics-linguistics","institutions":["University of California, Irvine"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chang26g_interspeech","category":"phonetics-linguistics","institutions":["University of California, Irvine"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3106","pdf":"https://www.isca-archive.org/interspeech_2026/chang26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chang26g_interspeech/markdown.md"},{"id":"chao26_interspeech","title":"RT-SEMamba: Real-Time Speech Enhancement Mamba via Progressive Knowledge Distillation","authors":["Rong Chao","Sung-Feng Huang","Moreno La Quatra","Sabato Marco Siniscalchi","Wen-Huang Cheng","Szu-Wei Fu","Yu Tsao"],"year":2026,"doi":"10.21437/Interspeech.2026-3197","isca_url":"https://www.isca-archive.org/interspeech_2026/chao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chao26_interspeech.pdf","session":"Real-Time, Low-Latency and Edge Speech Enhancement","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Academia Sinica","National Taiwan University","Kore University of Enna","University of Palermo","NVIDIA"],"code":{"url":"https://github.com/RoyChao19477/RT-SEMamba","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chao26_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Academia Sinica","National Taiwan University","Kore University of Enna","University of Palermo","NVIDIA"],"code":"https://github.com/RoyChao19477/RT-SEMamba","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3197","pdf":"https://www.isca-archive.org/interspeech_2026/chao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chao26_interspeech/markdown.md"},{"id":"charlot26_interspeech","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","authors":["Théo Charlot","Tarek Kunze","Maxime Poli","Alejandrina Cristia","Emmanuel Dupoux","Marvin Lavechin"],"year":2026,"doi":"10.21437/Interspeech.2026-2772","isca_url":"https://www.isca-archive.org/interspeech_2026/charlot26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/charlot26_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["self-supervised","speaker-diarization","multilingual"],"category":"speaker","labels":["multilingual","self-supervised"],"institutions":["École Normale Supérieure","École des Hautes Études en Sciences Sociales","CNRS","PSL University","Aix-Marseille University"],"funding":["Agence Nationale pour la Recherche","European Research Council","Simons Foundation International","Agence de l'Innovation de Défense"],"code":{"url":"https://github.com/LAAC-LSCP/VTC","stars":15,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"charlot26_interspeech","category":"speaker","labels":["multilingual","self-supervised"],"institutions":["École Normale Supérieure","École des Hautes Études en Sciences Sociales","CNRS","PSL University","Aix-Marseille University"],"code":"https://github.com/LAAC-LSCP/VTC","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2772","pdf":"https://www.isca-archive.org/interspeech_2026/charlot26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/charlot26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/charlot26_interspeech/markdown.md"},{"id":"charlot26b_interspeech","title":"Context-aware child-directed speech detection from long-form recordings","authors":["Théo Charlot","Tarek Kunze","Kaveri K. Sheth","Alejandrina Cristia","Marvin Lavechin"],"year":2026,"doi":"10.21437/Interspeech.2026-2780","isca_url":"https://www.isca-archive.org/interspeech_2026/charlot26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/charlot26b_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["speech-llm","self-supervised","multilingual"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["École Normale Supérieure","École des Hautes Études en Sciences Sociales","Centre National de la Recherche Scientifique","Université PSL","Aix-Marseille University"],"funding":["Agence Nationale pour la Recherche","European Research Council","Simons Foundation"],"code":{"url":"https://github.com/LAAC-LSCP/addressee","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"charlot26b_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["École Normale Supérieure","École des Hautes Études en Sciences Sociales","Centre National de la Recherche Scientifique","Université PSL","Aix-Marseille University"],"code":"https://github.com/LAAC-LSCP/addressee","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2780","pdf":"https://www.isca-archive.org/interspeech_2026/charlot26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/charlot26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/charlot26b_interspeech/markdown.md"},{"id":"chellaf26_interspeech","title":"Bridging Languages and Modalities: Lightweight Cross-Lingual Text and Speech Summarization for Low-Resource Scenarios","authors":["Chaimae Chellaf","Salima Mdhaffar","Yannick Estève","Stéphane Huet"],"year":2026,"doi":"10.21437/Interspeech.2026-2655","isca_url":"https://www.isca-archive.org/interspeech_2026/chellaf26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chellaf26_interspeech.pdf","session":"Corpus Creation, Summarization and Understanding","topics":["speech-translation","self-supervised","multilingual"],"category":"translation","labels":["low-resource","multilingual","efficient-on-device"],"institutions":["Avignon Universite","Lundi Matin"],"code":{"url":"https://huggingface.co/datasets/cchellaf/ABT-SpeechSUM","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chellaf26_interspeech","category":"translation","labels":["low-resource","multilingual","efficient-on-device"],"institutions":["Avignon Universite","Lundi Matin"],"code":"https://huggingface.co/datasets/cchellaf/ABT-SpeechSUM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2655","pdf":"https://www.isca-archive.org/interspeech_2026/chellaf26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chellaf26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chellaf26_interspeech/markdown.md"},{"id":"chen26_interspeech","title":"MoVE: Translating Laughter and Tears via Mixture of Vocalization Experts in Speech-to-Speech Translation","authors":["Szu-Chi Chen","I-Ning Tsai","Yi-Cheng Lin","Sung-Feng Huang","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-42","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26_interspeech.pdf","session":"Multilingual Speech 2","topics":["speech-translation","speech-llm","self-supervised"],"category":"translation","labels":["multilingual","generative-model"],"institutions":["National Taiwan University","NVIDIA"],"funding":["Ministry of Education"],"code":{"url":"https://47zzz.github.io/MoVE/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26_interspeech","category":"translation","labels":["multilingual","generative-model"],"institutions":["National Taiwan University","NVIDIA"],"code":"https://47zzz.github.io/MoVE/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-42","pdf":"https://www.isca-archive.org/interspeech_2026/chen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26_interspeech/markdown.md"},{"id":"chen26aa_interspeech","title":"PolyBench: A Benchmark for Compositional Reasoning in Polyphonic Audio","authors":["Yuanjian Chen","Yang Xiao","Han Yin","Xubo Liu","Jinjie Huang","Ting Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-2466","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26aa_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26aa_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["speech-llm","source-separation","evaluation"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Harbin University of Science and Technology","University of Melbourne","KAIST","University of Surrey"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26aa_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Harbin University of Science and Technology","University of Melbourne","KAIST","University of Surrey"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2466","pdf":"https://www.isca-archive.org/interspeech_2026/chen26aa_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26aa_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26aa_interspeech/markdown.md"},{"id":"chen26b_interspeech","title":"The Binding Effect: Analysis of How Multi-Dimensional Cues Form Gender Bias in Instruction TTS","authors":["Kuan-Yu Chen","Yi-Cheng Lin","Po-Chung Hsieh","Huang-Cheng Chou","Chih-Fan Hsu","Jeng-Lin Li","Hung-yi Lee","Jian-Jiun Ding"],"year":2026,"doi":"10.21437/Interspeech.2026-66","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26b_interspeech.pdf","session":"Speech Synthesis Evaluation 1","topics":["tts","self-supervised","evaluation"],"category":"tts","institutions":["National Taiwan University","Inventec Corporation","University of Southern California"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26b_interspeech","category":"tts","institutions":["National Taiwan University","Inventec Corporation","University of Southern California"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-66","pdf":"https://www.isca-archive.org/interspeech_2026/chen26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26b_interspeech/markdown.md"},{"id":"chen26ba_interspeech","title":"Probing Spatial Structure in Pretrained Audio Representations","authors":["Chuyang Chen","Sivan Ding","Adrian S. Roman","Juan P. Bello"],"year":2026,"doi":"10.21437/Interspeech.2026-2506","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26ba_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26ba_interspeech.pdf","session":"Evaluation, Benchmarking, and Reliability of Audio Systems","topics":["self-supervised","evaluation","speaker-diarization"],"category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["New York University"],"funding":["NYU / SONY Audio Institute for Music Business and Technology"],"code":{"url":"https://github.com/chuyangchencd/SARL","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26ba_interspeech","category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["New York University"],"code":"https://github.com/chuyangchencd/SARL","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2506","pdf":"https://www.isca-archive.org/interspeech_2026/chen26ba_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26ba_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26ba_interspeech/markdown.md"},{"id":"chen26c_interspeech","title":"Beyond Pitch: Multidimensional Cue Reweighting of Two High-Falling Tones in Pingdingshan Mandarin","authors":["Zhuo Chen","Bingliang Zhao","Xiyu Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-361","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26c_interspeech.pdf","session":"Tones","topics":["phonetics","speech-enhancement","evaluation"],"category":"phonetics-linguistics","institutions":["Peking University"],"funding":["National Social Science Foundation of China"],"code":{"url":"https://github.com/Dzaau/pingdingshanIS2026","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26c_interspeech","category":"phonetics-linguistics","institutions":["Peking University"],"code":"https://github.com/Dzaau/pingdingshanIS2026","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-361","pdf":"https://www.isca-archive.org/interspeech_2026/chen26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26c_interspeech/markdown.md"},{"id":"chen26ca_interspeech","title":"Inside the Latent Flow: Causal Deciphering of Attention Dynamics in Audio Separation Foundation Models","authors":["Yuxuan Chen","Haoyuan Yu","Peize He"],"year":2026,"doi":"10.21437/Interspeech.2026-2684","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26ca_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26ca_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["source-separation","self-supervised","speech-enhancement"],"category":"enhancement-separation","labels":["efficient-on-device","self-supervised"],"institutions":["Chinese University of Hong Kong, Shenzhen","Jilin University","Hunan University","University of Electronic Science and Technology of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26ca_interspeech","category":"enhancement-separation","labels":["efficient-on-device","self-supervised"],"institutions":["Chinese University of Hong Kong, Shenzhen","Jilin University","Hunan University","University of Electronic Science and Technology of China"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2684","pdf":"https://www.isca-archive.org/interspeech_2026/chen26ca_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26ca_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26ca_interspeech/markdown.md"},{"id":"chen26d_interspeech","title":"YODAS v3: Over 1 Million Hours of High-Bandwidth, Stereophonic, Multilingual Speech","authors":["William Chen","Shinnosuke Takamichi","Sayaka Shiota","Satoru Fukayama","Samuele Cornell","Shinji Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-386","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26d_interspeech.pdf","session":"Multilingual Speech 2","topics":["dataset","multilingual","self-supervised"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Carnegie Mellon University","Keio University","Tokyo Metropolitan University","National Institute of Advanced Industrial Science and Technology"],"funding":["ACCESS program","National Science Foundation","Cabinet Office, Government of Japan"],"code":{"url":"https://huggingface.co/datasets/espnet/yodas3","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26d_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Carnegie Mellon University","Keio University","Tokyo Metropolitan University","National Institute of Advanced Industrial Science and Technology"],"code":"https://huggingface.co/datasets/espnet/yodas3","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-386","pdf":"https://www.isca-archive.org/interspeech_2026/chen26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26d_interspeech/markdown.md"},{"id":"chen26da_interspeech","title":"Spectro-Temporal Interference Confounds Phase Encoding in Spatial Audio Foundation Models","authors":["Yuxuan Chen","Haoyuan Yu","Peize He"],"year":2026,"doi":"10.21437/Interspeech.2026-2873","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26da_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26da_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["self-supervised","evaluation","spatial-audio"],"category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["Chinese University of Hong Kong","Jilin University","Hunan University","University of Electronic Science and Technology of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26da_interspeech","category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["Chinese University of Hong Kong","Jilin University","Hunan University","University of Electronic Science and Technology of China"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2873","pdf":"https://www.isca-archive.org/interspeech_2026/chen26da_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26da_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26da_interspeech/markdown.md"},{"id":"chen26e_interspeech","title":"CAAD: Contrastive Audio-Aware Distillation for Efficient Speech Language Models","authors":["Chun Wei Chen","Tzu-Quan Lin","Ke-Han Lu","Wei-Ping Huang","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-645","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26e_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised"],"institutions":["National Taiwan University"],"funding":["Ministry of Education, Taiwan","Taiwan Centers of Excellence in Artificial Intelligence"],"code":{"url":"https://github.com/ChenWils/Contrastive_Audio-Aware_Distillation","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26e_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised"],"institutions":["National Taiwan University"],"code":"https://github.com/ChenWils/Contrastive_Audio-Aware_Distillation","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-645","pdf":"https://www.isca-archive.org/interspeech_2026/chen26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26e_interspeech/markdown.md"},{"id":"chen26ea_interspeech","title":"SFL-MTSC: Leveraging Semantic Frame-Level Multi-Task Self-Consistency for Robust Multi-Intent Spoken Language Understanding","authors":["Po-Yen Chen","Berlin Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-3369","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26ea_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26ea_interspeech.pdf","session":"Spoken Language Understanding","topics":["spoken-language-understanding","speech-llm","self-supervised"],"category":"speech-llm-dialogue","institutions":["National Taiwan Normal University"],"funding":["Realtek Semiconductor Corporation"],"code":{"url":"https://github.com/boyan1001/SFL-MTSC","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26ea_interspeech","category":"speech-llm-dialogue","institutions":["National Taiwan Normal University"],"code":"https://github.com/boyan1001/SFL-MTSC","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3369","pdf":"https://www.isca-archive.org/interspeech_2026/chen26ea_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26ea_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26ea_interspeech/markdown.md"},{"id":"chen26f_interspeech","title":"Toward Multimodal Industrial Fault Analysis: A Single-Speed Chain Conveyor Dataset with Audio and Vibration Signals","authors":["Zhang Chen","Yucong Zhang","Xiaoxiao Miao","Ming Li"],"year":2026,"doi":"10.21437/Interspeech.2026-838","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26f_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["speech-enhancement","self-supervised","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Duke Kunshan University","Chinese University of Hong Kong","Wuhan University"],"funding":["Science and Technology Program of Suzhou City"],"code":{"url":"https://github.com/yucongzh/SSCC-Fault-Benchmark","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26f_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Duke Kunshan University","Chinese University of Hong Kong","Wuhan University"],"code":"https://github.com/yucongzh/SSCC-Fault-Benchmark","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-838","pdf":"https://www.isca-archive.org/interspeech_2026/chen26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26f_interspeech/markdown.md"},{"id":"chen26fa_interspeech","title":"Bayesian Model-Based Assessment of Spatial and Source Priors in Sagittal-Plane Sound Localization","authors":["Yunda Chen","Nengheng Zheng"],"year":2026,"doi":"10.21437/Interspeech.2026-3494","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26fa_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26fa_interspeech.pdf","session":"Spatial Audio 3","topics":["paralinguistics","evaluation","self-supervised"],"category":"health-clinical","institutions":["Shenzhen University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26fa_interspeech","category":"health-clinical","institutions":["Shenzhen University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3494","pdf":"https://www.isca-archive.org/interspeech_2026/chen26fa_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26fa_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26fa_interspeech/markdown.md"},{"id":"chen26g_interspeech","title":"Using Phonological-Level Wav2Vec2 for Mandarin Automatic Mispronunciation Detection and Diagnosis","authors":["Jinghao Chen","Mostafa Shahin","Beena Ahmed"],"year":2026,"doi":"10.21437/Interspeech.2026-869","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26g_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["asr","speech-llm","low-resource"],"category":"applications-other","labels":["self-supervised"],"institutions":["UNSW"],"code":{"url":"https://github.com/Evanchan1923/MDD_SpeechAttribute","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26g_interspeech","category":"applications-other","labels":["self-supervised"],"institutions":["UNSW"],"code":"https://github.com/Evanchan1923/MDD_SpeechAttribute","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-869","pdf":"https://www.isca-archive.org/interspeech_2026/chen26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26g_interspeech/markdown.md"},{"id":"chen26h_interspeech","title":"Geometrically Constrained Decentralized Independent Vector Analysis for Distributed Microphone Arrays","authors":["Changda Chen","Yichen Yang","Wei Liu","Bing Zhu","Gongping Huang","Shoji Makino","Shuai Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1037","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26h_interspeech.pdf","session":"Spatial Audio 4","topics":["source-separation","speech-enhancement","multilingual"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Waseda University","Nanjing University","Northwestern Polytechnical University","Wuhan University"],"funding":["Ministry of Education of China","National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26h_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Waseda University","Nanjing University","Northwestern Polytechnical University","Wuhan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1037","pdf":"https://www.isca-archive.org/interspeech_2026/chen26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26h_interspeech/markdown.md"},{"id":"chen26i_interspeech","title":"G2PO: A Lightweight Lexicon-enhanced Framework for Open-Vocabulary Mandarin Polyphone Disambiguation","authors":["Feifan Chen","Chunhui Lu","Rui Zhou","Liming Song","Hongjun Kil","YoonChoon Hwang","Junkwang Oh"],"year":2026,"doi":"10.21437/Interspeech.2026-1064","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26i_interspeech.pdf","session":"Text Processing for Speech Synthesis","topics":["tts","low-resource","self-supervised"],"category":"tts","labels":["efficient-on-device"],"institutions":["Samsung"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26i_interspeech","category":"tts","labels":["efficient-on-device"],"institutions":["Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1064","pdf":"https://www.isca-archive.org/interspeech_2026/chen26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26i_interspeech/markdown.md"},{"id":"chen26j_interspeech","title":"Spiking Vocos: An Energy-Efficient Neural Vocoder","authors":["Yukun Chen","Zhaoxi Mu","Andong Li","Peilin Li","Xingyu Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-1086","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26j_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["tts","self-supervised","on-device"],"category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Xi'an Jiaotong University","Chinese Academy of Sciences"],"code":{"url":"https://github.com/pymaster17/Spiking-Vocos","stars":11,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26j_interspeech","category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Xi'an Jiaotong University","Chinese Academy of Sciences"],"code":"https://github.com/pymaster17/Spiking-Vocos","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1086","pdf":"https://www.isca-archive.org/interspeech_2026/chen26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26j_interspeech/markdown.md"},{"id":"chen26k_interspeech","title":"Causal Tracing of Audio-Text Fusion in Large Audio Language Models","authors":["Wei-Chih Chen","Chien-yu Huang","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1118","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26k_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["self-supervised","evaluation","speech-llm"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["National Taiwan University","Carnegie Mellon University"],"funding":["Ministry of Education","Taiwan Centers of Excellence in Artificial Intelligence","NTU Artificial Intelligence Center of Research Excellence"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26k_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["National Taiwan University","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1118","pdf":"https://www.isca-archive.org/interspeech_2026/chen26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26k_interspeech/markdown.md"},{"id":"chen26l_interspeech","title":"Leveraging Audio-LLMs to Filter Speech-to-Speech Training Data","authors":["Qixu Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1148","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26l_interspeech.pdf","session":"Multilingual Speech 2","topics":["speech-translation","self-supervised","speech-llm"],"category":"translation","labels":["multilingual","self-supervised"],"institutions":["Chinese University of Hong Kong, Shenzhen"],"funding":["National Natural Science Foundation of China","Program for Guangdong Introducing Innovative and Entrepreneurial Teams"],"code":{"url":"https://github.com/chin-alt/S2S-Filtering","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26l_interspeech","category":"translation","labels":["multilingual","self-supervised"],"institutions":["Chinese University of Hong Kong, Shenzhen"],"code":"https://github.com/chin-alt/S2S-Filtering","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1148","pdf":"https://www.isca-archive.org/interspeech_2026/chen26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26l_interspeech/markdown.md"},{"id":"chen26m_interspeech","title":"Modulation of Phonetic Realizations in Cantonese Dialogue with Human and AI Interlocutors","authors":["Xinyi Chen","Grace Wenling Cao","Yusheng Tian","Manson Chun Man Wong","Miko Hoi Tung Ng","Tan Lee","Peggy Pik Ki Mok"],"year":2026,"doi":"10.21437/Interspeech.2026-1194","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26m_interspeech.pdf","session":"Phonetic Aspects of TTS and ASR Systems","topics":["paralinguistics","tts","phonetics"],"category":"phonetics-linguistics","institutions":["Chinese University of Hong Kong","University College Dublin","Chinese University of Hong Kong, Shenzhen"],"funding":["Hong Kong RGC GRF","CUHK Research Committee Postdoctoral Fellowship Scheme","Hong Kong RGC Postdoctoral Fellowship"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26m_interspeech","category":"phonetics-linguistics","institutions":["Chinese University of Hong Kong","University College Dublin","Chinese University of Hong Kong, Shenzhen"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1194","pdf":"https://www.isca-archive.org/interspeech_2026/chen26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26m_interspeech/markdown.md"},{"id":"chen26n_interspeech","title":"Formant-Guided Speech Repair for Enhanced Comprehension of Dysarthric Speech","authors":["Xin-Yu Chen","Jing-Tong Tzeng","Carlos Busso","Chi-Chun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1217","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26n_interspeech.pdf","session":"Speech and Language Technologies in Healthcare","topics":["speech-enhancement","tts","asr"],"category":"health-clinical","labels":["generative-model"],"institutions":["National Tsing Hua University","Carnegie Mellon University"],"code":{"url":"https://github.com/xinyu0308/FAST-SR","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26n_interspeech","category":"health-clinical","labels":["generative-model"],"institutions":["National Tsing Hua University","Carnegie Mellon University"],"code":"https://github.com/xinyu0308/FAST-SR","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1217","pdf":"https://www.isca-archive.org/interspeech_2026/chen26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26n_interspeech/markdown.md"},{"id":"chen26o_interspeech","title":"CE-CoT: A Contrastive Empathetic Chain-of-Thought Training Strategy for Improving Emotion Consensus in Empathetic Speech LLMs","authors":["Jing-Han Chen","Ya-Tse Wu","Bo-Hao Su","Xin-Yu Chen","Krishna Somandepalli","Chi-Chun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1271","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26o_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26o_interspeech.pdf","session":"Empathetic Dialogue and Interaction Dynamics","topics":["speech-llm","paralinguistics","evaluation"],"category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["National Tsing Hua University","Google DeepMind"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26o_interspeech","category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["National Tsing Hua University","Google DeepMind"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1271","pdf":"https://www.isca-archive.org/interspeech_2026/chen26o_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26o_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26o_interspeech/markdown.md"},{"id":"chen26p_interspeech","title":"Bridging the Gap: A Hierarchical Framework for Cross-Modal Style Modeling in Expressive TTS","authors":["Jiale Chen","Jiaxun Li","Wei Li","Yuanpeng Wang","Yuehai Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1513","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26p_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26p_interspeech.pdf","session":"Flow Matching for Speech Synthesis","topics":["tts","speech-llm","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["Zhejiang University"],"code":{"url":"https://sunny00952.github.io/OTAFlow/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26p_interspeech","category":"tts","labels":["generative-model"],"institutions":["Zhejiang University"],"code":"https://sunny00952.github.io/OTAFlow/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1513","pdf":"https://www.isca-archive.org/interspeech_2026/chen26p_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26p_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26p_interspeech/markdown.md"},{"id":"chen26q_interspeech","title":"Streaming Open-Vocabulary Keyword Spotting via Role Swapping in Cross-Attention","authors":["Xi Chen","Haichuan Bai","Liming Song"],"year":2026,"doi":"10.21437/Interspeech.2026-1676","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26q_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26q_interspeech.pdf","session":"Information Extraction and Retrieval","topics":["keyword-spotting","self-supervised","on-device"],"category":"asr","labels":["streaming-real-time"],"institutions":["Samsung"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26q_interspeech","category":"asr","labels":["streaming-real-time"],"institutions":["Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1676","pdf":"https://www.isca-archive.org/interspeech_2026/chen26q_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26q_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26q_interspeech/markdown.md"},{"id":"chen26r_interspeech","title":"Explainable and Trustworthy Speech Emotion Recognition Using Confidence Score and Reinforcement Learning Rectified Speech Emotion Descriptors","authors":["Youjun Chen","Xurong Xie","Mengzhe Geng","Zengrui Jin","Jiajun Deng","Guinan Li","Shujie Hu","Huimeng Wang","Haoning Xu","Chengxi Deng","Bowen Zhang","Xunying Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-1683","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26r_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26r_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["speech-llm","emotion-recognition","self-supervised"],"category":"paralinguistics-emotion","institutions":["Chinese University of Hong Kong","Institute of Software, Chinese Academy of Sciences","National Research Council Canada","Tsinghua University"],"funding":["Hong Kong RGC GRF","National Natural Science Foundation of China","Beijing Natural Science Foundation-Xiaomi Innovation Joint Fund","Youth Innovation Promotion Association CAS"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26r_interspeech","category":"paralinguistics-emotion","institutions":["Chinese University of Hong Kong","Institute of Software, Chinese Academy of Sciences","National Research Council Canada","Tsinghua University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1683","pdf":"https://www.isca-archive.org/interspeech_2026/chen26r_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26r_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26r_interspeech/markdown.md"},{"id":"chen26s_interspeech","title":"LLM-Guided Reinforcement Learning for Audio-Visual Speech Enhancement","authors":["Chih-Ning Chen","Jen-Cheng Hou","Hsin-Min Wang","Shao-Yi Chien","Yu Tsao","Fan-Gang Zeng"],"year":2026,"doi":"10.21437/Interspeech.2026-1816","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26s_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26s_interspeech.pdf","session":"Audio-Visual and Generative Target Speaker Extraction","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","institutions":["National Taiwan University","Academia Sinica","University of California Irvine"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26s_interspeech","category":"enhancement-separation","institutions":["National Taiwan University","Academia Sinica","University of California Irvine"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1816","pdf":"https://www.isca-archive.org/interspeech_2026/chen26s_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26s_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26s_interspeech/markdown.md"},{"id":"chen26t_interspeech","title":"A Preclinical Study of Electrolaryngeal Voice Conversion for a Novel Nasal Electrolarynx: Feature Choice and Data Augmentation","authors":["Qi-Yan Chen","Ming-Chi Yen","Fo-Rui Li","Hsin-Te Hwang","Ching-Hung Lai","Shu-Wei Tsai","Ping-Cheng Yeh","Jyh-Shing Roger Jang","Yu Tsao","Hsin-Min Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1882","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26t_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26t_interspeech.pdf","session":"Beyond Speech Technologies in Healthcare","topics":["speech-enhancement","voice-conversion","low-resource"],"category":"tts","labels":["low-resource","generative-model"],"institutions":["Academia Sinica","National Taiwan University","National Central University","National Cheng Kung University Hospital"],"funding":["Taiwan National Science and Technology Council"],"code":{"url":"https://ymchiqq.github.io/nelvc_demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26t_interspeech","category":"tts","labels":["low-resource","generative-model"],"institutions":["Academia Sinica","National Taiwan University","National Central University","National Cheng Kung University Hospital"],"code":"https://ymchiqq.github.io/nelvc_demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1882","pdf":"https://www.isca-archive.org/interspeech_2026/chen26t_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26t_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26t_interspeech/markdown.md"},{"id":"chen26u_interspeech","title":"Latent-Mark: An Audio Watermark Robust to Neural Codec Compression","authors":["Yen-Shan Chen","Shih-Yu Lai","Ying-Jung Tsou","Yi-Cheng Lin","Bing-Yu Chen","Yun-Nung Chen","Hung-yi Lee","Shang-Tse Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1979","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26u_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26u_interspeech.pdf","session":"Spoofing, Deepfake Detection and Watermarking","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","institutions":["National Taiwan University","CyCraft","RIKEN","MoonShine Animation Studio"],"funding":["National Science and Technology Council"],"code":{"url":"https://github.com/yenshan0530/Latent-Mark","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26u_interspeech","category":"deepfake-security","institutions":["National Taiwan University","CyCraft","RIKEN","MoonShine Animation Studio"],"code":"https://github.com/yenshan0530/Latent-Mark","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1979","pdf":"https://www.isca-archive.org/interspeech_2026/chen26u_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26u_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26u_interspeech/markdown.md"},{"id":"chen26v_interspeech","title":"SARA: A Dual-Stream VAE for High-Fidelity Speech Generation via Integrating Semantic and Acoustic Representations","authors":["Peijie Chen","Wenhao Guan","Weijie Wu","Kaidi Wang","Daiyu Huang","Zhuanling Zha","Junbo Li","Jun Fang","Qingyang Hong","Lin Li"],"year":2026,"doi":"10.21437/Interspeech.2026-2082","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26v_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26v_interspeech.pdf","session":"Acoustic Signal Analysis and Generation","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["self-supervised","generative-model"],"institutions":["Xiamen University","DiDi Global Inc"],"funding":["National Natural Science Foundation of China","Innovation of Policing Science and Technology, Fujian province"],"code":{"url":"https://pppjchen.github.io/SARA","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26v_interspeech","category":"tts","labels":["self-supervised","generative-model"],"institutions":["Xiamen University","DiDi Global Inc"],"code":"https://pppjchen.github.io/SARA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2082","pdf":"https://www.isca-archive.org/interspeech_2026/chen26v_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26v_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26v_interspeech/markdown.md"},{"id":"chen26w_interspeech","title":"T-ORR: Text-Anchored Orthogonal Residual Rectification for Robust Multimodal Sarcasm Detection","authors":["Qi Chen","Junyi Chen","Jiabin Han","Hongjiao Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-2222","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26w_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26w_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["paralinguistics","self-supervised","evaluation"],"category":"paralinguistics-emotion","institutions":["Tianjin Foreign Studies University","Beijing Language and Culture University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26w_interspeech","category":"paralinguistics-emotion","institutions":["Tianjin Foreign Studies University","Beijing Language and Culture University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2222","pdf":"https://www.isca-archive.org/interspeech_2026/chen26w_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26w_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26w_interspeech/markdown.md"},{"id":"chen26x_interspeech","title":"Who is Talking to Me? Addressing Egocentric TTM with Speaker-aware Conversational Context","authors":["Fukun Chen","Xionghu Zhong","Minjie Cai"],"year":2026,"doi":"10.21437/Interspeech.2026-2358","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26x_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26x_interspeech.pdf","session":"Multimodal Spoken Dialogue Systems","topics":["speech-llm","multimodal","speaker-diarization"],"category":"speech-llm-dialogue","institutions":["Hunan University"],"code":{"url":"https://github.com/cfk1009/EgoTTM","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26x_interspeech","category":"speech-llm-dialogue","institutions":["Hunan University"],"code":"https://github.com/cfk1009/EgoTTM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2358","pdf":"https://www.isca-archive.org/interspeech_2026/chen26x_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26x_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26x_interspeech/markdown.md"},{"id":"chen26y_interspeech","title":"Improving Flow Matching based Text-to-Speech with Dual-Model Preference Optimization and Classifier-Free Guidance","authors":["Minchuan Chen","Chenchen Wan","Junjie Li","Peng Qi","Shaojun Wang","Jing Xiao"],"year":2026,"doi":"10.21437/Interspeech.2026-2412","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26y_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26y_interspeech.pdf","session":"Flow Matching for Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["Ping An Technology","Shanghai Jiao Tong University","Shanghai Jiao Tong University Chongqing Artificial Intelligence Research Institute"],"code":{"url":"https://minchuan2025.github.io/interspeech2026","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26y_interspeech","category":"tts","labels":["generative-model"],"institutions":["Ping An Technology","Shanghai Jiao Tong University","Shanghai Jiao Tong University Chongqing Artificial Intelligence Research Institute"],"code":"https://minchuan2025.github.io/interspeech2026","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2412","pdf":"https://www.isca-archive.org/interspeech_2026/chen26y_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26y_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26y_interspeech/markdown.md"},{"id":"chen26z_interspeech","title":"DiaMoE-TTS: A Unified IPA-Based Dialect TTS Framework with Parameter-Efficient Adaptation and Reward-Driven Optimization","authors":["Ziqi Chen","Gongyu Chen","Yihua Wang","Zihao Chen","Wei-Qiang Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-2447","isca_url":"https://www.isca-archive.org/interspeech_2026/chen26z_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chen26z_interspeech.pdf","session":"Text Processing for Speech Synthesis","topics":["tts","low-resource","multilingual"],"category":"tts","labels":["low-resource","generative-model"],"institutions":["Tsinghua University","Giant Network"],"code":{"url":"https://github.com/GiantAILab/DiaMoE-TTS","stars":254,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chen26z_interspeech","category":"tts","labels":["low-resource","generative-model"],"institutions":["Tsinghua University","Giant Network"],"code":"https://github.com/GiantAILab/DiaMoE-TTS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2447","pdf":"https://www.isca-archive.org/interspeech_2026/chen26z_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26z_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chen26z_interspeech/markdown.md"},{"id":"cheng26_interspeech","title":"Diffusion Reconstruction towards Generalizable Audio Deepfake Detection","authors":["Bo Cheng","Songjun Cao","Xiaoming Zhang","Jie Chen","Long Ma","Fei Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-158","isca_url":"https://www.isca-archive.org/interspeech_2026/cheng26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cheng26_interspeech.pdf","session":"Speech Deepfake Detection: Robustness, Generalization, Attribution","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["generative-model","robustness-noise"],"institutions":["Southern University of Science and Technology","Tencent Youtu Lab"],"funding":["Shenzhen Key Technology Program Funding","Center for Computational Science and Engineering at Southern University of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cheng26_interspeech","category":"deepfake-security","labels":["generative-model","robustness-noise"],"institutions":["Southern University of Science and Technology","Tencent Youtu Lab"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-158","pdf":"https://www.isca-archive.org/interspeech_2026/cheng26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cheng26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cheng26_interspeech/markdown.md"},{"id":"cheng26b_interspeech","title":"Active Noise Control With a Gain Constraint for Micro-Loudspeakers","authors":["Zhenhua Cheng","Yi Zhou","Yin Liu","Yu Zhao","Liming Shi"],"year":2026,"doi":"10.21437/Interspeech.2026-2202","isca_url":"https://www.isca-archive.org/interspeech_2026/cheng26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cheng26b_interspeech.pdf","session":"Active Noise and Echo Control, Sound Zones and Packet-Loss Concealment","topics":["speech-enhancement"],"category":"enhancement-separation","institutions":["Chongqing University of Posts and Telecommunications"],"funding":["National Key Research and Development Program of China","National Natural Science Foundation of China","Natural Science Foundation of Chongqing"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cheng26b_interspeech","category":"enhancement-separation","institutions":["Chongqing University of Posts and Telecommunications"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2202","pdf":"https://www.isca-archive.org/interspeech_2026/cheng26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cheng26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cheng26b_interspeech/markdown.md"},{"id":"chiba26_interspeech","title":"Speech-based Psychological Crisis Assessment using LLMs","authors":["Terumi Chiba","Yang Luo","Ziyun Cui","Yongsheng Tong","Chao Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-997","isca_url":"https://www.isca-archive.org/interspeech_2026/chiba26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chiba26_interspeech.pdf","session":"Medical Dialogue and Conversational Understanding","topics":["paralinguistics","speech-llm","health"],"category":"health-clinical","institutions":["Tsinghua University","Peking University Huilongguan Clinical Medical School","WHO Collaborating Centre for Research and Training in Suicide Prevention"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chiba26_interspeech","category":"health-clinical","institutions":["Tsinghua University","Peking University Huilongguan Clinical Medical School","WHO Collaborating Centre for Research and Training in Suicide Prevention"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-997","pdf":"https://www.isca-archive.org/interspeech_2026/chiba26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chiba26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chiba26_interspeech/markdown.md"},{"id":"chien26_interspeech","title":"Attentive Mamba: Channel-wise Local Attention for Speech Recognition","authors":["Jen-Tzung Chien","Fan-Che Feng","Ching-Hsien Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1708","isca_url":"https://www.isca-archive.org/interspeech_2026/chien26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chien26_interspeech.pdf","session":"Long-form Audio & New Attention Approaches","topics":["asr","self-supervised","speech-llm"],"category":"asr","institutions":["National Yang Ming Chiao Tung University","Industrial Technology Research Institute"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chien26_interspeech","category":"asr","institutions":["National Yang Ming Chiao Tung University","Industrial Technology Research Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1708","pdf":"https://www.isca-archive.org/interspeech_2026/chien26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chien26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chien26_interspeech/markdown.md"},{"id":"chien26b_interspeech","title":"From Bilevel to Trilevel: Joint Training for Speech Recognition","authors":["Jen-Tzung Chien","Yu-Chun Lin","Xiaodong Cui"],"year":2026,"doi":"10.21437/Interspeech.2026-1738","isca_url":"https://www.isca-archive.org/interspeech_2026/chien26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chien26b_interspeech.pdf","session":"New Training Methods for ASR","topics":["asr","self-supervised","speech-llm"],"category":"asr","labels":["self-supervised"],"institutions":["National Yang Ming Chiao Tung University","IBM"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chien26b_interspeech","category":"asr","labels":["self-supervised"],"institutions":["National Yang Ming Chiao Tung University","IBM"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1738","pdf":"https://www.isca-archive.org/interspeech_2026/chien26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chien26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chien26b_interspeech/markdown.md"},{"id":"chien26c_interspeech","title":"Two-Sided Fairness Transfer for Gender-Neutral Speech Emotion Recognition with Partially Observed Attributes","authors":["Woan-Shiuan Chien","Tomohiko Nakamura","Huan-Yu Chen","Satoru Fukayama","Hitoshi Suda","Jun Ogata","Chi-Chun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-3201","isca_url":"https://www.isca-archive.org/interspeech_2026/chien26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chien26c_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["speech-llm","paralinguistics","emotion-recognition"],"category":"paralinguistics-emotion","institutions":["National Yang Ming Chiao Tung University","National Institute of Advanced Industrial Science and Technology","National Tsing Hua University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chien26c_interspeech","category":"paralinguistics-emotion","institutions":["National Yang Ming Chiao Tung University","National Institute of Advanced Industrial Science and Technology","National Tsing Hua University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3201","pdf":"https://www.isca-archive.org/interspeech_2026/chien26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chien26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chien26c_interspeech/markdown.md"},{"id":"chin26_interspeech","title":"Effects of listener language experience, masker language, and cognitive load on word monitoring accuracy and response time","authors":["Jessica Chin","Laurence Bruggeman","Mark Antoniou"],"year":2026,"doi":"10.21437/Interspeech.2026-2950","isca_url":"https://www.isca-archive.org/interspeech_2026/chin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chin26_interspeech.pdf","session":"Brain Studies and Speech","topics":["paralinguistics","speech-enhancement","evaluation"],"category":"phonetics-linguistics","labels":["multilingual","robustness-noise"],"institutions":["Western Sydney University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chin26_interspeech","category":"phonetics-linguistics","labels":["multilingual","robustness-noise"],"institutions":["Western Sydney University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2950","pdf":"https://www.isca-archive.org/interspeech_2026/chin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chin26_interspeech/markdown.md"},{"id":"cho26_interspeech","title":"Acoustic Prompting via Stage-wise Modulation for Few-Shot Learning in Audio Language Models","authors":["Hyebin Cho","Jaehyuk Jang","Changick Kim","Joon Son Chung"],"year":2026,"doi":"10.21437/Interspeech.2026-885","isca_url":"https://www.isca-archive.org/interspeech_2026/cho26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cho26_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["self-supervised","speech-llm","low-resource"],"category":"speech-llm-dialogue","labels":["low-resource","self-supervised"],"institutions":["Korea Advanced Institute of Science and Technology"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"https://github.com/hyebin-c/aspl","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cho26_interspeech","category":"speech-llm-dialogue","labels":["low-resource","self-supervised"],"institutions":["Korea Advanced Institute of Science and Technology"],"code":"https://github.com/hyebin-c/aspl","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-885","pdf":"https://www.isca-archive.org/interspeech_2026/cho26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cho26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cho26_interspeech/markdown.md"},{"id":"cho26b_interspeech","title":"A Multimodal Semi-Supervised Framework for Automatic Construction of a Cross-Lingual Taigi Speech-Chinese Subtitle Corpus","authors":["Cheng-Hsiu Cho","Chih-Chung Kuo","Yu-Siang Lan","Chao-Shih Huang","Yan-Ming Lin","Yuan-Fu Liao"],"year":2026,"doi":"10.21437/Interspeech.2026-2096","isca_url":"https://www.isca-archive.org/interspeech_2026/cho26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cho26b_interspeech.pdf","session":"Translation","topics":["asr","speech-translation","self-supervised"],"category":"translation","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["National Yang Ming Chiao Tung University"],"code":{"url":"https://github.com/Speech-AI-Research-Center/taigi-speech2chinese-subtitle","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cho26b_interspeech","category":"translation","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["National Yang Ming Chiao Tung University"],"code":"https://github.com/Speech-AI-Research-Center/taigi-speech2chinese-subtitle","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2096","pdf":"https://www.isca-archive.org/interspeech_2026/cho26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cho26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cho26b_interspeech/markdown.md"},{"id":"choi26_interspeech","title":"Systematic PTQ Study of Integer and Floating-Point Formats for On-Device Whisper ASR","authors":["Woosuk Choi","Dohyeon Lee","Taehyung Kim","Hyukjun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-698","isca_url":"https://www.isca-archive.org/interspeech_2026/choi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/choi26_interspeech.pdf","session":"Resource Constrained Speech Recognition","topics":["asr","self-supervised","on-device"],"category":"asr","labels":["efficient-on-device","self-supervised"],"institutions":["LG Electronics","Sogang University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"choi26_interspeech","category":"asr","labels":["efficient-on-device","self-supervised"],"institutions":["LG Electronics","Sogang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-698","pdf":"https://www.isca-archive.org/interspeech_2026/choi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26_interspeech/markdown.md"},{"id":"choi26b_interspeech","title":"ZeSTA: Zero-Shot TTS Augmentation with Domain-Conditioned Training for Data-Efficient Personalized Speech Synthesis","authors":["Youngwon Choi","Jinwoo Oh","Hwayeon Kim","Hyeonyu Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-1269","isca_url":"https://www.isca-archive.org/interspeech_2026/choi26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/choi26b_interspeech.pdf","session":"Scaling and Zero-Shot Speech Synthesis","topics":["tts","self-supervised","low-resource"],"category":"tts","labels":["low-resource","generative-model"],"institutions":["Maum AI","Humelo"],"funding":["Culture, Sports and Tourism R&D Program","Startup Growth Technology Development Program"],"code":{"url":"https://zeroone-universe.github.io/zesta/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"choi26b_interspeech","category":"tts","labels":["low-resource","generative-model"],"institutions":["Maum AI","Humelo"],"code":"https://zeroone-universe.github.io/zesta/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1269","pdf":"https://www.isca-archive.org/interspeech_2026/choi26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26b_interspeech/markdown.md"},{"id":"choi26c_interspeech","title":"Considerate Listener Modeling for Korean Streaming Backchannel Prediction","authors":["Yong-Seok Choi","Seung Hi Kim","Sung Yup Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1854","isca_url":"https://www.isca-archive.org/interspeech_2026/choi26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/choi26c_interspeech.pdf","session":"Spoken Dialogue Systems","topics":["speech-llm","spoken-language-understanding","streaming"],"category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["Electronics and Telecommunications Research Institute"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"choi26c_interspeech","category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["Electronics and Telecommunications Research Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1854","pdf":"https://www.isca-archive.org/interspeech_2026/choi26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26c_interspeech/markdown.md"},{"id":"choi26d_interspeech","title":"ProsoCodec: Prosody-Oriented Speech Codec for Voice Conversion","authors":["Jeongsoo Choi","Ji-Hoon Kim","Shujie Hu","Joon Son Chung"],"year":2026,"doi":"10.21437/Interspeech.2026-2146","isca_url":"https://www.isca-archive.org/interspeech_2026/choi26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/choi26d_interspeech.pdf","session":"Controllable and Expressive Speech Synthesis","topics":["voice-conversion","speech-enhancement","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["KAIST","Chung-Ang University","Chinese University of Hong Kong"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"choi26d_interspeech","category":"tts","labels":["generative-model"],"institutions":["KAIST","Chung-Ang University","Chinese University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2146","pdf":"https://www.isca-archive.org/interspeech_2026/choi26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26d_interspeech/markdown.md"},{"id":"choi26e_interspeech","title":"SISER: Speaker-Invariant Speech Emotion Recognition with Entropy-Based Adversarial Training","authors":["Eunseo Choi","Hyunku Kang","Chanwoo Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-2186","isca_url":"https://www.isca-archive.org/interspeech_2026/choi26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/choi26e_interspeech.pdf","session":"Speech Emotion Recognition and Representation 3","topics":["speech-emotion-recognition","self-supervised","speaker-verification"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Korea University"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"https://github.com/slp-lab-research/siser.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"choi26e_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Korea University"],"code":"https://github.com/slp-lab-research/siser.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2186","pdf":"https://www.isca-archive.org/interspeech_2026/choi26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26e_interspeech/markdown.md"},{"id":"choi26f_interspeech","title":"IPA-Guided Dual Transcription for Data-Centric Speech Corpus Refinement","authors":["Jeong-Ju Choi","Young-Ik Kim","Jin NamGoong","Jaeyeon Jang"],"year":2026,"doi":"10.21437/Interspeech.2026-2221","isca_url":"https://www.isca-archive.org/interspeech_2026/choi26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/choi26f_interspeech.pdf","session":"Corpus Creation, Summarization and Understanding","topics":["asr","tts","dataset"],"category":"asr","institutions":["DenComm","Catholic University of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"choi26f_interspeech","category":"asr","institutions":["DenComm","Catholic University of Korea"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2221","pdf":"https://www.isca-archive.org/interspeech_2026/choi26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26f_interspeech/markdown.md"},{"id":"choi26g_interspeech","title":"SpkGuideDOA: Speaker-wise Representation Guidance for Multiple Moving Speaker Localization","authors":["Yongseok Choi","Davin Kim","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-3131","isca_url":"https://www.isca-archive.org/interspeech_2026/choi26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/choi26g_interspeech.pdf","session":"Spatial Audio 4","topics":["speaker-diarization","speech-enhancement","self-supervised"],"category":"enhancement-separation","institutions":["Hanyang University"],"funding":["National Research Foundation of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"choi26g_interspeech","category":"enhancement-separation","institutions":["Hanyang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3131","pdf":"https://www.isca-archive.org/interspeech_2026/choi26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/choi26g_interspeech/markdown.md"},{"id":"chongwhite26_interspeech","title":"An Immersive VR System for Experiencing Spatial Speech-in-Noise Challenges in Clinical Audiology","authors":["Nicky Chong-White","Thomas Ho"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/chongwhite26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chongwhite26_interspeech.pdf","session":"Speech and Language Processing for Health and Accessibility","topics":["paralinguistics","evaluation","on-device"],"category":"health-clinical","labels":["robustness-noise"],"institutions":["National Acoustic Laboratories"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chongwhite26_interspeech","category":"health-clinical","labels":["robustness-noise"],"institutions":["National Acoustic Laboratories"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/chongwhite26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/chongwhite26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chongwhite26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chongwhite26_interspeech/markdown.md"},{"id":"chou26_interspeech","title":"Hidden Priors in Speech LLMs: Speaker Identity Shapes Emotional Perception","authors":["Hsing-Hang Chou","Bo-Hao Su","Krishna Somandepalli","Chi-Chun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1238","isca_url":"https://www.isca-archive.org/interspeech_2026/chou26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chou26_interspeech.pdf","session":"Speaker Identity, States, and Traits in Paralinguistics","topics":["speech-llm","emotion-recognition","self-supervised"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["National Tsing Hua University","Google"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chou26_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["National Tsing Hua University","Google"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1238","pdf":"https://www.isca-archive.org/interspeech_2026/chou26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chou26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chou26_interspeech/markdown.md"},{"id":"choudhury26_interspeech","title":"Impact Analysis of Speech Representation Learning Models for Acoustic Side-Channel Attack","authors":["Nitin Choudhury","Bikrant Bikram Pratap Maurya","Arun Balaji Buduru","Orchid Chetia Phukan"],"year":2026,"doi":"10.21437/Interspeech.2026-3500","isca_url":"https://www.isca-archive.org/interspeech_2026/choudhury26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/choudhury26_interspeech.pdf","session":"Spoofing and Deepfake Detection 1","topics":["self-supervised","evaluation","dataset"],"category":"deepfake-security","labels":["self-supervised","dataset-or-benchmark-release","robustness-noise"],"institutions":["IIIT-Delhi","NTHU"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"choudhury26_interspeech","category":"deepfake-security","labels":["self-supervised","dataset-or-benchmark-release","robustness-noise"],"institutions":["IIIT-Delhi","NTHU"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3500","pdf":"https://www.isca-archive.org/interspeech_2026/choudhury26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/choudhury26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/choudhury26_interspeech/markdown.md"},{"id":"chowdhury26_interspeech","title":"Predicting Cognitive Load from Speech and Interaction Dynamics in Dyadic Conversations","authors":["Tahiya Chowdhury"],"year":2026,"doi":"10.21437/Interspeech.2026-3052","isca_url":"https://www.isca-archive.org/interspeech_2026/chowdhury26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chowdhury26_interspeech.pdf","session":"Empathetic Dialogue and Interaction Dynamics","topics":["paralinguistics","self-supervised","evaluation"],"category":"paralinguistics-emotion","institutions":["Colby College"],"funding":["Henry Luce Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chowdhury26_interspeech","category":"paralinguistics-emotion","institutions":["Colby College"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3052","pdf":"https://www.isca-archive.org/interspeech_2026/chowdhury26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chowdhury26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chowdhury26_interspeech/markdown.md"},{"id":"chung26_interspeech","title":"Localizing and Editing Knowledge in Large Audio-Language Models","authors":["Sung Kyun Chung","Jiaheng Dong","Qiuchi Hu","Jinuo Sun","Gongping Huang","Hong Jia","Ting Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-2066","isca_url":"https://www.isca-archive.org/interspeech_2026/chung26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chung26_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["University of Melbourne","University of Auckland","Wuhan University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chung26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["University of Melbourne","University of Auckland","Wuhan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2066","pdf":"https://www.isca-archive.org/interspeech_2026/chung26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chung26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chung26_interspeech/markdown.md"},{"id":"chung26b_interspeech","title":"Robust Audio-Visual Emotion Recognition via Conditional Transformer U-Nets with Frequency-Injected Visual Stream","authors":["Hanwook Chung"],"year":2026,"doi":"10.21437/Interspeech.2026-2770","isca_url":"https://www.isca-archive.org/interspeech_2026/chung26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chung26b_interspeech.pdf","session":"Speech Emotion Recognition and Representation 3","topics":["speech-enhancement","emotion-recognition","multimodal"],"category":"paralinguistics-emotion","labels":["robustness-noise"],"institutions":["Faurecia IRYStec"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chung26b_interspeech","category":"paralinguistics-emotion","labels":["robustness-noise"],"institutions":["Faurecia IRYStec"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2770","pdf":"https://www.isca-archive.org/interspeech_2026/chung26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chung26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chung26b_interspeech/markdown.md"},{"id":"chuprina26_interspeech","title":"Sorting Clusters into the Shape of the Word: Positional Distribution of Probabilities","authors":["Anastasia Chuprina"],"year":2026,"doi":"10.21437/Interspeech.2026-3400","isca_url":"https://www.isca-archive.org/interspeech_2026/chuprina26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/chuprina26_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["phonetics","dataset","evaluation"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["University of Cambridge"],"code":{"url":"https://osf.io/v6uc7/overview","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"chuprina26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["University of Cambridge"],"code":"https://osf.io/v6uc7/overview","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3400","pdf":"https://www.isca-archive.org/interspeech_2026/chuprina26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/chuprina26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/chuprina26_interspeech/markdown.md"},{"id":"constantin26_interspeech","title":"A multilingual composite speech index to assess passage reading in Huntington’s disease","authors":["Valentina G. Constantin","Vitoria S. Fahed","Emer P. Doheny","Ruth Filan","Carla Collazo","Joanna Krzysztofik","Elliot Mann","Philippa Morgan-Jones","Laura Mills","Cheney Drew","Anne E. Rosser","Rebecca Cousins","Grzegorz Witkowski","Esther Cubo","Monica Busse","Madeleine M. Lowery"],"year":2026,"doi":"10.21437/Interspeech.2026-2654","isca_url":"https://www.isca-archive.org/interspeech_2026/constantin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/constantin26_interspeech.pdf","session":"Pathological Speech Assessment 3","topics":["paralinguistics","evaluation","health"],"category":"health-clinical","labels":["multilingual"],"institutions":["University College Dublin","Universidad de Burgos","Hospital Universitario of Burgos","Institute of Psychiatry and Neurology","Military Institute of Aviation Medicine","Cardiff University","North Bristol NHS Trust","King's College London"],"funding":["EU Joint Programme - Neurodegenerative Disease Research","Research Ireland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"constantin26_interspeech","category":"health-clinical","labels":["multilingual"],"institutions":["University College Dublin","Universidad de Burgos","Hospital Universitario of Burgos","Institute of Psychiatry and Neurology","Military Institute of Aviation Medicine","Cardiff University","North Bristol NHS Trust","King's College London"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2654","pdf":"https://www.isca-archive.org/interspeech_2026/constantin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/constantin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/constantin26_interspeech/markdown.md"},{"id":"cooper26_interspeech","title":"A Large-Scale Dataset of Listener Impressions of Emotional TTS","authors":["Erica Cooper","Xiaoxue Gao","Takuma Okamoto","Tomoki Toda","Nancy Chen","Hisashi Kawai"],"year":2026,"doi":"10.21437/Interspeech.2026-1521","isca_url":"https://www.isca-archive.org/interspeech_2026/cooper26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cooper26_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["tts","evaluation","speech-quality-assessment"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["National Institute of Information and Communications Technology","Agency for Science, Technology and Research","Nagoya University"],"funding":["National Research Foundation, Singapore","Infocomm Media Development Authority, Singapore","National Large Language Models Funding Initiative","A*STAR","Japan Science and Technology Agency"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cooper26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["National Institute of Information and Communications Technology","Agency for Science, Technology and Research","Nagoya University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1521","pdf":"https://www.isca-archive.org/interspeech_2026/cooper26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cooper26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cooper26_interspeech/markdown.md"},{"id":"correa26_interspeech","title":"From Tokens to Faces: Investigating Discrete Speech Representations for 3D Facial Animation","authors":["Pedro R. Corrêa","Olivier Perrotin","Samir Sadok","Paula D. P. Costa","Thomas Hueber"],"year":2026,"doi":"10.21437/Interspeech.2026-1397","isca_url":"https://www.isca-archive.org/interspeech_2026/correa26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/correa26_interspeech.pdf","session":"Articulatory, EMG, and Visual Speech Generation","topics":["speech-llm","evaluation","self-supervised"],"category":"tts","labels":["self-supervised","generative-model"],"institutions":["State University of Campinas","Grenoble Alpes University","National Centre for Scientific Research","Grenoble Institute of Technology","Inria"],"funding":["Sao Paulo Research Foundation","Brazilian Institute of Data Science","Coordenacao de Aperfeicoamento de Pessoal de Nivel Superior"],"code":{"url":"https://github.com/uuembodiedsocialai/FaceDiffuser","stars":183,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"correa26_interspeech","category":"tts","labels":["self-supervised","generative-model"],"institutions":["State University of Campinas","Grenoble Alpes University","National Centre for Scientific Research","Grenoble Institute of Technology","Inria"],"code":"https://github.com/uuembodiedsocialai/FaceDiffuser","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1397","pdf":"https://www.isca-archive.org/interspeech_2026/correa26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/correa26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/correa26_interspeech/markdown.md"},{"id":"cortes26_interspeech","title":"The ArtComp dataset: Articulatory and Acoustic Measurements of Swedish in Speech with Naturally Manipulated Jaw Position","authors":["Elísabet Eir Cortes","Lisa Gustavsson","Ellen Marklund"],"year":2026,"doi":"10.21437/Interspeech.2026-2647","isca_url":"https://www.isca-archive.org/interspeech_2026/cortes26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cortes26_interspeech.pdf","session":"Methods and Data for Vocal Tract Shape and Articulation Analysis","topics":["phonetics","dataset","self-supervised"],"category":"phonetics-linguistics","labels":["dataset-or-benchmark-release"],"institutions":["Stockholm University"],"funding":["The L3WOproject","The Marcus and Amalia Wallenberg Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cortes26_interspeech","category":"phonetics-linguistics","labels":["dataset-or-benchmark-release"],"institutions":["Stockholm University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2647","pdf":"https://www.isca-archive.org/interspeech_2026/cortes26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cortes26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cortes26_interspeech/markdown.md"},{"id":"cotosolano26_interspeech","title":"Automating Sociophonetic Research in Under-Resourced Languages: A Case Study of Speech Rate in Cook Islands Māori","authors":["Rolando Coto-Solano","Sally Akevai Nicholas"],"year":2026,"doi":"10.21437/Interspeech.2026-3507","isca_url":"https://www.isca-archive.org/interspeech_2026/cotosolano26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cotosolano26_interspeech.pdf","session":"Pacific Voices: Speech Science and Technology for the Languages of the Pacific Ocean","topics":["sociophonetics","asr","low-resource"],"category":"phonetics-linguistics","labels":["low-resource"],"institutions":["Dartmouth College","University of Auckland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cotosolano26_interspeech","category":"phonetics-linguistics","labels":["low-resource"],"institutions":["Dartmouth College","University of Auckland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3507","pdf":"https://www.isca-archive.org/interspeech_2026/cotosolano26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cotosolano26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cotosolano26_interspeech/markdown.md"},{"id":"cox26_interspeech","title":"Learning task-specific subspaces via interventional post-training of speech foundation models","authors":["Jack Cox","Jon P Barker"],"year":2026,"doi":"10.21437/Interspeech.2026-2542","isca_url":"https://www.isca-archive.org/interspeech_2026/cox26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cox26_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["self-supervised","speaker-verification","keyword-spotting"],"category":"speaker","labels":["self-supervised"],"institutions":["University of Sheffield"],"funding":["UK Research and Innovation","Meta"],"code":{"url":"https://github.com/mjukus/interventional-post-training-speech","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cox26_interspeech","category":"speaker","labels":["self-supervised"],"institutions":["University of Sheffield"],"code":"https://github.com/mjukus/interventional-post-training-speech","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2542","pdf":"https://www.isca-archive.org/interspeech_2026/cox26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cox26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cox26_interspeech/markdown.md"},{"id":"cram26_interspeech","title":"Vowel Allophony Improves Maximum-Likelihood Classification of Warlpiri Consonants","authors":["Coralie Cram","John McGahay","Megha Sundara"],"year":2026,"doi":"10.21437/Interspeech.2026-3107","isca_url":"https://www.isca-archive.org/interspeech_2026/cram26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cram26_interspeech.pdf","session":"Speech Production and Perception 2","topics":["phonetics","low-resource","evaluation"],"category":"phonetics-linguistics","labels":["low-resource"],"institutions":["UCLA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cram26_interspeech","category":"phonetics-linguistics","labels":["low-resource"],"institutions":["UCLA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3107","pdf":"https://www.isca-archive.org/interspeech_2026/cram26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cram26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cram26_interspeech/markdown.md"},{"id":"cronenberg26_interspeech","title":"To glide or not to glide: Acoustic realization of the diphthong-hiatus contrast in Italian and Romanian","authors":["Johanna Cronenberg","Lori Lamel","Ioana Chitoran"],"year":2026,"doi":"10.21437/Interspeech.2026-2433","isca_url":"https://www.isca-archive.org/interspeech_2026/cronenberg26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cronenberg26_interspeech.pdf","session":"Diphthongs and Monophthongs","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Universite Paris Cite","CNRS","Universite Paris-Saclay","Institut Universitaire de France"],"funding":["ANR","IdEx program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cronenberg26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Universite Paris Cite","CNRS","Universite Paris-Saclay","Institut Universitaire de France"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2433","pdf":"https://www.isca-archive.org/interspeech_2026/cronenberg26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cronenberg26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cronenberg26_interspeech/markdown.md"},{"id":"cui26_interspeech","title":"TurnGuide: Enhancing Meaningful Full Duplex Spoken Interactions via Dynamic Turn-Level Text-Speech Interleaving","authors":["Wenqian Cui","Lei Zhu","Xiao-Hui Li","Zhihan Guo","Haoli Bai","Lu Hou","Irwin King"],"year":2026,"doi":"10.21437/Interspeech.2026-1141","isca_url":"https://www.isca-archive.org/interspeech_2026/cui26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cui26_interspeech.pdf","session":"LLMs and Conversational Interaction","topics":["speech-llm","self-supervised","multilingual"],"category":"speech-llm-dialogue","labels":["streaming-real-time","generative-model"],"institutions":["Chinese University of Hong Kong","Huawei Technologies"],"funding":["Research Grants Council of the Hong Kong Special Administrative Region, China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cui26_interspeech","category":"speech-llm-dialogue","labels":["streaming-real-time","generative-model"],"institutions":["Chinese University of Hong Kong","Huawei Technologies"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1141","pdf":"https://www.isca-archive.org/interspeech_2026/cui26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cui26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cui26_interspeech/markdown.md"},{"id":"cui26b_interspeech","title":"Dictionary-Free Discrete Key-Value Attention for Improving Speech Enhancement","authors":["Zihao Cui","Jinwei Huang","Tao Li","Rongxiu Zhong","Yingying Gao","Shilei Zhang","Chao Deng","Junlan Feng"],"year":2026,"doi":"10.21437/Interspeech.2026-2881","isca_url":"https://www.isca-archive.org/interspeech_2026/cui26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cui26b_interspeech.pdf","session":"SE Architectures, Adaptation and Audio Front-Ends","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","institutions":["China Mobile Jiutian Artificial Intelligence Technology (Beijing) Co., Ltd","Peking University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cui26b_interspeech","category":"enhancement-separation","institutions":["China Mobile Jiutian Artificial Intelligence Technology (Beijing) Co., Ltd","Peking University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2881","pdf":"https://www.isca-archive.org/interspeech_2026/cui26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cui26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cui26b_interspeech/markdown.md"},{"id":"cunningham26_interspeech","title":"Decolonizing Linguistic Policies in Automatic Speech Recognition: A Framework for Cross-Culturally Competent Speech AI","authors":["Jay L. Cunningham","Mark Atta Mensah","Richard Martinez","João Vieira da Silva Neto","Efi Dawodu"],"year":2026,"doi":"10.21437/Interspeech.2026-3351","isca_url":"https://www.isca-archive.org/interspeech_2026/cunningham26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/cunningham26_interspeech.pdf","session":"Multilingual Speech 1","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["multilingual"],"institutions":["DePaul University","York University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"cunningham26_interspeech","category":"asr","labels":["multilingual"],"institutions":["DePaul University","York University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3351","pdf":"https://www.isca-archive.org/interspeech_2026/cunningham26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/cunningham26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/cunningham26_interspeech/markdown.md"},{"id":"curetti26_interspeech","title":"Towards an understanding of prosodic cue weighting for turn-end classification in older adults with varying hearing abilities","authors":["Lorenza Zaira Curetti","Hae-Sung Jeon","Lauren V. Hadley"],"year":2026,"doi":"10.21437/Interspeech.2026-2534","isca_url":"https://www.isca-archive.org/interspeech_2026/curetti26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/curetti26_interspeech.pdf","session":"Assistive Technologies 1","topics":["paralinguistics","evaluation","phonetics"],"category":"phonetics-linguistics","institutions":["University of Nottingham","University of Lancashire"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"curetti26_interspeech","category":"phonetics-linguistics","institutions":["University of Nottingham","University of Lancashire"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2534","pdf":"https://www.isca-archive.org/interspeech_2026/curetti26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/curetti26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/curetti26_interspeech/markdown.md"},{"id":"dahal26_interspeech","title":"Mixture of Phonetic Experts Based Low-Rank Adaptation of Conformer Models for Accented English Speech Recognition","authors":["Santosh Dahal","Anmol Guragain","Tianchi Liu","Luis Fernando D'Haro","Kiran Chandra Dahal"],"year":2026,"doi":"10.21437/Interspeech.2026-322","isca_url":"https://www.isca-archive.org/interspeech_2026/dahal26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dahal26_interspeech.pdf","session":"ASR Under Real-World Constraints: Streaming, Adaptation, and Efficiency","topics":["asr","self-supervised","multilingual"],"category":"asr","institutions":["Universidad Politecnica de Madrid","National University of Singapore"],"funding":["BRAINS","MCIN","AEI","European Regional Development Fund","European Union","Comunidad de Madrid"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dahal26_interspeech","category":"asr","institutions":["Universidad Politecnica de Madrid","National University of Singapore"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-322","pdf":"https://www.isca-archive.org/interspeech_2026/dahal26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dahal26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dahal26_interspeech/markdown.md"},{"id":"dai26_interspeech","title":"Consistency-Regularized Dual-Branch Network with Performance-Aware Mean Teacher for Sound Event Detection","authors":["Lipeng Dai","Qing Wang","Wu Guo","Peng Gao","Zhijun Zhang","Kuiliang Li","Jinjie Fu"],"year":2026,"doi":"10.21437/Interspeech.2026-353","isca_url":"https://www.isca-archive.org/interspeech_2026/dai26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dai26_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["sound-event-detection","self-supervised","dataset"],"category":"audio-understanding","institutions":["University of Science and Technology of China"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dai26_interspeech","category":"audio-understanding","institutions":["University of Science and Technology of China"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-353","pdf":"https://www.isca-archive.org/interspeech_2026/dai26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dai26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dai26_interspeech/markdown.md"},{"id":"dai26b_interspeech","title":"Joint Learning Global-Local Speaker Classification to Enhance End-to-End Speaker Diarization and Recognition","authors":["Yuhang Dai","Haopeng Lin","Jiale Qian","Ruiqi Yan","Hao Meng","Hanke Xie","Hanlin Wen","Shunshun Yin","Ming Tao","Xie Chen","Lei Xie","Xinsheng Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-774","isca_url":"https://www.isca-archive.org/interspeech_2026/dai26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dai26b_interspeech.pdf","session":"Speaker Diarization 1","topics":["speech-llm","speaker-diarization","asr"],"category":"speaker","institutions":["Northwestern Polytechnical University","Soul AI","Shanghai Jiao Tong University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dai26b_interspeech","category":"speaker","institutions":["Northwestern Polytechnical University","Soul AI","Shanghai Jiao Tong University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-774","pdf":"https://www.isca-archive.org/interspeech_2026/dai26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dai26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dai26b_interspeech/markdown.md"},{"id":"dai26c_interspeech","title":"One-Step Token-to-Waveform Generation with MeanFlow in Latent Space","authors":["Zheqi Dai","Guangyan Zhang","Zhen Ye","Jingyu Li","Haolin He","Chunyat Wu","Yiwen Guo","Qiuqiang Kong"],"year":2026,"doi":"10.21437/Interspeech.2026-791","isca_url":"https://www.isca-archive.org/interspeech_2026/dai26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dai26c_interspeech.pdf","session":"Speech Synthesis: Speech Features, Codec and Representations","topics":["tts","speech-llm","self-supervised"],"category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Chinese University of Hong Kong","Tencent","Hong Kong University of Science and Technology"],"code":{"url":"https://github.com/dzq84/meantok","stars":13,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dai26c_interspeech","category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Chinese University of Hong Kong","Tencent","Hong Kong University of Science and Technology"],"code":"https://github.com/dzq84/meantok","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-791","pdf":"https://www.isca-archive.org/interspeech_2026/dai26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dai26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dai26c_interspeech/markdown.md"},{"id":"dang26_interspeech","title":"Rhythmic Patterning in Vietnamese: The Case of Quadrisyllabic Reduplicative Words","authors":["Phuong Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-1508","isca_url":"https://www.isca-archive.org/interspeech_2026/dang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dang26_interspeech.pdf","session":"Perception and Appraisal of Prosody","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["Ohio State University"],"funding":["Ilse Lehiste Memorial Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dang26_interspeech","category":"phonetics-linguistics","institutions":["Ohio State University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1508","pdf":"https://www.isca-archive.org/interspeech_2026/dang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dang26_interspeech/markdown.md"},{"id":"danner26_interspeech","title":"Vocal Tract Disparity and Potential Implications for Speaker Recognition","authors":["Tiena Danner","Valeriia Vyshnevetska","Daniel Friedrichs","Steven Moran"],"year":2026,"doi":"10.21437/Interspeech.2026-2627","isca_url":"https://www.isca-archive.org/interspeech_2026/danner26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/danner26_interspeech.pdf","session":"Methods and Data for Vocal Tract Shape and Articulation Analysis","topics":["speaker-verification","phonetics","evaluation"],"category":"speaker","institutions":["University of Neuchatel","University of Zurich","Zurich Forensic Science Institute"],"funding":["Swiss National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"danner26_interspeech","category":"speaker","institutions":["University of Neuchatel","University of Zurich","Zurich Forensic Science Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2627","pdf":"https://www.isca-archive.org/interspeech_2026/danner26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/danner26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/danner26_interspeech/markdown.md"},{"id":"dao26_interspeech","title":"Linguistic Bias Mitigation for Spoofing Detection via Gradient Reversal and A Variational Information Bottleneck","authors":["Anh-Tuan DAO","Driss Matrouf","Mickael Rouvier","Nicholas Evans"],"year":2026,"doi":"10.21437/Interspeech.2026-676","isca_url":"https://www.isca-archive.org/interspeech_2026/dao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dao26_interspeech.pdf","session":"Speaker Verification and Anti-Spoofing","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","institutions":["Avignon Universite","EURECOM"],"funding":["ANR BRUEL"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dao26_interspeech","category":"deepfake-security","institutions":["Avignon Universite","EURECOM"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-676","pdf":"https://www.isca-archive.org/interspeech_2026/dao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dao26_interspeech/markdown.md"},{"id":"daoxuanquang26_interspeech","title":"M-LAMA: Multimodal Automated Scoring of Long-form Spoken English","authors":["Minh Dao-Xuan-Quang","Son Dinh-Nguyen","Thi-Mai-Anh Bui","Phi-Le Nguyen"],"year":2026,"doi":"10.21437/Interspeech.2026-1542","isca_url":"https://www.isca-archive.org/interspeech_2026/daoxuanquang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/daoxuanquang26_interspeech.pdf","session":"Long-form Audio & New Attention Approaches","topics":["speech-llm","evaluation","paralinguistics"],"category":"applications-other","institutions":["Hanoi University of Science and Technology"],"code":{"url":"https://github.com/cssi87m/M-LAMA","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"daoxuanquang26_interspeech","category":"applications-other","institutions":["Hanoi University of Science and Technology"],"code":"https://github.com/cssi87m/M-LAMA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1542","pdf":"https://www.isca-archive.org/interspeech_2026/daoxuanquang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/daoxuanquang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/daoxuanquang26_interspeech/markdown.md"},{"id":"deguchi26_interspeech","title":"Non-Autoregressive Minimum Bayes' Risk Decoding for Fast Speech Recognition","authors":["Hiroyuki Deguchi","Takatomo Kano","Katsuki Chousa","Marc Delcroix"],"year":2026,"doi":"10.21437/Interspeech.2026-2971","isca_url":"https://www.isca-archive.org/interspeech_2026/deguchi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/deguchi26_interspeech.pdf","session":"Search Methods and Inference Algorithms","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["efficient-on-device"],"institutions":["NTT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"deguchi26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["NTT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2971","pdf":"https://www.isca-archive.org/interspeech_2026/deguchi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/deguchi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/deguchi26_interspeech/markdown.md"},{"id":"deng26_interspeech","title":"Decoding while Adapting: Zero-Shot Online Speaker Adaptation via Audio-Textual Prompts for Elderly Speech Recognition","authors":["Chengxi Deng","Xurong Xie","Shujie Hu","Mengzhe Geng","Tianzi Wang","Youjun Chen","Huimeng Wang","Haoning Xu","Jiajun Deng","Xunying Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-620","isca_url":"https://www.isca-archive.org/interspeech_2026/deng26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/deng26_interspeech.pdf","session":"Speech and Language Technologies in Healthcare","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Chinese University of Hong Kong","Institute of Software, Chinese Academy of Sciences","National Research Council Canada"],"funding":["Hong Kong RGC GRF","Basic Research Project of Institute of Software, Chinese Academy of Sciences","Youth Innovation Promotion Association CAS"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"deng26_interspeech","category":"asr","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Chinese University of Hong Kong","Institute of Software, Chinese Academy of Sciences","National Research Council Canada"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-620","pdf":"https://www.isca-archive.org/interspeech_2026/deng26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/deng26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/deng26_interspeech/markdown.md"},{"id":"deng26b_interspeech","title":"Codec-induced Mismatch, Speech Duration, and Speaker-dependent Effect in a DNN-based Forensic Speaker Recognition System","authors":["Guangmou Deng","Bruce Xiao Wang","Vincent Hughes"],"year":2026,"doi":"10.21437/Interspeech.2026-1004","isca_url":"https://www.isca-archive.org/interspeech_2026/deng26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/deng26b_interspeech.pdf","session":"Speaker Recognition and Verification","topics":["speaker-verification","evaluation","low-resource"],"category":"speaker","labels":["robustness-noise"],"institutions":["Hong Kong Polytechnic University","University of York"],"funding":["Hong Kong Polytechnic University","Research Grants Council of the Hong Kong Special Administrative Region"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"deng26b_interspeech","category":"speaker","labels":["robustness-noise"],"institutions":["Hong Kong Polytechnic University","University of York"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1004","pdf":"https://www.isca-archive.org/interspeech_2026/deng26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/deng26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/deng26b_interspeech/markdown.md"},{"id":"deng26c_interspeech","title":"Confidence Score Guided Incremental and Speaker Adaptive Pseudo-Labeling for Semi-Supervised Elderly Speech Recognition","authors":["Chengxi Deng","Xurong Xie","Shujie Hu","Jiajun Deng","Mengzhe Geng","Youjun Chen","Huimeng Wang","Haoning Xu","Guinan Li","Xunying Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-1611","isca_url":"https://www.isca-archive.org/interspeech_2026/deng26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/deng26c_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 2","topics":["asr","speaker-adaptation","semi-supervised"],"category":"asr","labels":["low-resource","multilingual"],"institutions":["Chinese University of Hong Kong","Chinese Academy of Sciences","National Research Council Canada"],"funding":["Hong Kong RGC GRF","Basic Research Project of Institute of Software, Chinese Academy of Sciences","Youth Innovation Promotion Association CAS"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"deng26c_interspeech","category":"asr","labels":["low-resource","multilingual"],"institutions":["Chinese University of Hong Kong","Chinese Academy of Sciences","National Research Council Canada"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1611","pdf":"https://www.isca-archive.org/interspeech_2026/deng26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/deng26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/deng26c_interspeech/markdown.md"},{"id":"deng26d_interspeech","title":"Joint Learning of Covariance Estimation and White Noise Gain for Robust MVDR Beamforming","authors":["Yongyi Deng","Hanchen Pei","Jianbo Ma","Gongping Huang","Jingdong Chen","Jacob Benesty"],"year":2026,"doi":"10.21437/Interspeech.2026-2212","isca_url":"https://www.isca-archive.org/interspeech_2026/deng26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/deng26d_interspeech.pdf","session":"Multi-Channel, Beamforming and Spatial Speech Enhancement","topics":["speech-enhancement","source-separation","self-supervised"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Wuhan University","Dolby Laboratories","Northwestern Polytechnical University","University of Quebec"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"deng26d_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Wuhan University","Dolby Laboratories","Northwestern Polytechnical University","University of Quebec"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2212","pdf":"https://www.isca-archive.org/interspeech_2026/deng26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/deng26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/deng26d_interspeech/markdown.md"},{"id":"dey26_interspeech","title":"Improving Adversarial Robustness in Spoken Language Identification through Self-Defensive Distillation","authors":["Spandan Dey"],"year":2026,"doi":"10.21437/Interspeech.2026-3091","isca_url":"https://www.isca-archive.org/interspeech_2026/dey26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dey26_interspeech.pdf","session":"Language and Dialect Recognition","topics":["spoken-language-understanding","self-supervised","evaluation"],"category":"speaker","labels":["multilingual","robustness-noise"],"institutions":["Samsung"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dey26_interspeech","category":"speaker","labels":["multilingual","robustness-noise"],"institutions":["Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3091","pdf":"https://www.isca-archive.org/interspeech_2026/dey26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dey26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dey26_interspeech/markdown.md"},{"id":"dey26b_interspeech","title":"Rethinking Organization Entity Modeling in End-to-End Acoustic Named Entity Recognition","authors":["Spandan Dey","Nidhi Mantri","Sambit Behera","Hirak Mondal","Sanjay Kurmi","Sreyasree Mandal","Atharv Joshi","Premjeet Singh","Gopal Agrawal"],"year":2026,"doi":"10.21437/Interspeech.2026-3115","isca_url":"https://www.isca-archive.org/interspeech_2026/dey26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dey26b_interspeech.pdf","session":"Information Extraction and Retrieval / Survey Talk","topics":["speech-llm","self-supervised","dataset"],"category":"asr","institutions":["Samsung"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dey26b_interspeech","category":"asr","institutions":["Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3115","pdf":"https://www.isca-archive.org/interspeech_2026/dey26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dey26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dey26b_interspeech/markdown.md"},{"id":"dhaka26_interspeech","title":"WER Are We (Really): How Well Do Top Open ASR Leaderboard Models Generalize to Nonstandard Speech?","authors":["Aditya Dhaka","Aarush Mathur","Dena Mujtaba","Hope Gerlach-Houck","Caryn Herring","Chelsea Johnson","J. Scott Yaruss","Nihar Mahapatra"],"year":2026,"doi":"10.21437/Interspeech.2026-3522","isca_url":"https://www.isca-archive.org/interspeech_2026/dhaka26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dhaka26_interspeech.pdf","session":"Spoken Language Processing: Evaluation and Metrics","topics":["asr","evaluation","low-resource"],"category":"asr","labels":["self-supervised"],"institutions":["Michigan State University","Western Michigan University","Friends: The National Association of Young People Who Stutter"],"funding":["U.S. National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dhaka26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Michigan State University","Western Michigan University","Friends: The National Association of Young People Who Stutter"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3522","pdf":"https://www.isca-archive.org/interspeech_2026/dhaka26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dhaka26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dhaka26_interspeech/markdown.md"},{"id":"dhar26_interspeech","title":"Adaptive Oscillatory Inductive Bias for Modeling Sharp Prosodic Dynamics in Diffusion-Based TTS","authors":["Sandipan Dhar","Nirmesh J. Shah","Ashishkumar P. Gudmalwar","Pankaj Wasnik"],"year":2026,"doi":"10.21437/Interspeech.2026-1655","isca_url":"https://www.isca-archive.org/interspeech_2026/dhar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dhar26_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["tts","self-supervised","prosody"],"category":"tts","labels":["generative-model"],"institutions":["Sony"],"code":{"url":"https://research.sri-media-analysis.com/interspeech26-oscilla-tts/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dhar26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Sony"],"code":"https://research.sri-media-analysis.com/interspeech26-oscilla-tts/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1655","pdf":"https://www.isca-archive.org/interspeech_2026/dhar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dhar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dhar26_interspeech/markdown.md"},{"id":"dian26_interspeech","title":"Reconciling Dynamic Data Analysis with Linguistic Reality: Comparing Legendre Polynomial Modelling and GAMM Applied to Prosodic Contact","authors":["Angelo Dian","Mary Baltazani","Spyros Armostis","Elinor Payne"],"year":2026,"doi":"10.21437/Interspeech.2026-2585","isca_url":"https://www.isca-archive.org/interspeech_2026/dian26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dian26_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["prosody","evaluation","phonetics"],"category":"phonetics-linguistics","institutions":["University of Oxford","University of Cyprus"],"funding":["Oxford University John Fell Fund","Economic and Social Research Council"],"code":{"url":"https://doi.org/10.5281/zenodo.20734758","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dian26_interspeech","category":"phonetics-linguistics","institutions":["University of Oxford","University of Cyprus"],"code":"https://doi.org/10.5281/zenodo.20734758","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2585","pdf":"https://www.isca-archive.org/interspeech_2026/dian26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dian26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dian26_interspeech/markdown.md"},{"id":"diecidue26_interspeech","title":"The silence of the weights: a structural pruning strategy for Attention-based audio signal architectures with second-order metrics","authors":["Andrea Diecidue","Carlo Alberto Barbano","Piero Fraternali","Mathieu Fontaine","Enzo Tartaglione"],"year":2026,"doi":"10.21437/Interspeech.2026-2026","isca_url":"https://www.isca-archive.org/interspeech_2026/diecidue26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/diecidue26_interspeech.pdf","session":"Acoustic Event Detection 4","topics":["asr","speech-translation","self-supervised"],"category":"audio-understanding","labels":["efficient-on-device"],"institutions":["Politecnico di Milano","University of Turin","Telecom Paris","Institut Polytechnique de Paris"],"funding":["French National Research Agency","Hi! PARIS","ANR/France 2030 program","Italian Ministry of University and Research","European Union"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"diecidue26_interspeech","category":"audio-understanding","labels":["efficient-on-device"],"institutions":["Politecnico di Milano","University of Turin","Telecom Paris","Institut Polytechnique de Paris"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2026","pdf":"https://www.isca-archive.org/interspeech_2026/diecidue26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/diecidue26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/diecidue26_interspeech/markdown.md"},{"id":"ding26_interspeech","title":"ImKWS: Test-Time Adaptation for Keyword Spotting with Class Imbalance","authors":["Hanyu Ding","Yang Xiao","Jiaheng Dong","Ting Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-258","isca_url":"https://www.isca-archive.org/interspeech_2026/ding26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ding26_interspeech.pdf","session":"Audio segmentation","topics":["keyword-spotting","self-supervised","on-device"],"category":"asr","labels":["robustness-noise"],"institutions":["Jiangsu University","University of Melbourne"],"code":{"url":"https://github.com/dhyzy123/ImKWS","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ding26_interspeech","category":"asr","labels":["robustness-noise"],"institutions":["Jiangsu University","University of Melbourne"],"code":"https://github.com/dhyzy123/ImKWS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-258","pdf":"https://www.isca-archive.org/interspeech_2026/ding26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ding26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ding26_interspeech/markdown.md"},{"id":"ding26b_interspeech","title":"Learning to Evade: Adaptive Attacks on Audio Watermarking","authors":["Weikang Ding","Hanqing Guo","Rui Duan","Guangjing Wang","Yuanda Wang","Mingzhe Chen","Qiben Yan"],"year":2026,"doi":"10.21437/Interspeech.2026-814","isca_url":"https://www.isca-archive.org/interspeech_2026/ding26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ding26b_interspeech.pdf","session":"Spoofing, Deepfake Detection and Watermarking","topics":["speech-enhancement","evaluation","self-supervised"],"category":"deepfake-security","institutions":["University of Missouri-Kansas City","University of Hawaii at Manoa","University of South Florida","Michigan State University","University of Miami"],"funding":["National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ding26b_interspeech","category":"deepfake-security","institutions":["University of Missouri-Kansas City","University of Hawaii at Manoa","University of South Florida","Michigan State University","University of Miami"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-814","pdf":"https://www.isca-archive.org/interspeech_2026/ding26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ding26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ding26b_interspeech/markdown.md"},{"id":"ding26c_interspeech","title":"Through-Wall Radar Speech Acquisition via Cascaded Attention Fusion","authors":["Ruotong Ding","Zhi-Wei Tan","V.G. Reju","Andy W. H. Khong"],"year":2026,"doi":"10.21437/Interspeech.2026-1034","isca_url":"https://www.isca-archive.org/interspeech_2026/ding26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ding26c_interspeech.pdf","session":"Multi-Channel Processing and Specialized Acquisition (UAV, Radar, Hearables)","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","institutions":["Nanyang Technological University","Cochin University of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ding26c_interspeech","category":"enhancement-separation","institutions":["Nanyang Technological University","Cochin University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1034","pdf":"https://www.isca-archive.org/interspeech_2026/ding26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ding26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ding26c_interspeech/markdown.md"},{"id":"ding26d_interspeech","title":"SGAD: A State-Guided Adaptive Decision Framework for Robust EEG-Based Auditory Attention Switch Decoding","authors":["Yuting Ding","Xuefei Wang","Ximin Chen","Chunlin Li","Fei Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1815","isca_url":"https://www.isca-archive.org/interspeech_2026/ding26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ding26d_interspeech.pdf","session":"Speech Production and Perception 1","topics":["spoken-language-understanding","paralinguistics","evaluation"],"category":"applications-other","labels":["streaming-real-time"],"institutions":["Southern University of Science and Technology","Capital Medical University"],"funding":["National Key Research and Development Program of China","National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ding26d_interspeech","category":"applications-other","labels":["streaming-real-time"],"institutions":["Southern University of Science and Technology","Capital Medical University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1815","pdf":"https://www.isca-archive.org/interspeech_2026/ding26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ding26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ding26d_interspeech/markdown.md"},{"id":"dinh26_interspeech","title":"Improving Audio Codec-based Speech Separation By Stacking Residual Vector Quantization Layers","authors":["Nhu Minh Phuong Dinh","Roland Hartanto","Koichi Shinoda"],"year":2026,"doi":"10.21437/Interspeech.2026-2296","isca_url":"https://www.isca-archive.org/interspeech_2026/dinh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dinh26_interspeech.pdf","session":"Source Separation 2","topics":["speech-separation","self-supervised","speech-llm"],"category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["Institute of Science Tokyo"],"funding":["JSPS KAKENHI"],"code":{"url":"https://phuongdnm.github.io/rvqgrid","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dinh26_interspeech","category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["Institute of Science Tokyo"],"code":"https://phuongdnm.github.io/rvqgrid","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2296","pdf":"https://www.isca-archive.org/interspeech_2026/dinh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dinh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dinh26_interspeech/markdown.md"},{"id":"dinh26b_interspeech","title":"LitCodec: ASR-Guided Streaming Speech Coding with Unified Quantization","authors":["Son Dang Dinh","Nguyen Thi Minh Anh","Nhat Tran Hong","Huyen Ngo Thi Thu"],"year":2026,"doi":"10.21437/Interspeech.2026-3474","isca_url":"https://www.isca-archive.org/interspeech_2026/dinh26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dinh26b_interspeech.pdf","session":"Neural Speech Codecs: Low-Bitrate and Disentangled Coding","topics":["speech-coding","self-supervised","asr"],"category":"speech-coding","labels":["efficient-on-device","streaming-real-time"],"institutions":["Torilab AI","Viettel Group","Hanoi University of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dinh26b_interspeech","category":"speech-coding","labels":["efficient-on-device","streaming-real-time"],"institutions":["Torilab AI","Viettel Group","Hanoi University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3474","pdf":"https://www.isca-archive.org/interspeech_2026/dinh26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dinh26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dinh26b_interspeech/markdown.md"},{"id":"diwan26_interspeech","title":"ParaSpeechCLAP: A Dual-Encoder Speech-Text Model for Rich Stylistic Language-Audio Pretraining","authors":["Anuj Diwan","Eunsol Choi","David Harwath"],"year":2026,"doi":"10.21437/Interspeech.2026-1437","isca_url":"https://www.isca-archive.org/interspeech_2026/diwan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/diwan26_interspeech.pdf","session":"Multimodal Emotion Recognition","topics":["tts","self-supervised","speech-llm"],"category":"tts","labels":["self-supervised"],"institutions":["University of Texas at Austin","New York University"],"code":{"url":"https://github.com/ajd12342/paraspeechclap","stars":26,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"diwan26_interspeech","category":"tts","labels":["self-supervised"],"institutions":["University of Texas at Austin","New York University"],"code":"https://github.com/ajd12342/paraspeechclap","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1437","pdf":"https://www.isca-archive.org/interspeech_2026/diwan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/diwan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/diwan26_interspeech/markdown.md"},{"id":"dixit26_interspeech","title":"AURA Score: A Metric for Holistic Audio Question Answering Evaluation","authors":["Satvik Dixit","Soham Deshmukh","Bhiksha Raj"],"year":2026,"doi":"10.21437/Interspeech.2026-3185","isca_url":"https://www.isca-archive.org/interspeech_2026/dixit26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dixit26_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["evaluation","speech-llm","self-supervised"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Carnegie Mellon University"],"funding":["National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dixit26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3185","pdf":"https://www.isca-archive.org/interspeech_2026/dixit26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dixit26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dixit26_interspeech/markdown.md"},{"id":"dong26_interspeech","title":"Membership Inference Attacks against Large Audio Language Models","authors":["Jia-Kai Dong","Yu-Xiang Lin","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-514","isca_url":"https://www.isca-archive.org/interspeech_2026/dong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dong26_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["speech-llm","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["National Taiwan University"],"funding":["Ministry of Education","Taiwan Centers of Excellence in Artificial Intelligence","NTU Artificial Intelligence Center of Research Excellence"],"code":{"url":"https://github.com/snooow1029/ALM_MIA","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dong26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["National Taiwan University"],"code":"https://github.com/snooow1029/ALM_MIA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-514","pdf":"https://www.isca-archive.org/interspeech_2026/dong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dong26_interspeech/markdown.md"},{"id":"dong26b_interspeech","title":"English Vowel Perceptual Training under Multitalker Babble: A Comparison of Humans and Large Language Models","authors":["Wenwei Dong","Alif Silpachai","Catia Cucchiarini","Helmer Strik"],"year":2026,"doi":"10.21437/Interspeech.2026-966","isca_url":"https://www.isca-archive.org/interspeech_2026/dong26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dong26b_interspeech.pdf","session":"Model of Speech Perception","topics":["speech-llm","evaluation","low-resource"],"category":"phonetics-linguistics","labels":["multilingual","self-supervised","robustness-noise"],"institutions":["Radboud University"],"funding":["China Scholarship Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dong26b_interspeech","category":"phonetics-linguistics","labels":["multilingual","self-supervised","robustness-noise"],"institutions":["Radboud University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-966","pdf":"https://www.isca-archive.org/interspeech_2026/dong26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dong26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dong26b_interspeech/markdown.md"},{"id":"dong26c_interspeech","title":"Can Speech LLMs Approximate Human Ratings of Accentedness and Comprehensibility? Evidence from Correlational and Feature-Based Analyses","authors":["Wenwei Dong","Catia Cucchiarini","Roeland van Hout","Helmer Strik"],"year":2026,"doi":"10.21437/Interspeech.2026-1991","isca_url":"https://www.isca-archive.org/interspeech_2026/dong26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dong26c_interspeech.pdf","session":"Paralinguistics","topics":["paralinguistics","evaluation","speech-llm"],"category":"resources-evaluation","institutions":["Radboud University"],"funding":["China Scholarship Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dong26c_interspeech","category":"resources-evaluation","institutions":["Radboud University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1991","pdf":"https://www.isca-archive.org/interspeech_2026/dong26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dong26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dong26c_interspeech/markdown.md"},{"id":"drimalla26_interspeech","title":"Automatic Detection of Stress from Speech in the Trier Social Stress Test","authors":["Hanna Drimalla","Wieland R. Cremer","Christine Kraus","Oliver T. Wolf"],"year":2026,"doi":"10.21437/Interspeech.2026-671","isca_url":"https://www.isca-archive.org/interspeech_2026/drimalla26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/drimalla26_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["paralinguistics","emotion-recognition","dataset"],"category":"paralinguistics-emotion","institutions":["Bielefeld University","Ruhr University Bochum"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"https://github.com/mbp-lab/tsst-speech-stress","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"drimalla26_interspeech","category":"paralinguistics-emotion","institutions":["Bielefeld University","Ruhr University Bochum"],"code":"https://github.com/mbp-lab/tsst-speech-stress","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-671","pdf":"https://www.isca-archive.org/interspeech_2026/drimalla26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/drimalla26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/drimalla26_interspeech/markdown.md"},{"id":"du26_interspeech","title":"Streaming T5-based Text-to-Speech Synthesis with Limited Lookahead","authors":["Muyang Du","Jason Roche","Junjie Lai"],"year":2026,"doi":"10.21437/Interspeech.2026-235","isca_url":"https://www.isca-archive.org/interspeech_2026/du26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/du26_interspeech.pdf","session":"Streaming Speech Synthesis","topics":["tts","streaming","self-supervised"],"category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["NVIDIA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"du26_interspeech","category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["NVIDIA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-235","pdf":"https://www.isca-archive.org/interspeech_2026/du26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/du26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/du26_interspeech/markdown.md"},{"id":"du26b_interspeech","title":"The role of phonation type in Chinese Jin tones: a study using acoustic metrics","authors":["Xiaojing Du"],"year":2026,"doi":"10.21437/Interspeech.2026-1368","isca_url":"https://www.isca-archive.org/interspeech_2026/du26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/du26b_interspeech.pdf","session":"Voice Quality Aspects of Speech","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["University of Cambridge"],"funding":["Cambridge Trust","China Scholarship Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"du26b_interspeech","category":"phonetics-linguistics","institutions":["University of Cambridge"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1368","pdf":"https://www.isca-archive.org/interspeech_2026/du26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/du26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/du26b_interspeech/markdown.md"},{"id":"du26c_interspeech","title":"Orthogonal Feature Projection and Manifold-Constrained Neural PLDA for the TidyVoice2026 Cross-Lingual Speaker Verification Challenge","authors":["Yuxuan Du","Xing He","Jingwen Yang","Xupeng Jia","Yankai Wang","Weili Jiang","Kai Gao","Boyu Zhao","Rong Zheng","Jing Deng"],"year":2026,"doi":"10.21437/Interspeech.2026-3003","isca_url":"https://www.isca-archive.org/interspeech_2026/du26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/du26c_interspeech.pdf","session":"Challenge - TidyVoice Challenge: Cross-Lingual Speaker Verification","topics":["speaker-verification","self-supervised","multilingual"],"category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Beijing Fosafer Information Technology Co., Ltd","University of Electronic Science and Technology of China","Institute of Forensic Science, Ministry of Public Security"],"funding":["Basic Scientific Research Special Funds for Central-level Public Welfare Research Institutions"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"du26c_interspeech","category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Beijing Fosafer Information Technology Co., Ltd","University of Electronic Science and Technology of China","Institute of Forensic Science, Ministry of Public Security"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3003","pdf":"https://www.isca-archive.org/interspeech_2026/du26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/du26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/du26c_interspeech/markdown.md"},{"id":"dudek26_interspeech","title":"Phoneme-Level Mispronunciation Screening in Polish-Speaking Children with an Explainable Assistant","authors":["Milosz Dudek","Daria Hemmerling","Kamil Kwarciak","Maciej Stroinski","Maria Pensko","Mateusz Kowalewski","Leonid Pavlovskyi","Sebastian Jurczak","Anna-Mariia Vitkovska","Zuzanna Miodonska","Natalia Mocko","Michal Krecichwost"],"year":2026,"doi":"10.21437/Interspeech.2026-1416","isca_url":"https://www.isca-archive.org/interspeech_2026/dudek26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dudek26_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["speech-enhancement","self-supervised","low-resource"],"category":"health-clinical","labels":["low-resource","self-supervised"],"institutions":["AGH University of Krakow","SoftServe","Silesian University of Technology","University of Silesia in Katowice"],"funding":["National Centre for Research and Development"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dudek26_interspeech","category":"health-clinical","labels":["low-resource","self-supervised"],"institutions":["AGH University of Krakow","SoftServe","Silesian University of Technology","University of Silesia in Katowice"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1416","pdf":"https://www.isca-archive.org/interspeech_2026/dudek26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dudek26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dudek26_interspeech/markdown.md"},{"id":"dufour26_interspeech","title":"A Large-Scale Per-Speaker Analysis of Re-identification Risk in Speech Anonymization","authors":["Orane Dufour","Paul Magron","Mickael Rouvier","Emmanuel Vincent"],"year":2026,"doi":"10.21437/Interspeech.2026-440","isca_url":"https://www.isca-archive.org/interspeech_2026/dufour26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dufour26_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["speech-anonymization","speaker-verification","evaluation"],"category":"deepfake-security","institutions":["Universite de Lorraine","CNRS","Inria","Avignon University"],"funding":["Agence Nationale de la Recherche","SpeechPrivacy"],"code":{"url":"https://github.com/OraneD/Speaker-Linkability","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dufour26_interspeech","category":"deepfake-security","institutions":["Universite de Lorraine","CNRS","Inria","Avignon University"],"code":"https://github.com/OraneD/Speaker-Linkability","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-440","pdf":"https://www.isca-archive.org/interspeech_2026/dufour26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dufour26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dufour26_interspeech/markdown.md"},{"id":"duraisamy26_interspeech","title":"Subject-Invariant Dynamic Graph Modeling for Cross-Subject EEG Imagined Speech Decoding","authors":["Saravanakumar Duraisamy","Luis A. Leiva"],"year":2026,"doi":"10.21437/Interspeech.2026-2884","isca_url":"https://www.isca-archive.org/interspeech_2026/duraisamy26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/duraisamy26_interspeech.pdf","session":"Neurophysiology of Speech","topics":["asr","self-supervised","health"],"category":"asr","institutions":["University of Luxembourg"],"funding":["Pathfinder program of the European Innovation Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"duraisamy26_interspeech","category":"asr","institutions":["University of Luxembourg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2884","pdf":"https://www.isca-archive.org/interspeech_2026/duraisamy26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/duraisamy26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/duraisamy26_interspeech/markdown.md"},{"id":"dutta26_interspeech","title":"Spashta Audio-Bench: Unified ASR and TTS Evaluation Framework across Indian Languages","authors":["Bikash Dutta","Siddhant Gahankari","Abhinav Kumar","Siddarth Modugu","Shalini Kapoor","Mayank Vatsa","Richa Singh"],"year":2026,"doi":"10.21437/Interspeech.2026-1777","isca_url":"https://www.isca-archive.org/interspeech_2026/dutta26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dutta26_interspeech.pdf","session":"Speech Benchmarks, Evaluation, and Resources","topics":["asr","tts","multilingual"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Indian Institute of Technology Jodhpur","EkStep Foundation"],"funding":["EkStep Foundation","IndiaAI","Meta"],"code":{"url":"https://iab-rubric.org/resources/codes/spashta-audio-bench","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dutta26_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Indian Institute of Technology Jodhpur","EkStep Foundation"],"code":"https://iab-rubric.org/resources/codes/spashta-audio-bench","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1777","pdf":"https://www.isca-archive.org/interspeech_2026/dutta26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dutta26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dutta26_interspeech/markdown.md"},{"id":"dvirniak26_interspeech","title":"Towards Robust Speech Deepfake Detection via Human-Inspired Reasoning","authors":["Artem Dvirniak","Evgeny Kushnir","Dmitrii Tarasov","Artem Iudin","Oleg Kiriukhin","Mikhail Pautov","Dmitrii Korzh","Oleg Rogov"],"year":2026,"doi":"10.21437/Interspeech.2026-1289","isca_url":"https://www.isca-archive.org/interspeech_2026/dvirniak26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/dvirniak26_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["speech-deepfake-detection","speech-llm","self-supervised"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["MIRAI","AXXX","HSE University","Applied AI Institute","Fusion Brain Lab","MTUCI","City University of Hong Kong","Trusted AI Research Center"],"funding":["Ministry of Economic Development of the Russian Federation"],"code":{"url":"https://github.com/dkorzh10/HIR-SDD","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"dvirniak26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["MIRAI","AXXX","HSE University","Applied AI Institute","Fusion Brain Lab","MTUCI","City University of Hong Kong","Trusted AI Research Center"],"code":"https://github.com/dkorzh10/HIR-SDD","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1289","pdf":"https://www.isca-archive.org/interspeech_2026/dvirniak26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/dvirniak26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/dvirniak26_interspeech/markdown.md"},{"id":"e26_interspeech","title":"Benchmarking Speech Systems for Frontline Health Conversations: The DISPLACE-M Challenge","authors":["Dhanya E","Ankita Meena","Manas Nanivadekar","Noumida A","Victor Azad","Ashwini Nagaraj Shenoy","Pratik Roy Chowdhuri","Shobhit Banga","Vanshika Chhabra","Chitralekha Bhati","Shareef babu Kalluri","Srikanth Raj Chetupalli","Deepu Vijayasenan","Sriram Ganapathy"],"year":2026,"doi":"10.21437/Interspeech.2026-3255","isca_url":"https://www.isca-archive.org/interspeech_2026/e26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/e26_interspeech.pdf","session":"Medical Dialogue and Conversational Understanding","topics":["asr","speaker-diarization","dataset"],"category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Indian Institute of Science","National Institute of Technology Karnataka","Josh Talks","Manipal Academy of Higher Education","Indian Institute of Technology Bombay"],"code":{"url":"https://www.codabench.org/competitions/13833/?secret_key=1b714e64-0f0d-4e0f-8a3c-be9b3d10f00c#","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"e26_interspeech","category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Indian Institute of Science","National Institute of Technology Karnataka","Josh Talks","Manipal Academy of Higher Education","Indian Institute of Technology Bombay"],"code":"https://www.codabench.org/competitions/13833/?secret_key=1b714e64-0f0d-4e0f-8a3c-be9b3d10f00c#","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3255","pdf":"https://www.isca-archive.org/interspeech_2026/e26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/e26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/e26_interspeech/markdown.md"},{"id":"edet26_interspeech","title":"Towards Digital Preservation of Efik: TTS for a Low-Resource African Language","authors":["Offiong Bassey Edet","Emmanuel Oyo-Ita","Archibong Okon Archibong","David Effanga Bassey","Mbuotidem Sunday Awak"],"year":2026,"doi":"10.21437/Interspeech.2026-1868","isca_url":"https://www.isca-archive.org/interspeech_2026/edet26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/edet26_interspeech.pdf","session":"Safeguards for Synthetic Speech: Ethical, Technical, and Legal Perspectives","topics":["tts","low-resource","multilingual"],"category":"tts","labels":["low-resource","dataset-or-benchmark-release","generative-model"],"institutions":["University of Cross River State","University of Calabar","ML Collective"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"edet26_interspeech","category":"tts","labels":["low-resource","dataset-or-benchmark-release","generative-model"],"institutions":["University of Cross River State","University of Calabar","ML Collective"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1868","pdf":"https://www.isca-archive.org/interspeech_2026/edet26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/edet26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/edet26_interspeech/markdown.md"},{"id":"edraki26_interspeech","title":"Energy Redistribution in the Spectro-Temporal Modulation Domain for Near-End Listening Enhancement","authors":["Amin Edraki","Amirhossein Hajavi","Irina Kezele","Yuanhao Yu"],"year":2026,"doi":"10.21437/Interspeech.2026-319","isca_url":"https://www.isca-archive.org/interspeech_2026/edraki26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/edraki26_interspeech.pdf","session":"Quality, Intelligibility and Evaluation of Speech and Codecs","topics":["speech-enhancement","paralinguistics","evaluation"],"category":"enhancement-separation","institutions":["Huawei"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"edraki26_interspeech","category":"enhancement-separation","institutions":["Huawei"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-319","pdf":"https://www.isca-archive.org/interspeech_2026/edraki26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/edraki26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/edraki26_interspeech/markdown.md"},{"id":"eeckt26_interspeech","title":"Parameter-Efficient Continual Learning for Automatic Speech Recognition","authors":["Steven Vander Eeckt","Hugo Van hamme"],"year":2026,"doi":"10.21437/Interspeech.2026-3169","isca_url":"https://www.isca-archive.org/interspeech_2026/eeckt26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/eeckt26_interspeech.pdf","session":"ASR Under Real-World Constraints: Streaming, Adaptation, and Efficiency","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["efficient-on-device"],"institutions":["KU Leuven"],"funding":["Research Foundation Flanders"],"code":{"url":"https://github.com/StevenVdEeckt/pecl-for-asr","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"eeckt26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["KU Leuven"],"code":"https://github.com/StevenVdEeckt/pecl-for-asr","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3169","pdf":"https://www.isca-archive.org/interspeech_2026/eeckt26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/eeckt26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/eeckt26_interspeech/markdown.md"},{"id":"elelu26_interspeech","title":"ConformalMOS: Uncertainty-Aware MOS Prediction with Conformal Intervals and Ordinal Modeling","authors":["Kehinde Elelu","Joshua E. Siegel","Mohammadali Saffary","Tashfain Ahmed","Simeon Babatunde","Ebuka Okpala"],"year":2026,"doi":"10.21437/Interspeech.2026-572","isca_url":"https://www.isca-archive.org/interspeech_2026/elelu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/elelu26_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["tts","voice-conversion","evaluation"],"category":"resources-evaluation","institutions":["Michigan State University","Clemson University"],"funding":["Michigan Translational Research and Commercialization Program","Michigan Strategic Fund","Michigan Economic Development Corporation","21st Century Jobs Trust Fund","State of Michigan","U.S. Economic Development Administration","U.S. Department of Commerce"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"elelu26_interspeech","category":"resources-evaluation","institutions":["Michigan State University","Clemson University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-572","pdf":"https://www.isca-archive.org/interspeech_2026/elelu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/elelu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/elelu26_interspeech/markdown.md"},{"id":"elisha26_interspeech","title":"Audio-Based Understanding of Audiobook Narration Appeal","authors":["Shahar Elisha","Mariano Beguerisse-Díaz","Emmanouil Benetos"],"year":2026,"doi":"10.21437/Interspeech.2026-453","isca_url":"https://www.isca-archive.org/interspeech_2026/elisha26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/elisha26_interspeech.pdf","session":"Speaker Identity, States, and Traits in Paralinguistics","topics":["speech-paralinguistics","dataset","evaluation"],"category":"paralinguistics-emotion","institutions":["Spotify","Queen Mary University of London"],"code":{"url":"https://github.com/spotify-research/audiobook-narrations-interspeech","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"elisha26_interspeech","category":"paralinguistics-emotion","institutions":["Spotify","Queen Mary University of London"],"code":"https://github.com/spotify-research/audiobook-narrations-interspeech","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-453","pdf":"https://www.isca-archive.org/interspeech_2026/elisha26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/elisha26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/elisha26_interspeech/markdown.md"},{"id":"eljasiak26_interspeech","title":"Foundational speech models evaluation on multilingual dementia prediction","authors":["Bartłomiej Eljasiak","Wojciech Szecówka","Piotr Masztalski","Agnieszka Pruszek","Teresa Brzozka","Zofia Marciniak","Justyna Krzywdziak","Michał K. Grzeszczyk","Łukasz Łazarski","Mateusz Matuszewski","Daria Hemmerling"],"year":2026,"doi":"10.21437/Interspeech.2026-2587","isca_url":"https://www.isca-archive.org/interspeech_2026/eljasiak26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/eljasiak26_interspeech.pdf","session":"Pathological Speech Assessment 1","topics":["speech-llm","health","multilingual"],"category":"health-clinical","labels":["multilingual","self-supervised"],"institutions":["Samsung R&D Institute","AGH University of Kraków","Harvard Medical School","Kozminski University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"eljasiak26_interspeech","category":"health-clinical","labels":["multilingual","self-supervised"],"institutions":["Samsung R&D Institute","AGH University of Kraków","Harvard Medical School","Kozminski University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2587","pdf":"https://www.isca-archive.org/interspeech_2026/eljasiak26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/eljasiak26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/eljasiak26_interspeech/markdown.md"},{"id":"ellinson26_interspeech","title":"HRTF-guided Binaural Target Speaker Extraction with Real-World Validation","authors":["Yoav Ellinson","Sharon Gannot"],"year":2026,"doi":"10.21437/Interspeech.2026-807","isca_url":"https://www.isca-archive.org/interspeech_2026/ellinson26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ellinson26_interspeech.pdf","session":"Target Speaker Extraction, Speech Separation and Audio Understanding","topics":["speech-enhancement","source-separation","self-supervised"],"category":"enhancement-separation","institutions":["Bar-Ilan University"],"funding":["Israel Science Foundation","German Research Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ellinson26_interspeech","category":"enhancement-separation","institutions":["Bar-Ilan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-807","pdf":"https://www.isca-archive.org/interspeech_2026/ellinson26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ellinson26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ellinson26_interspeech/markdown.md"},{"id":"elmerich26_interspeech","title":"NewAppVoice: Tools for Visualizing and Correcting Acoustic Measures","authors":["Amélie Elmerich","Yao Dong","Louise McKeever","Katia Chirkova","Lise Crevier-Buchman","Claire Pillot-Loiseau","Angelique Amelot"],"year":2026,"doi":"10.21437/Interspeech.2026-2560","isca_url":"https://www.isca-archive.org/interspeech_2026/elmerich26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/elmerich26_interspeech.pdf","session":"Emotion, Prosody, and Articulation","topics":["phonetics","speech-enhancement","low-resource"],"category":"phonetics-linguistics","labels":["low-resource"],"institutions":["CNRS","Sorbonne Nouvelle University","Foch Hospital","Paris-Saclay University"],"funding":["French National Research Agency"],"code":{"url":"https://osf.io/y6zce/overview?view_only=8890aae1298b4550a273a0067dd68281","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"elmerich26_interspeech","category":"phonetics-linguistics","labels":["low-resource"],"institutions":["CNRS","Sorbonne Nouvelle University","Foch Hospital","Paris-Saclay University"],"code":"https://osf.io/y6zce/overview?view_only=8890aae1298b4550a273a0067dd68281","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2560","pdf":"https://www.isca-archive.org/interspeech_2026/elmerich26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/elmerich26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/elmerich26_interspeech/markdown.md"},{"id":"elsetohy26_interspeech","title":"ArFake: A Robust Framework for Multi-Dialect Arabic Speech Spoofing Detection Benchmark","authors":["Mohamed Elsetohy","Alhassan Ehab","Ali Mekky","Besher Hassan","Shady Shehata"],"year":2026,"doi":"10.21437/Interspeech.2026-2665","isca_url":"https://www.isca-archive.org/interspeech_2026/elsetohy26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/elsetohy26_interspeech.pdf","session":"Spoofing and Deepfake Detection 3","topics":["audio-deepfake","speaker-verification","multilingual"],"category":"deepfake-security","labels":["multilingual","dataset-or-benchmark-release","generative-model"],"institutions":["Mohamed bin Zayed University of Artificial Intelligence","Queen's University","University of Waterloo"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"elsetohy26_interspeech","category":"deepfake-security","labels":["multilingual","dataset-or-benchmark-release","generative-model"],"institutions":["Mohamed bin Zayed University of Artificial Intelligence","Queen's University","University of Waterloo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2665","pdf":"https://www.isca-archive.org/interspeech_2026/elsetohy26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/elsetohy26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/elsetohy26_interspeech/markdown.md"},{"id":"eom26_interspeech","title":"Transcript-Free Flow-Matching Text-to-Speech via Speech Feature Conditioning","authors":["SooHwan Eom","Hee Suk Yoon","Eunseop Yoon","Mark Hasegawa-Johnson","Chang D. Yoo"],"year":2026,"doi":"10.21437/Interspeech.2026-3190","isca_url":"https://www.isca-archive.org/interspeech_2026/eom26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/eom26_interspeech.pdf","session":"Speech Synthesis: Speech Features, Codec and Representations","topics":["tts","self-supervised","low-resource"],"category":"tts","labels":["self-supervised","generative-model"],"institutions":["Korea Advanced Institute of Science and Technology","University of Illinois Urbana-Champaign"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"eom26_interspeech","category":"tts","labels":["self-supervised","generative-model"],"institutions":["Korea Advanced Institute of Science and Technology","University of Illinois Urbana-Champaign"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3190","pdf":"https://www.isca-archive.org/interspeech_2026/eom26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/eom26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/eom26_interspeech/markdown.md"},{"id":"evans26_interspeech","title":"Mapping Acceptable Pronunciation Range for te reo Māori through Perceptual, Acoustic, and Marker Evaluative Data","authors":["Zoe E Evans","C. I. Watson","Peter J Keegan","Jesin James","Piata Allen"],"year":2026,"doi":"10.21437/Interspeech.2026-1561","isca_url":"https://www.isca-archive.org/interspeech_2026/evans26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/evans26_interspeech.pdf","session":"Pacific Voices: Speech Science and Technology for the Languages of the Pacific Ocean","topics":["low-resource","evaluation","phonetics"],"category":"phonetics-linguistics","labels":["low-resource"],"institutions":["University of Auckland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"evans26_interspeech","category":"phonetics-linguistics","labels":["low-resource"],"institutions":["University of Auckland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1561","pdf":"https://www.isca-archive.org/interspeech_2026/evans26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/evans26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/evans26_interspeech/markdown.md"},{"id":"fan26_interspeech","title":"PrefSQA: Pairwise Preference Prediction for Speech Quality Assessment and the Critical Role of High Quality Datasets","authors":["Junyi Fan","Donald S. Williamson"],"year":2026,"doi":"10.21437/Interspeech.2026-1512","isca_url":"https://www.isca-archive.org/interspeech_2026/fan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fan26_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["speech-enhancement","evaluation","self-supervised"],"category":"resources-evaluation","institutions":["Ohio State University"],"funding":["Ohio Supercomputer Center","National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fan26_interspeech","category":"resources-evaluation","institutions":["Ohio State University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1512","pdf":"https://www.isca-archive.org/interspeech_2026/fan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fan26_interspeech/markdown.md"},{"id":"fan26b_interspeech","title":"Robust Multi-Tier Infant-Centered Audio Understanding with Whisper via Structured Speaker Conditioning","authors":["Xulin Fan","Jialu Li","Mohammad Nur Hossain Khan","Kexin Hu","Bashima Islam","Mark Hasegawa-Johnson","Nancy L. McElwain"],"year":2026,"doi":"10.21437/Interspeech.2026-2746","isca_url":"https://www.isca-archive.org/interspeech_2026/fan26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fan26b_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["speech-llm","speaker-diarization","paralinguistics"],"category":"speaker","labels":["self-supervised"],"institutions":["University of Illinois Urbana-Champaign","University of Arizona","Worcester Polytechnic Institute"],"funding":["National Institute on Drug Abuse"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fan26b_interspeech","category":"speaker","labels":["self-supervised"],"institutions":["University of Illinois Urbana-Champaign","University of Arizona","Worcester Polytechnic Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2746","pdf":"https://www.isca-archive.org/interspeech_2026/fan26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fan26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fan26b_interspeech/markdown.md"},{"id":"fan26c_interspeech","title":"Bayesian Generalized Additive Multilevel Models for Accurate ERP Latency Estimation under Moderate Downsampling","authors":["Zixia Fan","Ronny Kurniawan Ibrahim","Joshua Penney","Felicity Cox"],"year":2026,"doi":"10.21437/Interspeech.2026-2750","isca_url":"https://www.isca-archive.org/interspeech_2026/fan26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fan26c_interspeech.pdf","session":"Neurophysiology of Speech","topics":["paralinguistics","evaluation","self-supervised"],"category":"applications-other","institutions":["Macquarie University"],"funding":["China Scholarship Council","Macquarie University","Australian Research Council"],"code":{"url":"https://osf.io/fcgbk/overview?view_only=f13dacbe9bde4658a8f8c6660774117c","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fan26c_interspeech","category":"applications-other","institutions":["Macquarie University"],"code":"https://osf.io/fcgbk/overview?view_only=f13dacbe9bde4658a8f8c6660774117c","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2750","pdf":"https://www.isca-archive.org/interspeech_2026/fan26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fan26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fan26c_interspeech/markdown.md"},{"id":"fan26d_interspeech","title":"Cloud-Boosted Low-Compute Multi-Channel Speech Enhancement","authors":["Xulin Fan","Juan Azcarreta","Ashutosh Pandey","Jesus Alvarez","Ke Tan","Jacob Donley","Ritwik Giri","Buye Xu"],"year":2026,"doi":"10.21437/Interspeech.2026-2774","isca_url":"https://www.isca-archive.org/interspeech_2026/fan26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fan26d_interspeech.pdf","session":"Multi-Channel Processing and Specialized Acquisition (UAV, Radar, Hearables)","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["University of Illinois Urbana-Champaign","Meta"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fan26d_interspeech","category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["University of Illinois Urbana-Champaign","Meta"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2774","pdf":"https://www.isca-archive.org/interspeech_2026/fan26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fan26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fan26d_interspeech/markdown.md"},{"id":"fang26_interspeech","title":"Temporal Ensembling Threshold and Neighbor-Aware Label Mixup for Speaker Verification with Open-Set Noisy Labels","authors":["Zhihua Fang","Shumei Tao","Liang He"],"year":2026,"doi":"10.21437/Interspeech.2026-11","isca_url":"https://www.isca-archive.org/interspeech_2026/fang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fang26_interspeech.pdf","session":"Speaker Verification: Architectures, Losses, and LLMs","topics":["speaker-verification","self-supervised","dataset"],"category":"speaker","labels":["robustness-noise"],"institutions":["Xinjiang University","Tsinghua University","AGIBOT"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fang26_interspeech","category":"speaker","labels":["robustness-noise"],"institutions":["Xinjiang University","Tsinghua University","AGIBOT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-11","pdf":"https://www.isca-archive.org/interspeech_2026/fang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fang26_interspeech/markdown.md"},{"id":"fang26b_interspeech","title":"WhispEar: A Bidirectional Framework for Scaling Whispered Speech Conversion via Pseudo-Parallel Whisper Generation","authors":["Zihao Fang","Yingda Shen","Zifan Guan","Tongtong Song","Zhenyi Liu","Zhizheng Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-1827","isca_url":"https://www.isca-archive.org/interspeech_2026/fang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fang26b_interspeech.pdf","session":"Voice Conversion","topics":["speech-enhancement","voice-conversion","multilingual"],"category":"tts","labels":["self-supervised","generative-model"],"institutions":["Chinese University of Hong Kong, Shenzhen","Honor Device Co., Ltd","Shenzhen Research Institute of Big Data","Shenzhen Loop Area Institute","Amphion Technology Co., Ltd"],"funding":["Shenzhen Research Institute of Big Data","Program for Guangdong Introducing Innovative and Enterpreneurial Teams"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fang26b_interspeech","category":"tts","labels":["self-supervised","generative-model"],"institutions":["Chinese University of Hong Kong, Shenzhen","Honor Device Co., Ltd","Shenzhen Research Institute of Big Data","Shenzhen Loop Area Institute","Amphion Technology Co., Ltd"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1827","pdf":"https://www.isca-archive.org/interspeech_2026/fang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fang26b_interspeech/markdown.md"},{"id":"farsi26_interspeech","title":"Preserving the Iranian Turkic Language: Community-Driven ASR Datasets and Benchmarking for South Azerbaijani","authors":["Farhan Farsi","Shayan Bali","Jalil Nourmohammadi Khiarak","Mohammad Hossein Aref","Taher Akbari Saeed"],"year":2026,"doi":"10.21437/Interspeech.2026-1516","isca_url":"https://www.isca-archive.org/interspeech_2026/farsi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/farsi26_interspeech.pdf","session":"Multilingual Speech 1","topics":["asr","low-resource","multilingual"],"category":"asr","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["Amirkabir University of Technology","King's College London","Kartal OL Foundation"],"funding":["Kartal Ol Foundation","YoYo research group"],"code":{"url":"https://github.com/Kartalol/Kartalol-azb-asr","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"farsi26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["Amirkabir University of Technology","King's College London","Kartal OL Foundation"],"code":"https://github.com/Kartalol/Kartalol-azb-asr","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1516","pdf":"https://www.isca-archive.org/interspeech_2026/farsi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/farsi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/farsi26_interspeech/markdown.md"},{"id":"feghhi26_interspeech","title":"Lightbeam: An Accurate and Memory-Efficient CTC Decoder for Speech Neuroprostheses","authors":["Ebrahim Feghhi","Junlin Hu","Nima Hadidi","Jonathan Kao"],"year":2026,"doi":"10.21437/Interspeech.2026-2947","isca_url":"https://www.isca-archive.org/interspeech_2026/feghhi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/feghhi26_interspeech.pdf","session":"Assistive Technologies 1","topics":["speech-llm","decoding","low-resource"],"category":"asr","labels":["efficient-on-device"],"institutions":["University of California Los Angeles"],"funding":["National Science Foundation","National Institutes of Health"],"code":{"url":"https://doi.org/10.5281/zenodo.20564139","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"feghhi26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["University of California Los Angeles"],"code":"https://doi.org/10.5281/zenodo.20564139","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2947","pdf":"https://www.isca-archive.org/interspeech_2026/feghhi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/feghhi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/feghhi26_interspeech/markdown.md"},{"id":"feng26_interspeech","title":"MMGenre: Benchmarking Singing Voice Synthesis across Multiple Musical Genres","authors":["Wenhao Feng","Yuxun Tang","Jiatong Shi","Qin Jin"],"year":2026,"doi":"10.21437/Interspeech.2026-137","isca_url":"https://www.isca-archive.org/interspeech_2026/feng26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/feng26_interspeech.pdf","session":"Speech Synthesis Evaluation 2","topics":["tts","evaluation","dataset"],"category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Renmin University of China","Carnegie Mellon University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://fengjin1117.github.io/mmgenre-web/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"feng26_interspeech","category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Renmin University of China","Carnegie Mellon University"],"code":"https://fengjin1117.github.io/mmgenre-web/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-137","pdf":"https://www.isca-archive.org/interspeech_2026/feng26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/feng26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/feng26_interspeech/markdown.md"},{"id":"ferreira26_interspeech","title":"CAL-MOS: Bridging Layers with Adapters for Robust MOS Prediction Across Speech Foundation Models","authors":["Alef Iury Ferreira","Pedro Botelho","Fernanda Silva","Daniel Casanova","Rafael Faustino","Frederico Oliveira","Arlindo Galvão Filho","Anderson da Silva Soares"],"year":2026,"doi":"10.21437/Interspeech.2026-2960","isca_url":"https://www.isca-archive.org/interspeech_2026/ferreira26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ferreira26_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["evaluation","self-supervised","paralinguistics"],"category":"resources-evaluation","labels":["self-supervised"],"institutions":["Advanced Knowledge Center in Immersive Technologies","Federal University of Goias","Federal University of Rio Grande do Norte","Federal University of Technology"],"funding":["Advanced Knowledge Center in Immersive Technologies","PPI IoT of the MCTI","EMBRAPII"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ferreira26_interspeech","category":"resources-evaluation","labels":["self-supervised"],"institutions":["Advanced Knowledge Center in Immersive Technologies","Federal University of Goias","Federal University of Rio Grande do Norte","Federal University of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2960","pdf":"https://www.isca-archive.org/interspeech_2026/ferreira26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ferreira26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ferreira26_interspeech/markdown.md"},{"id":"fiedler26_interspeech","title":"Contrastive Time-Proximity Pre-Training for Speech-Based Heart Failure Monitoring","authors":["Tobias Fiedler","Mariam Fouad","Marcus Hott","Leonhard Riehle","Felix Hohendanner","Bruce Johnson","Nicholas Cummins","Bert Arnrich"],"year":2026,"doi":"10.21437/Interspeech.2026-424","isca_url":"https://www.isca-archive.org/interspeech_2026/fiedler26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fiedler26_interspeech.pdf","session":"Pathological Speech Assessment 1","topics":["paralinguistics","self-supervised","health"],"category":"health-clinical","labels":["multilingual","self-supervised","robustness-noise"],"institutions":["Noah Labs","University of Potsdam","German Heart Center of the Charite","Mayo Clinic","King's College London"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fiedler26_interspeech","category":"health-clinical","labels":["multilingual","self-supervised","robustness-noise"],"institutions":["Noah Labs","University of Potsdam","German Heart Center of the Charite","Mayo Clinic","King's College London"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-424","pdf":"https://www.isca-archive.org/interspeech_2026/fiedler26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fiedler26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fiedler26_interspeech/markdown.md"},{"id":"filippakopoulos26_interspeech","title":"Segregate, Refine, Integrate: Decomposing Multimodal Fusion for Sentiment Analysis","authors":["Alexios Filippakopoulos","Elias Kallioras","Nikolaos Xiros","Efthymios Georgiou","Alexandros Potamianos"],"year":2026,"doi":"10.21437/Interspeech.2026-1299","isca_url":"https://www.isca-archive.org/interspeech_2026/filippakopoulos26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/filippakopoulos26_interspeech.pdf","session":"Emotion, Prosody, and Articulation","topics":["paralinguistics","emotion-recognition","speech-llm"],"category":"paralinguistics-emotion","institutions":["National Technical University of Athens","Athena Research Center","University of Bern","Archimedes AI","Synaptic Bloom PBC"],"funding":["National Recovery and Resilience Plan Greece 2.0","European Union","NextGenerationEU Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"filippakopoulos26_interspeech","category":"paralinguistics-emotion","institutions":["National Technical University of Athens","Athena Research Center","University of Bern","Archimedes AI","Synaptic Bloom PBC"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1299","pdf":"https://www.isca-archive.org/interspeech_2026/filippakopoulos26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/filippakopoulos26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/filippakopoulos26_interspeech/markdown.md"},{"id":"firc26_interspeech","title":"The Hidden Cost of Pairwise Verification in Synthetic Speech Source Tracing","authors":["Anton Firc","Zbyněk Lička","Vojtěch Staněk","Kamil Malinka"],"year":2026,"doi":"10.21437/Interspeech.2026-120","isca_url":"https://www.isca-archive.org/interspeech_2026/firc26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/firc26_interspeech.pdf","session":"Audio Watermarking and Source Verification","topics":["audio-deepfake","evaluation","self-supervised"],"category":"deepfake-security","institutions":["Brno University of Technology"],"funding":["Brno University of Technology","Ministry of Education, Youth and Sports of the Czech Republic","e-INFRA CZ"],"code":{"url":"https://github.com/Security-FIT/hidden-cost-pairwise-verification","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"firc26_interspeech","category":"deepfake-security","institutions":["Brno University of Technology"],"code":"https://github.com/Security-FIT/hidden-cost-pairwise-verification","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-120","pdf":"https://www.isca-archive.org/interspeech_2026/firc26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/firc26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/firc26_interspeech/markdown.md"},{"id":"firc26b_interspeech","title":"SpAArSIST: Sparsified AASIST for Efficient and Reliable Anti-Spoofing","authors":["Anton Firc","Vojtěch Staněk","Zbyněk Lička","Kamil Malinka","Martin Perešíni"],"year":2026,"doi":"10.21437/Interspeech.2026-2430","isca_url":"https://www.isca-archive.org/interspeech_2026/firc26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/firc26b_interspeech.pdf","session":"Speaker Verification and Anti-Spoofing","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["efficient-on-device","self-supervised","robustness-noise"],"institutions":["Brno University of Technology"],"funding":["Brno University of Technology","Ministry of Education, Youth and Sports of the Czech Republic","e-INFRA CZ"],"code":{"url":"https://github.com/Security-FIT/SpAArSIST","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"firc26b_interspeech","category":"deepfake-security","labels":["efficient-on-device","self-supervised","robustness-noise"],"institutions":["Brno University of Technology"],"code":"https://github.com/Security-FIT/SpAArSIST","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2430","pdf":"https://www.isca-archive.org/interspeech_2026/firc26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/firc26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/firc26b_interspeech/markdown.md"},{"id":"fletcher26_interspeech","title":"Oral stop realisation in three French Polynesian languages","authors":["Janet Fletcher","Adele Gregory"],"year":2026,"doi":"10.21437/Interspeech.2026-899","isca_url":"https://www.isca-archive.org/interspeech_2026/fletcher26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fletcher26_interspeech.pdf","session":"Pacific Voices: Speech Science and Technology for the Languages of the Pacific Ocean","topics":["phonetics","low-resource","multilingual"],"category":"phonetics-linguistics","labels":["low-resource","multilingual"],"institutions":["University of Melbourne"],"funding":["Faculty of Arts at the University of Melbourne","mission de recherche","Australian Research Council Centre of Excellence for the Dynamics of Language"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fletcher26_interspeech","category":"phonetics-linguistics","labels":["low-resource","multilingual"],"institutions":["University of Melbourne"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-899","pdf":"https://www.isca-archive.org/interspeech_2026/fletcher26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fletcher26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fletcher26_interspeech/markdown.md"},{"id":"fong26_interspeech","title":"Ada-Mic: Orientation-Adaptive and Robust Close-to-Mic Speech Detection on Smartphone Using Generalized Cross-Correlation Features","authors":["Stuart Fong","Sky Qiao","Xuecong Sun","Seyed Shahabeddin Nabavi","Huanzhang Zhu","Jinran Zhu","Siqi Zhang","Amirhossein Hajavi","Rasoul Mohammadi Nasiri","Shuangliang Sun","Longshuai Xiao","Yuanhao Yu","Irina Kezele"],"year":2026,"doi":"10.21437/Interspeech.2026-224","isca_url":"https://www.isca-archive.org/interspeech_2026/fong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fong26_interspeech.pdf","session":"Spatial Audio 1","topics":["keyword-spotting","speech-llm","self-supervised"],"category":"asr","labels":["robustness-noise"],"institutions":["Huawei Technologies"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fong26_interspeech","category":"asr","labels":["robustness-noise"],"institutions":["Huawei Technologies"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-224","pdf":"https://www.isca-archive.org/interspeech_2026/fong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fong26_interspeech/markdown.md"},{"id":"fong26b_interspeech","title":"Towards Enabling Multilingual Multitask SpeechLLMs in Data-Scarce Settings","authors":["Seraphina Fong","Marco Matassoni","Alessio Brutti"],"year":2026,"doi":"10.21437/Interspeech.2026-1229","isca_url":"https://www.isca-archive.org/interspeech_2026/fong26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fong26b_interspeech.pdf","session":"Audio & Speech Language Models: Evaluation, Representations, and Emerging Capabilities","topics":["asr","speech-translation","spoken-language-understanding"],"category":"speech-llm-dialogue","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Trento","Fondazione Bruno Kessler"],"funding":["European Union"],"code":{"url":"https://github.com/X-LANCE/SLAM-LLM","stars":1067,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fong26b_interspeech","category":"speech-llm-dialogue","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Trento","Fondazione Bruno Kessler"],"code":"https://github.com/X-LANCE/SLAM-LLM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1229","pdf":"https://www.isca-archive.org/interspeech_2026/fong26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fong26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fong26b_interspeech/markdown.md"},{"id":"foo26_interspeech","title":"All That Glitters Is Not Audio: Rethinking Text Priors and Audio Reliance in Audio-Language Evaluation","authors":["Leonardo Haw-Yang Foo","Chih-Kai Yang","Chen-An Li","Ke-Han Lu","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-913","isca_url":"https://www.isca-archive.org/interspeech_2026/foo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/foo26_interspeech.pdf","session":"Audio Language Models: Reasoning, Reliability, and Multimodal Understanding","topics":["evaluation","self-supervised","speech-llm"],"category":"resources-evaluation","institutions":["National Taiwan University"],"funding":["Ministry of Education","Taiwan Centers of Excellence in Artificial Intelligence","NTU Artificial Intelligence Center of Research Excellence"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"foo26_interspeech","category":"resources-evaluation","institutions":["National Taiwan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-913","pdf":"https://www.isca-archive.org/interspeech_2026/foo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/foo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/foo26_interspeech/markdown.md"},{"id":"fortier26_interspeech","title":"Where Do Backdoors Live? A Component-Level Analysis of Backdoor Propagation in Speech Language Models","authors":["Alexandrine Fortier","Thomas Thebaud","Jesús Villalba-Lopez","Najim Dehak","Patrick Cardinal","Peter West"],"year":2026,"doi":"10.21437/Interspeech.2026-2813","isca_url":"https://www.isca-archive.org/interspeech_2026/fortier26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fortier26_interspeech.pdf","session":"Speaker Diarization and Recognition","topics":["speech-llm","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["University of British Columbia","Ecole de technologie superieure","Johns Hopkins University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fortier26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["University of British Columbia","Ecole de technologie superieure","Johns Hopkins University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2813","pdf":"https://www.isca-archive.org/interspeech_2026/fortier26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fortier26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fortier26_interspeech/markdown.md"},{"id":"francis26_interspeech","title":"No-Shot Text-to-Speech: Limitations of Zero-Shot TTS and its Evaluation Methods in Representing Queer and Transgender Voices","authors":["Juliana Francis","Robin Netzorg","Joakim Gustafson","Éva Székely"],"year":2026,"doi":"10.21437/Interspeech.2026-709","isca_url":"https://www.isca-archive.org/interspeech_2026/francis26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/francis26_interspeech.pdf","session":"Queer and Trans Speech Science and Technology","topics":["tts","evaluation","speaker-verification"],"category":"tts","labels":["generative-model"],"institutions":["KTH Royal Institute of Technology","University of California, Berkeley"],"funding":["WASP"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"francis26_interspeech","category":"tts","labels":["generative-model"],"institutions":["KTH Royal Institute of Technology","University of California, Berkeley"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-709","pdf":"https://www.isca-archive.org/interspeech_2026/francis26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/francis26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/francis26_interspeech/markdown.md"},{"id":"frangiadaki26_interspeech","title":"Automatic Lyric Transcription for Greek Songs: Scaling and Task Composition Effects in Whisper Adaptation","authors":["Maria Frangiadaki","Dimitrios Damianos","Kosmas Kritsis","Vassilis Katsouros"],"year":2026,"doi":"10.21437/Interspeech.2026-1371","isca_url":"https://www.isca-archive.org/interspeech_2026/frangiadaki26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/frangiadaki26_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["asr","dataset","low-resource"],"category":"asr","labels":["low-resource","self-supervised","dataset-or-benchmark-release"],"institutions":["Athena Research Center"],"funding":["European High-Performance Computing Joint Undertaking","Pharos AI Factory","Greek Ministry of Digital Governance and Artificial Intelligence"],"code":{"url":"https://github.com/athena-ilsp/lyrics-transcription","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"frangiadaki26_interspeech","category":"asr","labels":["low-resource","self-supervised","dataset-or-benchmark-release"],"institutions":["Athena Research Center"],"code":"https://github.com/athena-ilsp/lyrics-transcription","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1371","pdf":"https://www.isca-archive.org/interspeech_2026/frangiadaki26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/frangiadaki26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/frangiadaki26_interspeech/markdown.md"},{"id":"franz26_interspeech","title":"From Echo to Accuracy: Robust Voice Quality Assessment Using Blind Unsupervised Diffusion-based Dereverberation","authors":["Sven Franz","Tanja Grewe","Bernd T. Meyer","Jörg Bitzer"],"year":2026,"doi":"10.21437/Interspeech.2026-2608","isca_url":"https://www.isca-archive.org/interspeech_2026/franz26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/franz26_interspeech.pdf","session":"Pathological Speech Assessment 2","topics":["paralinguistics","speech-enhancement","evaluation"],"category":"health-clinical","labels":["generative-model","robustness-noise"],"institutions":["Jade University of Applied Sciences","Carl von Ossietzky Universitat Oldenburg"],"funding":["Lower Saxony Ministry for Science and Culture","Volkswagen Foundation","Deutsche Forschungsgemeinschaft"],"code":{"url":"https://svenfranz.github.io/Room-Acoustics-and-Objective-Voice-Quality-in-SLT/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"franz26_interspeech","category":"health-clinical","labels":["generative-model","robustness-noise"],"institutions":["Jade University of Applied Sciences","Carl von Ossietzky Universitat Oldenburg"],"code":"https://svenfranz.github.io/Room-Acoustics-and-Objective-Voice-Quality-in-SLT/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2608","pdf":"https://www.isca-archive.org/interspeech_2026/franz26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/franz26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/franz26_interspeech/markdown.md"},{"id":"friedrichs26_interspeech","title":"Acoustic Pharyngometry as an Auditable Anchor for Cross-Speaker EMA Normalization","authors":["Daniel Friedrichs","Valeriia Vyshnevetska"],"year":2026,"doi":"10.21437/Interspeech.2026-1883","isca_url":"https://www.isca-archive.org/interspeech_2026/friedrichs26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/friedrichs26_interspeech.pdf","session":"Methods and Data for Vocal Tract Shape and Articulation Analysis","topics":["phonetics","speech-enhancement","self-supervised"],"category":"phonetics-linguistics","institutions":["Zurich Forensic Science Institute","University of Zurich"],"funding":["Swiss National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"friedrichs26_interspeech","category":"phonetics-linguistics","institutions":["Zurich Forensic Science Institute","University of Zurich"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1883","pdf":"https://www.isca-archive.org/interspeech_2026/friedrichs26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/friedrichs26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/friedrichs26_interspeech/markdown.md"},{"id":"fu26_interspeech","title":"Acoustic and Semantic Feature Fusion Mapping for Lyric Intelligibility Prediction","authors":["Yuxiang Fu","Guodong Lin","Da Shen","Wenlin Huang","Weili Jiang","Kai Gao","Boyu Zhao","Wei-Qiang Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1014","isca_url":"https://www.isca-archive.org/interspeech_2026/fu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fu26_interspeech.pdf","session":"Quality, Intelligibility and Evaluation of Speech and Codecs","topics":["paralinguistics","evaluation","self-supervised"],"category":"resources-evaluation","labels":["self-supervised"],"institutions":["Tsinghua University","Institute of Forensic Science, Ministry of Public Security"],"code":{"url":"https://cadenzachallenge.org/docs/clip1/baseline","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fu26_interspeech","category":"resources-evaluation","labels":["self-supervised"],"institutions":["Tsinghua University","Institute of Forensic Science, Ministry of Public Security"],"code":"https://cadenzachallenge.org/docs/clip1/baseline","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1014","pdf":"https://www.isca-archive.org/interspeech_2026/fu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fu26_interspeech/markdown.md"},{"id":"fu26b_interspeech","title":"Learnable Schrödinger Bridge and Activations for Efficient Diffusion-based Speech Enhancement","authors":["Yihui Fu","Wouter Tirry","Tim Fingscheidt"],"year":2026,"doi":"10.21437/Interspeech.2026-2465","isca_url":"https://www.isca-archive.org/interspeech_2026/fu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fu26b_interspeech.pdf","session":"Neural Speech Enhancement: Survey, Diffusion and Flow Matching","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","generative-model"],"institutions":["TU Braunschweig","Goodix Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fu26b_interspeech","category":"enhancement-separation","labels":["efficient-on-device","generative-model"],"institutions":["TU Braunschweig","Goodix Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2465","pdf":"https://www.isca-archive.org/interspeech_2026/fu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fu26b_interspeech/markdown.md"},{"id":"fujita26_interspeech","title":"Scalable Direction-Following TTS via Voice Impression-Guided Pseudo Triplet Construction","authors":["Kenichi Fujita","Yusuke Ijima"],"year":2026,"doi":"10.21437/Interspeech.2026-919","isca_url":"https://www.isca-archive.org/interspeech_2026/fujita26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fujita26_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["NTT"],"code":{"url":"https://ntt-hilab-gensp.github.io/IS2026pseudo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fujita26_interspeech","category":"tts","labels":["generative-model"],"institutions":["NTT"],"code":"https://ntt-hilab-gensp.github.io/IS2026pseudo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-919","pdf":"https://www.isca-archive.org/interspeech_2026/fujita26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fujita26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fujita26_interspeech/markdown.md"},{"id":"fukuda26_interspeech","title":"What Makes Us Hate Our Own Voice? Large-scale experiments on Playback–Imagery Gaps and Individual--Speech Feature Effects","authors":["Koki Fukuda","Shinnosuke Takamichi"],"year":2026,"doi":"10.21437/Interspeech.2026-973","isca_url":"https://www.isca-archive.org/interspeech_2026/fukuda26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fukuda26_interspeech.pdf","session":"Speech Production and Perception 1","topics":["paralinguistics","evaluation","speech-enhancement"],"category":"paralinguistics-emotion","institutions":["Keio University","University of Tokyo"],"funding":["JST Moonshot R&D","JST FOREST Program"],"code":{"url":"https://github.com/takamichi-lab/selfvoice-playback-imagery-gap","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fukuda26_interspeech","category":"paralinguistics-emotion","institutions":["Keio University","University of Tokyo"],"code":"https://github.com/takamichi-lab/selfvoice-playback-imagery-gap","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-973","pdf":"https://www.isca-archive.org/interspeech_2026/fukuda26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fukuda26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fukuda26_interspeech/markdown.md"},{"id":"fukuda26b_interspeech","title":"Evaluating Large Language Models Abilities for Addressee, Turn-change, and Next Speaker Prediction in Meetings","authors":["Ryo Fukuda","Takatomo Kano","Siddhant Arora","Marc Delcroix","Naohiro Tawara","Atsunori Ogawa","Yuya Chiba","Atsushi Ando","William Chen","Shinji Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-2923","isca_url":"https://www.isca-archive.org/interspeech_2026/fukuda26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/fukuda26b_interspeech.pdf","session":"LLMs and Conversational Interaction","topics":["speech-llm","spoken-language-understanding","evaluation"],"category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["NTT","Carnegie Mellon University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"fukuda26b_interspeech","category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["NTT","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2923","pdf":"https://www.isca-archive.org/interspeech_2026/fukuda26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/fukuda26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/fukuda26b_interspeech/markdown.md"},{"id":"futami26_interspeech","title":"Merging the Knowledge of LLMs for Automatic Speech Recognition","authors":["Hayato Futami","Tatsuya Kawahara"],"year":2026,"doi":"10.21437/Interspeech.2026-2561","isca_url":"https://www.isca-archive.org/interspeech_2026/futami26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/futami26_interspeech.pdf","session":"Efficient Inference for ASR and Speech LMs","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["multilingual","efficient-on-device","self-supervised"],"institutions":["Kyoto University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"futami26_interspeech","category":"asr","labels":["multilingual","efficient-on-device","self-supervised"],"institutions":["Kyoto University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2561","pdf":"https://www.isca-archive.org/interspeech_2026/futami26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/futami26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/futami26_interspeech/markdown.md"},{"id":"gachot26_interspeech","title":"A Generalized Formalism of Auto-Regressive Decoding for Speech Processing","authors":["Julia Gachot","Philipp Allgeuer","Marie S. Bauer","Stefan Wermter"],"year":2026,"doi":"10.21437/Interspeech.2026-2768","isca_url":"https://www.isca-archive.org/interspeech_2026/gachot26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gachot26_interspeech.pdf","session":"Search Methods and Inference Algorithms","topics":["self-supervised","evaluation","speech-llm"],"category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["University of Hamburg"],"funding":["Horizon Europe","German Research Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gachot26_interspeech","category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["University of Hamburg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2768","pdf":"https://www.isca-archive.org/interspeech_2026/gachot26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gachot26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gachot26_interspeech/markdown.md"},{"id":"galhotra26_interspeech","title":"Advancing Infant Distress Detection: Two- and Three-Way Classification in Real-World Audio Environments","authors":["Yashaswi Galhotra","Priyanka Khante","Anna Madden-Rusnak","Kaya de Barbaro"],"year":2026,"doi":"10.21437/Interspeech.2026-3234","isca_url":"https://www.isca-archive.org/interspeech_2026/galhotra26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/galhotra26_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["paralinguistics","self-supervised","dataset"],"category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["University of Texas at Austin"],"code":{"url":"https://github.com/dailyactivitylab/InfantDistressClassification.git","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"galhotra26_interspeech","category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["University of Texas at Austin"],"code":"https://github.com/dailyactivitylab/InfantDistressClassification.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3234","pdf":"https://www.isca-archive.org/interspeech_2026/galhotra26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/galhotra26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/galhotra26_interspeech/markdown.md"},{"id":"gan26_interspeech","title":"L2 Speakers Accommodate Differently to AI and Human Voices Across Phonetic Features","authors":["Nan Gan","Elisa Pellegrino"],"year":2026,"doi":"10.21437/Interspeech.2026-2605","isca_url":"https://www.isca-archive.org/interspeech_2026/gan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gan26_interspeech.pdf","session":"Cross-Linguistic and L2 Phonetic Studies","topics":["speech-enhancement","tts","low-resource"],"category":"phonetics-linguistics","institutions":["University of Zurich"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gan26_interspeech","category":"phonetics-linguistics","institutions":["University of Zurich"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2605","pdf":"https://www.isca-archive.org/interspeech_2026/gan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gan26_interspeech/markdown.md"},{"id":"gangwar26_interspeech","title":"HybridCodec: Fast Dual-Stream, Semantically Enhanced Neural Audio Codec","authors":["Arjun Gangwar","S Umesh"],"year":2026,"doi":"10.21437/Interspeech.2026-3393","isca_url":"https://www.isca-archive.org/interspeech_2026/gangwar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gangwar26_interspeech.pdf","session":"Streaming Speech Synthesis","topics":["speech-llm","self-supervised","speech-translation"],"category":"speech-coding","labels":["efficient-on-device","self-supervised","generative-model"],"institutions":["Indian Institute of Technology, Madras"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gangwar26_interspeech","category":"speech-coding","labels":["efficient-on-device","self-supervised","generative-model"],"institutions":["Indian Institute of Technology, Madras"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3393","pdf":"https://www.isca-archive.org/interspeech_2026/gangwar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gangwar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gangwar26_interspeech/markdown.md"},{"id":"gao26_interspeech","title":"NoiseLoRA-SV: Hierarchical Noise-Conditioned Adaptation with Embedding Distillation for Robust Speaker Verification","authors":["Dai Gao","Chen Jiang","Sizhe Liu","Peng Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-64","isca_url":"https://www.isca-archive.org/interspeech_2026/gao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gao26_interspeech.pdf","session":"Speaker Verification: Advances in Speaker Embeddings","topics":["speaker-verification","self-supervised","speech-enhancement"],"category":"speaker","labels":["efficient-on-device","robustness-noise"],"institutions":["Soochow University","Beijing Jiaotong University","Renmin University of China","Qilu University of Technology","Shandong Academy of Sciences"],"code":{"url":"https://github.com/peter112231/NoiseLoRA-SV","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gao26_interspeech","category":"speaker","labels":["efficient-on-device","robustness-noise"],"institutions":["Soochow University","Beijing Jiaotong University","Renmin University of China","Qilu University of Technology","Shandong Academy of Sciences"],"code":"https://github.com/peter112231/NoiseLoRA-SV","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-64","pdf":"https://www.isca-archive.org/interspeech_2026/gao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gao26_interspeech/markdown.md"},{"id":"gao26b_interspeech","title":"HistoMatch: Unified Transient-Steady Assessment for Noise-Robust Semi-Supervised Speaker Verification","authors":["Shenghan Gao","Xueshuai Zhang","Pengyuan Zhang","Yonghong Yan"],"year":2026,"doi":"10.21437/Interspeech.2026-244","isca_url":"https://www.isca-archive.org/interspeech_2026/gao26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gao26b_interspeech.pdf","session":"Speaker Recognition and Verification","topics":["speaker-verification","self-supervised","low-resource"],"category":"speaker","institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gao26b_interspeech","category":"speaker","institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-244","pdf":"https://www.isca-archive.org/interspeech_2026/gao26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gao26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gao26b_interspeech/markdown.md"},{"id":"gao26c_interspeech","title":"UD-ASD: A Unified Diffusion Model for Anomalous Sound Detection","authors":["Pengxiang Gao","Yu Qiu","Yanzhi Song"],"year":2026,"doi":"10.21437/Interspeech.2026-482","isca_url":"https://www.isca-archive.org/interspeech_2026/gao26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gao26c_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["paralinguistics","self-supervised","evaluation"],"category":"audio-understanding","labels":["generative-model"],"institutions":["University of Science and Technology of China"],"funding":["NSF of China","Major Project of Science and Technology Innovation Tackling Plan of Anhui Province","Fundamental Research Funds for the Central Universities"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gao26c_interspeech","category":"audio-understanding","labels":["generative-model"],"institutions":["University of Science and Technology of China"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-482","pdf":"https://www.isca-archive.org/interspeech_2026/gao26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gao26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gao26c_interspeech/markdown.md"},{"id":"gao26d_interspeech","title":"Uncovering Latent Depression Severity for Binary Depression Detection via Advantage-weighting Ranking","authors":["Manning Gao","Tingyi Liu","Leheng Zhang","Haifeng Hu","Yuncheng Jiang","Sijie Mai"],"year":2026,"doi":"10.21437/Interspeech.2026-535","isca_url":"https://www.isca-archive.org/interspeech_2026/gao26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gao26d_interspeech.pdf","session":"Multimodal and Non-Speech Healthcare Applications","topics":["paralinguistics","emotion-recognition","self-supervised"],"category":"health-clinical","institutions":["South China Normal University","Sun Yat-sen University"],"funding":["Guangdong Philosophy and Social Sciences Planning Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gao26d_interspeech","category":"health-clinical","institutions":["South China Normal University","Sun Yat-sen University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-535","pdf":"https://www.isca-archive.org/interspeech_2026/gao26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gao26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gao26d_interspeech/markdown.md"},{"id":"gao26e_interspeech","title":"PhASE-Flow: Phonetic-Conditioned Acoustic Flow Matching in SSL Representation Domain for Speech Enhancement","authors":["Jun Gao","Xiaobin Rong","Yu Sun","Dahan Wang","Jing Lu"],"year":2026,"doi":"10.21437/Interspeech.2026-916","isca_url":"https://www.isca-archive.org/interspeech_2026/gao26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gao26e_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","self-supervised","generative-model"],"category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Nanjing University","Horizon Robotics","Samsung Electronics"],"funding":["National Natural Science Foundation of China","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gao26e_interspeech","category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Nanjing University","Horizon Robotics","Samsung Electronics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-916","pdf":"https://www.isca-archive.org/interspeech_2026/gao26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gao26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gao26e_interspeech/markdown.md"},{"id":"garnaik26_interspeech","title":"When Machines Speak Like Local Peers: Improving Conversational Experiences with Accent-Adaptive Voice Agents","authors":["Shubhangi S. R. Garnaik","JiHyun Jeong","Jihoon Ryoo"],"year":2026,"doi":"10.21437/Interspeech.2026-2197","isca_url":"https://www.isca-archive.org/interspeech_2026/garnaik26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/garnaik26_interspeech.pdf","session":"LLMs and Conversational Interaction","topics":["accent-adaptive-tts","speech-recognition","spoken-language-understanding"],"category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["State University of New York, Korea","Stony Brook University"],"funding":["Ministry of Science and ICT","National Research Foundation of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"garnaik26_interspeech","category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["State University of New York, Korea","Stony Brook University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2197","pdf":"https://www.isca-archive.org/interspeech_2026/garnaik26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/garnaik26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/garnaik26_interspeech/markdown.md"},{"id":"gaughan26_interspeech","title":"Do speech representational spaces encode language family structures?","authors":["Emily Gaughan","Peter Bell"],"year":2026,"doi":"10.21437/Interspeech.2026-2418","isca_url":"https://www.isca-archive.org/interspeech_2026/gaughan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gaughan26_interspeech.pdf","session":"Speech and Language Representation","topics":["multilingual","self-supervised","evaluation"],"category":"speaker","labels":["multilingual","self-supervised"],"institutions":["University of Edinburgh"],"funding":["UK Research and Innovation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gaughan26_interspeech","category":"speaker","labels":["multilingual","self-supervised"],"institutions":["University of Edinburgh"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2418","pdf":"https://www.isca-archive.org/interspeech_2026/gaughan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gaughan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gaughan26_interspeech/markdown.md"},{"id":"geng26_interspeech","title":"Beyond Acoustic Sparsity and Linguistic Bias: A Prompt-Free Paradigm for Mispronunciation Detection and Diagnosis","authors":["Haopeng Geng","Longfei Yang","Xi Chen","Haitong Sun","Daisuke Saito","Nobuaki Minematsu"],"year":2026,"doi":"10.21437/Interspeech.2026-711","isca_url":"https://www.isca-archive.org/interspeech_2026/geng26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/geng26_interspeech.pdf","session":"Pronunciation Diversity","topics":["speech-recognition","low-resource","evaluation"],"category":"asr","institutions":["University of Tokyo"],"code":{"url":"https://github.com/Secondtonumb/IF-MDD","stars":13,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"geng26_interspeech","category":"asr","institutions":["University of Tokyo"],"code":"https://github.com/Secondtonumb/IF-MDD","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-711","pdf":"https://www.isca-archive.org/interspeech_2026/geng26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/geng26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/geng26_interspeech/markdown.md"},{"id":"geng26b_interspeech","title":"Stabilizing Instruction Supervision for Instruct-TTS via Controllable Diversification and Drift Filtering","authors":["Yizhong Geng","Kecan Mao","Qifei Li","Cong Wang","Yingming Gao","Ruimin Wang","Chunfeng Wang","Hao Li","Ya Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1227","isca_url":"https://www.isca-archive.org/interspeech_2026/geng26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/geng26b_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["Beijing University of Posts and Telecommunications","Li Auto"],"funding":["National Key R&D Program of China","National Natural Science Foundation of China","National Language Commission","National Social Science Fund of China"],"code":{"url":"https://piedpiperg.github.io/instruct-tts-stabilizer/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"geng26b_interspeech","category":"tts","labels":["generative-model"],"institutions":["Beijing University of Posts and Telecommunications","Li Auto"],"code":"https://piedpiperg.github.io/instruct-tts-stabilizer/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1227","pdf":"https://www.isca-archive.org/interspeech_2026/geng26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/geng26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/geng26b_interspeech/markdown.md"},{"id":"geng26c_interspeech","title":"Beyond Uncertainty and Diversity: Temporal-Spectral Guided Active Learning for Audio","authors":["Hui Geng","Tianjiao Wan","Yi Su","Qisheng Xu","Hengzhu Liu","Kele Xu"],"year":2026,"doi":"10.21437/Interspeech.2026-1719","isca_url":"https://www.isca-archive.org/interspeech_2026/geng26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/geng26c_interspeech.pdf","session":"Acoustic Event Detection 4","topics":["self-supervised","speech-enhancement","dataset"],"category":"audio-understanding","labels":["low-resource"],"institutions":["National University of Defense Technology"],"funding":["National Science and Technology Major Project","National University of Defense Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"geng26c_interspeech","category":"audio-understanding","labels":["low-resource"],"institutions":["National University of Defense Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1719","pdf":"https://www.isca-archive.org/interspeech_2026/geng26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/geng26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/geng26c_interspeech/markdown.md"},{"id":"gering26_interspeech","title":"A System-Agnostic Approach to Modelling Interaction Quality in Spoken Dialogue Systems","authors":["Paul Gering","Roger K. Moore"],"year":2026,"doi":"10.21437/Interspeech.2026-1152","isca_url":"https://www.isca-archive.org/interspeech_2026/gering26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gering26_interspeech.pdf","session":"Spoken Dialogue Systems","topics":["spoken-language-understanding","self-supervised","evaluation"],"category":"resources-evaluation","institutions":["University of Sheffield"],"funding":["UK Research and Innovation","UKRI AI Centre for Doctoral Training in Speech and Language Technologies (SLT) and their Applications"],"code":{"url":"https://github.com/prgering/interaction-quality-modelling-sds","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gering26_interspeech","category":"resources-evaluation","institutions":["University of Sheffield"],"code":"https://github.com/prgering/interaction-quality-modelling-sds","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1152","pdf":"https://www.isca-archive.org/interspeech_2026/gering26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gering26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gering26_interspeech/markdown.md"},{"id":"getman26_interspeech","title":"Data Filtering Trade-offs in Self-Supervised Speech Representation Learning: A Study on Unconstrained Broadcast Audio","authors":["Yaroslav Getman","Tamás Grósz","Tommi Lehtonen","Mikko Kurimo"],"year":2026,"doi":"10.21437/Interspeech.2026-50","isca_url":"https://www.isca-archive.org/interspeech_2026/getman26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/getman26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["self-supervised","asr","multilingual"],"category":"asr","labels":["self-supervised"],"institutions":["Aalto University","South East Technological University","Finnish Arts and Culture Agency"],"funding":["Business Finland","Foundation for Aalto University Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"getman26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Aalto University","South East Technological University","Finnish Arts and Culture Agency"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-50","pdf":"https://www.isca-archive.org/interspeech_2026/getman26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/getman26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/getman26_interspeech/markdown.md"},{"id":"getman26b_interspeech","title":"Do Learned Layer Weights Reflect Pretrained Information Structure in Self-Supervised Speech Models?","authors":["Yaroslav Getman","Tamás Grósz","Mikko Kurimo"],"year":2026,"doi":"10.21437/Interspeech.2026-566","isca_url":"https://www.isca-archive.org/interspeech_2026/getman26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/getman26b_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["self-supervised","asr","evaluation"],"category":"asr","labels":["self-supervised"],"institutions":["Aalto University","South East Technological University"],"funding":["Business Finland","Foundation for Aalto University Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"getman26b_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Aalto University","South East Technological University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-566","pdf":"https://www.isca-archive.org/interspeech_2026/getman26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/getman26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/getman26b_interspeech/markdown.md"},{"id":"ghosh26_interspeech","title":"Phy-VC: Physics-Informed Voice Conversion for Privacy-Preserving Pathological Speech","authors":["Suhita Ghosh","Yamini Sinha","Melanie Jouaiti","Tim Wansiedler","Kim Hakenberg","Julian Karcher","Ingo Siegert","Sebastian Stober"],"year":2026,"doi":"10.21437/Interspeech.2026-378","isca_url":"https://www.isca-archive.org/interspeech_2026/ghosh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ghosh26_interspeech.pdf","session":"Clinical and Inclusive Speech Technology","topics":["speech-anonymisation","voice-conversion","health"],"category":"deepfake-security","labels":["generative-model"],"institutions":["Otto-von-Guericke University","University of Birmingham"],"funding":["Federal Ministry of Education and Research of Germany"],"code":{"url":"https://github.com/suhitaghosh10/phy-vc.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ghosh26_interspeech","category":"deepfake-security","labels":["generative-model"],"institutions":["Otto-von-Guericke University","University of Birmingham"],"code":"https://github.com/suhitaghosh10/phy-vc.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-378","pdf":"https://www.isca-archive.org/interspeech_2026/ghosh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26_interspeech/markdown.md"},{"id":"ghosh26b_interspeech","title":"LipAdapter: Text-to-Video Alignment is All You Need for Lip-to-Speech","authors":["Souvik Ghosh","C. V. Jawahar","Vinay Namboodiri"],"year":2026,"doi":"10.21437/Interspeech.2026-518","isca_url":"https://www.isca-archive.org/interspeech_2026/ghosh26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ghosh26b_interspeech.pdf","session":"Speech Production and Perception 1","topics":["speech-synthesis","self-supervised","multilingual"],"category":"tts","labels":["generative-model"],"institutions":["International Institute of Information Technology Hyderabad","University of Bath"],"funding":["MeitY, Government of India"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ghosh26b_interspeech","category":"tts","labels":["generative-model"],"institutions":["International Institute of Information Technology Hyderabad","University of Bath"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-518","pdf":"https://www.isca-archive.org/interspeech_2026/ghosh26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26b_interspeech/markdown.md"},{"id":"ghosh26c_interspeech","title":"V-Align: Visual Forced Alignment via Phoneme to Video Optimal Path Traversal","authors":["Souvik Ghosh","C. V. Jawahar","Vinay Namboodiri"],"year":2026,"doi":"10.21437/Interspeech.2026-560","isca_url":"https://www.isca-archive.org/interspeech_2026/ghosh26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ghosh26c_interspeech.pdf","session":"Speech signal analysis","topics":["speech-translation","self-supervised","evaluation"],"category":"asr","institutions":["IIIT Hyderabad","University of Bath"],"funding":["MeitY, Government of India"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ghosh26c_interspeech","category":"asr","institutions":["IIIT Hyderabad","University of Bath"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-560","pdf":"https://www.isca-archive.org/interspeech_2026/ghosh26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26c_interspeech/markdown.md"},{"id":"ghosh26d_interspeech","title":"ASR-Synchronized Speaker-Role Diarization","authors":["Arindam Ghosh","Mark Fuhs","Bongjun Kim","Anurag Chowdhury","Monika Woszczyna"],"year":2026,"doi":"10.21437/Interspeech.2026-880","isca_url":"https://www.isca-archive.org/interspeech_2026/ghosh26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ghosh26d_interspeech.pdf","session":"Speaker Diarization 2","topics":["asr","speaker-diarization","self-supervised"],"category":"speaker","institutions":["Solventum Health Information Systems"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ghosh26d_interspeech","category":"speaker","institutions":["Solventum Health Information Systems"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-880","pdf":"https://www.isca-archive.org/interspeech_2026/ghosh26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26d_interspeech/markdown.md"},{"id":"ghosh26e_interspeech","title":"AnySimLite: A Lightweight Few-Shot Similarity Encoder for On-Device Speech-Adjacent Classification","authors":["Sourav Ghosh","Yash Bhatia","Keshav Goyal","Sahil Singh Bagri","Mohamed Akram Ulla Shariff","Saravana Balaji Shanmugam"],"year":2026,"doi":"10.21437/Interspeech.2026-1316","isca_url":"https://www.isca-archive.org/interspeech_2026/ghosh26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ghosh26e_interspeech.pdf","session":"Information Extraction and Retrieval / Survey Talk","topics":["speech-llm","low-resource","on-device"],"category":"applications-other","labels":["low-resource","efficient-on-device"],"institutions":["Samsung"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ghosh26e_interspeech","category":"applications-other","labels":["low-resource","efficient-on-device"],"institutions":["Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1316","pdf":"https://www.isca-archive.org/interspeech_2026/ghosh26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26e_interspeech/markdown.md"},{"id":"ghosh26f_interspeech","title":"MagpieTTS-LF: Inference-Time Long-Form Speech Generation Without Training on Long-Form data","authors":["Subhankar Ghosh","Jason Li","Paarth Neekhara","Shehzeen Hussain","Ryan Langman","Xuesong Yang","Roy Fejgin"],"year":2026,"doi":"10.21437/Interspeech.2026-1461","isca_url":"https://www.isca-archive.org/interspeech_2026/ghosh26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ghosh26f_interspeech.pdf","session":"Long-Form Speech Synthesis","topics":["tts","speech-llm","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["NVIDIA"],"code":{"url":"https://github.com/NVIDIA-NeMo/NeMo","stars":18521,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ghosh26f_interspeech","category":"tts","labels":["generative-model"],"institutions":["NVIDIA"],"code":"https://github.com/NVIDIA-NeMo/NeMo","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1461","pdf":"https://www.isca-archive.org/interspeech_2026/ghosh26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ghosh26f_interspeech/markdown.md"},{"id":"gichamba26_interspeech","title":"Probing Low Frame Rate Degradation in Neural Audio Codecs","authors":["Alex Gichamba","Moise Busogi"],"year":2026,"doi":"10.21437/Interspeech.2026-3493","isca_url":"https://www.isca-archive.org/interspeech_2026/gichamba26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gichamba26_interspeech.pdf","session":"Quality, Intelligibility and Evaluation of Speech and Codecs","topics":["speech-coding","self-supervised","evaluation"],"category":"speech-coding","labels":["generative-model"],"institutions":["Carnegie Mellon University"],"funding":["African Engineering and Technology Network","Mastercard Foundation","Advanced Cyberinfrastructure Coordination Ecosystem: Services & Support","U.S. National Science Foundation"],"code":{"url":"https://wakandaai.github.io/low-frame-rate-codec/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gichamba26_interspeech","category":"speech-coding","labels":["generative-model"],"institutions":["Carnegie Mellon University"],"code":"https://wakandaai.github.io/low-frame-rate-codec/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3493","pdf":"https://www.isca-archive.org/interspeech_2026/gichamba26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gichamba26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gichamba26_interspeech/markdown.md"},{"id":"giovannini26_interspeech","title":"Exploring the Effect of the Visual Channel in Vocal Expression of Affect in an Irish (Gaelic) Synthetic Voice","authors":["Anna Maria Giovannini","Zihan Wang","Ailbhe Ní Chasaide","Christer Gobl"],"year":2026,"doi":"10.21437/Interspeech.2026-1363","isca_url":"https://www.isca-archive.org/interspeech_2026/giovannini26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/giovannini26_interspeech.pdf","session":"Multimodal Emotion Recognition","topics":["tts","paralinguistics","evaluation"],"category":"paralinguistics-emotion","institutions":["Trinity College Dublin"],"funding":["Irish Research Council","Department of Rural and Community Development and the Gaeltacht","National Library"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"giovannini26_interspeech","category":"paralinguistics-emotion","institutions":["Trinity College Dublin"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1363","pdf":"https://www.isca-archive.org/interspeech_2026/giovannini26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/giovannini26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/giovannini26_interspeech/markdown.md"},{"id":"girish26_interspeech","title":"Towards Detecting Neural Audio Codec Synthesized Heart Sounds","authors":["Girish","Orchid Chetia Phukan","Mohd Mujtaba Akhtar","Bhavinkumar Vinodbhai Kuwar","Swarup Ranjan Behera","Arun Balaji Buduru"],"year":2026,"doi":"10.21437/Interspeech.2026-2116","isca_url":"https://www.isca-archive.org/interspeech_2026/girish26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/girish26_interspeech.pdf","session":"Spoofing and Deepfake Detection 3","topics":["paralinguistics","self-supervised","evaluation"],"category":"deepfake-security","labels":["dataset-or-benchmark-release"],"institutions":["UPES","National Tsing Hua University","VBSPU","Indraprastha Institute of Information Technology Delhi"],"code":{"url":"https://helixometry.github.io/SHAC/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"girish26_interspeech","category":"deepfake-security","labels":["dataset-or-benchmark-release"],"institutions":["UPES","National Tsing Hua University","VBSPU","Indraprastha Institute of Information Technology Delhi"],"code":"https://helixometry.github.io/SHAC/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2116","pdf":"https://www.isca-archive.org/interspeech_2026/girish26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/girish26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/girish26_interspeech/markdown.md"},{"id":"girish26b_interspeech","title":"Synergizing Zero-Shot Cross-Lingual Alzheimer Detection with Language-Invariant Multimodal Bi-Geometric Adversarial Learning","authors":["Girish","Mohd Mujtaba Akhtar","Farhan Sheth","Muskaan Singh","Juliana Gerard","Paula McClean","Kongfatt Wong-Lin"],"year":2026,"doi":"10.21437/Interspeech.2026-2756","isca_url":"https://www.isca-archive.org/interspeech_2026/girish26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/girish26b_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 2","topics":["speech-llm","self-supervised","multilingual"],"category":"health-clinical","labels":["low-resource","multilingual","self-supervised"],"institutions":["Ulster University","Manipal University"],"funding":["Alzheimer's Research UK","United States-Ireland-Northern Ireland R&D Partnership Programme","EPSRC"],"code":{"url":"https://github.com/Helixometry/ORBIT.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"girish26b_interspeech","category":"health-clinical","labels":["low-resource","multilingual","self-supervised"],"institutions":["Ulster University","Manipal University"],"code":"https://github.com/Helixometry/ORBIT.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2756","pdf":"https://www.isca-archive.org/interspeech_2026/girish26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/girish26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/girish26b_interspeech/markdown.md"},{"id":"glasser26_interspeech","title":"Bridging the Speech AI Accessibility Gap for Deaf and Hard of Hearing People","authors":["Abraham Glasser","Christian Vogler","Raja Kushalnagar"],"year":2026,"doi":"10.21437/Interspeech.2026-3001","isca_url":"https://www.isca-archive.org/interspeech_2026/glasser26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/glasser26_interspeech.pdf","session":"Assistive Technologies 2","topics":["asr","tts","speech-translation"],"category":"applications-other","institutions":["Gallaudet University"],"funding":["National Institute on Disability, Independent Living, and Rehabilitation Research","National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"glasser26_interspeech","category":"applications-other","institutions":["Gallaudet University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3001","pdf":"https://www.isca-archive.org/interspeech_2026/glasser26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/glasser26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/glasser26_interspeech/markdown.md"},{"id":"gnevsheva26_interspeech","title":"Modelling diphthong dynamics: A GAMM-based analysis of Australian English diphthongs","authors":["Ksenia Gnevsheva","Canaan Lan","Gerard Docherty","Catherine Travis"],"year":2026,"doi":"10.21437/Interspeech.2026-2861","isca_url":"https://www.isca-archive.org/interspeech_2026/gnevsheva26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gnevsheva26_interspeech.pdf","session":"Diphthongs and Monophthongs","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["Australian National University","University of Melbourne","Griffith University"],"funding":["Voices of Regional Australia","ARC Centre of Excellence for the Dynamics of Language"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gnevsheva26_interspeech","category":"phonetics-linguistics","institutions":["Australian National University","University of Melbourne","Griffith University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2861","pdf":"https://www.isca-archive.org/interspeech_2026/gnevsheva26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gnevsheva26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gnevsheva26_interspeech/markdown.md"},{"id":"goes26_interspeech","title":"PhonemeCVAE: Contrastive Latent Clustering with Class-Conditioned Priors for Controllable Phoneme Interpolation","authors":["Nina Goes","Lars Meyer","Katrin Neumann","Paula Andrea Pérez-Toro","Andreas M. Kist"],"year":2026,"doi":"10.21437/Interspeech.2026-2277","isca_url":"https://www.isca-archive.org/interspeech_2026/goes26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/goes26_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["Friedrich-Alexander-Universitaet Erlangen-Nuernberg","Max Planck Institute for Human Cognitive and Brain Sciences","University Hospital Muenster"],"funding":["German Research Foundation","Leopold-Klinge-Stiftung"],"code":{"url":"https://github.com/ankilab/PhonemeCVAE.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"goes26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Friedrich-Alexander-Universitaet Erlangen-Nuernberg","Max Planck Institute for Human Cognitive and Brain Sciences","University Hospital Muenster"],"code":"https://github.com/ankilab/PhonemeCVAE.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2277","pdf":"https://www.isca-archive.org/interspeech_2026/goes26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/goes26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/goes26_interspeech/markdown.md"},{"id":"golmakani26_interspeech","title":"Acoustic token admixture for joint speaker and content anonymization","authors":["Ali Golmakani","Seyed Ahmad Hosseini","Omar Manil Bendali","Emmanuel Vincent","Brij Mohan Lal Srivastava"],"year":2026,"doi":"10.21437/Interspeech.2026-1949","isca_url":"https://www.isca-archive.org/interspeech_2026/golmakani26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/golmakani26_interspeech.pdf","session":"Speaker Privacy and Anonymization","topics":["speech-privacy","voice-conversion","self-supervised"],"category":"deepfake-security","institutions":["Nijta","Universite de Lorraine","CNRS","Inria","LORIA"],"code":{"url":"https://github.com/Nijta/acoustic-token-admixture","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"golmakani26_interspeech","category":"deepfake-security","institutions":["Nijta","Universite de Lorraine","CNRS","Inria","LORIA"],"code":"https://github.com/Nijta/acoustic-token-admixture","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1949","pdf":"https://www.isca-archive.org/interspeech_2026/golmakani26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/golmakani26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/golmakani26_interspeech/markdown.md"},{"id":"gomez26_interspeech","title":"Unsupervised Speech in the Wild Challenge: Learning Robust Multilingual Representations","authors":["Rafael Mosquera Gómez","Juan Felipe Rodríguez","Daniel Galvez","Sarah Luger"],"year":2026,"doi":"10.21437/Interspeech.2026-3113","isca_url":"https://www.isca-archive.org/interspeech_2026/gomez26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gomez26_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["self-supervised","multilingual","asr"],"category":"resources-evaluation","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release","robustness-noise"],"institutions":["Factored AI","MLCommons","NVIDIA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gomez26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release","robustness-noise"],"institutions":["Factored AI","MLCommons","NVIDIA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3113","pdf":"https://www.isca-archive.org/interspeech_2026/gomez26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gomez26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gomez26_interspeech/markdown.md"},{"id":"gong26_interspeech","title":"NCPSZ: A Nonlinear Control Network for Miniature Loudspeakers in Personal Sound Zone Applications","authors":["Chen Gong","Lei Zhou","Chen Huang","Hongqing Liu","Liming Shi","Lu Gan"],"year":2026,"doi":"10.21437/Interspeech.2026-182","isca_url":"https://www.isca-archive.org/interspeech_2026/gong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gong26_interspeech.pdf","session":"Active Noise and Echo Control, Sound Zones and Packet-Loss Concealment","topics":["speech-enhancement","on-device","self-supervised"],"category":"enhancement-separation","institutions":["Chongqing University of Posts and Telecommunications","Brunel University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gong26_interspeech","category":"enhancement-separation","institutions":["Chongqing University of Posts and Telecommunications","Brunel University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-182","pdf":"https://www.isca-archive.org/interspeech_2026/gong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gong26_interspeech/markdown.md"},{"id":"gong26b_interspeech","title":"Bagpiper-Edit: Zero-Shot Open-Ended Audio Editing via Rich-Caption","authors":["Xun Gong","Jinchuan Tian","Haoran Wang","William Chen","Shinji Watanabe","Yanmin Qian"],"year":2026,"doi":"10.21437/Interspeech.2026-631","isca_url":"https://www.isca-archive.org/interspeech_2026/gong26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gong26b_interspeech.pdf","session":"Voice Editing","topics":["speech-editing","self-supervised","speech-llm"],"category":"speech-llm-dialogue","labels":["self-supervised","generative-model"],"institutions":["Shanghai Jiao Tong University","Carnegie Mellon University"],"funding":["China NSFC","SJTU Med-X (Medicine & Engineering) Translational Research Grant"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gong26b_interspeech","category":"speech-llm-dialogue","labels":["self-supervised","generative-model"],"institutions":["Shanghai Jiao Tong University","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-631","pdf":"https://www.isca-archive.org/interspeech_2026/gong26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gong26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gong26b_interspeech/markdown.md"},{"id":"gonzalez26_interspeech","title":"Absorbing Discrete Diffusion for Speech Enhancement","authors":["Philippe Gonzalez"],"year":2026,"doi":"10.21437/Interspeech.2026-659","isca_url":"https://www.isca-archive.org/interspeech_2026/gonzalez26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gonzalez26_interspeech.pdf","session":"Neural Speech Enhancement: Survey, Diffusion and Flow Matching","topics":["speech-enhancement","self-supervised","neural-coding"],"category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["Technical University of Denmark"],"code":{"url":"https://philgzl.com/addse-demo","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gonzalez26_interspeech","category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["Technical University of Denmark"],"code":"https://philgzl.com/addse-demo","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-659","pdf":"https://www.isca-archive.org/interspeech_2026/gonzalez26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gonzalez26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gonzalez26_interspeech/markdown.md"},{"id":"gonzalez26b_interspeech","title":"Probing Linguistic Information in Speech Embeddings: A Diagnostic Analysis across Acoustic and Structural Domains","authors":["Simon Gonzalez","Tao Hoang","Hayden Ooi","Chloe Dean","Bradley Donnelly","Myung Kim","Latchman Singh","Jason Littlefield","Tim Cawley","Jennifer Biggs"],"year":2026,"doi":"10.21437/Interspeech.2026-905","isca_url":"https://www.isca-archive.org/interspeech_2026/gonzalez26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gonzalez26b_interspeech.pdf","session":"New Architecture and Analyses for ASR and Speech LMs","topics":["self-supervised","phonetics","multilingual"],"category":"phonetics-linguistics","labels":["self-supervised"],"institutions":["Defence Science and Technology Group"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gonzalez26b_interspeech","category":"phonetics-linguistics","labels":["self-supervised"],"institutions":["Defence Science and Technology Group"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-905","pdf":"https://www.isca-archive.org/interspeech_2026/gonzalez26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gonzalez26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gonzalez26b_interspeech/markdown.md"},{"id":"gonzalez26c_interspeech","title":"How Linguistic Dimension Interactions Shape Meaning Preservation in Multilingual ASR","authors":["Simon Gonzalez","Tao Hoang","Bradley Donnelly","Jason Littlefield","Myung Kim","Chloe Dean","Jennifer Biggs","Hayden Ooi","Latchman Singh","Tim Cawley"],"year":2026,"doi":"10.21437/Interspeech.2026-920","isca_url":"https://www.isca-archive.org/interspeech_2026/gonzalez26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gonzalez26c_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 2","topics":["asr","multilingual","evaluation"],"category":"asr","labels":["multilingual"],"institutions":["Defence Science and Technology Group"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gonzalez26c_interspeech","category":"asr","labels":["multilingual"],"institutions":["Defence Science and Technology Group"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-920","pdf":"https://www.isca-archive.org/interspeech_2026/gonzalez26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gonzalez26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gonzalez26c_interspeech/markdown.md"},{"id":"gonzalez26d_interspeech","title":"Minimum Token Thresholds and Stabilisation for Reliable Automatic Vowel Alignment: Empirical Study on TIMIT Vowels and MFA","authors":["Simon Gonzalez","Jason Littlefield","Tao Hoang","Chloe Dean","Hayden Ooi","Myung Kim","Bradley Donnelly","Latchman Singh","Jennifer Biggs","Tim Cawley"],"year":2026,"doi":"10.21437/Interspeech.2026-934","isca_url":"https://www.isca-archive.org/interspeech_2026/gonzalez26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gonzalez26d_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["phonetics","evaluation","dataset"],"category":"phonetics-linguistics","institutions":["Defence Science and Technology Group"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gonzalez26d_interspeech","category":"phonetics-linguistics","institutions":["Defence Science and Technology Group"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-934","pdf":"https://www.isca-archive.org/interspeech_2026/gonzalez26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gonzalez26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gonzalez26d_interspeech/markdown.md"},{"id":"gonzalezmachorro26_interspeech","title":"Towards Speech Impairment Prediction in German-Speaking Individuals with Amyotrophic Lateral Sclerosis","authors":["Monica Gonzalez-Machorro","Ricarda von Heynitz","Justin Hanslmeier","Finja Grimm","Alexandra-Iulia Deac","Anne Gründel","Isabell Cordts","Bjoern Schuller"],"year":2026,"doi":"10.21437/Interspeech.2026-1052","isca_url":"https://www.isca-archive.org/interspeech_2026/gonzalezmachorro26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gonzalezmachorro26_interspeech.pdf","session":"Clinically Useful Speech Representations 1","topics":["health","paralinguistics","evaluation"],"category":"health-clinical","institutions":["Technical University of Munich","audEERING","Munich Center for Machine Learning","Imperial College London"],"code":{"url":"https://github.com/monicagoma98/IS_AIMnd_2026","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gonzalezmachorro26_interspeech","category":"health-clinical","institutions":["Technical University of Munich","audEERING","Munich Center for Machine Learning","Imperial College London"],"code":"https://github.com/monicagoma98/IS_AIMnd_2026","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1052","pdf":"https://www.isca-archive.org/interspeech_2026/gonzalezmachorro26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gonzalezmachorro26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gonzalezmachorro26_interspeech/markdown.md"},{"id":"gopal26_interspeech","title":"Language-Aware Distillation for Multilingual Instruction-Following Speech LLMs with ASR-Only Supervision","authors":["Shreyas Gopal","Donghang Wu","Ashutosh Anshul","Yeo Yue Heng","Yizhou Peng","Haoyang Li","Hexin Liu","Eng Siong Chng"],"year":2026,"doi":"10.21437/Interspeech.2026-2446","isca_url":"https://www.isca-archive.org/interspeech_2026/gopal26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gopal26_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 2","topics":["speech-llm","multilingual","self-supervised"],"category":"speech-llm-dialogue","labels":["multilingual","self-supervised"],"institutions":["Nanyang Technological University","AI Singapore","National University of Singapore","Institute for Infocomm Research","Agency for Science, Technology and Research"],"funding":["WeBank-NTU Joint Research Institute on FinTech"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gopal26_interspeech","category":"speech-llm-dialogue","labels":["multilingual","self-supervised"],"institutions":["Nanyang Technological University","AI Singapore","National University of Singapore","Institute for Infocomm Research","Agency for Science, Technology and Research"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2446","pdf":"https://www.isca-archive.org/interspeech_2026/gopal26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gopal26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gopal26_interspeech/markdown.md"},{"id":"gorman26_interspeech","title":"Shifting Relational Paradigms for Affective Computing: Affective Resonance, Vitality Affects, and Vocal Interaction Fields","authors":["Cy Gorman","Yihang Yao"],"year":2026,"doi":"10.21437/Interspeech.2026-2829","isca_url":"https://www.isca-archive.org/interspeech_2026/gorman26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gorman26_interspeech.pdf","session":"Paralinguistics","topics":["self-supervised","paralinguistics","speech-llm"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Nurobodi"],"funding":["CSIRO","Innovate to Grow","ON Prime"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gorman26_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Nurobodi"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2829","pdf":"https://www.isca-archive.org/interspeech_2026/gorman26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gorman26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gorman26_interspeech/markdown.md"},{"id":"goto26_interspeech","title":"Online Predictive Coding for Dual-Mode Self-Supervised Speech Models","authors":["Keita Goto","Takashi Maekaku","Jin Sakuma","Jinchuan Tian","Yusuke Shinohara","Shinji Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-1997","isca_url":"https://www.isca-archive.org/interspeech_2026/goto26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/goto26_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["self-supervised","streaming-real-time"],"institutions":["LY Corporation","Carnegie Mellon University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"goto26_interspeech","category":"asr","labels":["self-supervised","streaming-real-time"],"institutions":["LY Corporation","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1997","pdf":"https://www.isca-archive.org/interspeech_2026/goto26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/goto26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/goto26_interspeech/markdown.md"},{"id":"gotz26_interspeech","title":"Improving Multichannel Speech Enhancement through Accurate Room-Acoustic Simulations","authors":["Georg Götz","Alessia Milo","Steinar Guðjónsson","Daniel Gert Nielsen","Jesper Pedersen","Finnur Pind"],"year":2026,"doi":"10.21437/Interspeech.2026-2512","isca_url":"https://www.isca-archive.org/interspeech_2026/gotz26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gotz26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Treble Technologies"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gotz26_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Treble Technologies"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2512","pdf":"https://www.isca-archive.org/interspeech_2026/gotz26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gotz26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gotz26_interspeech/markdown.md"},{"id":"gotz26b_interspeech","title":"Scalable Audio Scene Generation with the Treble SDK","authors":["Georg Götz","Konstantinos Gkanos","Steinar Guðjónsson","Daniel Gert Nielsen","Jesper Pedersen","Finnur Pind"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/gotz26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gotz26b_interspeech.pdf","session":"Speech Synthesis, Voice Conversion and Audio Generation","topics":["dataset","speech-enhancement","source-separation"],"category":"resources-evaluation","institutions":["Treble Technologies"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gotz26b_interspeech","category":"resources-evaluation","institutions":["Treble Technologies"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/gotz26b_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/gotz26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gotz26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gotz26b_interspeech/markdown.md"},{"id":"gourav26_interspeech","title":"DsNA(Digital sigNature for Audios): A Unique Method to Fingerprint Audio Files Generated by Text to Speech","authors":["Vishal Gourav","Phanindra Mankale"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/gourav26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gourav26_interspeech.pdf","session":"Speech Synthesis, Voice Conversion and Audio Generation","topics":["tts","evaluation","audio-deepfake"],"category":"deepfake-security","institutions":["Oracle"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gourav26_interspeech","category":"deepfake-security","institutions":["Oracle"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/gourav26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/gourav26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gourav26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gourav26_interspeech/markdown.md"},{"id":"granda26_interspeech","title":"Genealogical Priors in Self-Supervised Learning: Improving Speech Technology for Low-Resource Languages","authors":["Elizabeth Granda","Edson A. Luna","Andres F. Gonzalez"],"year":2026,"doi":"10.21437/Interspeech.2026-2886","isca_url":"https://www.isca-archive.org/interspeech_2026/granda26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/granda26_interspeech.pdf","session":"Low-Resource & Endangered Language Speech Processing","topics":["low-resource","multilingual","self-supervised"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Factored AI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"granda26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Factored AI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2886","pdf":"https://www.isca-archive.org/interspeech_2026/granda26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/granda26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/granda26_interspeech/markdown.md"},{"id":"grinberg26_interspeech","title":"ALARM: Audio–Language Alignment for Reasoning Models","authors":["Petr Grinberg","Hassan Shahmohammadi"],"year":2026,"doi":"10.21437/Interspeech.2026-759","isca_url":"https://www.isca-archive.org/interspeech_2026/grinberg26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/grinberg26_interspeech.pdf","session":"Speech Representations and Alignment","topics":["speech-llm","self-supervised","multilingual"],"category":"speech-llm-dialogue","labels":["multilingual","self-supervised"],"institutions":["EPFL","Sony"],"code":{"url":"https://github.com/Blinorot/ALARM","stars":16,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"grinberg26_interspeech","category":"speech-llm-dialogue","labels":["multilingual","self-supervised"],"institutions":["EPFL","Sony"],"code":"https://github.com/Blinorot/ALARM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-759","pdf":"https://www.isca-archive.org/interspeech_2026/grinberg26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/grinberg26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/grinberg26_interspeech/markdown.md"},{"id":"grundhuber26_interspeech","title":"Beyond Cross-Reconstruction: Probing-Based Disentanglement Evaluation for Acoustic Teleportation Codecs","authors":["Philipp Grundhuber","Emanuël A. P. Habets"],"year":2026,"doi":"10.21437/Interspeech.2026-2406","isca_url":"https://www.isca-archive.org/interspeech_2026/grundhuber26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/grundhuber26_interspeech.pdf","session":"Quality, Intelligibility and Evaluation of Speech and Codecs","topics":["self-supervised","speech-enhancement","evaluation"],"category":"speech-coding","institutions":["Fraunhofer Institute for Integrated Circuits","International Audio Laboratories Erlangen","Friedrich-Alexander-Universität Erlangen-Nürnberg"],"funding":["German Research Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"grundhuber26_interspeech","category":"speech-coding","institutions":["Fraunhofer Institute for Integrated Circuits","International Audio Laboratories Erlangen","Friedrich-Alexander-Universität Erlangen-Nürnberg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2406","pdf":"https://www.isca-archive.org/interspeech_2026/grundhuber26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/grundhuber26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/grundhuber26_interspeech/markdown.md"},{"id":"gu26_interspeech","title":"Mutual Cancellation between Masking Effects Benefits Speech Intelligibility","authors":["Yixin Gu","Yan Tang"],"year":2026,"doi":"10.21437/Interspeech.2026-1290","isca_url":"https://www.isca-archive.org/interspeech_2026/gu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gu26_interspeech.pdf","session":"Model of Speech Perception","topics":["speech-perception","speech-enhancement","evaluation"],"category":"phonetics-linguistics","labels":["multilingual","robustness-noise"],"institutions":["University of Illinois Urbana-Champaign","Beckman Institute for Advanced Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gu26_interspeech","category":"phonetics-linguistics","labels":["multilingual","robustness-noise"],"institutions":["University of Illinois Urbana-Champaign","Beckman Institute for Advanced Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1290","pdf":"https://www.isca-archive.org/interspeech_2026/gu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gu26_interspeech/markdown.md"},{"id":"gu26b_interspeech","title":"ProWhistress: An Enhanced Dual-Stream Transcription Architecture for Prosody-Aware Sentence Stress Detection","authors":["Hujian Gu","Li Tao","Fei Jiang","Ying Wang","Jingwei Qu","Zhaofang Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-1303","isca_url":"https://www.isca-archive.org/interspeech_2026/gu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gu26b_interspeech.pdf","session":"Prosody, Pronunciation and Specialized Speech Processing","topics":["paralinguistics","self-supervised","asr"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Southwest University","Chongqing Academy of Science and Technology"],"funding":["Chongqing Academy of Science and Technology Basic Research Funding"],"code":{"url":"https://github.com/Guhujian/ProWhistress.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gu26b_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Southwest University","Chongqing Academy of Science and Technology"],"code":"https://github.com/Guhujian/ProWhistress.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1303","pdf":"https://www.isca-archive.org/interspeech_2026/gu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gu26b_interspeech/markdown.md"},{"id":"guan26_interspeech","title":"Effects of body position on vowel formants in New Zealand English","authors":["Qing Guan","C. I. Watson","C. T. Justine Hui"],"year":2026,"doi":"10.21437/Interspeech.2026-1545","isca_url":"https://www.isca-archive.org/interspeech_2026/guan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/guan26_interspeech.pdf","session":"Speaker-Specific, Forensic and Segmental Characteristics of Typical and Atypical Speech","topics":["phonetics","prosody","evaluation"],"category":"phonetics-linguistics","institutions":["University of Auckland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"guan26_interspeech","category":"phonetics-linguistics","institutions":["University of Auckland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1545","pdf":"https://www.isca-archive.org/interspeech_2026/guan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/guan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/guan26_interspeech/markdown.md"},{"id":"guan26b_interspeech","title":"UniVoice: Unifying Autoregressive ASR and Flow-Matching based TTS with Large Language Models","authors":["Wenhao Guan","Zhikang Niu","Ziyue Jiang","Kaidi Wang","Peijie Chen","Qingyang Hong","Xie Chen","Lin Li"],"year":2026,"doi":"10.21437/Interspeech.2026-2194","isca_url":"https://www.isca-archive.org/interspeech_2026/guan26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/guan26b_interspeech.pdf","session":"Speech Production and Perception 2","topics":["asr","tts","self-supervised"],"category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["Xiamen University","Shanghai Innovation Institute","Shanghai Jiao Tong University","Zhejiang University"],"funding":["National Natural Science Foundation of China","Innovation of Policing Science and Technology, Fujian province"],"code":{"url":"https://github.com/gwh22/UniVoice","stars":122,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"guan26b_interspeech","category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["Xiamen University","Shanghai Innovation Institute","Shanghai Jiao Tong University","Zhejiang University"],"code":"https://github.com/gwh22/UniVoice","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2194","pdf":"https://www.isca-archive.org/interspeech_2026/guan26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/guan26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/guan26b_interspeech/markdown.md"},{"id":"guo26_interspeech","title":"GLAD: Global-Local Aware Dynamic Mixture-of-Experts for Multi-Talker ASR","authors":["Yujie Guo","Jiaming Zhou","Yuhang Jia","Shiwan Zhao","Yong Qin"],"year":2026,"doi":"10.21437/Interspeech.2026-1022","isca_url":"https://www.isca-archive.org/interspeech_2026/guo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/guo26_interspeech.pdf","session":"Multi-Talker ASR & Speaker Diarization","topics":["asr","speech-llm","self-supervised"],"category":"asr","institutions":["Nankai University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://github.com/NKU-HLT/GLAD","stars":7,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"guo26_interspeech","category":"asr","institutions":["Nankai University"],"code":"https://github.com/NKU-HLT/GLAD","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1022","pdf":"https://www.isca-archive.org/interspeech_2026/guo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/guo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/guo26_interspeech/markdown.md"},{"id":"guo26b_interspeech","title":"COALA: Robust Contextualized Speech-augmented Language Modeling for ASR via Contrastive Regularizer and Biasing Score Estimation","authors":["Jhih-Rong Guo","Bi-Cheng Yan","Tien-Hong Lo","Berlin Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1097","isca_url":"https://www.isca-archive.org/interspeech_2026/guo26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/guo26b_interspeech.pdf","session":"Robust ASR: Uncertainty and Confidence","topics":["asr","self-supervised","speech-llm"],"category":"asr","institutions":["National Taiwan Normal University"],"funding":["Realtek Semiconductor Corporation"],"code":{"url":"https://github.com/Guo0911/COALA","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"guo26b_interspeech","category":"asr","institutions":["National Taiwan Normal University"],"code":"https://github.com/Guo0911/COALA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1097","pdf":"https://www.isca-archive.org/interspeech_2026/guo26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/guo26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/guo26b_interspeech/markdown.md"},{"id":"guo26c_interspeech","title":"Adaptive Federated Fine-Tuning of Self-Supervised Speech Representations","authors":["Xin Guo","Chunrui Zhao","Hong Jia","Ting Dang","Gongping Huang","Xianrui Zheng","Yan Gao"],"year":2026,"doi":"10.21437/Interspeech.2026-2122","isca_url":"https://www.isca-archive.org/interspeech_2026/guo26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/guo26c_interspeech.pdf","session":"Self-supervised Speech Representation Learning","topics":["self-supervised","speech-llm","low-resource"],"category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised"],"institutions":["Wuhan University","University of Auckland","University of Melbourne","University of Cambridge","Flower Labs"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"guo26c_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised"],"institutions":["Wuhan University","University of Auckland","University of Melbourne","University of Cambridge","Flower Labs"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2122","pdf":"https://www.isca-archive.org/interspeech_2026/guo26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/guo26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/guo26c_interspeech/markdown.md"},{"id":"guo26d_interspeech","title":"Lightweight Convolutional Front-ends for Real-time Framewise Phoneme Recognition in Cochlear Implants","authors":["Yuchu Guo","Leslie M. Collins","Boyla O. Mainsah"],"year":2026,"doi":"10.21437/Interspeech.2026-2689","isca_url":"https://www.isca-archive.org/interspeech_2026/guo26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/guo26d_interspeech.pdf","session":"Assistive Technologies 1","topics":["asr","speech-enhancement","on-device"],"category":"health-clinical","labels":["efficient-on-device","streaming-real-time"],"institutions":["Duke Kunshan University","Duke University"],"funding":["Duke Summer Research Program for Duke Kunshan University Undergraduates"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"guo26d_interspeech","category":"health-clinical","labels":["efficient-on-device","streaming-real-time"],"institutions":["Duke Kunshan University","Duke University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2689","pdf":"https://www.isca-archive.org/interspeech_2026/guo26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/guo26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/guo26d_interspeech/markdown.md"},{"id":"guo26e_interspeech","title":"DEBATE: A Dataset for Disentangling Textual Ambiguity in Mandarin Through Speech","authors":["Haotian Guo","Jing Han","Yongfeng Tu","Shihao Gao","Weihao Gan","Zixing Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-3146","isca_url":"https://www.isca-archive.org/interspeech_2026/guo26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/guo26e_interspeech.pdf","session":"Spoken Language Understanding","topics":["speech-llm","evaluation","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Hunan University","Malanshan Audio & Video Laboratory","Yuelushan Center for Industrial Innovation"],"funding":["Malanshan Audio & Video Laboratory","National Natural Science Foundation of China","National Science and Technology Major Project of China","Science and Technology Innovation Program of Hunan Province","Guangdong Basic and Applied Basic Research Foundation","Shenzhen Natural Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"guo26e_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Hunan University","Malanshan Audio & Video Laboratory","Yuelushan Center for Industrial Innovation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3146","pdf":"https://www.isca-archive.org/interspeech_2026/guo26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/guo26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/guo26e_interspeech/markdown.md"},{"id":"gupta26_interspeech","title":"Closing the Modality Gap via Simplex-Constrained Representations","authors":["Shubham Gupta","Siva Reddy","Perouz Taslakian","Valentina Zantedeschi","Cem Subakan"],"year":2026,"doi":"10.21437/Interspeech.2026-2849","isca_url":"https://www.isca-archive.org/interspeech_2026/gupta26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/gupta26_interspeech.pdf","session":"Model of Speech Perception","topics":["self-supervised","multilingual","speech-llm"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Mila – Québec AI Institute","ServiceNow","McGill University","Université Laval"],"funding":["Natural Sciences and Engineering Research Council of Canada","Digital Research Alliance of Canada"],"code":{"url":"https://github.com/ServiceNow/retreever","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"gupta26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Mila – Québec AI Institute","ServiceNow","McGill University","Université Laval"],"code":"https://github.com/ServiceNow/retreever","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2849","pdf":"https://www.isca-archive.org/interspeech_2026/gupta26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/gupta26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/gupta26_interspeech/markdown.md"},{"id":"hacker26_interspeech","title":"Common Cold Corpus: Health-Aware Robustness Study of Modern Speaker Embeddings Under Physiological Domain Shift","authors":["Anabell Hacker","Ingo Siegert"],"year":2026,"doi":"10.21437/Interspeech.2026-1188","isca_url":"https://www.isca-archive.org/interspeech_2026/hacker26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hacker26_interspeech.pdf","session":"Pathological Speech Assessment 1","topics":["speaker-verification","paralinguistics","health"],"category":"speaker","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["Otto von Guericke University Magdeburg","University Hospital Magdeburg"],"funding":["BMFTR","European Union"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hacker26_interspeech","category":"speaker","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["Otto von Guericke University Magdeburg","University Hospital Magdeburg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1188","pdf":"https://www.isca-archive.org/interspeech_2026/hacker26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hacker26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hacker26_interspeech/markdown.md"},{"id":"haghbin26_interspeech","title":"From Black-Box to Clinical Insight: A Multi-Stage Explainable Framework for Speech-Based Cognitive Impairment Detection","authors":["Yasaman Haghbin","Sina Rashidi","Ali Zolnour","Fatemeh Taherinezhad","Ali Fartoot","Hossein Azadmaleki","James M. Noble","Maryam Dadkhah","Maryam Zolnoori"],"year":2026,"doi":"10.21437/Interspeech.2026-1252","isca_url":"https://www.isca-archive.org/interspeech_2026/haghbin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/haghbin26_interspeech.pdf","session":"Clinically Useful Speech Representations 1","topics":["speech-llm","paralinguistics","health"],"category":"health-clinical","institutions":["Columbia University","Chalmers University of Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"haghbin26_interspeech","category":"health-clinical","institutions":["Columbia University","Chalmers University of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1252","pdf":"https://www.isca-archive.org/interspeech_2026/haghbin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/haghbin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/haghbin26_interspeech/markdown.md"},{"id":"haghbin26b_interspeech","title":"Natural Speech Encodes Early Markers of Cognitive Decline: Evidence from Clinical Conversations","authors":["Yasaman Haghbin","Sina Rashidi","Ali Zolnour","Margaret McDonald","Maryam Zolnoori"],"year":2026,"doi":"10.21437/Interspeech.2026-1860","isca_url":"https://www.isca-archive.org/interspeech_2026/haghbin26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/haghbin26b_interspeech.pdf","session":"Clinically Useful Speech Representations 1","topics":["speech-llm","paralinguistics","health"],"category":"health-clinical","institutions":["Columbia University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"haghbin26b_interspeech","category":"health-clinical","institutions":["Columbia University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1860","pdf":"https://www.isca-archive.org/interspeech_2026/haghbin26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/haghbin26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/haghbin26b_interspeech/markdown.md"},{"id":"halmai26_interspeech","title":"How Language-Independent Are Emotional Attributes? A Study on Training Data Scaling and Cross-Lingual Generalization","authors":["Dániel Halmai","Gábor Gosztolya"],"year":2026,"doi":"10.21437/Interspeech.2026-2143","isca_url":"https://www.isca-archive.org/interspeech_2026/halmai26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/halmai26_interspeech.pdf","session":"Speech Emotion Recognition and Representation 3","topics":["speech-emotion-recognition","self-supervised","multilingual"],"category":"paralinguistics-emotion","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Szeged","HUN-REN"],"funding":["Ministry of Culture and Innovation of Hungary","National Research, Development and Innovation Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"halmai26_interspeech","category":"paralinguistics-emotion","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Szeged","HUN-REN"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2143","pdf":"https://www.isca-archive.org/interspeech_2026/halmai26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/halmai26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/halmai26_interspeech/markdown.md"},{"id":"halpern26_interspeech","title":"PathBench: Speech Intelligibility Benchmark for Automatic Pathological Speech Assessment","authors":["Bence Mark Halpern","Thomas Tienkamp","Defne Abur","Tomoki Toda"],"year":2026,"doi":"10.21437/Interspeech.2026-946","isca_url":"https://www.isca-archive.org/interspeech_2026/halpern26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/halpern26_interspeech.pdf","session":"Pathological Speech Assessment 4","topics":["speech-recognition","paralinguistics","evaluation"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Nagoya University","University of Groningen","University of Cologne"],"funding":["Dutch Research Council","JSPS KAKENHI","BRIDGE Program"],"code":{"url":"https://github.com/karkirowle/pathbench","stars":9,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"halpern26_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Nagoya University","University of Groningen","University of Cologne"],"code":"https://github.com/karkirowle/pathbench","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-946","pdf":"https://www.isca-archive.org/interspeech_2026/halpern26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/halpern26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/halpern26_interspeech/markdown.md"},{"id":"ham26_interspeech","title":"Continuous 2D Spectral—Temporal Transformer for Speaker Verification","authors":["Seongwook Ham","Thien-Phuc Doan"],"year":2026,"doi":"10.21437/Interspeech.2026-963","isca_url":"https://www.isca-archive.org/interspeech_2026/ham26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ham26_interspeech.pdf","session":"Speaker Recognition and Verification","topics":["speaker-verification","self-supervised"],"category":"speaker","institutions":["Soongsil University"],"code":{"url":"https://github.com/roadroller0501/C2D-ST","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ham26_interspeech","category":"speaker","institutions":["Soongsil University"],"code":"https://github.com/roadroller0501/C2D-ST","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-963","pdf":"https://www.isca-archive.org/interspeech_2026/ham26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ham26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ham26_interspeech/markdown.md"},{"id":"han26_interspeech","title":"A Transcript-anchored Pipeline With Large Language Models For Detecting Inappropriate Pauses In Dysarthric Speech","authors":["Minsu Han","Insung Lee","Taeyoung Jeong","Myoung-Wan Koo"],"year":2026,"doi":"10.21437/Interspeech.2026-534","isca_url":"https://www.isca-archive.org/interspeech_2026/han26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/han26_interspeech.pdf","session":"Speech, Voice and Language Disorders","topics":["paralinguistics","speech-recognition","evaluation"],"category":"health-clinical","labels":["self-supervised"],"institutions":["LG Electronics","Sogang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"han26_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["LG Electronics","Sogang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-534","pdf":"https://www.isca-archive.org/interspeech_2026/han26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/han26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/han26_interspeech/markdown.md"},{"id":"han26b_interspeech","title":"Branch-wise Complementary Attention for Acoustic Scene Classification","authors":["Seung-Gyu Han","Jinwoo Jung","Pil Moo Byun","Won-Gook Choi","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-865","isca_url":"https://www.isca-archive.org/interspeech_2026/han26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/han26b_interspeech.pdf","session":"Spatial Audio 3","topics":["asre","paralinguistics","on-device"],"category":"audio-understanding","institutions":["Hanyang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"han26b_interspeech","category":"audio-understanding","institutions":["Hanyang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-865","pdf":"https://www.isca-archive.org/interspeech_2026/han26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/han26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/han26b_interspeech/markdown.md"},{"id":"han26c_interspeech","title":"Exploring Hesitation as a Signal for Spoken Grammatical Error Correction","authors":["Seunghoon Han","Minyoung Kyoung","Hyungbae Jeon"],"year":2026,"doi":"10.21437/Interspeech.2026-1869","isca_url":"https://www.isca-archive.org/interspeech_2026/han26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/han26c_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["spoken-language-understanding","self-supervised","evaluation"],"category":"applications-other","institutions":["Tutorus Labs"],"funding":["Culture, Sports and Tourism R&D Program","Korea Creative Content Agency","Ministry of Culture, Sports and Tourism"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"han26c_interspeech","category":"applications-other","institutions":["Tutorus Labs"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1869","pdf":"https://www.isca-archive.org/interspeech_2026/han26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/han26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/han26c_interspeech/markdown.md"},{"id":"han26d_interspeech","title":"Imitation Learning for Elder-Facing Speech Synthesis","authors":["Dongrui Han","Weidong Chen","Jiawen Kang","Mingyu Cui","Helen Meng","Xixin Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-2107","isca_url":"https://www.isca-archive.org/interspeech_2026/han26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/han26d_interspeech.pdf","session":"Voice Conversion","topics":["tts","self-supervised","low-resource"],"category":"tts","labels":["generative-model"],"institutions":["Chinese University of Hong Kong","Tencent"],"funding":["National Natural Science Foundation of China","Centre for Perceptual and Interactive Intelligence","Innovation and Technology Commission of the Hong Kong Special Administrative Region Government"],"code":{"url":"https://dongru1.github.io/demo/im-efss/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"han26d_interspeech","category":"tts","labels":["generative-model"],"institutions":["Chinese University of Hong Kong","Tencent"],"code":"https://dongru1.github.io/demo/im-efss/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2107","pdf":"https://www.isca-archive.org/interspeech_2026/han26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/han26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/han26d_interspeech/markdown.md"},{"id":"han26e_interspeech","title":"Soft-Gating Score-Level Fusion for Spoofing-Aware Speaker Verification","authors":["Seongkyu Han","Yowon Lee","Thien-Phuc Doan","Thien An Nguyen","Souhwan Jung"],"year":2026,"doi":"10.21437/Interspeech.2026-2294","isca_url":"https://www.isca-archive.org/interspeech_2026/han26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/han26e_interspeech.pdf","session":"Spoofing and Deepfake Detection 3","topics":["speaker-verification","evaluation","dataset"],"category":"speaker","institutions":["Soongsil University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Information Technology Research Center","Ministry of Science and ICT","National Research Foundation of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"han26e_interspeech","category":"speaker","institutions":["Soongsil University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2294","pdf":"https://www.isca-archive.org/interspeech_2026/han26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/han26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/han26e_interspeech/markdown.md"},{"id":"han26f_interspeech","title":"TAP-ETS: Time Aligned Phoneme Guiding for EMG-to-Speech Synthesis","authors":["Dongyub Han","Injune Hwang","Jaejun Lee","Jiwon Lee","Kyogu Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-3485","isca_url":"https://www.isca-archive.org/interspeech_2026/han26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/han26f_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["speech-synthesis","self-supervised","multilingual"],"category":"tts","labels":["generative-model"],"institutions":["Seoul National University"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation","Advanced GPU Utilization Support Program"],"code":{"url":"https://github.com/ongdyub/TAP-ETS","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"han26f_interspeech","category":"tts","labels":["generative-model"],"institutions":["Seoul National University"],"code":"https://github.com/ongdyub/TAP-ETS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3485","pdf":"https://www.isca-archive.org/interspeech_2026/han26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/han26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/han26f_interspeech/markdown.md"},{"id":"hanif26_interspeech","title":"ZEBRA: Zero-Shot Entropy-Regularized Prompt Learning for Base-to-Novel Generalization in Audio-Language Models","authors":["Asif Hanif","Mohammad Yaqub"],"year":2026,"doi":"10.21437/Interspeech.2026-261","isca_url":"https://www.isca-archive.org/interspeech_2026/hanif26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hanif26_interspeech.pdf","session":"Multimodal Speech Processing and Speech LLM Systems","topics":["self-supervised","speech-llm","evaluation"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Mohamed bin Zayed University of Artificial Intelligence"],"code":{"url":"https://github.com/asif-hanif/zebra","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hanif26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Mohamed bin Zayed University of Artificial Intelligence"],"code":"https://github.com/asif-hanif/zebra","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-261","pdf":"https://www.isca-archive.org/interspeech_2026/hanif26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hanif26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hanif26_interspeech/markdown.md"},{"id":"hansen26_interspeech","title":"Balancing Speech, Language and Hearing Science with Machine Learning Modeling in the Age of AI: “Know your Problem, Data, and Solution”","authors":["John Hansen"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/hansen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hansen26_interspeech.pdf","session":"Keynote1 - John Hansen: Balancing Speech, Language and Hearing Science with Machine Learning Modeling in the Age of AI: “Know your Problem, Data, and Solution”","topics":["evaluation"],"category":"resources-evaluation","institutions":["University of Texas at Dallas"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hansen26_interspeech","category":"resources-evaluation","institutions":["University of Texas at Dallas"],"updated":"2026-09-28","confidence":"abstract-only","source":"https://www.isca-archive.org/interspeech_2026/hansen26_interspeech.html"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hansen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hansen26_interspeech/markdown.md"},{"id":"hao26_interspeech","title":"YingMusic-Singer: Controllable Singing Voice Synthesis with Flexible Lyric Manipulation and Annotation-free Melody Guidance","authors":["Chunbo Hao","Junjie Zheng","Guobin Ma","Yuepeng Jiang","Huakang Chen","Wenjie Tian","Gongyu Chen","Zihao Chen","Lei Xie"],"year":2026,"doi":"10.21437/Interspeech.2026-1547","isca_url":"https://www.isca-archive.org/interspeech_2026/hao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hao26_interspeech.pdf","session":"Singing Voice and Music Generation","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Northwestern Polytechnical University","Giant Network"],"code":{"url":"https://github.com/ASLP-lab/YingMusic-Singer-Plus","stars":115,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hao26_interspeech","category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Northwestern Polytechnical University","Giant Network"],"code":"https://github.com/ASLP-lab/YingMusic-Singer-Plus","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1547","pdf":"https://www.isca-archive.org/interspeech_2026/hao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hao26_interspeech/markdown.md"},{"id":"hao26b_interspeech","title":"Can Large Language Models Reliably Correct Errors in Low-Resource ASR? A Contamination-Aware Case Study on West Frisian","authors":["Yun Hao","Reihaneh Amooie","Wietse de Vries","Rik van Noord","Martijn Wieling"],"year":2026,"doi":"10.21437/Interspeech.2026-1659","isca_url":"https://www.isca-archive.org/interspeech_2026/hao26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hao26b_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 1","topics":["asr","speech-llm","evaluation"],"category":"asr","labels":["low-resource"],"institutions":["University of Groningen"],"funding":["China Scholarship Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hao26b_interspeech","category":"asr","labels":["low-resource"],"institutions":["University of Groningen"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1659","pdf":"https://www.isca-archive.org/interspeech_2026/hao26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hao26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hao26b_interspeech/markdown.md"},{"id":"hartmann26_interspeech","title":"Towards a Stochastic DNN Approximation of Cochlear Implant Auditory Models","authors":["Theresa Hartmann","Ian C. Bruce","Benjamin Lentz","Rainer Martin","Anil Nagathil"],"year":2026,"doi":"10.21437/Interspeech.2026-1872","isca_url":"https://www.isca-archive.org/interspeech_2026/hartmann26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hartmann26_interspeech.pdf","session":"Assistive Technologies 2","topics":["speech-enhancement","self-supervised","health"],"category":"health-clinical","labels":["efficient-on-device","generative-model"],"institutions":["Ruhr University Bochum","McMaster University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hartmann26_interspeech","category":"health-clinical","labels":["efficient-on-device","generative-model"],"institutions":["Ruhr University Bochum","McMaster University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1872","pdf":"https://www.isca-archive.org/interspeech_2026/hartmann26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hartmann26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hartmann26_interspeech/markdown.md"},{"id":"hasan26_interspeech","title":"Dual-Stream DNN-KAN Networks with Bangla-Specific Features for Speech Emotion Recognition","authors":["Kazi Reyazul Hasan","Muhammad Abdullah Adnan"],"year":2026,"doi":"10.21437/Interspeech.2026-3374","isca_url":"https://www.isca-archive.org/interspeech_2026/hasan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hasan26_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["speech-emotion-recognition","low-resource","self-supervised"],"category":"paralinguistics-emotion","labels":["low-resource"],"institutions":["Bangladesh University of Engineering and Technology"],"funding":["Basic Research Grant from Bangladesh University of Engineering and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hasan26_interspeech","category":"paralinguistics-emotion","labels":["low-resource"],"institutions":["Bangladesh University of Engineering and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3374","pdf":"https://www.isca-archive.org/interspeech_2026/hasan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hasan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hasan26_interspeech/markdown.md"},{"id":"hassan26_interspeech","title":"SCANS: Supervised Contrastive Temporal Alignment of Neural Responses and Speech Stimuli","authors":["K M Naimul Hassan","Donald S. Williamson"],"year":2026,"doi":"10.21437/Interspeech.2026-2651","isca_url":"https://www.isca-archive.org/interspeech_2026/hassan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hassan26_interspeech.pdf","session":"Brain Studies and Speech","topics":["speech-llm","self-supervised","evaluation"],"category":"applications-other","institutions":["Ohio State University"],"funding":["National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hassan26_interspeech","category":"applications-other","institutions":["Ohio State University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2651","pdf":"https://www.isca-archive.org/interspeech_2026/hassan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hassan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hassan26_interspeech/markdown.md"},{"id":"hasumi26_interspeech","title":"Aligning MusicLLM with Emotion using Instruction Tuning and Feedback-Driven Alignment","authors":["Takuya Hasumi","Welly Naptali"],"year":2026,"doi":"10.21437/Interspeech.2026-2293","isca_url":"https://www.isca-archive.org/interspeech_2026/hasumi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hasumi26_interspeech.pdf","session":"Audio signal analysis","topics":["speech-llm","paralinguistics","self-supervised"],"category":"paralinguistics-emotion","institutions":["LY Corporation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hasumi26_interspeech","category":"paralinguistics-emotion","institutions":["LY Corporation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2293","pdf":"https://www.isca-archive.org/interspeech_2026/hasumi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hasumi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hasumi26_interspeech/markdown.md"},{"id":"havare26_interspeech","title":"CodeVaani: A Multilingual, Voice-Based Code Learning Assistant","authors":["Jayant Havare","Srikanth Tamilselvam","Ashish Mittal","Shalaka Thorat","Soham Jadia","Varsha Apte","Ganesh Ramakrishnan"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/havare26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/havare26_interspeech.pdf","session":"Speech and Language Learning Technologies","topics":["speech-recognition","speech-llm","multilingual"],"category":"speech-llm-dialogue","labels":["low-resource","multilingual"],"institutions":["Indian Institute of Technology Bombay","IBM","Google DeepMind"],"code":{"url":"https://tinyurl.com/icse2026-artifacts","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"havare26_interspeech","category":"speech-llm-dialogue","labels":["low-resource","multilingual"],"institutions":["Indian Institute of Technology Bombay","IBM","Google DeepMind"],"code":"https://tinyurl.com/icse2026-artifacts","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/havare26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/havare26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/havare26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/havare26_interspeech/markdown.md"},{"id":"hay26_interspeech","title":"What does it mean to ‘know a word’?","authors":["Jennifer Hay"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/hay26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hay26_interspeech.pdf","session":"Keynote3 - Jennifer Hay: What does it mean to ‘know a word’?","topics":["phonetics","low-resource"],"category":"phonetics-linguistics","institutions":["University of Canterbury"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hay26_interspeech","category":"phonetics-linguistics","institutions":["University of Canterbury"],"updated":"2026-09-28","confidence":"abstract-only","source":"https://www.isca-archive.org/interspeech_2026/hay26_interspeech.html"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hay26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hay26_interspeech/markdown.md"},{"id":"hayden26_interspeech","title":"Accent-Emotion Entanglement in LM-Based Text-to-Speech Systems","authors":["Matthew Hayden","Jinzuomu Zhong","Korin Richmond"],"year":2026,"doi":"10.21437/Interspeech.2026-2749","isca_url":"https://www.isca-archive.org/interspeech_2026/hayden26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hayden26_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["tts","evaluation","paralinguistics"],"category":"tts","labels":["generative-model"],"institutions":["University of Edinburgh"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hayden26_interspeech","category":"tts","labels":["generative-model"],"institutions":["University of Edinburgh"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2749","pdf":"https://www.isca-archive.org/interspeech_2026/hayden26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hayden26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hayden26_interspeech/markdown.md"},{"id":"he26_interspeech","title":"TTBA: Spatial Prompted Text to Binaural Audio Generation Using Transformer","authors":["Changjun He","Lianyu Zhou","Yukun Qian","Shiyun Xu","Wenjie Zhang","Mingjiang Wang","Weiping Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-602","isca_url":"https://www.isca-archive.org/interspeech_2026/he26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/he26_interspeech.pdf","session":"Spatial Audio 3","topics":["spatial-audio","speech-llm","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["Harbin Institute of Technology"],"funding":["National Natural Science Foundation of China","Guangdong Basic and Applied Basic Research Foundation","Shenzhen Higher Education Institutions Stability Support Program","Key Research and Development Program of Xinjiang Uygur Autonomous Region"],"code":{"url":"https://spatialaudiodemo.github.io/TTBA/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"he26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Harbin Institute of Technology"],"code":"https://spatialaudiodemo.github.io/TTBA/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-602","pdf":"https://www.isca-archive.org/interspeech_2026/he26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/he26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/he26_interspeech/markdown.md"},{"id":"he26b_interspeech","title":"Spec2Spatial: A Time-Frequency Spatial Attention Network for Binaural Audio Synthesis","authors":["Changjun He","Wenjie Zhang","Shiyun Xu","Lianyu Zhou","Weiping Chen","Mingjiang Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-607","isca_url":"https://www.isca-archive.org/interspeech_2026/he26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/he26b_interspeech.pdf","session":"Spatial Audio 3","topics":["source-separation","evaluation","dataset"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Harbin Institute of Technology"],"funding":["National Natural Science Foundation of China","Guangdong Basic and Applied Basic Research Foundation","Shenzhen Higher Education Institutions Stability Support Program","Key Research and Development Program of Xinjiang Uygur Autonomous Region"],"code":{"url":"https://SpatialAudioDemo.github.io/Spec2Spatial/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"he26b_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Harbin Institute of Technology"],"code":"https://SpatialAudioDemo.github.io/Spec2Spatial/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-607","pdf":"https://www.isca-archive.org/interspeech_2026/he26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/he26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/he26b_interspeech/markdown.md"},{"id":"he26c_interspeech","title":"LLM-HB: Language-Aware LLM-Guided Hotword Biasing for Code-Switching ASR","authors":["Yuxuan He","Genshun Wan","Pengcheng Li","Gongping Huang","Jian-Qing Gao","Yanmin Qian"],"year":2026,"doi":"10.21437/Interspeech.2026-1113","isca_url":"https://www.isca-archive.org/interspeech_2026/he26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/he26c_interspeech.pdf","session":"Code-Switching ASR","topics":["asr","speech-llm","multilingual"],"category":"asr","labels":["multilingual"],"institutions":["Shanghai Jiao Tong University","iFLYTEK","Wuhan University"],"funding":["China NSFC","SJTU Med-X Translational Research Grant"],"code":{"url":"https://anonymous.4open.science/r/IS26-ASRU2019-Hotword-List/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"he26c_interspeech","category":"asr","labels":["multilingual"],"institutions":["Shanghai Jiao Tong University","iFLYTEK","Wuhan University"],"code":"https://anonymous.4open.science/r/IS26-ASRU2019-Hotword-List/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1113","pdf":"https://www.isca-archive.org/interspeech_2026/he26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/he26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/he26c_interspeech/markdown.md"},{"id":"he26d_interspeech","title":"Disentangling Acoustic Cues in Alzheimer’s Pathology and Perception: The Roles of Language and Gender","authors":["Liu He","Yuanchao Li","Yin-Long Liu","Rui Feng","Yiming Wang","Jiaxin Chen","Yizhe Wang","Jiahong Yuan"],"year":2026,"doi":"10.21437/Interspeech.2026-1149","isca_url":"https://www.isca-archive.org/interspeech_2026/he26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/he26d_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["paralinguistics","evaluation","multilingual"],"category":"health-clinical","labels":["multilingual"],"institutions":["University of Science and Technology of China","University of Edinburgh"],"funding":["National Social Science Foundation of China","Super-computing Center of the University of Science and Technology of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"he26d_interspeech","category":"health-clinical","labels":["multilingual"],"institutions":["University of Science and Technology of China","University of Edinburgh"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1149","pdf":"https://www.isca-archive.org/interspeech_2026/he26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/he26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/he26d_interspeech/markdown.md"},{"id":"he26e_interspeech","title":"Audio-DeepThinker: Progressive Reasoning-Aware Reinforcement Learning for High-Quality Chain-of-Thought Emergence in Audio Language Models","authors":["Xiang He","Chenxing Li","Jinting Wang","Yan Rong","Tianxin Xie","Zeyu Xie","Wenfu Wang","Li Liu","Dong Yu"],"year":2026,"doi":"10.21437/Interspeech.2026-1720","isca_url":"https://www.isca-archive.org/interspeech_2026/he26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/he26e_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Tencent","Hong Kong University of Science and Technology"],"funding":["National Natural Science Foundation of China","Guangdong Basic and Applied Basic Research Foundation","Tencent AI Lab Rhino-Bird Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"he26e_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Tencent","Hong Kong University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1720","pdf":"https://www.isca-archive.org/interspeech_2026/he26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/he26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/he26e_interspeech/markdown.md"},{"id":"he26f_interspeech","title":"Task-Aware Joint Pruning and Distillation for Efficient Audio Deepfake Detection","authors":["Miao He","Peng Cheng","Zhongjie Ba","Qing Wen","Li Lu","Xin Yang","Kui Ren"],"year":2026,"doi":"10.21437/Interspeech.2026-1766","isca_url":"https://www.isca-archive.org/interspeech_2026/he26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/he26f_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["audio-deepfake","self-supervised","on-device"],"category":"deepfake-security","labels":["efficient-on-device","self-supervised"],"institutions":["Zhejiang University","Hangzhou High-Tech Zone (Binjiang) Institute of Blockchain and Data Security","Shanghai Institute for Advanced Study","China University of Petroleum (East China)"],"funding":["Shanghai Municipal Special Program for Basic Research on General AI Foundation Models","National Natural Science Foundation of China","Zhejiang Provincial Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"he26f_interspeech","category":"deepfake-security","labels":["efficient-on-device","self-supervised"],"institutions":["Zhejiang University","Hangzhou High-Tech Zone (Binjiang) Institute of Blockchain and Data Security","Shanghai Institute for Advanced Study","China University of Petroleum (East China)"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1766","pdf":"https://www.isca-archive.org/interspeech_2026/he26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/he26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/he26f_interspeech/markdown.md"},{"id":"he26g_interspeech","title":"MOV-AAD: A Large-Scale Multimodal Dataset for Auditory Attention Decoding During Moving Conversations","authors":["Xiaomin He","Vishal Choudhari","Tristan J. Spratt","Aarya Raghavan","Richard T. Lee","Nima Mesgarani"],"year":2026,"doi":"10.21437/Interspeech.2026-3556","isca_url":"https://www.isca-archive.org/interspeech_2026/he26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/he26g_interspeech.pdf","session":"Audio-Visual and Multimodal Perception","topics":["auditory-attention-decoding","dataset","self-supervised"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Columbia University"],"funding":["National Institute on Deafness and Other Communication Disorders","Marie-Josee and Henry R. Kravis Foundation"],"code":{"url":"https://github.com/naplab/MOV-AAD","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"he26g_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Columbia University"],"code":"https://github.com/naplab/MOV-AAD","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3556","pdf":"https://www.isca-archive.org/interspeech_2026/he26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/he26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/he26g_interspeech/markdown.md"},{"id":"hegde26_interspeech","title":"Aligning Audio Captions with Human Preferences","authors":["Kartik Hegde","Rehana Mahfuz","Yinyi Guo","Erik Visser"],"year":2026,"doi":"10.21437/Interspeech.2026-2052","isca_url":"https://www.isca-archive.org/interspeech_2026/hegde26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hegde26_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["audio-captioning","self-supervised","evaluation"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["Qualcomm"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hegde26_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["Qualcomm"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2052","pdf":"https://www.isca-archive.org/interspeech_2026/hegde26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hegde26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hegde26_interspeech/markdown.md"},{"id":"heng26_interspeech","title":"Improving Code-Switching ASR with Code-Mixing Guided Synthetic Speech","authors":["Yeo Yue Heng","Haoyang Li","Yizhou Peng","Shreyas Gopal","Hexin Liu","Leibny Paola Garcia-Perera","Sailor Hardik","Jeremy H. M. Wong","Eng Siong Chng"],"year":2026,"doi":"10.21437/Interspeech.2026-642","isca_url":"https://www.isca-archive.org/interspeech_2026/heng26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/heng26_interspeech.pdf","session":"Code-Switching ASR","topics":["speech-recognition","self-supervised","multilingual"],"category":"asr","labels":["low-resource","multilingual","generative-model"],"institutions":["Nanyang Technological University","Agency for Science, Technology and Research","Johns Hopkins University","Google"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"heng26_interspeech","category":"asr","labels":["low-resource","multilingual","generative-model"],"institutions":["Nanyang Technological University","Agency for Science, Technology and Research","Johns Hopkins University","Google"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-642","pdf":"https://www.isca-archive.org/interspeech_2026/heng26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/heng26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/heng26_interspeech/markdown.md"},{"id":"heo26_interspeech","title":"Leveraging Diarization Labels for Robust Score Calibration in Target Speaker Tagging via Gaussian Mixture Modeling","authors":["Hee-Soo Heo","Minjae Lee","Youngki Kwon","Han-Gyu Kim","Bong-Jin Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-896","isca_url":"https://www.isca-archive.org/interspeech_2026/heo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/heo26_interspeech.pdf","session":"Speaker Diarization 1","topics":["speaker-verification","speaker-diarization","speech-enhancement"],"category":"speaker","institutions":["NAVER Cloud Corporation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"heo26_interspeech","category":"speaker","institutions":["NAVER Cloud Corporation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-896","pdf":"https://www.isca-archive.org/interspeech_2026/heo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/heo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/heo26_interspeech/markdown.md"},{"id":"heo26b_interspeech","title":"Tracing the Origins: Legacy Codec Identification in Neural Audio Transcoding","authors":["Wonje Heo","Shinee Youn","Yooshin Kim","Chuck Chae","Donghoon Shin"],"year":2026,"doi":"10.21437/Interspeech.2026-2354","isca_url":"https://www.isca-archive.org/interspeech_2026/heo26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/heo26b_interspeech.pdf","session":"Speech and Language Representation","topics":["audio-deepfake","evaluation","self-supervised"],"category":"deepfake-security","institutions":["DGIST"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"heo26b_interspeech","category":"deepfake-security","institutions":["DGIST"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2354","pdf":"https://www.isca-archive.org/interspeech_2026/heo26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/heo26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/heo26b_interspeech/markdown.md"},{"id":"hernandez26_interspeech","title":"Adapting Self-Supervised Speech Representations for Cross-Lingual Dysarthria Detection in Parkinson's Disease","authors":["Abner Hernandez","Eunjung Yeo","Kwanghee Choi","Chin-Jou Li","Zhengjun Yue","Rohan Kumar Das","Jan Rusz","Mathew Magimai Doss","Juan Rafael Orozco-Arroyave","Tomás Arias-Vergara","Andreas Maier","Elmar Nöth","David R. Mortensen","David Harwath","Paula Andrea Pérez-Toro"],"year":2026,"doi":"10.21437/Interspeech.2026-773","isca_url":"https://www.isca-archive.org/interspeech_2026/hernandez26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hernandez26_interspeech.pdf","session":"Pathological Speech Assessment 1","topics":["paralinguistics","self-supervised","multilingual"],"category":"health-clinical","labels":["low-resource","multilingual","self-supervised"],"institutions":["FAU Erlangen-Nurnberg","UT Austin","Carnegie Mellon University","Czech Technical University in Prague","Idiap Research Institute","Universidad de Antioquia","Shenzhen Loop Area Institute","Fortemedia"],"code":{"url":"https://github.com/abnerLing/language-shift-dysarthria","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hernandez26_interspeech","category":"health-clinical","labels":["low-resource","multilingual","self-supervised"],"institutions":["FAU Erlangen-Nurnberg","UT Austin","Carnegie Mellon University","Czech Technical University in Prague","Idiap Research Institute","Universidad de Antioquia","Shenzhen Loop Area Institute","Fortemedia"],"code":"https://github.com/abnerLing/language-shift-dysarthria","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-773","pdf":"https://www.isca-archive.org/interspeech_2026/hernandez26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hernandez26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hernandez26_interspeech/markdown.md"},{"id":"hernandez26b_interspeech","title":"Multilingual Phonological Feature Recognition with Self-Supervised Speech Models","authors":["Abner Hernandez","Tomás Arias-Vergara","Daiqi Liu","Andreas Maier","Paula Andrea Pérez-Toro"],"year":2026,"doi":"10.21437/Interspeech.2026-2735","isca_url":"https://www.isca-archive.org/interspeech_2026/hernandez26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hernandez26b_interspeech.pdf","session":"Phonetic Aspects of TTS and ASR Systems","topics":["self-supervised","multilingual","phonetics"],"category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["Friedrich-Alexander-Universitat Erlangen-Nurnberg","Universidad de Antioquia"],"code":{"url":"https://github.com/abnerLing/PhonoQ-2.0","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hernandez26b_interspeech","category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["Friedrich-Alexander-Universitat Erlangen-Nurnberg","Universidad de Antioquia"],"code":"https://github.com/abnerLing/PhonoQ-2.0","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2735","pdf":"https://www.isca-archive.org/interspeech_2026/hernandez26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hernandez26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hernandez26b_interspeech/markdown.md"},{"id":"higuchi26_interspeech","title":"Incremental End-to-End Spoken Dialogue State Tracking with a Multimodal LLM and Reinforcement Learning","authors":["Tomoya Higuchi","Michimasa Inaba"],"year":2026,"doi":"10.21437/Interspeech.2026-592","isca_url":"https://www.isca-archive.org/interspeech_2026/higuchi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/higuchi26_interspeech.pdf","session":"Spoken Language Understanding","topics":["spoken-language-understanding","speech-llm","self-supervised"],"category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["University of Electro-Communications"],"code":{"url":"https://github.com/UEC-InabaLab/IncrementalRLForSpokenDST","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"higuchi26_interspeech","category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["University of Electro-Communications"],"code":"https://github.com/UEC-InabaLab/IncrementalRLForSpokenDST","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-592","pdf":"https://www.isca-archive.org/interspeech_2026/higuchi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/higuchi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/higuchi26_interspeech/markdown.md"},{"id":"hilmes26_interspeech","title":"Positional Encoding in the Context of Memristor-Based Analog Computation for Automatic Speech Recognition","authors":["Benedikt Hilmes","Nick Rossenbach","Ralf Schlüter"],"year":2026,"doi":"10.21437/Interspeech.2026-683","isca_url":"https://www.isca-archive.org/interspeech_2026/hilmes26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hilmes26_interspeech.pdf","session":"Resource Constrained Speech Recognition","topics":["asr","self-supervised","on-device"],"category":"asr","labels":["efficient-on-device"],"institutions":["RWTH Aachen University","AppTek"],"funding":["NeuroSys","Federal Ministry of Research, Technology and Space BMFTR","RESCALE","Federal Ministry for the Environment, Nature Conservation, Nuclear Safety and Consumer Protection","Federal Ministry of Education and Research"],"code":{"url":"https://github.com/rwth-i6/returnn-experiments/tree/master/2026-memristor-pe","stars":163,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hilmes26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["RWTH Aachen University","AppTek"],"code":"https://github.com/rwth-i6/returnn-experiments/tree/master/2026-memristor-pe","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-683","pdf":"https://www.isca-archive.org/interspeech_2026/hilmes26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hilmes26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hilmes26_interspeech/markdown.md"},{"id":"hirose26_interspeech","title":"Self-adaptive Gradient Conflict Mitigator for Continuous-Time Diffusion Models","authors":["Takumi Hirose","Zhiyang Li","Nakamasa Inoue"],"year":2026,"doi":"10.21437/Interspeech.2026-3059","isca_url":"https://www.isca-archive.org/interspeech_2026/hirose26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hirose26_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["uncategorized"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Institute of Science Tokyo"],"funding":["JSPS KAKENHI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hirose26_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Institute of Science Tokyo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3059","pdf":"https://www.isca-archive.org/interspeech_2026/hirose26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hirose26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hirose26_interspeech/markdown.md"},{"id":"hjuler26_interspeech","title":"Listenability of Synthetic Speech: On the Effect of Linguistic Registers in Text-to-Speech Input","authors":["Maja Jønck Hjuler","Tuyet Katie Nhi Tran","Laurianne Sitbon"],"year":2026,"doi":"10.21437/Interspeech.2026-894","isca_url":"https://www.isca-archive.org/interspeech_2026/hjuler26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hjuler26_interspeech.pdf","session":"Text Processing for Speech Synthesis","topics":["tts","evaluation","speech-llm"],"category":"tts","institutions":["Queensland University of Technology","University Grenoble Alpes","CNRS","Grenoble INP"],"funding":["Australian Research Council","European Union","Marie Skłodowska-Curie"],"code":{"url":"https://github.com/MajaHjuler/TTS_Listenability","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hjuler26_interspeech","category":"tts","institutions":["Queensland University of Technology","University Grenoble Alpes","CNRS","Grenoble INP"],"code":"https://github.com/MajaHjuler/TTS_Listenability","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-894","pdf":"https://www.isca-archive.org/interspeech_2026/hjuler26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hjuler26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hjuler26_interspeech/markdown.md"},{"id":"ho26_interspeech","title":"An Investigation on Combining Geometry and Consistency Constraints into Phase Estimation for Speech Enhancement","authors":["Chun-Wei Ho","Pin-Jui Ku","Hao Yen","Sabato Marco Siniscalchi","Yu Tsao","Chin-Hui Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-621","isca_url":"https://www.isca-archive.org/interspeech_2026/ho26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ho26_interspeech.pdf","session":"Multi-Channel Processing and Specialized Acquisition (UAV, Radar, Hearables)","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","institutions":["Georgia Institute of Technology","Universita degli Studi di Palermo","Academia Sinica"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ho26_interspeech","category":"enhancement-separation","institutions":["Georgia Institute of Technology","Universita degli Studi di Palermo","Academia Sinica"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-621","pdf":"https://www.isca-archive.org/interspeech_2026/ho26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ho26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ho26_interspeech/markdown.md"},{"id":"hoang26_interspeech","title":"Towards Efficient Simultaneous Inverse Text Normalization with Pretrained Text-to-Text Language Model and Read-Tag-Write Policy","authors":["Kiet Anh Hoang","Khanh Le","Bao Nguyen","Linh Pham","Dung Vo","Thai Tran","Mai Nguyen","Tri Nguyen","Vu Le"],"year":2026,"doi":"10.21437/Interspeech.2026-1060","isca_url":"https://www.isca-archive.org/interspeech_2026/hoang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hoang26_interspeech.pdf","session":"Robust and Efficient ASR","topics":["asr","self-supervised","streaming"],"category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["UNEY"],"funding":["UNEY"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hoang26_interspeech","category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["UNEY"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1060","pdf":"https://www.isca-archive.org/interspeech_2026/hoang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hoang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hoang26_interspeech/markdown.md"},{"id":"hoffner26_interspeech","title":"Deep learning-based predictions of perceived listening effort and intelligibility across enhanced, synthetic, natural, and binaural speech","authors":["Dirk Eike Hoffner","Hartmut Schoon","Rainer Huber","Jan Rennies","Bernd T. Meyer"],"year":2026,"doi":"10.21437/Interspeech.2026-1891","isca_url":"https://www.isca-archive.org/interspeech_2026/hoffner26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hoffner26_interspeech.pdf","session":"Model of Speech Perception","topics":["paralinguistics","evaluation","self-supervised"],"category":"resources-evaluation","labels":["self-supervised","robustness-noise"],"institutions":["Carl von Ossietzky Universitat Oldenburg","Cluster of Excellence Hearing4all","Fraunhofer Institute for Digital Media Technology"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hoffner26_interspeech","category":"resources-evaluation","labels":["self-supervised","robustness-noise"],"institutions":["Carl von Ossietzky Universitat Oldenburg","Cluster of Excellence Hearing4all","Fraunhofer Institute for Digital Media Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1891","pdf":"https://www.isca-archive.org/interspeech_2026/hoffner26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hoffner26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hoffner26_interspeech/markdown.md"},{"id":"holt26_interspeech","title":"Talker Discrimination and Identification in 7-12-year-old Children: Effects of Talker Gender and Phonological Ability","authors":["Rebecca Holt","Parisa McGirr","Chi Yhun Lo","Anita Szakay"],"year":2026,"doi":"10.21437/Interspeech.2026-3053","isca_url":"https://www.isca-archive.org/interspeech_2026/holt26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/holt26_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["phonetics","prosody","evaluation"],"category":"phonetics-linguistics","institutions":["Macquarie University"],"funding":["Macquarie University","Australian Linguistic Society"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"holt26_interspeech","category":"phonetics-linguistics","institutions":["Macquarie University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3053","pdf":"https://www.isca-archive.org/interspeech_2026/holt26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/holt26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/holt26_interspeech/markdown.md"},{"id":"hong26_interspeech","title":"Convolutional Dynamic Rotary Positional Encoding","authors":["Euijin Hong","Mengchun Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1312","isca_url":"https://www.isca-archive.org/interspeech_2026/hong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hong26_interspeech.pdf","session":"New Architecture and Analyses for ASR and Speech LMs","topics":["asr","self-supervised","evaluation"],"category":"asr","institutions":["Carnegie Mellon University","University of Pittsburgh"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hong26_interspeech","category":"asr","institutions":["Carnegie Mellon University","University of Pittsburgh"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1312","pdf":"https://www.isca-archive.org/interspeech_2026/hong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hong26_interspeech/markdown.md"},{"id":"hope26_interspeech","title":"Lived Experiences of Power and Agency: Achieving Voice Sovereignty in Assistive Speech Technology for Nonbinary Users","authors":["Maxwell Hope","Juliana Francis","Joakim Gustafson","Éva Székely"],"year":2026,"doi":"10.21437/Interspeech.2026-2663","isca_url":"https://www.isca-archive.org/interspeech_2026/hope26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hope26_interspeech.pdf","session":"Queer and Trans Speech Science and Technology","topics":["tts","speech-llm","evaluation"],"category":"tts","institutions":["University of Delaware","KTH Royal Institute of Technology"],"funding":["WASP"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hope26_interspeech","category":"tts","institutions":["University of Delaware","KTH Royal Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2663","pdf":"https://www.isca-archive.org/interspeech_2026/hope26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hope26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hope26_interspeech/markdown.md"},{"id":"hori26_interspeech","title":"Plan and Double-Check: Streaming Multimodal Q-Former for Online Robot Action Generation","authors":["Chiori Hori","Ryosuke Korekata","Motonari Kambara","Yoshiki Masuyama","Siddarth Jain","Radu Corcodel","Diego Romeres","Jonathan Le Roux"],"year":2026,"doi":"10.21437/Interspeech.2026-2999","isca_url":"https://www.isca-archive.org/interspeech_2026/hori26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hori26_interspeech.pdf","session":"Multimodal Spoken Dialogue Systems","topics":["speech-llm","spoken-language-understanding","multilingual"],"category":"speech-llm-dialogue","labels":["efficient-on-device","streaming-real-time"],"institutions":["Mitsubishi Electric Research Laboratories","Keio University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hori26_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","streaming-real-time"],"institutions":["Mitsubishi Electric Research Laboratories","Keio University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2999","pdf":"https://www.isca-archive.org/interspeech_2026/hori26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hori26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hori26_interspeech/markdown.md"},{"id":"horiguchi26_interspeech","title":"Tight Boundary Prediction in Speaker Diarization Using Causal-Anticausal Consistency","authors":["Shota Horiguchi","Marc Delcroix","Naohiro Tawara","Takanori Ashihara","Atsushi Ando"],"year":2026,"doi":"10.21437/Interspeech.2026-45","isca_url":"https://www.isca-archive.org/interspeech_2026/horiguchi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/horiguchi26_interspeech.pdf","session":"Speaker Diarization and Recognition","topics":["speaker-diarization","self-supervised","speech-enhancement"],"category":"speaker","institutions":["NTT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"horiguchi26_interspeech","category":"speaker","institutions":["NTT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-45","pdf":"https://www.isca-archive.org/interspeech_2026/horiguchi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/horiguchi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/horiguchi26_interspeech/markdown.md"},{"id":"horii26_interspeech","title":"How does children's pronunciation develop? Capturing syllabic change with children's growth using unsupervised syllable discovery","authors":["Koharu Horii","Naohiro Tawara","Atsunori Ogawa","Shoko Araki"],"year":2026,"doi":"10.21437/Interspeech.2026-3104","isca_url":"https://www.isca-archive.org/interspeech_2026/horii26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/horii26_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["self-supervised","phonetics","dataset"],"category":"phonetics-linguistics","institutions":["NTT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"horii26_interspeech","category":"phonetics-linguistics","institutions":["NTT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3104","pdf":"https://www.isca-archive.org/interspeech_2026/horii26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/horii26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/horii26_interspeech/markdown.md"},{"id":"hosseinikivanani26_interspeech","title":"Speaker or Language? Explaining Variance in Charismatic Prosody Across Luxembourgish and French","authors":["Nina Hosseini-Kivanani","Nafiseh Taghva","Peter Gilles","Oliver Niebuhr"],"year":2026,"doi":"10.21437/Interspeech.2026-26","isca_url":"https://www.isca-archive.org/interspeech_2026/hosseinikivanani26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hosseinikivanani26_interspeech.pdf","session":"Multilingual and Cross-Lingual Paralinguistic Analysis and Processing","topics":["paralinguistics","multilingual","phonetics"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Radio Television Luxembourg","University of Luxembourg","Shiraz University","University of Southern Denmark"],"funding":["Luxembourg National Research Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hosseinikivanani26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Radio Television Luxembourg","University of Luxembourg","Shiraz University","University of Southern Denmark"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-26","pdf":"https://www.isca-archive.org/interspeech_2026/hosseinikivanani26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hosseinikivanani26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hosseinikivanani26_interspeech/markdown.md"},{"id":"hosseinikivanani26b_interspeech","title":"Speaker-Specific and Language-Dependent Temporal Organization in Bilingual Political Speech","authors":["Nina Hosseini-Kivanani","Nafiseh Taghva","Peter Gilles","Oliver Niebuhr"],"year":2026,"doi":"10.21437/Interspeech.2026-1335","isca_url":"https://www.isca-archive.org/interspeech_2026/hosseinikivanani26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hosseinikivanani26b_interspeech.pdf","session":"Speaker-Specific, Forensic and Segmental Characteristics of Typical and Atypical Speech","topics":["prosody","multilingual","evaluation"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["University of Luxembourg","Radio Television Luxembourg","Shiraz University","University of Southern Denmark"],"funding":["Luxembourg National Research Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hosseinikivanani26b_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["University of Luxembourg","Radio Television Luxembourg","Shiraz University","University of Southern Denmark"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1335","pdf":"https://www.isca-archive.org/interspeech_2026/hosseinikivanani26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hosseinikivanani26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hosseinikivanani26b_interspeech/markdown.md"},{"id":"hou26_interspeech","title":"UGPCB: Uncertainty-Gated Phonetic Contextual Biasing for Improving Hotword Recognition in Large Speech Models","authors":["Yong-Jie Hou","Yun-Fei Shao","Wei-Qiang Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1577","isca_url":"https://www.isca-archive.org/interspeech_2026/hou26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hou26_interspeech.pdf","session":"Robust ASR: Uncertainty and Confidence","topics":["asr","speech-llm","low-resource"],"category":"asr","labels":["self-supervised"],"institutions":["Tsinghua University","Tiangong University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hou26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Tsinghua University","Tiangong University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1577","pdf":"https://www.isca-archive.org/interspeech_2026/hou26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hou26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hou26_interspeech/markdown.md"},{"id":"hou26b_interspeech","title":"Correct Then Detect: Zero-Shot FVMC Annotation for Child Language Sample Analysis","authors":["Shuwei Hou","Wei Bo","Varun Shijo","Chuhui Liu","Manav Kanaganapalli","Ling-Yu Guo","Wenyao Xu"],"year":2026,"doi":"10.21437/Interspeech.2026-2593","isca_url":"https://www.isca-archive.org/interspeech_2026/hou26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hou26b_interspeech.pdf","session":"Child Speech and Health","topics":["spoken-language-understanding","evaluation","low-resource"],"category":"resources-evaluation","institutions":["University at Buffalo"],"funding":["National Science Foundation","U.S. Department of Education"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hou26b_interspeech","category":"resources-evaluation","institutions":["University at Buffalo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2593","pdf":"https://www.isca-archive.org/interspeech_2026/hou26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hou26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hou26b_interspeech/markdown.md"},{"id":"hovsepyan26_interspeech","title":"Exploratory analysis of yellow mongoose vocalization: detection from in-the-wild recordings and call classification","authors":["Sevada Hovsepyan","Imen Ben Mahmoud","Vanessa Rüegg","Marta Manser","Mathew Magimai Doss"],"year":2026,"doi":"10.21437/Interspeech.2026-2168","isca_url":"https://www.isca-archive.org/interspeech_2026/hovsepyan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hovsepyan26_interspeech.pdf","session":"Acoustic Event Detection 3","topics":["paralinguistics","self-supervised","evaluation"],"category":"audio-understanding","institutions":["Idiap Research Institute","University of Zurich"],"funding":["NCCR Evolving Language","Swiss National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hovsepyan26_interspeech","category":"audio-understanding","institutions":["Idiap Research Institute","University of Zurich"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2168","pdf":"https://www.isca-archive.org/interspeech_2026/hovsepyan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hovsepyan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hovsepyan26_interspeech/markdown.md"},{"id":"hsu26_interspeech","title":"Entity Binding Failures in Speech LLM Reasoning: Diagnosis and Chain-of-Thought Intervention","authors":["Ming-Hao Hsu","Xiaohai Tian","Jun Zhang","Zhizheng Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-640","isca_url":"https://www.isca-archive.org/interspeech_2026/hsu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hsu26_interspeech.pdf","session":"Reasoning with Speech/Audio Language Models","topics":["speech-llm","evaluation","self-supervised"],"category":"speech-llm-dialogue","institutions":["Chinese University of Hong Kong","ByteDance","Shenzhen Loop Area Institute","Amphion Technology Co., Ltd"],"funding":["Shenzhen Science and Technology Program","Internal Project Fund from Shenzhen Research Institute of Big Data"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hsu26_interspeech","category":"speech-llm-dialogue","institutions":["Chinese University of Hong Kong","ByteDance","Shenzhen Loop Area Institute","Amphion Technology Co., Ltd"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-640","pdf":"https://www.isca-archive.org/interspeech_2026/hsu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hsu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hsu26_interspeech/markdown.md"},{"id":"hu26_interspeech","title":"ArtNet: A JEPA-Like Articulatory Predictive Framework for Robust Zero-Shot Phoneme Recognition","authors":["Zeqian Hu","Fuliang Weng","Shu Shang","Yaqian Zhou"],"year":2026,"doi":"10.21437/Interspeech.2026-304","isca_url":"https://www.isca-archive.org/interspeech_2026/hu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hu26_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Fudan University","Logos & Dialogos"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hu26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Fudan University","Logos & Dialogos"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-304","pdf":"https://www.isca-archive.org/interspeech_2026/hu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26_interspeech/markdown.md"},{"id":"hu26b_interspeech","title":"OmniCodec: Low Frame Rate Universal Audio Codec with Semantic–Acoustic Disentanglement","authors":["Jingbin Hu","Haoyu Zhang","Dake Guo","Qirui Zhan","Wenhao Li","Huakang Chen","Guobin Ma","Hanke Xie","Chengyou Wang","Pengyuan Xie","Chuan Xie","Qiang Zhang","Lei Xie"],"year":2026,"doi":"10.21437/Interspeech.2026-494","isca_url":"https://www.isca-archive.org/interspeech_2026/hu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hu26b_interspeech.pdf","session":"Neural Audio Codec Architectures","topics":["speech-llm","self-supervised","tts"],"category":"speech-coding","labels":["self-supervised","generative-model"],"institutions":["Northwestern Polytechnical University","Shanghai Lingguang Zhaxian Technology"],"code":{"url":"https://github.com/ASLP-lab/OmniCodec","stars":50,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hu26b_interspeech","category":"speech-coding","labels":["self-supervised","generative-model"],"institutions":["Northwestern Polytechnical University","Shanghai Lingguang Zhaxian Technology"],"code":"https://github.com/ASLP-lab/OmniCodec","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-494","pdf":"https://www.isca-archive.org/interspeech_2026/hu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26b_interspeech/markdown.md"},{"id":"hu26c_interspeech","title":"Personalized Keyword Spotting for User-Defined Keywords Leveraging Text-Independent Speaker Verification","authors":["Ming-Hsiang Hu","Kuan-Tang Huang","Chien-Chun Wang","Hung-Shin Lee","Berlin Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1130","isca_url":"https://www.isca-archive.org/interspeech_2026/hu26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hu26c_interspeech.pdf","session":"Information Extraction and Retrieval / Survey Talk","topics":["keyword-spotting","speaker-verification","self-supervised"],"category":"asr","institutions":["National Taiwan Normal University","E.SUN Financial Holding Co., Ltd","United Link Co., Ltd"],"funding":["Realtek Semiconductor Corporation"],"code":{"url":"https://github.com/Padawan101/ZP-KWS","stars":7,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hu26c_interspeech","category":"asr","institutions":["National Taiwan Normal University","E.SUN Financial Holding Co., Ltd","United Link Co., Ltd"],"code":"https://github.com/Padawan101/ZP-KWS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1130","pdf":"https://www.isca-archive.org/interspeech_2026/hu26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26c_interspeech/markdown.md"},{"id":"hu26d_interspeech","title":"TF-MoE: Time-Frequency Mixture-of-Experts for Efficient Speech Separation","authors":["Qinzhe Hu","Chenda Li","Wangyou Zhang","Shujie Liu","Yan Lu","Yanmin Qian"],"year":2026,"doi":"10.21437/Interspeech.2026-1307","isca_url":"https://www.isca-archive.org/interspeech_2026/hu26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hu26d_interspeech.pdf","session":"Source Separation 1","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["Shanghai Jiao Tong University","Microsoft"],"funding":["China STI 2030–Major Projects","National Natural Science Foundation of China","SJTU Med-X Translational Research Grant"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hu26d_interspeech","category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["Shanghai Jiao Tong University","Microsoft"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1307","pdf":"https://www.isca-archive.org/interspeech_2026/hu26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26d_interspeech/markdown.md"},{"id":"hu26e_interspeech","title":"Joint Fullband-Subband Modeling for High-Resolution SingFake Detection","authors":["Chia-Yu Hu","Xuanjun Chen","Sung-Feng Huang","Haibin Wu","Hung-yi Lee","Jyh-Shing Roger Jang"],"year":2026,"doi":"10.21437/Interspeech.2026-1614","isca_url":"https://www.isca-archive.org/interspeech_2026/hu26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hu26e_interspeech.pdf","session":"Spoofing, Deepfake Detection and Watermarking","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","institutions":["National Taiwan University","NVIDIA"],"funding":["Ministry of Education of Taiwan","Taiwan Centers of Excellence in Artificial Intelligence"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hu26e_interspeech","category":"deepfake-security","institutions":["National Taiwan University","NVIDIA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1614","pdf":"https://www.isca-archive.org/interspeech_2026/hu26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26e_interspeech/markdown.md"},{"id":"hu26f_interspeech","title":"ABSE-NET: A Lightweight Neural Model for Active Binaural Speech Enhancement in Open-Fit Hearing Aids","authors":["De Hu","Xue Du","Qingying Zhao","Qintuya Si"],"year":2026,"doi":"10.21437/Interspeech.2026-1660","isca_url":"https://www.isca-archive.org/interspeech_2026/hu26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hu26f_interspeech.pdf","session":"Speech Enhancement and Restoration","topics":["speech-enhancement","on-device","self-supervised"],"category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["Inner Mongolia University"],"funding":["National Natural Science Foundation of China","Natural Science Foundation of Inner Mongolia Autonomous Region"],"code":{"url":"https://github.com/Bream101/ABSE-NET","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hu26f_interspeech","category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["Inner Mongolia University"],"code":"https://github.com/Bream101/ABSE-NET","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1660","pdf":"https://www.isca-archive.org/interspeech_2026/hu26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26f_interspeech/markdown.md"},{"id":"hu26g_interspeech","title":"Singing Voice Conversion via Shared Speaker Space and Min-Pooling Adversarially Enhanced Flow Matching","authors":["Yuye Hu","Ayiduosi Tuohan","Tianqi Ning","CuiCui Zhu","Hao Huang"],"year":2026,"doi":"10.21437/Interspeech.2026-2090","isca_url":"https://www.isca-archive.org/interspeech_2026/hu26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hu26g_interspeech.pdf","session":"Singing Voice and Music Generation","topics":["voice-conversion","singing-voice","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["Xinjiang University","Xinjiang Key Laboratory of Multi-lingual Information Technology","Joint International Research Laboratory of Silk Road Multilingual Cognitive Computing"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://linoteye.github.io/minflowsvc/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hu26g_interspeech","category":"tts","labels":["generative-model"],"institutions":["Xinjiang University","Xinjiang Key Laboratory of Multi-lingual Information Technology","Joint International Research Laboratory of Silk Road Multilingual Cognitive Computing"],"code":"https://linoteye.github.io/minflowsvc/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2090","pdf":"https://www.isca-archive.org/interspeech_2026/hu26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hu26g_interspeech/markdown.md"},{"id":"huang26_interspeech","title":"MVCL-DAF++: Enhancing Multimodal Intent Recognition via Prototype-Aware Contrastive Alignment and Coarse-to-Fine Dynamic Attention Fusion","authors":["Haofeng Huang","Bin Li","Yifei Han","Long Zhang","Yangfan He","Yaxin Xue"],"year":2026,"doi":"10.21437/Interspeech.2026-267","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26_interspeech.pdf","session":"Spoken Language Understanding","topics":["spoken-language-understanding","self-supervised","multilingual"],"category":"speech-llm-dialogue","institutions":["University of Shanghai for Science and Technology","Chinese Academy of Sciences","University of Minnesota","University of Leeds"],"funding":["Shenzhen Medical Research Fund","National Key Laboratory of the CAS on Medical Imaging Science and Technology System","Key Research and Development Program of Guangdong Province","Shenzhen STIB programs","Xisike Clinical Oncology Research Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26_interspeech","category":"speech-llm-dialogue","institutions":["University of Shanghai for Science and Technology","Chinese Academy of Sciences","University of Minnesota","University of Leeds"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-267","pdf":"https://www.isca-archive.org/interspeech_2026/huang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26_interspeech/markdown.md"},{"id":"huang26b_interspeech","title":"GISNO: Neural Operator-based HRTF Personalization from 3D Meshes via Differentiable Helmholtz Rendering","authors":["Chen Huang","Lei Zhou","Hongqing Liu","Lu Gan","Liming Shi"],"year":2026,"doi":"10.21437/Interspeech.2026-366","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26b_interspeech.pdf","session":"Spatial Audio 3","topics":["paralinguistics","evaluation","self-supervised"],"category":"applications-other","institutions":["Chongqing University of Posts and Telecommunications","Brunel University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26b_interspeech","category":"applications-other","institutions":["Chongqing University of Posts and Telecommunications","Brunel University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-366","pdf":"https://www.isca-archive.org/interspeech_2026/huang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26b_interspeech/markdown.md"},{"id":"huang26c_interspeech","title":"Rethinking Entropy Minimization in Test-Time Adaptation for Autoregressive Models","authors":["Wei-Ping Huang","Chee-En Yu","Guan-Ting Lin","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-944","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26c_interspeech.pdf","session":"Robust and Efficient ASR","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["National Taiwan University"],"funding":["Ministry of Education","Taiwan Centers of Excellence in Artificial Intelligence","NTU Artificial Intelligence Center of Research Excellence"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26c_interspeech","category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["National Taiwan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-944","pdf":"https://www.isca-archive.org/interspeech_2026/huang26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26c_interspeech/markdown.md"},{"id":"huang26d_interspeech","title":"Before the Turn: Investigating Motion Cues Preceding Speech in Dyadic Interaction","authors":["Ying-Hsuan Huang","Woan-Shiuan Chien","Huan-Yu Chen","Chi-Chun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1243","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26d_interspeech.pdf","session":"Turn-taking","topics":["speech-llm","dataset","evaluation"],"category":"speech-llm-dialogue","institutions":["National Tsing Hua University","National Yang Ming Chiao Tung University"],"code":{"url":"https://github.com/Dawn2745/MCBF","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26d_interspeech","category":"speech-llm-dialogue","institutions":["National Tsing Hua University","National Yang Ming Chiao Tung University"],"code":"https://github.com/Dawn2745/MCBF","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1243","pdf":"https://www.isca-archive.org/interspeech_2026/huang26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26d_interspeech/markdown.md"},{"id":"huang26e_interspeech","title":"Stress Detection Across Daily Activities: A Context-Aware Multimodal Framework with Trajectory and Ambient Speech","authors":["Wei-Heng Huang","Woan-Shiuan Chien","Chi-Chun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1262","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26e_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["paralinguistics","multimodal","self-supervised"],"category":"paralinguistics-emotion","institutions":["National Tsing Hua University","National Yang Ming Chiao Tung University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26e_interspeech","category":"paralinguistics-emotion","institutions":["National Tsing Hua University","National Yang Ming Chiao Tung University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1262","pdf":"https://www.isca-archive.org/interspeech_2026/huang26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26e_interspeech/markdown.md"},{"id":"huang26f_interspeech","title":"CodecMOS-Accent: A MOS Benchmark of Resynthesized and TTS Speech from Neural Codecs Across English Accents","authors":["Wen-Chin Huang","Nicholas Sanders","Erica Cooper"],"year":2026,"doi":"10.21437/Interspeech.2026-1273","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26f_interspeech.pdf","session":"Speech Synthesis Evaluation 2","topics":["tts","speech-enhancement","self-supervised"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Nagoya University","University of Edinburgh","National Institute of Information and Communications Technology"],"funding":["JSPS KAKENHI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26f_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Nagoya University","University of Edinburgh","National Institute of Information and Communications Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1273","pdf":"https://www.isca-archive.org/interspeech_2026/huang26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26f_interspeech/markdown.md"},{"id":"huang26g_interspeech","title":"MSR-HuBERT: Self-supervised Pre-training for Adaptation to Multiple Sampling Rates","authors":["Zikang Huang","Meng Ge","Tianrui Wang","Xuanchen Li","Xiaobao Wang","Longbiao Wang","Jianwu Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-1354","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26g_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["self-supervised","asr","speech-enhancement"],"category":"asr","labels":["self-supervised"],"institutions":["Tianjin University","Huiyan Technology Company","Shenzhen Institute of Advanced Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26g_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Tianjin University","Huiyan Technology Company","Shenzhen Institute of Advanced Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1354","pdf":"https://www.isca-archive.org/interspeech_2026/huang26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26g_interspeech/markdown.md"},{"id":"huang26h_interspeech","title":"RAS: a Reliability Oriented Metric for Automatic Speech Recognition","authors":["Wenbin Huang","Yuhang Qiu","Bohan Li","Yiwei Guo","Jing Peng","Hankun Wang","Xie Chen","Kai Yu"],"year":2026,"doi":"10.21437/Interspeech.2026-1409","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26h_interspeech.pdf","session":"Spoken Language Processing: Evaluation and Metrics","topics":["asr","multilingual","evaluation"],"category":"asr","labels":["multilingual"],"institutions":["Shanghai Jiao Tong University"],"funding":["China NSFC Project"],"code":{"url":"https://github.com/HartmannPsi/Reliability-Aware-Score","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26h_interspeech","category":"asr","labels":["multilingual"],"institutions":["Shanghai Jiao Tong University"],"code":"https://github.com/HartmannPsi/Reliability-Aware-Score","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1409","pdf":"https://www.isca-archive.org/interspeech_2026/huang26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26h_interspeech/markdown.md"},{"id":"huang26i_interspeech","title":"Align-Consistency: Improving Non-autoregressive and Semi-supervised ASR with Consistency Regularization","authors":["Wanting Huang","Weiran Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1471","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26i_interspeech.pdf","session":"Robust ASR: Uncertainty and Confidence","topics":["asr","self-supervised"],"category":"asr","institutions":["University of Iowa"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26i_interspeech","category":"asr","institutions":["University of Iowa"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1471","pdf":"https://www.isca-archive.org/interspeech_2026/huang26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26i_interspeech/markdown.md"},{"id":"huang26j_interspeech","title":"SignMatch: Aligning Pose Latent Diffusion via Multi-dimensional Rewards for Sign Language Video Generation","authors":["Rongjie Huang","Weidong Chen","Helen Meng","Xixin Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-1546","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26j_interspeech.pdf","session":"Prosody, Pronunciation and Specialized Speech Processing","topics":["tts","self-supervised","evaluation"],"category":"applications-other","labels":["generative-model"],"institutions":["Chinese University of Hong Kong"],"funding":["Centre for Perceptual and Interactive Intelligence","InnoHK"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26j_interspeech","category":"applications-other","labels":["generative-model"],"institutions":["Chinese University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1546","pdf":"https://www.isca-archive.org/interspeech_2026/huang26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26j_interspeech/markdown.md"},{"id":"huang26k_interspeech","title":"Noise-Aware In-Context Learning for Hallucination Mitigation in ALLMs","authors":["Qixuan Huang","Khalid Zaman","Masashi Unoki"],"year":2026,"doi":"10.21437/Interspeech.2026-1610","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26k_interspeech.pdf","session":"Reasoning with Speech/Audio Language Models","topics":["speech-llm","evaluation","self-supervised"],"category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release"],"institutions":["Japan Advanced Institute of Science and Technology"],"funding":["JSPS KAKENHI"],"code":{"url":"https://github.com/OrgHuang/NAICL-Clotho1k.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26k_interspeech","category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release"],"institutions":["Japan Advanced Institute of Science and Technology"],"code":"https://github.com/OrgHuang/NAICL-Clotho1k.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1610","pdf":"https://www.isca-archive.org/interspeech_2026/huang26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26k_interspeech/markdown.md"},{"id":"huang26l_interspeech","title":"Unified Neural Speech Coding for Multiple Sampling Rates","authors":["Jiankai Huang","Junteng Zhang","Lizhong Wang","Liang Wen","Ming Lu","Zhan Ma"],"year":2026,"doi":"10.21437/Interspeech.2026-1641","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26l_interspeech.pdf","session":"Neural Speech Codecs: Low-Bitrate and Disentangled Coding","topics":["speech-coding","self-supervised","multilingual"],"category":"speech-coding","institutions":["Nanjing University","Samsung"],"funding":["National Natural Science Foundation of China","Fundamental Research Funds for the Central Universities","Interdisciplinary Research Center for Future Intelligent Chips","Yachen Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26l_interspeech","category":"speech-coding","institutions":["Nanjing University","Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1641","pdf":"https://www.isca-archive.org/interspeech_2026/huang26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26l_interspeech/markdown.md"},{"id":"huang26m_interspeech","title":"On the Robustness of Speaker Embeddings for Cross-Domain Speaker Retrieval","authors":["Chuanqi Huang","Wei Xie","Xilu Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1796","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26m_interspeech.pdf","session":"Speaker Verification: Advances in Speaker Embeddings","topics":["speaker-verification","evaluation","self-supervised"],"category":"speaker","labels":["robustness-noise"],"institutions":["Guangxi University","Guangxi Key Laboratory of Multimedia Communications and Network Technology","University of Surrey"],"funding":["Guangxi Natural Science Foundation","Guangxi Science and Technology Base and Talent Special Project","Project for Enhancing Young and Middle-aged Teacher’s Research Basis Ability in Universities of Guangxi"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26m_interspeech","category":"speaker","labels":["robustness-noise"],"institutions":["Guangxi University","Guangxi Key Laboratory of Multimedia Communications and Network Technology","University of Surrey"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1796","pdf":"https://www.isca-archive.org/interspeech_2026/huang26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26m_interspeech/markdown.md"},{"id":"huang26n_interspeech","title":"EmoEUS: Uncertainty Supervision for Multimodal Emotion Recognition in Conversation","authors":["Zilong Huang","Kong Aik Lee","Junjie Li","Zhe Li","Man-Wai Mak"],"year":2026,"doi":"10.21437/Interspeech.2026-1996","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26n_interspeech.pdf","session":"Multimodal Spoken Dialogue Systems","topics":["paralinguistics","emotion-recognition","multimodal"],"category":"paralinguistics-emotion","institutions":["Hong Kong Polytechnic University","University of Hong Kong"],"funding":["Innovation Technology Co. Ltd"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26n_interspeech","category":"paralinguistics-emotion","institutions":["Hong Kong Polytechnic University","University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1996","pdf":"https://www.isca-archive.org/interspeech_2026/huang26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26n_interspeech/markdown.md"},{"id":"huang26o_interspeech","title":"Neural Directional Coding: Joint Spatial Coding and Filtering with Configurable Directivity Patterns","authors":["Weilong Huang","Emanuël A. P. Habets"],"year":2026,"doi":"10.21437/Interspeech.2026-2234","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26o_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26o_interspeech.pdf","session":"Multi-Channel, Beamforming and Spatial Speech Enhancement","topics":["speech-enhancement","source-separation","evaluation"],"category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["International Audio Laboratories Erlangen","Fraunhofer IIS","Friedrich-Alexander-Universitat Erlangen-Nurnberg"],"funding":["German Research Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26o_interspeech","category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["International Audio Laboratories Erlangen","Fraunhofer IIS","Friedrich-Alexander-Universitat Erlangen-Nurnberg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2234","pdf":"https://www.isca-archive.org/interspeech_2026/huang26o_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26o_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26o_interspeech/markdown.md"},{"id":"huang26p_interspeech","title":"EII-SCL: Harnessing Emotional Inertia for Multimodal Emotion Recognition in Conversation","authors":["Zilong Huang","Kong Aik Lee","Chong-xin Gan","Zezhong Jin","Ruichen Zuo","Man-Wai Mak"],"year":2026,"doi":"10.21437/Interspeech.2026-3532","isca_url":"https://www.isca-archive.org/interspeech_2026/huang26p_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huang26p_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["paralinguistics","emotion-recognition","self-supervised"],"category":"paralinguistics-emotion","institutions":["Hong Kong Polytechnic University"],"funding":["Research Platform for Advanced Audio and Speech Signal Processing"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huang26p_interspeech","category":"paralinguistics-emotion","institutions":["Hong Kong Polytechnic University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3532","pdf":"https://www.isca-archive.org/interspeech_2026/huang26p_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26p_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huang26p_interspeech/markdown.md"},{"id":"huo26_interspeech","title":"Do speech foundation models really learn words?","authors":["Robin Huo","Ewan Dunbar"],"year":2026,"doi":"10.21437/Interspeech.2026-2676","isca_url":"https://www.isca-archive.org/interspeech_2026/huo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/huo26_interspeech.pdf","session":"New Architecture and Analyses for ASR and Speech LMs","topics":["self-supervised","evaluation","phonetics"],"category":"asr","labels":["self-supervised"],"institutions":["University of Toronto"],"funding":["Natural Sciences and Engineering Research Council of Canada","Data Sciences Institute","Linguistics Graduate Research Award at the University of Toronto"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"huo26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["University of Toronto"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2676","pdf":"https://www.isca-archive.org/interspeech_2026/huo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/huo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/huo26_interspeech/markdown.md"},{"id":"husain26_interspeech","title":"Beyond WER: Entity and Disfluency Recall in Accented Conversational ASR","authors":["Fiza Husain","Ankit Pandey","Yash Singh"],"year":2026,"doi":"10.21437/Interspeech.2026-786","isca_url":"https://www.isca-archive.org/interspeech_2026/husain26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/husain26_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["Stimuler"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"husain26_interspeech","category":"asr","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["Stimuler"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-786","pdf":"https://www.isca-archive.org/interspeech_2026/husain26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/husain26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/husain26_interspeech/markdown.md"},{"id":"hutchinson26_interspeech","title":"Working Together on Technologies: A Case Study of Collaboration in Aotearoa","authors":["Ben Hutchinson","Gabriella Conlon","Greg Duncum","Carrie Jones","Tāne Karamaina","Hautahi Kingi","Jamie Towler","Caroline Rainsford","Ngahiwi Apanui-Barr"],"year":2026,"doi":"10.21437/Interspeech.2026-1573","isca_url":"https://www.isca-archive.org/interspeech_2026/hutchinson26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hutchinson26_interspeech.pdf","session":"Pacific Voices: Speech Science and Technology for the Languages of the Pacific Ocean","topics":["tts","low-resource","multilingual"],"category":"tts","labels":["low-resource","multilingual","generative-model"],"institutions":["Google","Te Taura Whiri i te Reo Māori"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hutchinson26_interspeech","category":"tts","labels":["low-resource","multilingual","generative-model"],"institutions":["Google","Te Taura Whiri i te Reo Māori"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1573","pdf":"https://www.isca-archive.org/interspeech_2026/hutchinson26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hutchinson26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hutchinson26_interspeech/markdown.md"},{"id":"hwang26_interspeech","title":"MF-EDM: Graph-based Multimodal Fusion and Emotional Dynamics Modeling for Emotion Recognition in Conversation","authors":["Sooyeon Hwang","Eunseong Kwon","Gahgene Gweon"],"year":2026,"doi":"10.21437/Interspeech.2026-1875","isca_url":"https://www.isca-archive.org/interspeech_2026/hwang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/hwang26_interspeech.pdf","session":"Multimodal Spoken Dialogue Systems","topics":["speech-llm","paralinguistics","emotion-recognition"],"category":"paralinguistics-emotion","institutions":["Seoul National University"],"funding":["Institute for Information & communications Technology Promotion","Ministry of Science and ICT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"hwang26_interspeech","category":"paralinguistics-emotion","institutions":["Seoul National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1875","pdf":"https://www.isca-archive.org/interspeech_2026/hwang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/hwang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/hwang26_interspeech/markdown.md"},{"id":"ibrahimov26_interspeech","title":"On the Role of the Tongue Region in Ultrasound-to-Acoustic Mapping","authors":["Ibrahim Ibrahimov","Gábor Gosztolya","Csaba Zainkó"],"year":2026,"doi":"10.21437/Interspeech.2026-1071","isca_url":"https://www.isca-archive.org/interspeech_2026/ibrahimov26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ibrahimov26_interspeech.pdf","session":"Beyond Speech Technologies in Healthcare","topics":["speech-enhancement","evaluation","dataset"],"category":"applications-other","institutions":["Budapest University of Technology and Economics","HUN-REN-SZTE Research Group on Artificial Intelligence","University of Szeged"],"funding":["Ministry of Culture and Innovation of Hungary","National Research, Development and Innovation Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ibrahimov26_interspeech","category":"applications-other","institutions":["Budapest University of Technology and Economics","HUN-REN-SZTE Research Group on Artificial Intelligence","University of Szeged"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1071","pdf":"https://www.isca-archive.org/interspeech_2026/ibrahimov26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ibrahimov26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ibrahimov26_interspeech/markdown.md"},{"id":"ieong26_interspeech","title":"Nudging Hidden States: Training-Free Model Steering for Chain-of-Thought Reasoning in Large Audio-Language Models","authors":["Lok-Lam Ieong","Chia-Chien Chen","Chih-Kai Yang","Yu-Han Huang","An-Yu Cheng","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-554","isca_url":"https://www.isca-archive.org/interspeech_2026/ieong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ieong26_interspeech.pdf","session":"Audio Language Models: Reasoning, Reliability, and Multimodal Understanding","topics":["speech-llm","spoken-language-understanding","multilingual"],"category":"speech-llm-dialogue","institutions":["National Taiwan University","NTU Artificial Intelligence Center of Research Excellence"],"funding":["Ministry of Education","Taiwan Centers of Excellence in Artificial Intelligence project","NTU Artificial Intelligence Center of Research Excellence"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ieong26_interspeech","category":"speech-llm-dialogue","institutions":["National Taiwan University","NTU Artificial Intelligence Center of Research Excellence"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-554","pdf":"https://www.isca-archive.org/interspeech_2026/ieong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ieong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ieong26_interspeech/markdown.md"},{"id":"ijjada26_interspeech","title":"WaveNorm: A Low-Complexity Time-Domain Neural Adaptive Gain Control for Real-Time Speech Applications","authors":["Deepika Ijjada","Charan Kumar Reddy B","Ashwini Hanaganti","Priyanka Devrao Jadhav","Varsha Uppalanchi","Balaji Padmanaban","Nivedita Chennupati","Karunakar Reddy Pucchakayala","Harish Rajamani","Naveen Ambati"],"year":2026,"doi":"10.21437/Interspeech.2026-2115","isca_url":"https://www.isca-archive.org/interspeech_2026/ijjada26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ijjada26_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","on-device","self-supervised"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Meeami Technologies"],"code":{"url":"https://github.com/wavenorm123/WaveNorm_AGC","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ijjada26_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Meeami Technologies"],"code":"https://github.com/wavenorm123/WaveNorm_AGC","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2115","pdf":"https://www.isca-archive.org/interspeech_2026/ijjada26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ijjada26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ijjada26_interspeech/markdown.md"},{"id":"ijjada26b_interspeech","title":"WaveNorm: Real-Time Neural AGC for Noise-Robust Speech Enhancement on Resource-Constrained Edge Devices","authors":["Deepika Ijjada","Charan Kumar Reddy B","Ashwini Hanaganti","Priyanka Devrao Jadhav","Varsha Uppalanchi","Nivedita Chennupati","Karunakar Reddy Pucchakayala","Balaji Padmanaban","Harish Rajamani","Naveen Ambati"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/ijjada26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ijjada26b_interspeech.pdf","session":"Speech Recognition, Enhancement and Real-Time Systems","topics":["speech-enhancement","on-device","self-supervised"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time","robustness-noise"],"institutions":["Meeami Technologies"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ijjada26b_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time","robustness-noise"],"institutions":["Meeami Technologies"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/ijjada26b_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/ijjada26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ijjada26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ijjada26b_interspeech/markdown.md"},{"id":"ilerisoy26_interspeech","title":"Zero-Shot Respiratory Sound Classification through LLM-Augmented Audio-Text Alignment","authors":["Mustafa Talha İlerisoy","Hung Manh Pham","Mathias Funk","Mykola Pechenizkiy","Aaqib Saeed"],"year":2026,"doi":"10.21437/Interspeech.2026-2235","isca_url":"https://www.isca-archive.org/interspeech_2026/ilerisoy26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ilerisoy26_interspeech.pdf","session":"Multimodal and Non-Speech Healthcare Applications","topics":["zero-shot","multilingual","self-supervised"],"category":"health-clinical","labels":["low-resource","self-supervised"],"institutions":["Eindhoven University of Technology","Singapore Management University"],"funding":["NWO AiNed Fellowship Grant","Google.org","Google Cloud Research Credits program"],"code":{"url":"https://github.com/mtilerisoy/REACH","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ilerisoy26_interspeech","category":"health-clinical","labels":["low-resource","self-supervised"],"institutions":["Eindhoven University of Technology","Singapore Management University"],"code":"https://github.com/mtilerisoy/REACH","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2235","pdf":"https://www.isca-archive.org/interspeech_2026/ilerisoy26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ilerisoy26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ilerisoy26_interspeech/markdown.md"},{"id":"im26_interspeech","title":"PF-D2M: A Pose-free Diffusion Model for Universal Dance-to-Music Generation","authors":["Jaekwon Im","Natalia Polouliakh","Taketo Akama"],"year":2026,"doi":"10.21437/Interspeech.2026-248","isca_url":"https://www.isca-archive.org/interspeech_2026/im26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/im26_interspeech.pdf","session":"Streaming Speech Synthesis","topics":["speech-llm","self-supervised","dataset"],"category":"audio-understanding","labels":["generative-model"],"institutions":["KAIST","Sony Computer Science Laboratories"],"code":{"url":"https://jakeoneijk.github.io/pfd2m_project","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"im26_interspeech","category":"audio-understanding","labels":["generative-model"],"institutions":["KAIST","Sony Computer Science Laboratories"],"code":"https://jakeoneijk.github.io/pfd2m_project","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-248","pdf":"https://www.isca-archive.org/interspeech_2026/im26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/im26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/im26_interspeech/markdown.md"},{"id":"irigoyen26_interspeech","title":"Pruning as Regularization: Sensitivity-Aware One-Shot Pruning in ASR","authors":["Julian Irigoyen","Arthur Söhler","Andreas Søeborg Kirkedal"],"year":2026,"doi":"10.21437/Interspeech.2026-3411","isca_url":"https://www.isca-archive.org/interspeech_2026/irigoyen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/irigoyen26_interspeech.pdf","session":"Resource Constrained Speech Recognition","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["efficient-on-device"],"institutions":["Danske Bank","Copenhagen Business School","Jabra"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"irigoyen26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["Danske Bank","Copenhagen Business School","Jabra"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3411","pdf":"https://www.isca-archive.org/interspeech_2026/irigoyen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/irigoyen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/irigoyen26_interspeech/markdown.md"},{"id":"ishihara26_interspeech","title":"Sub-band Cepstral Analysis of Speaker-Specific Information: A Case Study of Japanese Word /saN/","authors":["Shunichi Ishihara","Frantz Clermont","Can Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-303","isca_url":"https://www.isca-archive.org/interspeech_2026/ishihara26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ishihara26_interspeech.pdf","session":"Speech and Language Representation","topics":["speaker-verification","phonetics","evaluation"],"category":"speaker","institutions":["Australian National University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ishihara26_interspeech","category":"speaker","institutions":["Australian National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-303","pdf":"https://www.isca-archive.org/interspeech_2026/ishihara26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ishihara26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ishihara26_interspeech/markdown.md"},{"id":"ishihara26b_interspeech","title":"Band-Limited Cepstral Analysis of Speaker Sensitivity in Forensic Voice Comparison","authors":["Shunichi Ishihara","Satoru Tsuge","Frantz Clermont"],"year":2026,"doi":"10.21437/Interspeech.2026-1028","isca_url":"https://www.isca-archive.org/interspeech_2026/ishihara26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ishihara26b_interspeech.pdf","session":"Speaker-Specific, Forensic and Segmental Characteristics of Typical and Atypical Speech","topics":["speaker-verification","forensic-voice-comparison","evaluation"],"category":"speaker","institutions":["Australian National University","Daido University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ishihara26b_interspeech","category":"speaker","institutions":["Australian National University","Daido University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1028","pdf":"https://www.isca-archive.org/interspeech_2026/ishihara26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ishihara26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ishihara26b_interspeech/markdown.md"},{"id":"islam26_interspeech","title":"CAPS: A Cascaded Reconstruction Model to Power Saving in Hearables Using Sub-Nyquist Sampling with Bandwidth Extension","authors":["Tarikul Islam","Sajid F. Dipto","Luke B. Baja-Ricketts","David C. Vergano","Anomadarshi Barua"],"year":2026,"doi":"10.21437/Interspeech.2026-506","isca_url":"https://www.isca-archive.org/interspeech_2026/islam26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/islam26_interspeech.pdf","session":"Multi-Channel Processing and Specialized Acquisition (UAV, Radar, Hearables)","topics":["speech-enhancement","low-resource","on-device"],"category":"speech-coding","labels":["efficient-on-device","streaming-real-time"],"institutions":["George Mason University"],"funding":["Office of Naval Research"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"islam26_interspeech","category":"speech-coding","labels":["efficient-on-device","streaming-real-time"],"institutions":["George Mason University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-506","pdf":"https://www.isca-archive.org/interspeech_2026/islam26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/islam26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/islam26_interspeech/markdown.md"},{"id":"issam26_interspeech","title":"Cross-Modal Robustness Transfer (CMRT): Training Robust Speech Translation Models Using Adversarial Text","authors":["Abderrahmane Issam","Yusuf Can Semerci","Jan Scholtes","Gerasimos Spanakis"],"year":2026,"doi":"10.21437/Interspeech.2026-2278","isca_url":"https://www.isca-archive.org/interspeech_2026/issam26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/issam26_interspeech.pdf","session":"Multilingual Speech 2","topics":["speech-translation","self-supervised","multilingual"],"category":"translation","labels":["multilingual"],"institutions":["Maastricht University"],"funding":["European Union Horizon Europe program","SURF Cooperative"],"code":{"url":"https://github.com/issam9/CMRT","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"issam26_interspeech","category":"translation","labels":["multilingual"],"institutions":["Maastricht University"],"code":"https://github.com/issam9/CMRT","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2278","pdf":"https://www.isca-archive.org/interspeech_2026/issam26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/issam26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/issam26_interspeech/markdown.md"},{"id":"ito26_interspeech","title":"Unified Prosody Restoration Using Diffusion Models for Controllable Text-to-Speech Synthesis","authors":["Yuki Ito","Junki Ohmura","Hayato Futami","Toshiyuki Sekiya","Toshiyuki Kumakura"],"year":2026,"doi":"10.21437/Interspeech.2026-2942","isca_url":"https://www.isca-archive.org/interspeech_2026/ito26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ito26_interspeech.pdf","session":"Speech Synthesis: Speech Features, Codec and Representations","topics":["tts","prosody","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["Sony Group Corporation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ito26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Sony Group Corporation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2942","pdf":"https://www.isca-archive.org/interspeech_2026/ito26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ito26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ito26_interspeech/markdown.md"},{"id":"jabeen26_interspeech","title":"The (non-)universality of prominence and Intonation Phrases: German and Hungarian listeners' perception of an unfamiliar language","authors":["Farhat Jabeen","Ákos Buza","Ella Reimann"],"year":2026,"doi":"10.21437/Interspeech.2026-707","isca_url":"https://www.isca-archive.org/interspeech_2026/jabeen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jabeen26_interspeech.pdf","session":"Prominence, Stress and Focus","topics":["phonetics","prosody","evaluation"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Bielefeld University"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jabeen26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Bielefeld University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-707","pdf":"https://www.isca-archive.org/interspeech_2026/jabeen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jabeen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jabeen26_interspeech/markdown.md"},{"id":"jain26_interspeech","title":"The Lipreading Gap: Do VSR Models Perceive Visual Speech Like Human Lipreaders?","authors":["Rishabh Jain","Naomi Harte"],"year":2026,"doi":"10.21437/Interspeech.2026-2498","isca_url":"https://www.isca-archive.org/interspeech_2026/jain26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jain26_interspeech.pdf","session":"Audio-Visual and Multimodal Perception","topics":["visual-speech-recognition","evaluation","phonetics"],"category":"asr","institutions":["Trinity College Dublin"],"funding":["Research Ireland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jain26_interspeech","category":"asr","institutions":["Trinity College Dublin"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2498","pdf":"https://www.isca-archive.org/interspeech_2026/jain26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jain26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jain26_interspeech/markdown.md"},{"id":"jang26_interspeech","title":"End-to-End Model Compression for Personalized Neural Speech Codecs","authors":["Inseon Jang","Minje Kim","Wootaek Lim","Seungkwon Beack"],"year":2026,"doi":"10.21437/Interspeech.2026-878","isca_url":"https://www.isca-archive.org/interspeech_2026/jang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jang26_interspeech.pdf","session":"Quality, Intelligibility and Evaluation of Speech and Codecs","topics":["speech-coding","self-supervised","on-device"],"category":"speech-coding","labels":["efficient-on-device"],"institutions":["Electronics and Telecommunications Research Institute","University of Illinois Urbana-Champaign"],"funding":["Electronics and Telecommunications Research Institute"],"code":{"url":"https://minjekim.com/research-projects/PNSC#interspeech2026","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jang26_interspeech","category":"speech-coding","labels":["efficient-on-device"],"institutions":["Electronics and Telecommunications Research Institute","University of Illinois Urbana-Champaign"],"code":"https://minjekim.com/research-projects/PNSC#interspeech2026","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-878","pdf":"https://www.isca-archive.org/interspeech_2026/jang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jang26_interspeech/markdown.md"},{"id":"jang26b_interspeech","title":"DP-BCT: A Dual-Path model for predicting BackChannel Timing","authors":["Jin Yea Jang","Saim Shin","Gahgene Gweon"],"year":2026,"doi":"10.21437/Interspeech.2026-3216","isca_url":"https://www.isca-archive.org/interspeech_2026/jang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jang26b_interspeech.pdf","session":"Spoken Dialogue Systems","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","institutions":["Seoul National University","Korea Electronics Technology Institute"],"funding":["Ministry of Science and ICT, Republic of Korea","Artificial Intelligence Graduate School Program, Seoul National University","Ministry of Culture, Sports and Tourism, Republic of Korea","Korea Electronics Technology Institute"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jang26b_interspeech","category":"speech-llm-dialogue","institutions":["Seoul National University","Korea Electronics Technology Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3216","pdf":"https://www.isca-archive.org/interspeech_2026/jang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jang26b_interspeech/markdown.md"},{"id":"jasinski26_interspeech","title":"From Text Metrics to Model Internals: A Study of Whisper ASR Hallucination Detection","authors":["Jan Jasiński","Mateusz Barański","Julitta Bartolewska","Marcin Witkowski","Konrad Kowalczyk"],"year":2026,"doi":"10.21437/Interspeech.2026-338","isca_url":"https://www.isca-archive.org/interspeech_2026/jasinski26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jasinski26_interspeech.pdf","session":"Robust ASR: Hallucinations and Biases","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["self-supervised"],"institutions":["AGH University of Krakow"],"funding":["National Science Centre","Excellence initiative - research university"],"code":{"url":"https://github.com/DSP-AGH/asr_hallucination_detection_prompts","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jasinski26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["AGH University of Krakow"],"code":"https://github.com/DSP-AGH/asr_hallucination_detection_prompts","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-338","pdf":"https://www.isca-archive.org/interspeech_2026/jasinski26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jasinski26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jasinski26_interspeech/markdown.md"},{"id":"jeon26_interspeech","title":"Disentangling Depression from Cognitive Decline in Elderly Speech Using Concurrent Clinical Assessments","authors":["Woori Jeon","Seunghee Ha","Sang-Kyu Lee","Ji Hye Yoon","Tae-Jin Yoon","Seung Jin Lee","Woojae Han","Jungmin So"],"year":2026,"doi":"10.21437/Interspeech.2026-1247","isca_url":"https://www.isca-archive.org/interspeech_2026/jeon26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jeon26_interspeech.pdf","session":"Clinically Useful Speech Representations 1","topics":["paralinguistics","emotion-recognition","health"],"category":"health-clinical","institutions":["Sogang University","Hallym University","Sungshin Women's University"],"funding":["Ministry of Education of the Republic of Korea","National Research Foundation of Korea","Artificial Intelligence Innovation Graduate School grant funded by the Korea government (MSIT)","National Information Society Agency funded by the Ministry of Science, ICT through the Big Data Platform and Center Construction Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jeon26_interspeech","category":"health-clinical","institutions":["Sogang University","Hallym University","Sungshin Women's University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1247","pdf":"https://www.isca-archive.org/interspeech_2026/jeon26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jeon26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jeon26_interspeech/markdown.md"},{"id":"jeon26b_interspeech","title":"Ego-Noise-Aware Spatial Filtering for Reliable UAV Audition in Extreme Low-SNR Conditions","authors":["Chanhong Jeon","Jeongmin Lee","Hyungjoo Seo","Kyuhong Shim","Taewook Kang"],"year":2026,"doi":"10.21437/Interspeech.2026-1530","isca_url":"https://www.isca-archive.org/interspeech_2026/jeon26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jeon26b_interspeech.pdf","session":"Multi-Channel Processing and Specialized Acquisition (UAV, Radar, Hearables)","topics":["speech-enhancement","spatial-filtering","adaptive-beamforming"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Sungkyunkwan University","University of Illinois Urbana-Champaign"],"funding":["National Research Foundation","IITP","KIAT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jeon26b_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Sungkyunkwan University","University of Illinois Urbana-Champaign"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1530","pdf":"https://www.isca-archive.org/interspeech_2026/jeon26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jeon26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jeon26b_interspeech/markdown.md"},{"id":"jeon26c_interspeech","title":"Not All Frames Are Equal: Difference-Aware Quantization for Ultra-Low-Bit ASR","authors":["Woori Jeon","Jungmin So"],"year":2026,"doi":"10.21437/Interspeech.2026-1569","isca_url":"https://www.isca-archive.org/interspeech_2026/jeon26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jeon26c_interspeech.pdf","session":"Resource Constrained Speech Recognition","topics":["asr","self-supervised"],"category":"asr","labels":["efficient-on-device","self-supervised"],"institutions":["Sogang University"],"funding":["Ministry of Education of the Republic of Korea","National Research Foundation of Korea","Artificial Intelligence Innovation Graduate School grant funded by the Korea government (MSIT)"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jeon26c_interspeech","category":"asr","labels":["efficient-on-device","self-supervised"],"institutions":["Sogang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1569","pdf":"https://www.isca-archive.org/interspeech_2026/jeon26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jeon26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jeon26c_interspeech/markdown.md"},{"id":"jeon26d_interspeech","title":"ParaPairAudioBench: Paralinguistic Pairwise Audio Benchmark for LALM-as-a-Judge","authors":["Jisu Jeon","Seungyeon Jwa","Joosung Lee","Jinhyeon Kim","Woojin Chung","Hwiyeol Jo","Jeonghoon Kim","Jonghyun Choi","Soyoon Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-2021","isca_url":"https://www.isca-archive.org/interspeech_2026/jeon26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jeon26d_interspeech.pdf","session":"Paralinguistics","topics":["evaluation","self-supervised","paralinguistics"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Hongik University","Seoul National University","NAVER Cloud","KAIST"],"code":{"url":"https://github.com/jsujeon/ParaPairAudioBench","stars":5,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jeon26d_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Hongik University","Seoul National University","NAVER Cloud","KAIST"],"code":"https://github.com/jsujeon/ParaPairAudioBench","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2021","pdf":"https://www.isca-archive.org/interspeech_2026/jeon26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jeon26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jeon26d_interspeech/markdown.md"},{"id":"jeong26_interspeech","title":"An Empirical Analysis of Task-Induced Encoder Bias in Fréchet Audio Distance","authors":["Wonwoo Jeong"],"year":2026,"doi":"10.21437/Interspeech.2026-1549","isca_url":"https://www.isca-archive.org/interspeech_2026/jeong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jeong26_interspeech.pdf","session":"Speech and Audio Quality Assessment","topics":["evaluation","self-supervised","speech-enhancement"],"category":"resources-evaluation","labels":["self-supervised"],"institutions":["Sogang University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jeong26_interspeech","category":"resources-evaluation","labels":["self-supervised"],"institutions":["Sogang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1549","pdf":"https://www.isca-archive.org/interspeech_2026/jeong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jeong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jeong26_interspeech/markdown.md"},{"id":"jeong26b_interspeech","title":"Cross-lingual Retrieval-Augmented Classification for Dysarthria Severity Assessment","authors":["Taeyoung Jeong","Insung Lee","Du-Seong Chang","Myoung-Wan Koo"],"year":2026,"doi":"10.21437/Interspeech.2026-2697","isca_url":"https://www.isca-archive.org/interspeech_2026/jeong26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jeong26b_interspeech.pdf","session":"Pathological Speech Assessment 4","topics":["paralinguistics","low-resource","multilingual"],"category":"health-clinical","labels":["low-resource","multilingual"],"institutions":["Sogang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jeong26b_interspeech","category":"health-clinical","labels":["low-resource","multilingual"],"institutions":["Sogang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2697","pdf":"https://www.isca-archive.org/interspeech_2026/jeong26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jeong26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jeong26b_interspeech/markdown.md"},{"id":"ji26_interspeech","title":"Automatic Curation of Large-Scale, High-Quality, Multi-Category Music Source Separation Dataset","authors":["Yu Ji","Shuo Yang","Yuetonghui Xu","Mengmei Liu","Qiang Ji","zerui Han"],"year":2026,"doi":"10.21437/Interspeech.2026-190","isca_url":"https://www.isca-archive.org/interspeech_2026/ji26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ji26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["source-separation","dataset","self-supervised"],"category":"enhancement-separation","labels":["dataset-or-benchmark-release"],"institutions":["Xiaomi","Central Conservatory of Music"],"code":{"url":"https://github.com/scottishfold0621/ACMID","stars":26,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ji26_interspeech","category":"enhancement-separation","labels":["dataset-or-benchmark-release"],"institutions":["Xiaomi","Central Conservatory of Music"],"code":"https://github.com/scottishfold0621/ACMID","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-190","pdf":"https://www.isca-archive.org/interspeech_2026/ji26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ji26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ji26_interspeech/markdown.md"},{"id":"jia26_interspeech","title":"Augmenting Dysarthric Speech Severity Assessment with MOS Supervision","authors":["Kaimeng Jia","Minzhu Tu","Zengrui Jin","Siyin Wang","Chao Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1300","isca_url":"https://www.isca-archive.org/interspeech_2026/jia26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jia26_interspeech.pdf","session":"Pathological Speech Assessment 3","topics":["paralinguistics","speech-enhancement","self-supervised"],"category":"health-clinical","labels":["low-resource"],"institutions":["Tsinghua University","Beijing University of Posts and Telecommunications"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jia26_interspeech","category":"health-clinical","labels":["low-resource"],"institutions":["Tsinghua University","Beijing University of Posts and Telecommunications"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1300","pdf":"https://www.isca-archive.org/interspeech_2026/jia26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jia26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jia26_interspeech/markdown.md"},{"id":"jia26b_interspeech","title":"Interpretable Audio Editing Evaluation via Chain-of-Thought Difference-Commonality Reasoning with Multimodal LLMs","authors":["Yuhang Jia","Xu Zhang","Yang Chen","Hui Wang","Enzhi Wang","Yong Qin"],"year":2026,"doi":"10.21437/Interspeech.2026-3176","isca_url":"https://www.isca-archive.org/interspeech_2026/jia26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jia26b_interspeech.pdf","session":"Evaluation, Benchmarking, and Reliability of Audio Systems","topics":["evaluation","speech-llm","self-supervised"],"category":"resources-evaluation","labels":["self-supervised"],"institutions":["Nankai University"],"funding":["National Key R&D Program of China","NSF China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jia26b_interspeech","category":"resources-evaluation","labels":["self-supervised"],"institutions":["Nankai University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3176","pdf":"https://www.isca-archive.org/interspeech_2026/jia26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jia26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jia26b_interspeech/markdown.md"},{"id":"jiang26_interspeech","title":"DiffRhythm 2: Efficient and High Fidelity Song Generation via Block Flow Matching","authors":["Yuepeng Jiang","Huakang Chen","Ziqian Ning","Jixun Yao","zerui Han","Di Wu","Meng Meng","Jian Luan","Zhonghua Fu","Lei Xie"],"year":2026,"doi":"10.21437/Interspeech.2026-128","isca_url":"https://www.isca-archive.org/interspeech_2026/jiang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jiang26_interspeech.pdf","session":"Generative Audio and Music","topics":["tts","speech-llm","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["Northwestern Polytechnical University","Xiaomi"],"code":{"url":"https://github.com/xiaomi-research/diffrhythm2","stars":123,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jiang26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Northwestern Polytechnical University","Xiaomi"],"code":"https://github.com/xiaomi-research/diffrhythm2","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-128","pdf":"https://www.isca-archive.org/interspeech_2026/jiang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26_interspeech/markdown.md"},{"id":"jiang26b_interspeech","title":"VoCodec: A Low-bitrate Streamable Neural Speech Codec with Voicing-driven Quantization","authors":["Xiao-Hang Jiang","Yang Ai","Rui-Chen Zheng","Lirong Dai","Zhen-Hua Ling","Ji Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-466","isca_url":"https://www.isca-archive.org/interspeech_2026/jiang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jiang26b_interspeech.pdf","session":"Neural Speech Codecs: Low-Bitrate and Disentangled Coding","topics":["speech-coding","self-supervised"],"category":"speech-coding","labels":["streaming-real-time","generative-model"],"institutions":["University of Science and Technology of China","Tsinghua University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://pb20000090.github.io/VoCodec/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jiang26b_interspeech","category":"speech-coding","labels":["streaming-real-time","generative-model"],"institutions":["University of Science and Technology of China","Tsinghua University"],"code":"https://pb20000090.github.io/VoCodec/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-466","pdf":"https://www.isca-archive.org/interspeech_2026/jiang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26b_interspeech/markdown.md"},{"id":"jiang26c_interspeech","title":"Beyond WER: A Paired Acoustic Stress Test for Ambient Clinical Scribes","authors":["Xiao-Hang Jiang","Hanjie Guo","Ying-Si Liang","Yang Ai","Zhen-Hua Ling","Lei Jiang","Zhi-Yang He"],"year":2026,"doi":"10.21437/Interspeech.2026-606","isca_url":"https://www.isca-archive.org/interspeech_2026/jiang26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jiang26c_interspeech.pdf","session":"Medical Dialogue and Conversational Understanding","topics":["asr","speech-llm","evaluation"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["University of Science and Technology of China","iFLYTEK"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jiang26c_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["University of Science and Technology of China","iFLYTEK"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-606","pdf":"https://www.isca-archive.org/interspeech_2026/jiang26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26c_interspeech/markdown.md"},{"id":"jiang26d_interspeech","title":"FreeSonic: Training-Free Temporal-Aware Decoupled Attention for Precise Audio Editing","authors":["Yuxuan Jiang","Mingyang Han","Yusheng Dai","Andong Wang","Tianhong Zhou","Jiaxin Ye","Dongxiao Wang","Haoxiang Shi","Boyu Li","Jun Song","Cheng Yu","Bo Zheng","Weibei Dou","Zehua Chen","Jun Zhu"],"year":2026,"doi":"10.21437/Interspeech.2026-1121","isca_url":"https://www.isca-archive.org/interspeech_2026/jiang26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jiang26d_interspeech.pdf","session":"Audio Foundation Models and Generation","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Tsinghua University","Alibaba Group","Monash University","Renmin University of China","Fudan University"],"funding":["Ministry of Education of China","National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jiang26d_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Tsinghua University","Alibaba Group","Monash University","Renmin University of China","Fudan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1121","pdf":"https://www.isca-archive.org/interspeech_2026/jiang26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26d_interspeech/markdown.md"},{"id":"jiang26e_interspeech","title":"Cognitive-Heuristic Guided Multimodal Data Augmentation for Alzheimer’s Disease Detection Using LLM and TTS","authors":["Yu Jiang","Cheng Gong","Bin Wen","Ruihao Jing","Tianrui Wang","Shansong Liu","Boyu Zhu","Yuheng Lu","Xuanchen Li","Xiao Wei","Chunyu Qiang","Xiao-Lei Zhang","Longbiao Wang","Jianwu Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-1724","isca_url":"https://www.isca-archive.org/interspeech_2026/jiang26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jiang26e_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["health","speech-llm","tts"],"category":"health-clinical","labels":["low-resource","generative-model"],"institutions":["Tianjin University","China Telecom"],"code":{"url":"https://jiangyu1205.github.io/Data-Augmentation-for-Alzheimer-s/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jiang26e_interspeech","category":"health-clinical","labels":["low-resource","generative-model"],"institutions":["Tianjin University","China Telecom"],"code":"https://jiangyu1205.github.io/Data-Augmentation-for-Alzheimer-s/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1724","pdf":"https://www.isca-archive.org/interspeech_2026/jiang26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26e_interspeech/markdown.md"},{"id":"jiang26f_interspeech","title":"Earnings25: A Comprehensive 500-Hour Speech Benchmark for Finance","authors":["Denglin Jiang","Haoran Zhou","Anshul Wadhawan","Brendan Fahy","Vinay Ramesh","David Weisberg","Dmitriy Derkachevskiy","Helen Sheehan","Srivas Prasad","Michele Franceschini"],"year":2026,"doi":"10.21437/Interspeech.2026-2642","isca_url":"https://www.isca-archive.org/interspeech_2026/jiang26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jiang26f_interspeech.pdf","session":"Datasets","topics":["asr","evaluation","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Bloomberg"],"code":{"url":"https://doi.org/10.5281/zenodo.18762168","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jiang26f_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Bloomberg"],"code":"https://doi.org/10.5281/zenodo.18762168","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2642","pdf":"https://www.isca-archive.org/interspeech_2026/jiang26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26f_interspeech/markdown.md"},{"id":"jiang26g_interspeech","title":"Integrating Facial Generation into Full-Duplex Spoken Dialogue Systems","authors":["Jingjing Jiang","Atsumoto Ohashi","Ryuichiro Higashinaka"],"year":2026,"doi":"10.21437/Interspeech.2026-3114","isca_url":"https://www.isca-archive.org/interspeech_2026/jiang26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jiang26g_interspeech.pdf","session":"Spoken Dialogue Systems","topics":["speech-llm","multimodal","dataset"],"category":"speech-llm-dialogue","labels":["streaming-real-time","generative-model"],"institutions":["Nagoya University"],"funding":["JST Moonshot R&D"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jiang26g_interspeech","category":"speech-llm-dialogue","labels":["streaming-real-time","generative-model"],"institutions":["Nagoya University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3114","pdf":"https://www.isca-archive.org/interspeech_2026/jiang26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26g_interspeech/markdown.md"},{"id":"jiang26h_interspeech","title":"An Ultra-Low-Bitrate Neural Speech Codec with Plain-to-Pseudo Synergistic Vector Quantization","authors":["Xiao-Hang Jiang","Yang Ai","Fei Liu","Rui-Chen Zheng","Jian-Qing Gao","Zhen-Hua Ling","Ji Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-3506","isca_url":"https://www.isca-archive.org/interspeech_2026/jiang26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jiang26h_interspeech.pdf","session":"Neural Speech Codecs: Low-Bitrate and Disentangled Coding","topics":["speech-coding","self-supervised","low-resource"],"category":"speech-coding","labels":["generative-model"],"institutions":["University of Science and Technology of China","iFLYTEK","Tsinghua University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://pb20000090.github.io/P2PSynCodec/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jiang26h_interspeech","category":"speech-coding","labels":["generative-model"],"institutions":["University of Science and Technology of China","iFLYTEK","Tsinghua University"],"code":"https://pb20000090.github.io/P2PSynCodec/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3506","pdf":"https://www.isca-archive.org/interspeech_2026/jiang26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jiang26h_interspeech/markdown.md"},{"id":"jin26_interspeech","title":"Beyond Residual Connections: Manifold-Constrained Hyper-Connections for Robust Speaker Representation Learning","authors":["Zezhong Jin","Xiaoyu Wang","Zhe Li","Chong-xin Gan","Zilong Huang","Man-Wai Mak","Kong Aik Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-634","isca_url":"https://www.isca-archive.org/interspeech_2026/jin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jin26_interspeech.pdf","session":"Speaker Recognition and Verification","topics":["speaker-verification","self-supervised"],"category":"speaker","institutions":["Hong Kong Polytechnic University","Baidu","University of Hong Kong"],"funding":["Research Grants Council of the Hong Kong SAR","The Hong Kong Polytechnic University"],"code":{"url":"https://github.com/modelscope/3D-Speaker","stars":3157,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jin26_interspeech","category":"speaker","institutions":["Hong Kong Polytechnic University","Baidu","University of Hong Kong"],"code":"https://github.com/modelscope/3D-Speaker","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-634","pdf":"https://www.isca-archive.org/interspeech_2026/jin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jin26_interspeech/markdown.md"},{"id":"jin26b_interspeech","title":"Learning to Rescale: On-the-Fly Sequence Length Adaptation in Non-Autoregressive Speech Synthesis","authors":["Jiawei Jin","Ren Wang","Zhiyu Cui","Shun Lei","Yixuan Zhou","Zhiyong Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-1069","isca_url":"https://www.isca-archive.org/interspeech_2026/jin26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jin26b_interspeech.pdf","session":"Scaling and Zero-Shot Speech Synthesis","topics":["tts","self-supervised","speech-llm"],"category":"tts","labels":["generative-model"],"institutions":["Tsinghua University","Ant Group"],"funding":["National Natural Science Foundation of China","National Social Science Foundation of China","Ant Group"],"code":{"url":"https://thuhcsi.github.io/ElasticDLM/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jin26b_interspeech","category":"tts","labels":["generative-model"],"institutions":["Tsinghua University","Ant Group"],"code":"https://thuhcsi.github.io/ElasticDLM/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1069","pdf":"https://www.isca-archive.org/interspeech_2026/jin26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jin26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jin26b_interspeech/markdown.md"},{"id":"jing26_interspeech","title":"EmoSURA: Towards Accurate Evaluation of Detailed and Long-Context Emotional Speech Captions","authors":["Xin Jing","Andreas Triantafyllopoulos","Jiadong Wang","Shahin Amiriparian","Jun Luo","Bjoern Schuller"],"year":2026,"doi":"10.21437/Interspeech.2026-1046","isca_url":"https://www.isca-archive.org/interspeech_2026/jing26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jing26_interspeech.pdf","session":"Spoken Language Processing: Evaluation and Metrics","topics":["paralinguistics","emotion-recognition","evaluation"],"category":"resources-evaluation","institutions":["Technical University of Munich","Munich Center for Machine Learning","Huawei","Imperial College London"],"code":{"url":"https://github.com/KeiKinn/EmoSURA","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jing26_interspeech","category":"resources-evaluation","institutions":["Technical University of Munich","Munich Center for Machine Learning","Huawei","Imperial College London"],"code":"https://github.com/KeiKinn/EmoSURA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1046","pdf":"https://www.isca-archive.org/interspeech_2026/jing26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jing26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jing26_interspeech/markdown.md"},{"id":"jing26b_interspeech","title":"Tongue-Shape Strategies for Standard Mandarin Retroflex Sibilants: A Preliminary Ultrasound and Unsupervised Clustering Study","authors":["Zixi Jing","C. T. Justine Hui","Karen Huang","C. I. Watson"],"year":2026,"doi":"10.21437/Interspeech.2026-1495","isca_url":"https://www.isca-archive.org/interspeech_2026/jing26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jing26b_interspeech.pdf","session":"Methods and Data for Vocal Tract Shape and Articulation Analysis","topics":["phonetics","dataset","evaluation"],"category":"phonetics-linguistics","institutions":["University of Auckland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jing26b_interspeech","category":"phonetics-linguistics","institutions":["University of Auckland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1495","pdf":"https://www.isca-archive.org/interspeech_2026/jing26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jing26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jing26b_interspeech/markdown.md"},{"id":"joo26_interspeech","title":"Cross-Lingual Compositional Learning for Code-Switched Lip Reading","authors":["Jeonghyeon Joo","Jiyoung Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1163","isca_url":"https://www.isca-archive.org/interspeech_2026/joo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/joo26_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 1","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["Ewha Womans University"],"funding":["National Research Foundation of Korea"],"code":{"url":"https://github.com/ewha-mmai/CoCoVSR","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"joo26_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["Ewha Womans University"],"code":"https://github.com/ewha-mmai/CoCoVSR","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1163","pdf":"https://www.isca-archive.org/interspeech_2026/joo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/joo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/joo26_interspeech/markdown.md"},{"id":"joshi26_interspeech","title":"IndicContextEval: A Benchmark for Evaluating Context Utilisation in Audio Large Language Models Across 8 Indic Languages","authors":["Sakshi Joshi","Dhruv Subhash Rathi","Sanskar Singh","Eldho Ittan George","R J Hari","Kaushal Bhogale","Mitesh M Khapra"],"year":2026,"doi":"10.21437/Interspeech.2026-3272","isca_url":"https://www.isca-archive.org/interspeech_2026/joshi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/joshi26_interspeech.pdf","session":"Audio Language Models","topics":["asr","multilingual","evaluation"],"category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Indian Institute of Technology Madras","Sarvam AI"],"funding":["EkStep Foundation","Nilekani Philanthropies"],"code":{"url":"https://github.com/AI4Bharat/IndicContextEval","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"joshi26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Indian Institute of Technology Madras","Sarvam AI"],"code":"https://github.com/AI4Bharat/IndicContextEval","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3272","pdf":"https://www.isca-archive.org/interspeech_2026/joshi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/joshi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/joshi26_interspeech/markdown.md"},{"id":"joyce26_interspeech","title":"Demonstration of Embedded Systems for Clinical Speech Analysis","authors":["Jeremiah B Joyce","Erik Clemens","Sanjeev Mishra","Josh Boesche","David Johnson","Marie Reyes"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/joyce26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/joyce26_interspeech.pdf","session":"Speech and Language Processing for Health and Accessibility","topics":["asr","speaker-diarization","paralinguistics"],"category":"health-clinical","labels":["efficient-on-device","streaming-real-time"],"institutions":["Mayo Clinic"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"joyce26_interspeech","category":"health-clinical","labels":["efficient-on-device","streaming-real-time"],"institutions":["Mayo Clinic"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/joyce26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/joyce26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/joyce26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/joyce26_interspeech/markdown.md"},{"id":"jung26_interspeech","title":"Listening Between the Lines: Joint Learning of ASR Embeddings and LLM-Augmented Linguistics for Dementia Detection","authors":["Olivier Jiyoun Jung","Jonghyeon Park","Myungwoo Oh"],"year":2026,"doi":"10.21437/Interspeech.2026-939","isca_url":"https://www.isca-archive.org/interspeech_2026/jung26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jung26_interspeech.pdf","session":"Pathological Speech Assessment 2","topics":["speech-llm","paralinguistic","evaluation"],"category":"health-clinical","labels":["self-supervised"],"institutions":["Ewha Womans University","NAVER Cloud"],"code":{"url":"https://github.com/vivivic/is26dementia","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jung26_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["Ewha Womans University","NAVER Cloud"],"code":"https://github.com/vivivic/is26dementia","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-939","pdf":"https://www.isca-archive.org/interspeech_2026/jung26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jung26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jung26_interspeech/markdown.md"},{"id":"jung26b_interspeech","title":"Hierarchical Permutation Consistency Learning for Self-Conditioned End-to-End Speaker Diarization","authors":["Bongsu Jung","Wooil Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-1198","isca_url":"https://www.isca-archive.org/interspeech_2026/jung26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jung26b_interspeech.pdf","session":"Speaker Diarization 2","topics":["speaker-diarization","self-supervised"],"category":"speaker","institutions":["Incheon National University"],"funding":["KOITA","MSIT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jung26b_interspeech","category":"speaker","institutions":["Incheon National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1198","pdf":"https://www.isca-archive.org/interspeech_2026/jung26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jung26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jung26b_interspeech/markdown.md"},{"id":"jung26c_interspeech","title":"Edit the Moment, Keep the Rest: Time-Localized Audio Editing via Instruction","authors":["Jinwoo Jung","Gihun Son","Won-Gook Choi","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-3044","isca_url":"https://www.isca-archive.org/interspeech_2026/jung26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/jung26c_interspeech.pdf","session":"Acoustic Signal Analysis and Generation","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Hanyang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"https://jinwoo0302.github.io/emkr-demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"jung26c_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Hanyang University"],"code":"https://jinwoo0302.github.io/emkr-demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3044","pdf":"https://www.isca-archive.org/interspeech_2026/jung26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/jung26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/jung26c_interspeech/markdown.md"},{"id":"juvekar26_interspeech","title":"Vividh-ASR: A Complexity-Tiered Benchmark and Optimization Dynamics for Robust Indic Speech Recognition","authors":["Kush Juvekar","Kavya Manohar","Aditya Srinivas Menon","Arghya Bhattacharya","Kumarmanas Nethil"],"year":2026,"doi":"10.21437/Interspeech.2026-3408","isca_url":"https://www.isca-archive.org/interspeech_2026/juvekar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/juvekar26_interspeech.pdf","session":"Multilingual & Low-Resource ASR","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["Adalat AI"],"code":{"url":"https://huggingface.co/collections/adalat-ai/vividh-asr","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"juvekar26_interspeech","category":"asr","labels":["multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["Adalat AI"],"code":"https://huggingface.co/collections/adalat-ai/vividh-asr","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3408","pdf":"https://www.isca-archive.org/interspeech_2026/juvekar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/juvekar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/juvekar26_interspeech/markdown.md"},{"id":"kachare26_interspeech","title":"ClinAware: Speech Enhancement Needs Clinical Awareness","authors":["Pramod H. Kachare","Chetana Amancharla"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/kachare26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kachare26_interspeech.pdf","session":"Speech and Language Processing for Health and Accessibility","topics":["speech-enhancement","health","evaluation"],"category":"health-clinical","institutions":["Infosys"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kachare26_interspeech","category":"health-clinical","institutions":["Infosys"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/kachare26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/kachare26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kachare26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kachare26_interspeech/markdown.md"},{"id":"kaffeza26_interspeech","title":"The Illusion of Balanced Multimodal Sentiment Analysis: Beyond the Limits of Optimization-Based Methods","authors":["Ioanna Kaffeza","Efthymios Georgiou","Alexandros Potamianos"],"year":2026,"doi":"10.21437/Interspeech.2026-2556","isca_url":"https://www.isca-archive.org/interspeech_2026/kaffeza26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kaffeza26_interspeech.pdf","session":"Audio-Visual and Multimodal Perception","topics":["paralinguistics","self-supervised","evaluation"],"category":"paralinguistics-emotion","institutions":["Mines Paris-PSL University","University of Bern","National Technical University of Athens","Archimedes AI","Synaptic Bloom"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kaffeza26_interspeech","category":"paralinguistics-emotion","institutions":["Mines Paris-PSL University","University of Bern","National Technical University of Athens","Archimedes AI","Synaptic Bloom"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2556","pdf":"https://www.isca-archive.org/interspeech_2026/kaffeza26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kaffeza26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kaffeza26_interspeech/markdown.md"},{"id":"kagoshima26_interspeech","title":"POP-SED: Prototype Orthogonal Projection for Robust Few-shot Sound Event Detection","authors":["Takehiko Kagoshima"],"year":2026,"doi":"10.21437/Interspeech.2026-153","isca_url":"https://www.isca-archive.org/interspeech_2026/kagoshima26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kagoshima26_interspeech.pdf","session":"Acoustic Event Detection 3","topics":["few-shot","sound-event-detection","self-supervised"],"category":"audio-understanding","labels":["low-resource","robustness-noise"],"institutions":["Toshiba"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kagoshima26_interspeech","category":"audio-understanding","labels":["low-resource","robustness-noise"],"institutions":["Toshiba"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-153","pdf":"https://www.isca-archive.org/interspeech_2026/kagoshima26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kagoshima26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kagoshima26_interspeech/markdown.md"},{"id":"kagotani26_interspeech","title":"Effects of distributional bias in vowels and consonants on the Speech-to-Song Illusion","authors":["Haruki Kagotani","Hiroko Terasawa","Makiko Sadakata"],"year":2026,"doi":"10.21437/Interspeech.2026-2956","isca_url":"https://www.isca-archive.org/interspeech_2026/kagotani26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kagotani26_interspeech.pdf","session":"Model of Speech Perception","topics":["paralinguistics","phonetics","prosody"],"category":"phonetics-linguistics","institutions":["University of Tsukuba","University of Amsterdam"],"funding":["JSPS KAKENHI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kagotani26_interspeech","category":"phonetics-linguistics","institutions":["University of Tsukuba","University of Amsterdam"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2956","pdf":"https://www.isca-archive.org/interspeech_2026/kagotani26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kagotani26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kagotani26_interspeech/markdown.md"},{"id":"kamath26_interspeech","title":"Sensitivity Analysis of Generative Spatial Audio Metrics : A Study on Responsiveness, Smoothness, and Symmetry","authors":["Purnima Kamath","Adrian S. Roman","Koichi Saito","Yuki Mitsufuji","Juan P. Bello"],"year":2026,"doi":"10.21437/Interspeech.2026-252","isca_url":"https://www.isca-archive.org/interspeech_2026/kamath26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kamath26_interspeech.pdf","session":"Spatial Audio 4","topics":["spatial-audio","evaluation","self-supervised"],"category":"resources-evaluation","labels":["robustness-noise"],"institutions":["New York University","Sony AI","Sony Group Corporation"],"funding":["NYU / SONY Audio Institute for Music Business and Technology"],"code":{"url":"https://github.com/pkamath2/sa_sensitivity","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kamath26_interspeech","category":"resources-evaluation","labels":["robustness-noise"],"institutions":["New York University","Sony AI","Sony Group Corporation"],"code":"https://github.com/pkamath2/sa_sensitivity","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-252","pdf":"https://www.isca-archive.org/interspeech_2026/kamath26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kamath26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kamath26_interspeech/markdown.md"},{"id":"kamel26_interspeech","title":"Spectral Masking and Interpolation Attack (SMIA): A Black-box Adversarial Attack against Voice Authentication and Anti-Spoofing Systems","authors":["Kamel Kamel","Hridoy Sankar Dutta","Keshav Sood","Sunil Aryal"],"year":2026,"doi":"10.21437/Interspeech.2026-2736","isca_url":"https://www.isca-archive.org/interspeech_2026/kamel26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kamel26_interspeech.pdf","session":"Speaker Verification and Anti-Spoofing","topics":["speech-enhancement","speaker-verification","evaluation"],"category":"deepfake-security","institutions":["Deakin University"],"funding":["Air Force Office of Scientific Research","Deakin University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kamel26_interspeech","category":"deepfake-security","institutions":["Deakin University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2736","pdf":"https://www.isca-archive.org/interspeech_2026/kamel26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kamel26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kamel26_interspeech/markdown.md"},{"id":"kando26_interspeech","title":"On the Effect of Segmentation Width and Cluster Size on Speech Resynthesis and Continuation in Generative Spoken Language Models","authors":["Shunsuke Kando","Wataru Nakata","Shinnosuke Takamichi","Yusuke Miyao"],"year":2026,"doi":"10.21437/Interspeech.2026-999","isca_url":"https://www.isca-archive.org/interspeech_2026/kando26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kando26_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["tts","self-supervised","speech-llm"],"category":"speech-llm-dialogue","labels":["self-supervised","generative-model"],"institutions":["University of Tokyo","Keio University"],"funding":["JST ACT-X","JSPS KAKENHI"],"code":{"url":"https://github.com/gifdog97/espnet/tree/master/egs2/ljspeech/tts1/myscripts","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kando26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised","generative-model"],"institutions":["University of Tokyo","Keio University"],"code":"https://github.com/gifdog97/espnet/tree/master/egs2/ljspeech/tts1/myscripts","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-999","pdf":"https://www.isca-archive.org/interspeech_2026/kando26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kando26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kando26_interspeech/markdown.md"},{"id":"kaneko26_interspeech","title":"MeanVoiceFlow2: Joint Optimization of Mean Flow and Content Encoder for Fast One-Step Zero-Shot Voice Conversion","authors":["Takuhiro Kaneko","Hirokazu Kameoka","Kou Tanaka","Yuto Kondo"],"year":2026,"doi":"10.21437/Interspeech.2026-1596","isca_url":"https://www.isca-archive.org/interspeech_2026/kaneko26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kaneko26_interspeech.pdf","session":"Voice Conversion","topics":["voice-conversion","self-supervised","on-device"],"category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["NTT"],"code":{"url":"https://www.kecl.ntt.co.jp/people/kaneko.takuhiro/projects/meanvoiceflow2/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kaneko26_interspeech","category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["NTT"],"code":"https://www.kecl.ntt.co.jp/people/kaneko.takuhiro/projects/meanvoiceflow2/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1596","pdf":"https://www.isca-archive.org/interspeech_2026/kaneko26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kaneko26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kaneko26_interspeech/markdown.md"},{"id":"kang26_interspeech","title":"Beyond Short Segments : Expanding Speaker Embeddings with Vector Archives","authors":["Hyunku Kang","Minkyu Cho","Chanwoo Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-3192","isca_url":"https://www.isca-archive.org/interspeech_2026/kang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kang26_interspeech.pdf","session":"Speaker Verification: Advances in Speaker Embeddings","topics":["speaker-verification","self-supervised"],"category":"speaker","labels":["self-supervised"],"institutions":["Korea University"],"funding":["Institute of Information Communications Technology Planning Evaluation","National Research Foundation of Korea","Ministry of Science and ICT","Ministry of SMEs and Startups","Supreme Prosecutor’s Office"],"code":{"url":"https://github.com/slp-lab-research/vam_ecapa","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kang26_interspeech","category":"speaker","labels":["self-supervised"],"institutions":["Korea University"],"code":"https://github.com/slp-lab-research/vam_ecapa","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3192","pdf":"https://www.isca-archive.org/interspeech_2026/kang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kang26_interspeech/markdown.md"},{"id":"kang26b_interspeech","title":"What Do Neural Networks Learn for TDOA Estimation? A Cross-Architecture Probing Study","authors":["Yaozhong Kang","Jiang Wang","Runwu Shi","Takeshi Ashizawa","Benjamin Yen","Kazuhiro Nakadai"],"year":2026,"doi":"10.21437/Interspeech.2026-3246","isca_url":"https://www.isca-archive.org/interspeech_2026/kang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kang26b_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["source-separation","evaluation","speech-enhancement"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Institute of Science Tokyo"],"code":{"url":"https://github.com/york1to/cross-power-is-all-you-need","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kang26b_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Institute of Science Tokyo"],"code":"https://github.com/york1to/cross-power-is-all-you-need","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3246","pdf":"https://www.isca-archive.org/interspeech_2026/kang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kang26b_interspeech/markdown.md"},{"id":"karani26_interspeech","title":"mmWave Radar Aware Dual-Conditioned GAN for Speech Reconstruction of Signals With Low SNR","authors":["JASH KARANI","Adithya Chittem","Deepan Roy","Sandeep Joshi"],"year":2026,"doi":"10.21437/Interspeech.2026-2330","isca_url":"https://www.isca-archive.org/interspeech_2026/karani26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/karani26_interspeech.pdf","session":"Multi-Channel Processing and Specialized Acquisition (UAV, Radar, Hearables)","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["Birla Institute of Technology and Science, Pilani"],"funding":["Anusandhan National Research Foundation"],"code":{"url":"https://github.com/chitadi/RADGAN","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"karani26_interspeech","category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["Birla Institute of Technology and Science, Pilani"],"code":"https://github.com/chitadi/RADGAN","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2330","pdf":"https://www.isca-archive.org/interspeech_2026/karani26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/karani26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/karani26_interspeech/markdown.md"},{"id":"kashiwagi26_interspeech","title":"Speaker-Aware Hypothesis Clustering and Merging for Target-Speaker-free and Target-Speaker Multi-Talker ASR","authors":["Yosuke Kashiwagi","Osamu Take","Hayato Futami","Emiru Tsunoo","Siddhant Arora","Shinji Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-1604","isca_url":"https://www.isca-archive.org/interspeech_2026/kashiwagi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kashiwagi26_interspeech.pdf","session":"Multi-Talker ASR & Speaker Diarization","topics":["asr","speaker-verification","self-supervised"],"category":"asr","labels":["self-supervised"],"institutions":["Sony","Carnegie Mellon University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kashiwagi26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Sony","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1604","pdf":"https://www.isca-archive.org/interspeech_2026/kashiwagi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kashiwagi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kashiwagi26_interspeech/markdown.md"},{"id":"kashyap26_interspeech","title":"Quantifying Dimensional Independence in Speech: An Information-Theoretic Framework for Disentangled Representation Learning","authors":["Bipasha Kashyap","Bjoern Schuller","Pubudu N. Pathirana"],"year":2026,"doi":"10.21437/Interspeech.2026-1654","isca_url":"https://www.isca-archive.org/interspeech_2026/kashyap26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kashyap26_interspeech.pdf","session":"Speech and Language Representation","topics":["self-supervised","evaluation","speech-llm"],"category":"resources-evaluation","institutions":["Deakin University","Technical University of Munich","Imperial College London"],"funding":["Networked Sensing and Biomedical Engineering Research Lab, Deakin University"],"code":{"url":"https://github.com/avrpixel-design/P1","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kashyap26_interspeech","category":"resources-evaluation","institutions":["Deakin University","Technical University of Munich","Imperial College London"],"code":"https://github.com/avrpixel-design/P1","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1654","pdf":"https://www.isca-archive.org/interspeech_2026/kashyap26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kashyap26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kashyap26_interspeech/markdown.md"},{"id":"kato26_interspeech","title":"Coco-VC: Degradation-Robust Streaming Voice Conversion System on the Listener Side","authors":["Ryo Kato","Ryutaro Matsunaga","Toshio Imamura","Akinori Maeda","Shuhei Takahara","Shinnosuke Takamichi"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/kato26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kato26_interspeech.pdf","session":"Speech Synthesis, Voice Conversion and Audio Generation","topics":["speech-enhancement","voice-conversion","self-supervised"],"category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model","robustness-noise"],"institutions":["SoftBank","University of Tokyo","Keio University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kato26_interspeech","category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model","robustness-noise"],"institutions":["SoftBank","University of Tokyo","Keio University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/kato26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/kato26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kato26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kato26_interspeech/markdown.md"},{"id":"kausar26_interspeech","title":"DisenEEG-Net: Disentangling EEG features via sufficient information bottleneck and adversarial learning for cross-subject auditory attention detection","authors":["Tasleem Kausar","Yuan Liao","Haoqi Hu","Siqi Cai","Haizhou Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1703","isca_url":"https://www.isca-archive.org/interspeech_2026/kausar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kausar26_interspeech.pdf","session":"Neurophysiology of Speech","topics":["asr","self-supervised","health"],"category":"asr","institutions":["Chinese University of Hong Kong, Shenzhen","Harbin Institute of Technology"],"funding":["National Natural Science Foundation of China","Shenzhen Science and Technology Program","Program for Guangdong Introducing Innovative and Enterpreneurial Teams","Shenzhen Stability Science Program","Shenzhen Key Lab of Multi-Modal Cognitive Computing","German Research Foundation","Guangdong Provincial Key Laboratory of Big Data Computing","The Chinese University of Hong Kong, Shenzhen"],"code":{"url":"https://github.com/hello1233-maker/DisenEEG-Net","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kausar26_interspeech","category":"asr","institutions":["Chinese University of Hong Kong, Shenzhen","Harbin Institute of Technology"],"code":"https://github.com/hello1233-maker/DisenEEG-Net","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1703","pdf":"https://www.isca-archive.org/interspeech_2026/kausar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kausar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kausar26_interspeech/markdown.md"},{"id":"kawamura26_interspeech","title":"PASQA: Pitch-Accent-Focused Speech Quality Assessment Model Trained on Synthetic Speech with Accent Errors","authors":["Masaya Kawamura","Yuma Shirahata","Kentaro Mitsui","Reo Shimizu"],"year":2026,"doi":"10.21437/Interspeech.2026-1662","isca_url":"https://www.isca-archive.org/interspeech_2026/kawamura26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kawamura26_interspeech.pdf","session":"Speech and Audio Quality Assessment","topics":["tts","speech-quality-estimation","self-supervised"],"category":"resources-evaluation","institutions":["LY Corporation"],"code":{"url":"https://github.com/lycorp-jp/PASQA","stars":24,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kawamura26_interspeech","category":"resources-evaluation","institutions":["LY Corporation"],"code":"https://github.com/lycorp-jp/PASQA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1662","pdf":"https://www.isca-archive.org/interspeech_2026/kawamura26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kawamura26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kawamura26_interspeech/markdown.md"},{"id":"kc26_interspeech","title":"ECAPA-TDNN-based Speaker Embedding Framework for Voice Mimicry Assessment","authors":["Bhasi K.C.","Rajeev Rajan"],"year":2026,"doi":"10.21437/Interspeech.2026-1535","isca_url":"https://www.isca-archive.org/interspeech_2026/kc26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kc26_interspeech.pdf","session":"Speech and Language Representation","topics":["speaker-verification","paralinguistics","evaluation"],"category":"speaker","institutions":["Government Engineering College Barton Hill","Government Engineering College Idukki","APJ Abdul Kalam Technological University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kc26_interspeech","category":"speaker","institutions":["Government Engineering College Barton Hill","Government Engineering College Idukki","APJ Abdul Kalam Technological University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1535","pdf":"https://www.isca-archive.org/interspeech_2026/kc26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kc26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kc26_interspeech/markdown.md"},{"id":"keetha26_interspeech","title":"Progressive Learning for Robust Speaker Representation","authors":["Nikhil Keetha","Hima Jyothi R","Nivedita Chennupati","Balaji Padmanaban","Harish Rajamani","Naveen Ambati"],"year":2026,"doi":"10.21437/Interspeech.2026-2097","isca_url":"https://www.isca-archive.org/interspeech_2026/keetha26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/keetha26_interspeech.pdf","session":"Challenge - TidyVoice Challenge: Cross-Lingual Speaker Verification","topics":["speaker-verification","speaker-diarization","multilingual"],"category":"speaker","labels":["multilingual"],"institutions":["Meeami Technologies"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"keetha26_interspeech","category":"speaker","labels":["multilingual"],"institutions":["Meeami Technologies"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2097","pdf":"https://www.isca-archive.org/interspeech_2026/keetha26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/keetha26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/keetha26_interspeech/markdown.md"},{"id":"keetha26b_interspeech","title":"A light weight Continuous Speaker Verification System for Real time Monitoring","authors":["Nikhil Keetha","Hima Jyothi R","Nivedita Chennupati","Balaji Padmanaban","Harish Rajamani","Naveen Ambati"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/keetha26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/keetha26b_interspeech.pdf","session":"Speech Recognition, Enhancement and Real-Time Systems","topics":["speaker-verification","multilingual","on-device"],"category":"speaker","labels":["efficient-on-device","streaming-real-time"],"institutions":["Meeami Technologies"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"keetha26b_interspeech","category":"speaker","labels":["efficient-on-device","streaming-real-time"],"institutions":["Meeami Technologies"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/keetha26b_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/keetha26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/keetha26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/keetha26b_interspeech/markdown.md"},{"id":"kenne26_interspeech","title":"Multi-Level Privacy-Preserving Dementia Detection from Speech via Targeted Adversarial Obfuscation and Representation Learning","authors":["Henriette Flore Kenne","Raphael Anaadumba","Mohammad Arif Ul Alam"],"year":2026,"doi":"10.21437/Interspeech.2026-2868","isca_url":"https://www.isca-archive.org/interspeech_2026/kenne26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kenne26_interspeech.pdf","session":"Speaker-Specific, Forensic and Segmental Characteristics of Typical and Atypical Speech","topics":["speech-enhancement","speaker-verification","health"],"category":"health-clinical","institutions":["University of Massachusetts Lowell"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kenne26_interspeech","category":"health-clinical","institutions":["University of Massachusetts Lowell"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2868","pdf":"https://www.isca-archive.org/interspeech_2026/kenne26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kenne26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kenne26_interspeech/markdown.md"},{"id":"kesiraju26_interspeech","title":"FLiP: Towards understanding and interpreting multimodal multilingual sentence embeddings","authors":["Santosh Kesiraju","Bolaji Yusuf","Šimon Sedláček","Oldřich Plchot","Petr Schwarz"],"year":2026,"doi":"10.21437/Interspeech.2026-3315","isca_url":"https://www.isca-archive.org/interspeech_2026/kesiraju26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kesiraju26_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["self-supervised","multilingual","speech-llm"],"category":"applications-other","labels":["multilingual","self-supervised"],"institutions":["Brno University of Technology"],"funding":["Ministry of Education, Youth and Sports of the Czech Republic","European Union"],"code":{"url":"https://github.com/BUTSpeechFIT/FLiP","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kesiraju26_interspeech","category":"applications-other","labels":["multilingual","self-supervised"],"institutions":["Brno University of Technology"],"code":"https://github.com/BUTSpeechFIT/FLiP","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3315","pdf":"https://www.isca-archive.org/interspeech_2026/kesiraju26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kesiraju26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kesiraju26_interspeech/markdown.md"},{"id":"khaloo26_interspeech","title":"Perceptual and Acoustic Correlates of Racial Identity in Text-to-Speech Voices","authors":["Noah Khaloo","Nicole Holliday","Sarah Creel"],"year":2026,"doi":"10.21437/Interspeech.2026-1479","isca_url":"https://www.isca-archive.org/interspeech_2026/khaloo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/khaloo26_interspeech.pdf","session":"Phonetic Aspects of TTS and ASR Systems","topics":["tts","paralinguistics","evaluation"],"category":"paralinguistics-emotion","institutions":["University of California, San Diego","University of California, Berkeley"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"khaloo26_interspeech","category":"paralinguistics-emotion","institutions":["University of California, San Diego","University of California, Berkeley"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1479","pdf":"https://www.isca-archive.org/interspeech_2026/khaloo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/khaloo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/khaloo26_interspeech/markdown.md"},{"id":"khan26_interspeech","title":"Dual-Branch Gated Fusion for Open-Set Audio Deepfake Source Tracing","authors":["Awais Khan","Kutub Uddin","Khalid Malik"],"year":2026,"doi":"10.21437/Interspeech.2026-3008","isca_url":"https://www.isca-archive.org/interspeech_2026/khan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/khan26_interspeech.pdf","session":"Speech Deepfake Detection: Robustness, Generalization, Attribution","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["University of Michigan","ProbeTruth Inc"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"khan26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["University of Michigan","ProbeTruth Inc"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3008","pdf":"https://www.isca-archive.org/interspeech_2026/khan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/khan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/khan26_interspeech/markdown.md"},{"id":"khan26b_interspeech","title":"I Am No One: Style-Aware Paraphrasing for Text Anonymization","authors":["Ahmed Sohair Khan","Estrid He","Monica Wachowicz","Elham Naghizade"],"year":2026,"doi":"10.21437/Interspeech.2026-3175","isca_url":"https://www.isca-archive.org/interspeech_2026/khan26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/khan26b_interspeech.pdf","session":"New Architecture and Analyses for ASR and Speech LMs","topics":["self-supervised","speech-llm","evaluation"],"category":"deepfake-security","labels":["generative-model"],"institutions":["RMIT University"],"code":{"url":"https://github.com/ahmedsohair/SAPTA26","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"khan26b_interspeech","category":"deepfake-security","labels":["generative-model"],"institutions":["RMIT University"],"code":"https://github.com/ahmedsohair/SAPTA26","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3175","pdf":"https://www.isca-archive.org/interspeech_2026/khan26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/khan26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/khan26b_interspeech/markdown.md"},{"id":"khanagha26_interspeech","title":"Your U-Net Dereverberation Model is Secretly an RIR Encoder","authors":["Sina Khanagha","Timo Gerkmann"],"year":2026,"doi":"10.21437/Interspeech.2026-2707","isca_url":"https://www.isca-archive.org/interspeech_2026/khanagha26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/khanagha26_interspeech.pdf","session":"Dereverberation, Bandwidth Extension and Restoration","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["self-supervised"],"institutions":["University of Hamburg"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"https://github.com/sp-uhh/rir-encoder","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"khanagha26_interspeech","category":"enhancement-separation","labels":["self-supervised"],"institutions":["University of Hamburg"],"code":"https://github.com/sp-uhh/rir-encoder","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2707","pdf":"https://www.isca-archive.org/interspeech_2026/khanagha26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/khanagha26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/khanagha26_interspeech/markdown.md"},{"id":"khanom26_interspeech","title":"SpiroPhonia: Non-Invasive Respiratory Health Assessment from Spontaneous Speech","authors":["Roksana Khanom","Shafia Supty","Nirupam Roy","Ashok Agrawala"],"year":2026,"doi":"10.21437/Interspeech.2026-2571","isca_url":"https://www.isca-archive.org/interspeech_2026/khanom26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/khanom26_interspeech.pdf","session":"Child Speech and Health","topics":["health","paralinguistics","evaluation"],"category":"health-clinical","institutions":["University of Maryland","DR. M R Khan Shishu Hospital & Institute of Child Health"],"funding":["National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"khanom26_interspeech","category":"health-clinical","institutions":["University of Maryland","DR. M R Khan Shishu Hospital & Institute of Child Health"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2571","pdf":"https://www.isca-archive.org/interspeech_2026/khanom26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/khanom26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/khanom26_interspeech/markdown.md"},{"id":"khaymonenko26_interspeech","title":"Scalable Keyword Spotting via Modular Network Expansion","authors":["Viktor Khaymonenko","Dzmitry Saladukha","Aliaksei Rak","Alexander Rostov"],"year":2026,"doi":"10.21437/Interspeech.2026-987","isca_url":"https://www.isca-archive.org/interspeech_2026/khaymonenko26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/khaymonenko26_interspeech.pdf","session":"Information Extraction and Retrieval","topics":["keyword-spotting","self-supervised","on-device"],"category":"asr","labels":["efficient-on-device"],"institutions":["Yandex"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"khaymonenko26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["Yandex"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-987","pdf":"https://www.isca-archive.org/interspeech_2026/khaymonenko26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/khaymonenko26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/khaymonenko26_interspeech/markdown.md"},{"id":"kheir26_interspeech","title":"DeepFense: A Unified, Modular, and Extensible Framework for Robust Audio Deepfake Detection","authors":["Yassine El Kheir","Arnab Das","Yixuan Xiao","Xin Wang","Feidi Kallel","Enes Erdem Erdogan","Ngoc Thang Vu","Tim Polzehl","Sebastian Möller"],"year":2026,"doi":"10.21437/Interspeech.2026-1366","isca_url":"https://www.isca-archive.org/interspeech_2026/kheir26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kheir26_interspeech.pdf","session":"Spoofing, Deepfake Detection and Watermarking","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["German Research Center for Artificial Intelligence","University of Stuttgart","National Institute of Informatics","Technical University of Berlin"],"funding":["Federal Ministry of Research, Technology and Space","Investitionsbank Berlin","JST PRESTO"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kheir26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["German Research Center for Artificial Intelligence","University of Stuttgart","National Institute of Informatics","Technical University of Berlin"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1366","pdf":"https://www.isca-archive.org/interspeech_2026/kheir26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kheir26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kheir26_interspeech/markdown.md"},{"id":"kheir26b_interspeech","title":"IQRA 2026: Interspeech Challenge on Automatic Assessment Pronunciation for Modern Standard Arabic (MSA)","authors":["Yassine El Kheir","Ahmed Ali","Ahmed Ali","Ahmed Ali","Ahmed Ali","Ahmed Ali","Ahmed Ali","Ahmed Ali"],"year":2026,"doi":"10.21437/Interspeech.2026-2445","isca_url":"https://www.isca-archive.org/interspeech_2026/kheir26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kheir26b_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["low-resource","multilingual","self-supervised"],"category":"applications-other","labels":["low-resource","self-supervised","dataset-or-benchmark-release"],"institutions":["German Research Center for Artificial Intelligence","Technical University of Berlin","University of Sheffield","University of New South Wales","Alexandria University","Qatar Computing Research Institute","Taibah University","HUMAIN"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kheir26b_interspeech","category":"applications-other","labels":["low-resource","self-supervised","dataset-or-benchmark-release"],"institutions":["German Research Center for Artificial Intelligence","Technical University of Berlin","University of Sheffield","University of New South Wales","Alexandria University","Qatar Computing Research Institute","Taibah University","HUMAIN"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2445","pdf":"https://www.isca-archive.org/interspeech_2026/kheir26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kheir26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kheir26b_interspeech/markdown.md"},{"id":"kilpatrick26_interspeech","title":"Word Iconicity and Phonological Surprisal as Predictors of Age of Acquisition","authors":["Alexander Kilpatrick","Rikke Bundgaard-Nielsen"],"year":2026,"doi":"10.21437/Interspeech.2026-1551","isca_url":"https://www.isca-archive.org/interspeech_2026/kilpatrick26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kilpatrick26_interspeech.pdf","session":"Modeling L1 Acquisition","topics":["phonetics","evaluation","dataset"],"category":"phonetics-linguistics","institutions":["University of Aizu","University of Melbourne"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kilpatrick26_interspeech","category":"phonetics-linguistics","institutions":["University of Aizu","University of Melbourne"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1551","pdf":"https://www.isca-archive.org/interspeech_2026/kilpatrick26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kilpatrick26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kilpatrick26_interspeech/markdown.md"},{"id":"kim26_interspeech","title":"ZipL-Dialog: Memory-Efficient Long-Form Spoken Dialog Synthesis via Latent Flow Matching","authors":["Jihwan Kim","Nam Soo Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-185","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26_interspeech.pdf","session":"Long-Form Speech Synthesis","topics":["tts","speech-llm","self-supervised"],"category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Seoul National University","KT Corporation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26_interspeech","category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Seoul National University","KT Corporation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-185","pdf":"https://www.isca-archive.org/interspeech_2026/kim26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26_interspeech/markdown.md"},{"id":"kim26b_interspeech","title":"The Role of Laryngeal Position in the Articulation of American English Velar Stop Consonants","authors":["Daejin Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-207","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26b_interspeech.pdf","session":"Speaker-Specific, Forensic and Segmental Characteristics of Typical and Atypical Speech","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["University of New Mexico"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26b_interspeech","category":"phonetics-linguistics","institutions":["University of New Mexico"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-207","pdf":"https://www.isca-archive.org/interspeech_2026/kim26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26b_interspeech/markdown.md"},{"id":"kim26c_interspeech","title":"Mixture Consistency Learning for Robust Speaker Verification in Noisy Environments","authors":["Seung-bin Kim","Chan-yeong Lim","Jungwoo Heo","Hyun-seo Shin","Kyo-Won Koo","Jisoo Son","Kyung-Wha Kim","Ha-Jin Yu"],"year":2026,"doi":"10.21437/Interspeech.2026-364","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26c_interspeech.pdf","session":"Speaker Verification: Architectures, Losses, and LLMs","topics":["speaker-verification","speech-enhancement","self-supervised"],"category":"speaker","labels":["robustness-noise"],"institutions":["University of Seoul","Supreme Prosecutor’s Office"],"funding":["Supreme Prosecutors' Office"],"code":{"url":"https://github.com/kimho1wq/MCL-SV","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26c_interspeech","category":"speaker","labels":["robustness-noise"],"institutions":["University of Seoul","Supreme Prosecutor’s Office"],"code":"https://github.com/kimho1wq/MCL-SV","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-364","pdf":"https://www.isca-archive.org/interspeech_2026/kim26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26c_interspeech/markdown.md"},{"id":"kim26d_interspeech","title":"Privacy-Preserving Speaker Verification with Multi-Granularity Feature Obfuscation","authors":["Hanseul Kim","Nam In Park","Chanjun Chun"],"year":2026,"doi":"10.21437/Interspeech.2026-437","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26d_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["speaker-verification","voice-conversion","self-supervised"],"category":"deepfake-security","institutions":["Chosun University","National Forensic Service","Glosori Inc"],"funding":["Innopolis Foundation","Commercialization Promotion Agency for R&D Outcomes"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26d_interspeech","category":"deepfake-security","institutions":["Chosun University","National Forensic Service","Glosori Inc"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-437","pdf":"https://www.isca-archive.org/interspeech_2026/kim26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26d_interspeech/markdown.md"},{"id":"kim26e_interspeech","title":"Temporal Transition-Aware Multi-Head Modeling for Partially Spoofed Audio Detection and Localization","authors":["Yunsu Kim","Juyeob Lee","Eunil Park"],"year":2026,"doi":"10.21437/Interspeech.2026-474","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26e_interspeech.pdf","session":"Spoofing, Deepfake Detection and Watermarking","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","institutions":["Sungkyunkwan University","University of Toronto","Jaume I University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea government (MSIT)"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26e_interspeech","category":"deepfake-security","institutions":["Sungkyunkwan University","University of Toronto","Jaume I University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-474","pdf":"https://www.isca-archive.org/interspeech_2026/kim26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26e_interspeech/markdown.md"},{"id":"kim26f_interspeech","title":"ArtBoost: Synthetic Articulatory Data Augmentation for Acoustic-to-Articulatory Inversion","authors":["Hyung Kyu Kim","Byungchan Hwang","Hak Gu Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-664","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26f_interspeech.pdf","session":"Modeling Articulation","topics":["asr","self-supervised","dataset"],"category":"phonetics-linguistics","labels":["low-resource"],"institutions":["Chung-Ang University"],"funding":["Ministry of Science and ICT","Institute for Information & Communications Technology Planning & Evaluation","Ministry of Culture, Sports and Tourism","Korea Creative Content Agency"],"code":{"url":"https://cau-irislab.github.io/Interspeech26-ArtBoost/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26f_interspeech","category":"phonetics-linguistics","labels":["low-resource"],"institutions":["Chung-Ang University"],"code":"https://cau-irislab.github.io/Interspeech26-ArtBoost/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-664","pdf":"https://www.isca-archive.org/interspeech_2026/kim26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26f_interspeech/markdown.md"},{"id":"kim26g_interspeech","title":"Latency-Configurable Streaming Speech Enhancement via Asymmetric Temporal Padding","authors":["Yunsik Kim","Yoonyoung Chung"],"year":2026,"doi":"10.21437/Interspeech.2026-817","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26g_interspeech.pdf","session":"Real-Time, Low-Latency and Edge Speech Enhancement","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Pohang University of Science and Technology","Intus"],"funding":["National Research Foundation of Korea","Ministry of Science and ICT","Institute of Information & Communications Technology Planning & Evaluation","Regional Innovation System & Education project","High-Performance Computing Support Project"],"code":{"url":"https://github.com/yskim3271/LaCo-SENet","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26g_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Pohang University of Science and Technology","Intus"],"code":"https://github.com/yskim3271/LaCo-SENet","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-817","pdf":"https://www.isca-archive.org/interspeech_2026/kim26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26g_interspeech/markdown.md"},{"id":"kim26h_interspeech","title":"Attention-Guided Reliability Scaling for Contrastive Decoding in Robust Audio-Visual Speech Recognition","authors":["YoungChae Kim","Da-Hee Yang","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-929","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26h_interspeech.pdf","session":"Long-form Audio & New Attention Approaches","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["robustness-noise"],"institutions":["Hanyang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea government (MSIT)","Artificial Intelligence Graduate School Program (Hanyang University)"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26h_interspeech","category":"asr","labels":["robustness-noise"],"institutions":["Hanyang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-929","pdf":"https://www.isca-archive.org/interspeech_2026/kim26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26h_interspeech/markdown.md"},{"id":"kim26i_interspeech","title":"Revisiting Label-Free Speaker Embedding Enhancement with vMF Profile Likelihood","authors":["Seunghwan Kim","Jinyong Kim","Sooyoung Yang","Youngjin Ko","Myungjoo Kang"],"year":2026,"doi":"10.21437/Interspeech.2026-1146","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26i_interspeech.pdf","session":"Speaker Verification: Advances in Speaker Embeddings","topics":["speaker-verification","self-supervised","speech-enhancement"],"category":"speaker","labels":["robustness-noise"],"institutions":["Seoul National University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea government","National Research Foundation of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26i_interspeech","category":"speaker","labels":["robustness-noise"],"institutions":["Seoul National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1146","pdf":"https://www.isca-archive.org/interspeech_2026/kim26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26i_interspeech/markdown.md"},{"id":"kim26j_interspeech","title":"MeCo: One-Step MeanFlow-based Corrector for Multi-Channel Speech Separation","authors":["Dohwan Kim","Jung-Woo Choi"],"year":2026,"doi":"10.21437/Interspeech.2026-1150","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26j_interspeech.pdf","session":"Source Separation 2","topics":["speech-enhancement","source-separation","self-supervised"],"category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["KAIST"],"funding":["National Research Foundation of Korea","Ministry of Science and ICT of Korea","Ministry of Education of Korea"],"code":{"url":"https://github.com/rlaehghks5/MECO","stars":13,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26j_interspeech","category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["KAIST"],"code":"https://github.com/rlaehghks5/MECO","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1150","pdf":"https://www.isca-archive.org/interspeech_2026/kim26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26j_interspeech/markdown.md"},{"id":"kim26k_interspeech","title":"Quality Adaptive Angular Margin Learning for Respiratory Sound Classification","authors":["Yoon Tae Kim","Heejoon Koo","Miika Toikkanen","June-Woo Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-1213","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26k_interspeech.pdf","session":"Acoustic Event Detection 3","topics":["paralinguistics","self-supervised","evaluation"],"category":"health-clinical","labels":["robustness-noise"],"institutions":["MODULABS","Wonkwang University"],"funding":["Ministry of Education","Jeonbuk State","National Research Foundation of Korea"],"code":{"url":"https://github.com/RSC-Toolkit/QLung","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26k_interspeech","category":"health-clinical","labels":["robustness-noise"],"institutions":["MODULABS","Wonkwang University"],"code":"https://github.com/RSC-Toolkit/QLung","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1213","pdf":"https://www.isca-archive.org/interspeech_2026/kim26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26k_interspeech/markdown.md"},{"id":"kim26l_interspeech","title":"Improving Generalization in Speech Deepfake Detection via Orthogonality-Constrained Common-Specific Feature Decorrelation","authors":["Donghee Kim","Wooil Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-1483","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26l_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["robustness-noise"],"institutions":["Incheon National University"],"funding":["KOITA","Ministry of Science and ICT","Incheon National University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26l_interspeech","category":"deepfake-security","labels":["robustness-noise"],"institutions":["Incheon National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1483","pdf":"https://www.isca-archive.org/interspeech_2026/kim26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26l_interspeech/markdown.md"},{"id":"kim26m_interspeech","title":"Physics-Aware Deepfake Detection via Distance–Speech Consistency","authors":["Kyeongrae Kim","Kim Sung-Bin","Oh Hyun-Bin","Tae-Hyun Oh"],"year":2026,"doi":"10.21437/Interspeech.2026-1541","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26m_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","institutions":["KAIST","POSTECH"],"funding":["IITP","Ministry of Science and ICT","KAIST Undergraduate Research Program"],"code":{"url":"https://byulharang.github.io/PADD/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26m_interspeech","category":"deepfake-security","institutions":["KAIST","POSTECH"],"code":"https://byulharang.github.io/PADD/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1541","pdf":"https://www.isca-archive.org/interspeech_2026/kim26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26m_interspeech/markdown.md"},{"id":"kim26n_interspeech","title":"Cross-Modal Consistency-Aware Structured Pruning for Efficient Speech Enhancement with Air- and Bone-Conduction Microphones","authors":["Yeeun Kim","Yonghun Song","Yunsik Kim","Yoonyoung Chung"],"year":2026,"doi":"10.21437/Interspeech.2026-1548","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26n_interspeech.pdf","session":"Multi-Channel Processing and Specialized Acquisition (UAV, Radar, Hearables)","topics":["speech-enhancement","multilingual","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Pohang University of Science and Technology","Intus"],"funding":["National Research Foundation","Institute of Information & Communications Technology Planning & Evaluation","High-Performance Computing Support Project","Regional Innovation System & Education project"],"code":{"url":"https://github.com/KYE-Postech/CCAP","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26n_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Pohang University of Science and Technology","Intus"],"code":"https://github.com/KYE-Postech/CCAP","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1548","pdf":"https://www.isca-archive.org/interspeech_2026/kim26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26n_interspeech/markdown.md"},{"id":"kim26o_interspeech","title":"AdaTT: Text-Guided Instrument Timbre Transfer with Target-Adaptive Structural Control","authors":["Dabin Kim","Junwon Lee","Juhan Nam"],"year":2026,"doi":"10.21437/Interspeech.2026-1828","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26o_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26o_interspeech.pdf","session":"Acoustic Signal Analysis and Generation","topics":["tts","speech-enhancement","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["KAIST"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea government (MSIT)","Artificial Intelligence Graduate School Program (KAIST)","National Research Foundation of Korea"],"code":{"url":"https://dabinkim0.github.io/adatt/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26o_interspeech","category":"tts","labels":["generative-model"],"institutions":["KAIST"],"code":"https://dabinkim0.github.io/adatt/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1828","pdf":"https://www.isca-archive.org/interspeech_2026/kim26o_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26o_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26o_interspeech/markdown.md"},{"id":"kim26p_interspeech","title":"SALT: Selective Allophone-Level Tokenization for Korean Text-to-Speech Synthesis","authors":["Kwangsung Kim","Cellik Adams","EunKyoung Jo"],"year":2026,"doi":"10.21437/Interspeech.2026-2055","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26p_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26p_interspeech.pdf","session":"Text Processing for Speech Synthesis","topics":["tts","low-resource","self-supervised"],"category":"tts","labels":["low-resource","generative-model"],"institutions":["CBS","Sogang University"],"funding":["Institute of Information Communications Technology Planning Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26p_interspeech","category":"tts","labels":["low-resource","generative-model"],"institutions":["CBS","Sogang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2055","pdf":"https://www.isca-archive.org/interspeech_2026/kim26p_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26p_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26p_interspeech/markdown.md"},{"id":"kim26q_interspeech","title":"A Reranker for Orchestrating Heterogeneous Speech and Text Retrievers","authors":["Inho Kim","Sumyeong Ahn"],"year":2026,"doi":"10.21437/Interspeech.2026-2154","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26q_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26q_interspeech.pdf","session":"Audio Language Models: Reasoning, Reliability, and Multimodal Understanding","topics":["speech-llm","dataset","evaluation"],"category":"speech-llm-dialogue","institutions":["Korea Institute of Energy Technology"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26q_interspeech","category":"speech-llm-dialogue","institutions":["Korea Institute of Energy Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2154","pdf":"https://www.isca-archive.org/interspeech_2026/kim26q_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26q_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26q_interspeech/markdown.md"},{"id":"kim26r_interspeech","title":"GradHarmony: A Gradient Alignment and Magnitude Normalization Strategy for Audio Deepfake Detection","authors":["Inho Kim","Thien-Phuc Doan","Souhwan Jung"],"year":2026,"doi":"10.21437/Interspeech.2026-2216","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26r_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26r_interspeech.pdf","session":"Spoofing and Deepfake Detection 3","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["robustness-noise"],"institutions":["Soongsil University"],"funding":["Korea Institute of Police Technology","Korean National Police Agency","National Research Foundation of Korea","Ministry of Science and ICT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26r_interspeech","category":"deepfake-security","labels":["robustness-noise"],"institutions":["Soongsil University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2216","pdf":"https://www.isca-archive.org/interspeech_2026/kim26r_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26r_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26r_interspeech/markdown.md"},{"id":"kim26s_interspeech","title":"Do Modern Video-LLMs Need to Listen? A Benchmark Audit and Scalable Remedy","authors":["Geewook Kim","Minjoon Seo"],"year":2026,"doi":"10.21437/Interspeech.2026-2532","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26s_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26s_interspeech.pdf","session":"Audio-Visual Grounding, Synchronization & Video Understanding","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","labels":["efficient-on-device"],"institutions":["NAVER Cloud","KAIST"],"code":{"url":"https://github.com/naver-ai/unimambamia-av","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26s_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device"],"institutions":["NAVER Cloud","KAIST"],"code":"https://github.com/naver-ai/unimambamia-av","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2532","pdf":"https://www.isca-archive.org/interspeech_2026/kim26s_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26s_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26s_interspeech/markdown.md"},{"id":"kim26t_interspeech","title":"Fast Speech Foundation Model Distillation Using Interleaved Stacking","authors":["Eungbeom Kim","Kyogu Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-3071","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26t_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26t_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["self-supervised","speech-llm","low-resource"],"category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised"],"institutions":["Seoul National University"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation","National IT Industry Promotion Agency"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26t_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised"],"institutions":["Seoul National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3071","pdf":"https://www.isca-archive.org/interspeech_2026/kim26t_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26t_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26t_interspeech/markdown.md"},{"id":"kim26u_interspeech","title":"ETC-TTS: Emotion Trajectory Learning for Controllable Emotional Text-to-Speech","authors":["Gaeun Kim","Jaeuk Lee","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-3088","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26u_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26u_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["tts","paralinguistics","emotion-recognition"],"category":"tts","labels":["generative-model"],"institutions":["Hanyang University"],"funding":["Institute of Information & communications Technology Planning & Evaluation (IITP)","Korea government (MSIT)","Artificial Intelligence Graduate School Program (Hanyang University)"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26u_interspeech","category":"tts","labels":["generative-model"],"institutions":["Hanyang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3088","pdf":"https://www.isca-archive.org/interspeech_2026/kim26u_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26u_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26u_interspeech/markdown.md"},{"id":"kim26v_interspeech","title":"SDP-Codec: A Speaker-Decoupled Speech Codec with Pitch Injection for Low-Bitrate Coding and Zero-Shot Voice Conversion","authors":["Hounsu Kim","Juhan Nam"],"year":2026,"doi":"10.21437/Interspeech.2026-3108","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26v_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26v_interspeech.pdf","session":"Neural Audio Codec Architectures","topics":["speech-codec","voice-conversion","self-supervised"],"category":"speech-coding","labels":["self-supervised","generative-model"],"institutions":["KAIST"],"funding":["National Research Foundation of Korea"],"code":{"url":"https://github.com/hanshounsu/sdpcodec-open/","stars":9,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26v_interspeech","category":"speech-coding","labels":["self-supervised","generative-model"],"institutions":["KAIST"],"code":"https://github.com/hanshounsu/sdpcodec-open/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3108","pdf":"https://www.isca-archive.org/interspeech_2026/kim26v_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26v_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26v_interspeech/markdown.md"},{"id":"kim26w_interspeech","title":"Scaling Self-Supervised Speech Models Uncovers Deep Linguistic Relationships: Evidence from the Pacific Cluster","authors":["Minu Kim","Hoirin Kim","David R. Mortensen"],"year":2026,"doi":"10.21437/Interspeech.2026-3205","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26w_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26w_interspeech.pdf","session":"Pacific Voices: Speech Science and Technology for the Languages of the Pacific Ocean","topics":["self-supervised","multilingual","asr"],"category":"speaker","labels":["low-resource","multilingual","self-supervised"],"institutions":["KAIST","University of Southern California","Carnegie Mellon University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26w_interspeech","category":"speaker","labels":["low-resource","multilingual","self-supervised"],"institutions":["KAIST","University of Southern California","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3205","pdf":"https://www.isca-archive.org/interspeech_2026/kim26w_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26w_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26w_interspeech/markdown.md"},{"id":"kim26x_interspeech","title":"A Hierarchical Feature Engineering Framework for Automated Classification of Phonotraumatic and Non-Phonotraumatic Vocal Hyperfunction","authors":["June-Woo Kim","Kangwook Jang","Minu Kim","Hyunju Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-3437","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26x_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26x_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["paralinguistics","health","evaluation"],"category":"health-clinical","institutions":["Wonkwang University","Gwangju Institute of Science and Technology","KAIST"],"funding":["InnoCORE program of the Ministry of Science and ICT","Regional Innovation System and Education program through the Jeonbuk RISE Center"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26x_interspeech","category":"health-clinical","institutions":["Wonkwang University","Gwangju Institute of Science and Technology","KAIST"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3437","pdf":"https://www.isca-archive.org/interspeech_2026/kim26x_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26x_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26x_interspeech/markdown.md"},{"id":"kim26y_interspeech","title":"VividAC: Visually Informed and Visually Interacted Audio Captioning for Enhancing Audio-Visual Question Answering","authors":["Mingi Kim","Jaehoon Go","Jinkwon Hwang","Sangyeon Cho","Sunjae Yoon","Junyeong Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-3442","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26y_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26y_interspeech.pdf","session":"Multimodal Spoken Dialogue Systems","topics":["speech-llm","evaluation","multilingual"],"category":"audio-understanding","institutions":["Chung-Ang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","National Research Foundation of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26y_interspeech","category":"audio-understanding","institutions":["Chung-Ang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3442","pdf":"https://www.isca-archive.org/interspeech_2026/kim26y_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26y_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26y_interspeech/markdown.md"},{"id":"kim26z_interspeech","title":"AudioGround: Fine-Grained Temporal Grounding in Audio via Deterministic Boundary Supervision","authors":["Mingi Kim","Minchol Kwon","Junyeong Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-3467","isca_url":"https://www.isca-archive.org/interspeech_2026/kim26z_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kim26z_interspeech.pdf","session":"Reasoning with Speech/Audio Language Models","topics":["speech-llm","evaluation","dataset"],"category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release"],"institutions":["Chung-Ang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea government"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kim26z_interspeech","category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release"],"institutions":["Chung-Ang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3467","pdf":"https://www.isca-archive.org/interspeech_2026/kim26z_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26z_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kim26z_interspeech/markdown.md"},{"id":"kirby26_interspeech","title":"Perceptual compensation for tonal context in self-supervised speech models","authors":["James Kirby","Ioana Krehan","Michele Gubian"],"year":2026,"doi":"10.21437/Interspeech.2026-2409","isca_url":"https://www.isca-archive.org/interspeech_2026/kirby26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kirby26_interspeech.pdf","session":"Tones","topics":["self-supervised","evaluation","phonetics"],"category":"phonetics-linguistics","labels":["self-supervised"],"institutions":["LMU Munich"],"code":{"url":"https://github.com/kehanlu/mandarin-wav2vec2","stars":44,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kirby26_interspeech","category":"phonetics-linguistics","labels":["self-supervised"],"institutions":["LMU Munich"],"code":"https://github.com/kehanlu/mandarin-wav2vec2","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2409","pdf":"https://www.isca-archive.org/interspeech_2026/kirby26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kirby26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kirby26_interspeech/markdown.md"},{"id":"kirkham26_interspeech","title":"PyPhonPlan: Simulating phonetic planning with dynamic neural fields and task dynamics","authors":["Sam Kirkham"],"year":2026,"doi":"10.21437/Interspeech.2026-1804","isca_url":"https://www.isca-archive.org/interspeech_2026/kirkham26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kirkham26_interspeech.pdf","session":"Modeling Articulation","topics":["speech-production","prosody","self-supervised"],"category":"phonetics-linguistics","institutions":["Lancaster University"],"funding":["Arts and Humanities Research Council","The Royal Society","The Leverhulme Trust"],"code":{"url":"https://github.com/samkirkham/PyPhonPlan","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kirkham26_interspeech","category":"phonetics-linguistics","institutions":["Lancaster University"],"code":"https://github.com/samkirkham/PyPhonPlan","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1804","pdf":"https://www.isca-archive.org/interspeech_2026/kirkham26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kirkham26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kirkham26_interspeech/markdown.md"},{"id":"kishi26_interspeech","title":"Do speech foundation models perceive speaker similarity as humans do?","authors":["Minoru Kishi","Hayato Yagi","Shinnosuke Takamichi","Yuki Saito"],"year":2026,"doi":"10.21437/Interspeech.2026-1172","isca_url":"https://www.isca-archive.org/interspeech_2026/kishi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kishi26_interspeech.pdf","session":"Speech Production and Perception 1","topics":["speaker-verification","self-supervised","evaluation"],"category":"speaker","labels":["self-supervised"],"institutions":["Keio University","University of Tokyo"],"funding":["JST FOREST","JSPS KAKENHI","Moonshot R&D"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kishi26_interspeech","category":"speaker","labels":["self-supervised"],"institutions":["Keio University","University of Tokyo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1172","pdf":"https://www.isca-archive.org/interspeech_2026/kishi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kishi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kishi26_interspeech/markdown.md"},{"id":"kiyama26_interspeech","title":"Voice Onset Time Categorical Perception in Mandarin-Speaking People Who Stutter: A Zoom-In Nonword Study","authors":["Yusuke Kiyama","Xiyu Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-2907","isca_url":"https://www.isca-archive.org/interspeech_2026/kiyama26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kiyama26_interspeech.pdf","session":"Speech, Voice and Language Disorders","topics":["phonetics","evaluation","paralinguistics"],"category":"phonetics-linguistics","labels":["streaming-real-time"],"institutions":["Peking University"],"funding":["National Social Science Fund of China","Beijing Social Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kiyama26_interspeech","category":"phonetics-linguistics","labels":["streaming-real-time"],"institutions":["Peking University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2907","pdf":"https://www.isca-archive.org/interspeech_2026/kiyama26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kiyama26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kiyama26_interspeech/markdown.md"},{"id":"klement26_interspeech","title":"Analysing Adversarial Priors for Data-driven Unsupervised Speech Enhancement","authors":["Dominik Klement","Matthew Maciejewski","Sanjeev Khudanpur","Honza Černocký","Lukáš Burget"],"year":2026,"doi":"10.21437/Interspeech.2026-2511","isca_url":"https://www.isca-archive.org/interspeech_2026/klement26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/klement26_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","self-supervised","source-separation"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Brno University of Technology","Johns Hopkins University"],"funding":["National Science Foundation","Technology Agency of the Czech Republic","Czech Ministry of Education, Youth and Sports"],"code":{"url":"https://github.com/BUTSpeechFIT/USE_DDP","stars":9,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"klement26_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Brno University of Technology","Johns Hopkins University"],"code":"https://github.com/BUTSpeechFIT/USE_DDP","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2511","pdf":"https://www.isca-archive.org/interspeech_2026/klement26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/klement26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/klement26_interspeech/markdown.md"},{"id":"klimi26_interspeech","title":"Beyond Standard Greek: Adapting Whisper for Greek Dialects through Curriculum Multitask Learning","authors":["Antigoni Klimi","Dimitrios Damianos","Stavros Bompolas","Vivian Stamou","Stella Markantonatou","Vassilis Katsouros","Georgios Paraskevopoulos"],"year":2026,"doi":"10.21437/Interspeech.2026-2567","isca_url":"https://www.isca-archive.org/interspeech_2026/klimi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/klimi26_interspeech.pdf","session":"Multilingual & Low-Resource ASR","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Athena Research Center"],"funding":["European High-Performance Computing Joint Undertaking","Greek Ministry of Digital Governance and Artificial Intelligence"],"code":{"url":"https://github.com/athena-ilsp/Dialect-Adaptation","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"klimi26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Athena Research Center"],"code":"https://github.com/athena-ilsp/Dialect-Adaptation","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2567","pdf":"https://www.isca-archive.org/interspeech_2026/klimi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/klimi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/klimi26_interspeech/markdown.md"},{"id":"ko26_interspeech","title":"Mispronunciation Modeling via PPG-Based Phone Editing: A Data Augmentation Framework for Dysarthric Speech Recognition","authors":["Tsai-Hsiu Ko","You-Cyuan Jhuo","Chung-Hsien Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-891","isca_url":"https://www.isca-archive.org/interspeech_2026/ko26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ko26_interspeech.pdf","session":"Speech, Voice and Language Disorders","topics":["asr","voice-conversion","dataset"],"category":"asr","labels":["low-resource"],"institutions":["National Cheng Kung University"],"funding":["National Science and Technology Council of Taiwan"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ko26_interspeech","category":"asr","labels":["low-resource"],"institutions":["National Cheng Kung University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-891","pdf":"https://www.isca-archive.org/interspeech_2026/ko26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ko26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ko26_interspeech/markdown.md"},{"id":"ko26b_interspeech","title":"SCOLoRA: Similarity Conditioned Signed Orthogonal LoRA for Continual Speaker Adaptation","authors":["Ye-Eun Ko","Jae-Hong Lee","Dong-Hyun Kim","Jin-Seong Choi","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-3243","isca_url":"https://www.isca-archive.org/interspeech_2026/ko26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ko26b_interspeech.pdf","session":"ASR Under Real-World Constraints: Streaming, Adaptation, and Efficiency","topics":["asr","speaker-verification","self-supervised"],"category":"asr","institutions":["Hanyang University","Hankuk University of Foreign Studies"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ko26b_interspeech","category":"asr","institutions":["Hanyang University","Hankuk University of Foreign Studies"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3243","pdf":"https://www.isca-archive.org/interspeech_2026/ko26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ko26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ko26b_interspeech/markdown.md"},{"id":"koch26_interspeech","title":"Collecting Prosody in the Wild: A Content-Controlled, Privacy-First Smartphone Protocol and Empirical Evaluation","authors":["Timo K. Koch","Florian Bemmann","Ramona Schoedel","Markus Buehner","Clemens Stachl"],"year":2026,"doi":"10.21437/Interspeech.2026-2417","isca_url":"https://www.isca-archive.org/interspeech_2026/koch26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/koch26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["paralinguistic","self-supervised","evaluation"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["University of St. Gallen","LMU Munich","University of Mannheim","Charlotte Fresenius Hochschule"],"funding":["Leibniz Institute for Psychology","Swiss National Science Foundation","German Academic Scholarship Foundation"],"code":{"url":"https://github.com/Timo-Ko/prosody_in_the_wild","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"koch26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["University of St. Gallen","LMU Munich","University of Mannheim","Charlotte Fresenius Hochschule"],"code":"https://github.com/Timo-Ko/prosody_in_the_wild","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2417","pdf":"https://www.isca-archive.org/interspeech_2026/koch26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/koch26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/koch26_interspeech/markdown.md"},{"id":"koguchi26_interspeech","title":"Instantaneous Pitch Estimation via Wave-U-Net-Based Fundamental Waveform Enhancement","authors":["Junya Koguchi","Tomoki Koriyama"],"year":2026,"doi":"10.21437/Interspeech.2026-3202","isca_url":"https://www.isca-archive.org/interspeech_2026/koguchi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/koguchi26_interspeech.pdf","session":"Audio signal analysis","topics":["paralinguistics","self-supervised","speech-enhancement"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["CyberAgent"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"koguchi26_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["CyberAgent"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3202","pdf":"https://www.isca-archive.org/interspeech_2026/koguchi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/koguchi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/koguchi26_interspeech/markdown.md"},{"id":"koh26_interspeech","title":"A Multi-Agent Framework to Automate Feedback Generation for IELTS Speaking Test using Multimodal SpeechLMs","authors":["Hui Xin Koh","Yang Wang","Chenghua Lin"],"year":2026,"doi":"10.21437/Interspeech.2026-1417","isca_url":"https://www.isca-archive.org/interspeech_2026/koh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/koh26_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["speech-llm","spoken-language-understanding","evaluation"],"category":"applications-other","labels":["self-supervised"],"institutions":["University of Manchester"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"koh26_interspeech","category":"applications-other","labels":["self-supervised"],"institutions":["University of Manchester"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1417","pdf":"https://www.isca-archive.org/interspeech_2026/koh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/koh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/koh26_interspeech/markdown.md"},{"id":"kolani26_interspeech","title":"Phonikud: Overcoming Phonetic Underspecification for Hebrew Text-To-Speech","authors":["Yakov Kolani","Maxim Melichov","Cobi Calev","Morris Alper"],"year":2026,"doi":"10.21437/Interspeech.2026-604","isca_url":"https://www.isca-archive.org/interspeech_2026/kolani26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kolani26_interspeech.pdf","session":"Text Processing for Speech Synthesis","topics":["tts","asr","low-resource"],"category":"tts","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Reichman University","Cisco Systems","Tel Aviv University","Carnegie Mellon University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kolani26_interspeech","category":"tts","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Reichman University","Cisco Systems","Tel Aviv University","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-604","pdf":"https://www.isca-archive.org/interspeech_2026/kolani26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kolani26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kolani26_interspeech/markdown.md"},{"id":"kolos26_interspeech","title":"Controlled Generation of Synthetic Speaker Vectors for Voice Anonymization","authors":["Ekaterina Kolos","Sarina Meyer","Ngoc Thang Vu"],"year":2026,"doi":"10.21437/Interspeech.2026-1464","isca_url":"https://www.isca-archive.org/interspeech_2026/kolos26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kolos26_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["speech-enhancement","self-supervised","evaluation"],"category":"deepfake-security","labels":["generative-model"],"institutions":["University of Stuttgart"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"https://github.com/katja-kolos/synthetic-speaker-vectors","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kolos26_interspeech","category":"deepfake-security","labels":["generative-model"],"institutions":["University of Stuttgart"],"code":"https://github.com/katja-kolos/synthetic-speaker-vectors","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1464","pdf":"https://www.isca-archive.org/interspeech_2026/kolos26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kolos26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kolos26_interspeech/markdown.md"},{"id":"koluguri26_interspeech","title":"Preference-ASR: A Preference-Aware Test Set for Benchmarking ASR in the Era of Speech LLMs","authors":["Nithin Rao Koluguri","Sasha Meister","Nikolay Karpov","Piotr Żelasko","Desh Raj","Jagadeesh Balam","Boris Ginsburg"],"year":2026,"doi":"10.21437/Interspeech.2026-728","isca_url":"https://www.isca-archive.org/interspeech_2026/koluguri26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/koluguri26_interspeech.pdf","session":"Speech Benchmarks, Evaluation, and Resources","topics":["asr","speech-llm","evaluation"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["NVIDIA"],"code":{"url":"https://github.com/nithinraok/preference-asr-bench","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"koluguri26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["NVIDIA"],"code":"https://github.com/nithinraok/preference-asr-bench","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-728","pdf":"https://www.isca-archive.org/interspeech_2026/koluguri26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/koluguri26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/koluguri26_interspeech/markdown.md"},{"id":"koo26_interspeech","title":"VeRe-Flow: Guiding Flow Matching toward Clean Speech via Velocity Contrastive Regularization and Representation Alignment for Noise-Robust Bandwidth Expansion","authors":["Sujin Koo","Sangyoon Kim","Ji Sub Um","Hoirin Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-712","isca_url":"https://www.isca-archive.org/interspeech_2026/koo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/koo26_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["MAGO","KAIST"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"https://vere-flow.github.io/VeRe-Flow-Demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"koo26_interspeech","category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["MAGO","KAIST"],"code":"https://vere-flow.github.io/VeRe-Flow-Demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-712","pdf":"https://www.isca-archive.org/interspeech_2026/koo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/koo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/koo26_interspeech/markdown.md"},{"id":"kopar26_interspeech","title":"Beyond Binary: Speech Representations Across the Cognitive Score Hierarchy","authors":["Serli Kopar","Roshan P. Rane","Christian Mychajliw","Lydia Federmann","Gerhard Eschweiler","Daniela Berg","Sam Gijsen","Paula Andrea Pérez-Toro","Kerstin Ritter"],"year":2026,"doi":"10.21437/Interspeech.2026-2725","isca_url":"https://www.isca-archive.org/interspeech_2026/kopar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kopar26_interspeech.pdf","session":"Clinically Useful Speech Representations 1","topics":["paralinguistics","self-supervised","health"],"category":"health-clinical","labels":["self-supervised"],"institutions":["Hertie Institute for AI in Brain Health","Tübingen AI Center","Humboldt-Universität zu Berlin","Tübingen University Hospital","Tübingen Center for Mental Health","German Center for Mental Health","University Medical Center Schleswig-Holstein","Hertie Institute for Clinical Brain Research","Friedrich-Alexander-Universität Erlangen-Nürnberg","Charité–Universitätsmedizin"],"funding":["Gemeinnützigen Hertie-Stiftung","Deutsche Forschungsgemeinschaft","Germany's Excellence Strategy","International Max Planck Research School for Intelligent Systems"],"code":{"url":"https://github.com/neselidondurma/beyond-binary-mci","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kopar26_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["Hertie Institute for AI in Brain Health","Tübingen AI Center","Humboldt-Universität zu Berlin","Tübingen University Hospital","Tübingen Center for Mental Health","German Center for Mental Health","University Medical Center Schleswig-Holstein","Hertie Institute for Clinical Brain Research","Friedrich-Alexander-Universität Erlangen-Nürnberg","Charité–Universitätsmedizin"],"code":"https://github.com/neselidondurma/beyond-binary-mci","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2725","pdf":"https://www.isca-archive.org/interspeech_2026/kopar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kopar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kopar26_interspeech/markdown.md"},{"id":"kopuklu26_interspeech","title":"RT-ASDNet: Unified, Real-Time Active Speaker Detection","authors":["Okan Köpüklü"],"year":2026,"doi":"10.21437/Interspeech.2026-448","isca_url":"https://www.isca-archive.org/interspeech_2026/kopuklu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kopuklu26_interspeech.pdf","session":"Speaker Diarization 1","topics":["speaker-diarization","self-supervised"],"category":"speaker","labels":["efficient-on-device","streaming-real-time"],"institutions":["Microsoft"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kopuklu26_interspeech","category":"speaker","labels":["efficient-on-device","streaming-real-time"],"institutions":["Microsoft"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-448","pdf":"https://www.isca-archive.org/interspeech_2026/kopuklu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kopuklu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kopuklu26_interspeech/markdown.md"},{"id":"kordt26_interspeech","title":"Learning to Hear Hesitation: Continual Learning for Disfluency-Aware ASR","authors":["Henri-Leon Kordt","Theresa Pekarek Rosin","Jae Hee Lee","Stefan Wermter"],"year":2026,"doi":"10.21437/Interspeech.2026-2080","isca_url":"https://www.isca-archive.org/interspeech_2026/kordt26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kordt26_interspeech.pdf","session":"Speech, Voice and Language Disorders","topics":["asr","self-supervised","health"],"category":"asr","labels":["self-supervised"],"institutions":["University of Hamburg"],"funding":["Horizon Europe","German Research Foundation","National Institutes of Health"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kordt26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["University of Hamburg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2080","pdf":"https://www.isca-archive.org/interspeech_2026/kordt26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kordt26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kordt26_interspeech/markdown.md"},{"id":"koriyama26_interspeech","title":"Benchmarking Large Language Models for Grapheme-to-Phoneme Conversion: A Japanese Case Study","authors":["Tomoki Koriyama"],"year":2026,"doi":"10.21437/Interspeech.2026-1800","isca_url":"https://www.isca-archive.org/interspeech_2026/koriyama26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/koriyama26_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["tts","evaluation","speech-llm"],"category":"tts","institutions":["CyberAgent"],"code":{"url":"https://github.com/CyberAgentAILab/jvs_nonpara_kana","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"koriyama26_interspeech","category":"tts","institutions":["CyberAgent"],"code":"https://github.com/CyberAgentAILab/jvs_nonpara_kana","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1800","pdf":"https://www.isca-archive.org/interspeech_2026/koriyama26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/koriyama26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/koriyama26_interspeech/markdown.md"},{"id":"koshino26_interspeech","title":"Automatic generation of audio comic from manga images","authors":["Sota Koshino","Shotaro Ueji","Shinnosuke Takamichi","Tomohiko Nakamura"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/koshino26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/koshino26_interspeech.pdf","session":"Speech Synthesis, Voice Conversion and Audio Generation","topics":["speech-synthesis","speech-llm","multimodal"],"category":"tts","labels":["generative-model"],"institutions":["Keio University","National Institute of Advanced Industrial Science and Technology"],"funding":["AIST policy-budget project \"Research and Development of Generative AI Foundation Models for the Physical Domain\"","JSPS KAKENHI","JST FOREST Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"koshino26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Keio University","National Institute of Advanced Industrial Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/koshino26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/koshino26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/koshino26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/koshino26_interspeech/markdown.md"},{"id":"kostenok26_interspeech","title":"Calibration-Reasoning Framework for Descriptive Speech Quality Assessment","authors":["Elizaveta Kostenok","Mathieu Salzmann","Milos Cernak"],"year":2026,"doi":"10.21437/Interspeech.2026-2362","isca_url":"https://www.isca-archive.org/interspeech_2026/kostenok26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kostenok26_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["speech-enhancement","evaluation","self-supervised"],"category":"resources-evaluation","institutions":["EPFL","Logitech"],"code":{"url":"https://github.com/KostenokLisa/calibration-reasoning-framework","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kostenok26_interspeech","category":"resources-evaluation","institutions":["EPFL","Logitech"],"code":"https://github.com/KostenokLisa/calibration-reasoning-framework","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2362","pdf":"https://www.isca-archive.org/interspeech_2026/kostenok26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kostenok26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kostenok26_interspeech/markdown.md"},{"id":"kothare26_interspeech","title":"Speech-based Digital Biomarkers can Accelerate ALS Clinical Trials: Insights from Time-to-Event and Hazard Rate Analysis","authors":["Hardik Kothare","Michael Neumann","Vikram Ramanarayanan"],"year":2026,"doi":"10.21437/Interspeech.2026-2847","isca_url":"https://www.isca-archive.org/interspeech_2026/kothare26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kothare26_interspeech.pdf","session":"Clinically Useful Speech Representations 1","topics":["paralinguistics","health","evaluation"],"category":"health-clinical","institutions":["Modality.AI"],"funding":["National Institutes of Health"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kothare26_interspeech","category":"health-clinical","institutions":["Modality.AI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2847","pdf":"https://www.isca-archive.org/interspeech_2026/kothare26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kothare26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kothare26_interspeech/markdown.md"},{"id":"kothare26b_interspeech","title":"Speech and Video Biomarkers Exhibit Reduced Within-Subject Variability in Early Parkinson’s Disease and Resistance to Placebo and Hawthorne Effects","authors":["Hardik Kothare","Oliver Roesler","Lakshmi Arbatti","Michael Neumann","Karl Kieburtz","Andrew McGarry","Craig Thompson","Vikram Ramanarayanan"],"year":2026,"doi":"10.21437/Interspeech.2026-2850","isca_url":"https://www.isca-archive.org/interspeech_2026/kothare26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kothare26b_interspeech.pdf","session":"Multimodal and Non-Speech Healthcare Applications","topics":["paralinguistics","health","evaluation"],"category":"health-clinical","institutions":["Modality.AI","CLINTREX","Cerevance","University of California, San Francisco"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kothare26b_interspeech","category":"health-clinical","institutions":["Modality.AI","CLINTREX","Cerevance","University of California, San Francisco"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2850","pdf":"https://www.isca-archive.org/interspeech_2026/kothare26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kothare26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kothare26b_interspeech/markdown.md"},{"id":"kothari26_interspeech","title":"Multilingual Multi-Speaker Unit Vocoders: A Systematic Analysis of Discrete Speech Representations","authors":["Naman Kothari","Arjun Gangwar","Adarsh Arigala","S Umesh"],"year":2026,"doi":"10.21437/Interspeech.2026-3330","isca_url":"https://www.isca-archive.org/interspeech_2026/kothari26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kothari26_interspeech.pdf","session":"Multilingual Speech 2","topics":["tts","multilingual","self-supervised"],"category":"tts","labels":["multilingual","generative-model"],"institutions":["National Institute of Technology, Tiruchirappalli","Indian Institute of Technology, Madras"],"code":{"url":"https://github.com/UnitBigVGAN","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kothari26_interspeech","category":"tts","labels":["multilingual","generative-model"],"institutions":["National Institute of Technology, Tiruchirappalli","Indian Institute of Technology, Madras"],"code":"https://github.com/UnitBigVGAN","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3330","pdf":"https://www.isca-archive.org/interspeech_2026/kothari26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kothari26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kothari26_interspeech/markdown.md"},{"id":"koudounas26_interspeech","title":"Synthetic Pathological Speech at Scale: A Flow Matching Approach for Clinical Data Augmentation","authors":["Alkis Koudounas","Moreno La Quatra","Quentin Jodelet","Hayato Futami","Valerio Mario Salerno","Sabato Marco Siniscalchi","Emiru Tsunoo"],"year":2026,"doi":"10.21437/Interspeech.2026-2313","isca_url":"https://www.isca-archive.org/interspeech_2026/koudounas26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/koudounas26_interspeech.pdf","session":"Pathological Speech Assessment 3","topics":["speech-enhancement","self-supervised","health"],"category":"health-clinical","labels":["low-resource","generative-model"],"institutions":["Sony Group Corporation","Kore University of Enna","Universita degli Studi di Palermo"],"funding":["D.A.R.E. - Digital Lifelong Prevention"],"code":{"url":"https://github.com/koudounasalkis/Pathology-F5TTS","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"koudounas26_interspeech","category":"health-clinical","labels":["low-resource","generative-model"],"institutions":["Sony Group Corporation","Kore University of Enna","Universita degli Studi di Palermo"],"code":"https://github.com/koudounasalkis/Pathology-F5TTS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2313","pdf":"https://www.isca-archive.org/interspeech_2026/koudounas26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/koudounas26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/koudounas26_interspeech/markdown.md"},{"id":"koudounas26b_interspeech","title":"Hallucination Benchmark for Speech Foundation Models","authors":["Alkis Koudounas","Moreno La Quatra","Manuel Giollo","Sabato Marco Siniscalchi","Elena Baralis"],"year":2026,"doi":"10.21437/Interspeech.2026-2347","isca_url":"https://www.isca-archive.org/interspeech_2026/koudounas26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/koudounas26b_interspeech.pdf","session":"Benchmarking Foundation Models","topics":["asr","evaluation","speech-llm"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Politecnico di Torino","Kore University of Enna","Amazon","Universita degli Studi di Palermo"],"code":{"url":"https://github.com/SALT-Research/SHALLOW","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"koudounas26b_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Politecnico di Torino","Kore University of Enna","Amazon","Universita degli Studi di Palermo"],"code":"https://github.com/SALT-Research/SHALLOW","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2347","pdf":"https://www.isca-archive.org/interspeech_2026/koudounas26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/koudounas26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/koudounas26b_interspeech/markdown.md"},{"id":"kovalev26_interspeech","title":"SEAM: Shortcut-Aware Real-Time Detection of Scripted vs. Spontaneous Speech for Interview Guardrails","authors":["Vsevolod Kovalev","Pranay Manocha"],"year":2026,"doi":"10.21437/Interspeech.2026-1480","isca_url":"https://www.isca-archive.org/interspeech_2026/kovalev26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kovalev26_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["speech-llm","self-supervised","evaluation"],"category":"paralinguistics-emotion","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Symbal AI","Princeton University"],"code":{"url":"https://github.com/vsevolod-kovalev/seam","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kovalev26_interspeech","category":"paralinguistics-emotion","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Symbal AI","Princeton University"],"code":"https://github.com/vsevolod-kovalev/seam","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1480","pdf":"https://www.isca-archive.org/interspeech_2026/kovalev26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kovalev26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kovalev26_interspeech/markdown.md"},{"id":"krishnan26_interspeech","title":"On Optimizing Multimodal Jailbreaks for Spoken Language Models","authors":["Aravind Krishnan","Karolina Stańczak","Dietrich Klakow"],"year":2026,"doi":"10.21437/Interspeech.2026-309","isca_url":"https://www.isca-archive.org/interspeech_2026/krishnan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/krishnan26_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["speech-llm","evaluation","self-supervised"],"category":"speech-llm-dialogue","institutions":["Saarland University","DFKI","ETH Zurich"],"funding":["European Defence Fund","ETH AI Center"],"code":{"url":"https://repos.lsv.uni-saarland.de/akrishnan/multimodal-jailbreak-slm","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"krishnan26_interspeech","category":"speech-llm-dialogue","institutions":["Saarland University","DFKI","ETH Zurich"],"code":"https://repos.lsv.uni-saarland.de/akrishnan/multimodal-jailbreak-slm","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-309","pdf":"https://www.isca-archive.org/interspeech_2026/krishnan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/krishnan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/krishnan26_interspeech/markdown.md"},{"id":"krysinska26_interspeech","title":"Transitional Objective Learning with Connectionist Temporal Classification in Phoneme Recognition","authors":["Izabela Krysińska","Mikołaj Morzy","Agnieszka Pludra"],"year":2026,"doi":"10.21437/Interspeech.2026-1040","isca_url":"https://www.isca-archive.org/interspeech_2026/krysinska26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/krysinska26_interspeech.pdf","session":"Robust ASR: Uncertainty and Confidence","topics":["asr","self-supervised","phonetics"],"category":"asr","institutions":["Poznan University of Technology","Pearson Central Europe"],"funding":["National Science Centre, Poland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"krysinska26_interspeech","category":"asr","institutions":["Poznan University of Technology","Pearson Central Europe"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1040","pdf":"https://www.isca-archive.org/interspeech_2026/krysinska26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/krysinska26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/krysinska26_interspeech/markdown.md"},{"id":"krzywdziak26_interspeech","title":"Task-Conditioned Audio-Text-Image Fusion for Cognitive Score Estimation from Speech-Based Assessments","authors":["Justyna Krzywdziak","Władysław Średniawa","Agnieszka Pruszek","Wojciech Szecówka","Michał K. Grzeszczyk","Teresa Brzozka","Bartłomiej Eljasiak","Zofia Marciniak","Łukasz Łazarski","Mateusz Matuszewski","Daria Hemmerling"],"year":2026,"doi":"10.21437/Interspeech.2026-1424","isca_url":"https://www.isca-archive.org/interspeech_2026/krzywdziak26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/krzywdziak26_interspeech.pdf","session":"Clinically Useful Speech Representations 2","topics":["speech-llm","paralinguistics","health"],"category":"health-clinical","institutions":["AGH University of Krakow","Samsung"],"funding":["Excellence Initiative - Research University program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"krzywdziak26_interspeech","category":"health-clinical","institutions":["AGH University of Krakow","Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1424","pdf":"https://www.isca-archive.org/interspeech_2026/krzywdziak26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/krzywdziak26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/krzywdziak26_interspeech/markdown.md"},{"id":"ku26_interspeech","title":"Audiovisual CXMI: Scene-based Context Tagging for Spoken Language Translation Evaluation","authors":["Dayeon Ku","Hwayoung Park","Hong Kook Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-2175","isca_url":"https://www.isca-archive.org/interspeech_2026/ku26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ku26_interspeech.pdf","session":"Multilingual Speech 2","topics":["speech-translation","multilingual","evaluation"],"category":"resources-evaluation","labels":["multilingual"],"institutions":["Gwangju Institute of Science and Technology","AunionAI"],"funding":["Korea Ministry of SMEs and Startups","National Research Foundation of Korea","Korea Ministry of Trade, Industry & Energy","Ministry of Science and ICT, Korea","Gwangju Metropolitan City"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ku26_interspeech","category":"resources-evaluation","labels":["multilingual"],"institutions":["Gwangju Institute of Science and Technology","AunionAI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2175","pdf":"https://www.isca-archive.org/interspeech_2026/ku26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ku26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ku26_interspeech/markdown.md"},{"id":"kuan26_interspeech","title":"Improving Text-to-Audio Instruction Following via Fine-Grained Feedback from Audio-Aware Large Language Models","authors":["Chun-Yi Kuan","Siwon Kim","Byeonggeun Kim","Suyoun Kim","Bo-Ru Lu","Qingming Tang","Ankur Gandhe","Hung-yi Lee","Chieh-Chi Kao","Chao Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1111","isca_url":"https://www.isca-archive.org/interspeech_2026/kuan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kuan26_interspeech.pdf","session":"Text-to-Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["National Taiwan University","Amazon"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kuan26_interspeech","category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["National Taiwan University","Amazon"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1111","pdf":"https://www.isca-archive.org/interspeech_2026/kuan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kuan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kuan26_interspeech/markdown.md"},{"id":"kubo26_interspeech","title":"Building Tailored Speech Recognizers for Japanese Speaking Assessment","authors":["Yotaro Kubo","Richard Sproat","Chihiro Taguchi","Llion Jones"],"year":2026,"doi":"10.21437/Interspeech.2026-1672","isca_url":"https://www.isca-archive.org/interspeech_2026/kubo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kubo26_interspeech.pdf","session":"Prosody, Pronunciation and Specialized Speech Processing","topics":["asr","paralinguistics","low-resource"],"category":"asr","labels":["streaming-real-time"],"institutions":["Sakana AI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kubo26_interspeech","category":"asr","labels":["streaming-real-time"],"institutions":["Sakana AI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1672","pdf":"https://www.isca-archive.org/interspeech_2026/kubo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kubo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kubo26_interspeech/markdown.md"},{"id":"kulkarni26_interspeech","title":"A Closer Look at Failure Modes in Temporal Understanding of Large Audio-Language Models","authors":["Apoorva Kulkarni","Kaousheik Jayakumar","Sreyan Ghosh","Sarah Wiegreffe","Dinesh Manocha","Ramani Duraiswami"],"year":2026,"doi":"10.21437/Interspeech.2026-3070","isca_url":"https://www.isca-archive.org/interspeech_2026/kulkarni26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kulkarni26_interspeech.pdf","session":"Reasoning with Speech/Audio Language Models","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["University of Maryland, College Park"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kulkarni26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["University of Maryland, College Park"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3070","pdf":"https://www.isca-archive.org/interspeech_2026/kulkarni26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kulkarni26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kulkarni26_interspeech/markdown.md"},{"id":"kumar26_interspeech","title":"Listening with Attention: Entropy-Guided Explainability for Transformer-Based Audio Models","authors":["Ravi Kumar","Utkarsh Grover","Xiaomin Lin","Agoritsa Polyzou"],"year":2026,"doi":"10.21437/Interspeech.2026-593","isca_url":"https://www.isca-archive.org/interspeech_2026/kumar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kumar26_interspeech.pdf","session":"Long-form Audio & New Attention Approaches","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["self-supervised"],"institutions":["Florida International University","University of South Florida"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kumar26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Florida International University","University of South Florida"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-593","pdf":"https://www.isca-archive.org/interspeech_2026/kumar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26_interspeech/markdown.md"},{"id":"kumar26b_interspeech","title":"ML-KD-DRI-GAN: Teacher-Guided Denoising and Triplet-Adversarial Training for Robust Spoken Language Understanding","authors":["Ankit Kumar","Munir Georges"],"year":2026,"doi":"10.21437/Interspeech.2026-670","isca_url":"https://www.isca-archive.org/interspeech_2026/kumar26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kumar26b_interspeech.pdf","session":"Audio & Speech Language Models: Evaluation, Representations, and Emerging Capabilities","topics":["asr","spoken-language-understanding","self-supervised"],"category":"speech-llm-dialogue","labels":["generative-model","robustness-noise"],"institutions":["Galgotias University","Technische Hochschule Ingolstadt"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kumar26b_interspeech","category":"speech-llm-dialogue","labels":["generative-model","robustness-noise"],"institutions":["Galgotias University","Technische Hochschule Ingolstadt"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-670","pdf":"https://www.isca-archive.org/interspeech_2026/kumar26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26b_interspeech/markdown.md"},{"id":"kumar26c_interspeech","title":"Overcoming Decoder Inconsistencies in Whisper for Dravidian and Low-Resource Languages","authors":["Chowdam Venkata Kumar","Kumud Tripathi","Pankaj Wasnik"],"year":2026,"doi":"10.21437/Interspeech.2026-1007","isca_url":"https://www.isca-archive.org/interspeech_2026/kumar26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kumar26c_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["asr","multilingual","low-resource"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Sony"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kumar26c_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Sony"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1007","pdf":"https://www.isca-archive.org/interspeech_2026/kumar26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26c_interspeech/markdown.md"},{"id":"kumar26d_interspeech","title":"Search-GRT: Guided Retrieval Training of Search Agents to Optimize for Complex Question Answering","authors":["Aounon Kumar","Sudipta Paul","Vivek Kulkarni","Vijay Srinivasan","Srinivas Chappidi"],"year":2026,"doi":"10.21437/Interspeech.2026-2006","isca_url":"https://www.isca-archive.org/interspeech_2026/kumar26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kumar26d_interspeech.pdf","session":"Reasoning with Speech/Audio Language Models","topics":["asr","self-supervised","speech-llm"],"category":"speech-llm-dialogue","institutions":["Samsung Electronics"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kumar26d_interspeech","category":"speech-llm-dialogue","institutions":["Samsung Electronics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2006","pdf":"https://www.isca-archive.org/interspeech_2026/kumar26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26d_interspeech/markdown.md"},{"id":"kumar26e_interspeech","title":"Lightweight Cross-Lingual Speaker Adaptation for Indic TTS","authors":["Tarun Kumar","Keshav Agarwal","Pawan Goyal","Laxmidhar Behera","Hema A. Murthy"],"year":2026,"doi":"10.21437/Interspeech.2026-2050","isca_url":"https://www.isca-archive.org/interspeech_2026/kumar26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kumar26e_interspeech.pdf","session":"Low-Resource Speech Synthesis","topics":["tts","speaker-verification","low-resource"],"category":"tts","labels":["low-resource","multilingual","efficient-on-device","generative-model"],"institutions":["Indian Institute of Technology Mandi","Indian Institute of Technology Madras","Indian Institute of Technology Kharagpur"],"funding":["Ministry of Electronics and Information Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kumar26e_interspeech","category":"tts","labels":["low-resource","multilingual","efficient-on-device","generative-model"],"institutions":["Indian Institute of Technology Mandi","Indian Institute of Technology Madras","Indian Institute of Technology Kharagpur"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2050","pdf":"https://www.isca-archive.org/interspeech_2026/kumar26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26e_interspeech/markdown.md"},{"id":"kumar26f_interspeech","title":"Who Synthesized This? Joint Deepfake Detection and Generative Source Attribution","authors":["Vishal Kumar","Vinayak Abrol","Mathew Magimai Doss"],"year":2026,"doi":"10.21437/Interspeech.2026-2442","isca_url":"https://www.isca-archive.org/interspeech_2026/kumar26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kumar26f_interspeech.pdf","session":"Safeguards for Synthetic Speech: Ethical, Technical, and Legal Perspectives","topics":["audio-deepfake","speaker-verification","self-supervised"],"category":"deepfake-security","labels":["low-resource","self-supervised"],"institutions":["IIIT-Delhi","IDIAP"],"funding":["Swiss National Science Foundation","Nebius Research Grant","Infosys Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kumar26f_interspeech","category":"deepfake-security","labels":["low-resource","self-supervised"],"institutions":["IIIT-Delhi","IDIAP"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2442","pdf":"https://www.isca-archive.org/interspeech_2026/kumar26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26f_interspeech/markdown.md"},{"id":"kumar26g_interspeech","title":"An auscultation location specific study on the relationship between expiratory-to-inspiratory acoustic patterns and spirometric airflow limitation across age and gender in asthmatic patients","authors":["Dheeraj Harish Kumar","Sanjana MC","Perumal Keerthi Priya","K V Nikhath Khanam","Uma Maheshwari Krishnaswamy","Prasanta Kumar Ghosh"],"year":2026,"doi":"10.21437/Interspeech.2026-2602","isca_url":"https://www.isca-archive.org/interspeech_2026/kumar26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kumar26g_interspeech.pdf","session":"Pathological Speech Assessment 1","topics":["paralinguistics","health","evaluation"],"category":"health-clinical","institutions":["Indian Institute of Science","St. Johns National Academy of Health Sciences"],"funding":["Department of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kumar26g_interspeech","category":"health-clinical","institutions":["Indian Institute of Science","St. Johns National Academy of Health Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2602","pdf":"https://www.isca-archive.org/interspeech_2026/kumar26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26g_interspeech/markdown.md"},{"id":"kumar26h_interspeech","title":"VāṇīSetu: A Human-AI Collaborative Framework for Scalable Conversational Speech Corpus Creation in Low-Resource Settings","authors":["Rishabh Kumar","Dhruv Kudale","Chriss Philip Saji","Abhinav Painuli","John Nirmal","Ganesh Ramakrishnan"],"year":2026,"doi":"10.21437/Interspeech.2026-2607","isca_url":"https://www.isca-archive.org/interspeech_2026/kumar26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kumar26h_interspeech.pdf","session":"Corpus Creation, Summarization and Understanding","topics":["asr","dataset","low-resource"],"category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Indian Institute of Technology Bombay","University of Science and Technology","University of Delhi","BharatGen"],"funding":["BharatGen"],"code":{"url":"https://github.com/KrishiVaani/KrishiVaani","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kumar26h_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Indian Institute of Technology Bombay","University of Science and Technology","University of Delhi","BharatGen"],"code":"https://github.com/KrishiVaani/KrishiVaani","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2607","pdf":"https://www.isca-archive.org/interspeech_2026/kumar26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kumar26h_interspeech/markdown.md"},{"id":"kutsakov26_interspeech","title":"GigaChat Audio: Time-aware Large Audio Language Model","authors":["Aleksandr Kutsakov","Mariia Sadovina","Georgii Gospodinov","Alexandr Maximenko","Oleg Kutuzov","Pavel Bogomolov","Fyodor Minkin"],"year":2026,"doi":"10.21437/Interspeech.2026-2343","isca_url":"https://www.isca-archive.org/interspeech_2026/kutsakov26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kutsakov26_interspeech.pdf","session":"Spoken Language Understanding","topics":["speech-llm","dataset","evaluation"],"category":"speech-llm-dialogue","institutions":["SaluteDevices"],"code":{"url":"https://huggingface.co/aisage/GigaChat3.1-Audio-10B-A1.8B","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kutsakov26_interspeech","category":"speech-llm-dialogue","institutions":["SaluteDevices"],"code":"https://huggingface.co/aisage/GigaChat3.1-Audio-10B-A1.8B","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2343","pdf":"https://www.isca-archive.org/interspeech_2026/kutsakov26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kutsakov26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kutsakov26_interspeech/markdown.md"},{"id":"kuwar26_interspeech","title":"VINAYAKA: Multilingual Audio-Visual Hate Speech Detection via Cross-Modal Fusion in Hyperbolic Space","authors":["Bhavinkumar Vinodbhai Kuwar","Orchid Chetia Phukan","Rajesh Sharma"],"year":2026,"doi":"10.21437/Interspeech.2026-2262","isca_url":"https://www.isca-archive.org/interspeech_2026/kuwar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kuwar26_interspeech.pdf","session":"Audio-Visual Grounding, Synchronization & Video Understanding","topics":["speech-llm","multilingual","self-supervised"],"category":"paralinguistics-emotion","labels":["multilingual"],"institutions":["Plaksha University","Indraprastha Institute of Information Technology Delhi","National Tsing Hua University","University of Tartu"],"code":{"url":"https://bhavin-19.github.io/vinayaka-interspeech26/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kuwar26_interspeech","category":"paralinguistics-emotion","labels":["multilingual"],"institutions":["Plaksha University","Indraprastha Institute of Information Technology Delhi","National Tsing Hua University","University of Tartu"],"code":"https://bhavin-19.github.io/vinayaka-interspeech26/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2262","pdf":"https://www.isca-archive.org/interspeech_2026/kuwar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kuwar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kuwar26_interspeech/markdown.md"},{"id":"kuzmenko26_interspeech","title":"GigaAM Multilingual: Foundation Model for Underrepresented Languages","authors":["Andrei Kuzmenko","Alexandr Maximenko","Aleksandr Kutsakov","Georgii Gospodinov","Dmitrii Bolotov","Oleg Kutuzov","Pavel Bogomolov","Fyodor Minkin"],"year":2026,"doi":"10.21437/Interspeech.2026-2483","isca_url":"https://www.isca-archive.org/interspeech_2026/kuzmenko26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kuzmenko26_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["SaluteDevices"],"code":{"url":"https://github.com/salute-developers/GigaAM","stars":831,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kuzmenko26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["SaluteDevices"],"code":"https://github.com/salute-developers/GigaAM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2483","pdf":"https://www.isca-archive.org/interspeech_2026/kuzmenko26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kuzmenko26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kuzmenko26_interspeech/markdown.md"},{"id":"kuzmin26_interspeech","title":"StreamVoiceAnon+: Emotion-Preserving Streaming Speaker Anonymization via Frame-Level Acoustic Distillation","authors":["Nikita Kuzmin","Kong Aik Lee","Eng Siong Chng"],"year":2026,"doi":"10.21437/Interspeech.2026-3105","isca_url":"https://www.isca-archive.org/interspeech_2026/kuzmin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kuzmin26_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["speaker-anonymization","emotion-recognition","self-supervised"],"category":"deepfake-security","labels":["streaming-real-time"],"institutions":["Nanyang Technological University","Agency for Science, Technology and Research","Hong Kong Polytechnic University"],"code":{"url":"https://paniquex.github.io/streamvoiceanon-plus/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kuzmin26_interspeech","category":"deepfake-security","labels":["streaming-real-time"],"institutions":["Nanyang Technological University","Agency for Science, Technology and Research","Hong Kong Polytechnic University"],"code":"https://paniquex.github.io/streamvoiceanon-plus/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3105","pdf":"https://www.isca-archive.org/interspeech_2026/kuzmin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kuzmin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kuzmin26_interspeech/markdown.md"},{"id":"kuzmin26b_interspeech","title":"Privacy-Preserving End-to-End Full-Duplex Speech Dialogue Models","authors":["Nikita Kuzmin","Tao Zhong","Jiajun Deng","Yingke Zhu","Tristan Tsoi","Tianxiang Cao","Simon Lui","Kong Aik Lee","Eng Siong Chng"],"year":2026,"doi":"10.21437/Interspeech.2026-3181","isca_url":"https://www.isca-archive.org/interspeech_2026/kuzmin26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kuzmin26b_interspeech.pdf","session":"Speaker Privacy and Anonymization","topics":["speech-llm","speaker-verification","low-resource"],"category":"deepfake-security","labels":["streaming-real-time","generative-model"],"institutions":["Nanyang Technological University","A*STAR","Huawei","Leibniz Research Center","Chinese University of Hong Kong","Hong Kong Polytechnic University"],"code":{"url":"https://github.com/Plachtaa/StreamVoiceAnon","stars":95,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kuzmin26b_interspeech","category":"deepfake-security","labels":["streaming-real-time","generative-model"],"institutions":["Nanyang Technological University","A*STAR","Huawei","Leibniz Research Center","Chinese University of Hong Kong","Hong Kong Polytechnic University"],"code":"https://github.com/Plachtaa/StreamVoiceAnon","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3181","pdf":"https://www.isca-archive.org/interspeech_2026/kuzmin26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kuzmin26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kuzmin26b_interspeech/markdown.md"},{"id":"kuznetsov26_interspeech","title":"FastWave: Optimized Diffusion Model for Audio Super-Resolution","authors":["Nikita Kuznetsov","Maksim Kaledin"],"year":2026,"doi":"10.21437/Interspeech.2026-2721","isca_url":"https://www.isca-archive.org/interspeech_2026/kuznetsov26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kuznetsov26_interspeech.pdf","session":"Dereverberation, Bandwidth Extension and Restoration","topics":["speech-enhancement","self-supervised","on-device"],"category":"speech-coding","labels":["efficient-on-device","generative-model"],"institutions":["HSE University","VK LLC"],"code":{"url":"https://github.com/Nikait/FastWave","stars":21,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kuznetsov26_interspeech","category":"speech-coding","labels":["efficient-on-device","generative-model"],"institutions":["HSE University","VK LLC"],"code":"https://github.com/Nikait/FastWave","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2721","pdf":"https://www.isca-archive.org/interspeech_2026/kuznetsov26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kuznetsov26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kuznetsov26_interspeech/markdown.md"},{"id":"kwak26_interspeech","title":"Plug-and-Steer: Decoupling Separation and Selection in Audio-Visual Target Speaker Extraction","authors":["Doyeop Kwak","Suyeon Lee","Joon Son Chung"],"year":2026,"doi":"10.21437/Interspeech.2026-706","isca_url":"https://www.isca-archive.org/interspeech_2026/kwak26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kwak26_interspeech.pdf","session":"Audio-Visual and Generative Target Speaker Extraction","topics":["speech-separation","self-supervised","multimodal"],"category":"enhancement-separation","institutions":["Korea Advanced Institute of Science and Technology"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kwak26_interspeech","category":"enhancement-separation","institutions":["Korea Advanced Institute of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-706","pdf":"https://www.isca-archive.org/interspeech_2026/kwak26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kwak26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kwak26_interspeech/markdown.md"},{"id":"kwon26_interspeech","title":"Delayed-Commitment Online Speaker Tracking for Robust Many-Speaker Diarization","authors":["Youngki Kwon","Hee-Soo Heo","Minjae Lee","Han-Gyu Kim","Bong-Jin Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-898","isca_url":"https://www.isca-archive.org/interspeech_2026/kwon26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kwon26_interspeech.pdf","session":"Speaker Diarization 2","topics":["speaker-diarization","self-supervised","evaluation"],"category":"speaker","labels":["streaming-real-time"],"institutions":["NAVER Cloud Corporation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kwon26_interspeech","category":"speaker","labels":["streaming-real-time"],"institutions":["NAVER Cloud Corporation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-898","pdf":"https://www.isca-archive.org/interspeech_2026/kwon26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kwon26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kwon26_interspeech/markdown.md"},{"id":"kwon26b_interspeech","title":"Investigating ASR for Low-Intelligibility Dysarthric Speech","authors":["Jinuk Kwon","Beiming Cao","Carl Sigmond","Kristin Teplansky","Jun Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1326","isca_url":"https://www.isca-archive.org/interspeech_2026/kwon26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/kwon26b_interspeech.pdf","session":"Assistive Technologies 2","topics":["asr","low-resource","self-supervised"],"category":"asr","labels":["low-resource","self-supervised","dataset-or-benchmark-release"],"institutions":["University of Texas at Austin"],"funding":["National Institute on Deafness and Other Communication Disorders","National Institutes of Health"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"kwon26b_interspeech","category":"asr","labels":["low-resource","self-supervised","dataset-or-benchmark-release"],"institutions":["University of Texas at Austin"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1326","pdf":"https://www.isca-archive.org/interspeech_2026/kwon26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/kwon26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/kwon26b_interspeech/markdown.md"},{"id":"labrak26_interspeech","title":"Generating Synthetic Doctor-Patient Conversations for Long-form Audio Summarization","authors":["Yanis Labrak","David Grünert","Severin Baroudi","Jiyun Chun","Pawel Cyrta","Sergio Burdisso","Ahmed Hassoon","David Liu","Adam Rothschild","Reed Van Deusen","Petr Motlicek","Andrew Perrault","Ricard Marxer","Thomas Schaaf"],"year":2026,"doi":"10.21437/Interspeech.2026-2901","isca_url":"https://www.isca-archive.org/interspeech_2026/labrak26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/labrak26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["speech-llm","dataset","spoken-language-understanding"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Idiap Research Institute","University of Zurich","Ohio State University","Universite de Toulon","Aix Marseille Univ","LIS","CNRS","Stenograf","Johns Hopkins University Bloomberg School of Public Health","Colorado School of Mines","Allegheny Health Network","University of Pittsburgh Medical Center","ILLS","Solventum","Carnegie Mellon University"],"funding":["Jelinek Memorial Summer Workshop on Speech and Language Technologies","European Union Horizon 2020","Agence de l’Innovation Defense","French National Research Agency"],"code":{"url":"https://huggingface.co/datasets/Play-Your-Part/Synth-DoPaCo","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"labrak26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Idiap Research Institute","University of Zurich","Ohio State University","Universite de Toulon","Aix Marseille Univ","LIS","CNRS","Stenograf","Johns Hopkins University Bloomberg School of Public Health","Colorado School of Mines","Allegheny Health Network","University of Pittsburgh Medical Center","ILLS","Solventum","Carnegie Mellon University"],"code":"https://huggingface.co/datasets/Play-Your-Part/Synth-DoPaCo","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2901","pdf":"https://www.isca-archive.org/interspeech_2026/labrak26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/labrak26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/labrak26_interspeech/markdown.md"},{"id":"lahtinen26_interspeech","title":"Looking for Affect in Spontaneous Finnish Speech through Linguistic Interpretability","authors":["Kalle Lahtinen","Liisa Mustanoja","Okko Räsänen"],"year":2026,"doi":"10.21437/Interspeech.2026-2452","isca_url":"https://www.isca-archive.org/interspeech_2026/lahtinen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lahtinen26_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["speech-emotion-recognition","multilingual","paralinguistics"],"category":"paralinguistics-emotion","institutions":["Tampere University"],"funding":["Jane and Aatos Erkko Foundation"],"code":{"url":"https://github.com/SPEECHCOG/LookingForAffect/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lahtinen26_interspeech","category":"paralinguistics-emotion","institutions":["Tampere University"],"code":"https://github.com/SPEECHCOG/LookingForAffect/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2452","pdf":"https://www.isca-archive.org/interspeech_2026/lahtinen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lahtinen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lahtinen26_interspeech/markdown.md"},{"id":"lai26_interspeech","title":"Lung-CL: Spectrum-aware Distillation and Generative Replay for Continual Learning based buffer-free Respiratory Sound Classification","authors":["Qinben Lai","Lukui Shi","Jiaxin Zhao","Wenjuan Wang","Shuai Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-1199","isca_url":"https://www.isca-archive.org/interspeech_2026/lai26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lai26_interspeech.pdf","session":"Pathological Speech Assessment 1","topics":["paralinguistics","self-supervised","health"],"category":"health-clinical","institutions":["Hebei University of Technology","Macao Polytechnic University"],"funding":["National Natural Science Foundation","Hebei Province Natural Science Foundation Funded Project"],"code":{"url":"https://github.com/ben100118/Lung-CL","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lai26_interspeech","category":"health-clinical","institutions":["Hebei University of Technology","Macao Polytechnic University"],"code":"https://github.com/ben100118/Lung-CL","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1199","pdf":"https://www.isca-archive.org/interspeech_2026/lai26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lai26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lai26_interspeech/markdown.md"},{"id":"lakshmi26_interspeech","title":"The First Dravidian Speech Datasets for Transphobic and Homophobic Hate Speech: Creation, Annotation, and Multimodal Benchmarking","authors":["K N Lakshmi","K Hemavardhan Reddy","A Venkata Satya","Kota Venkata Vamshidhar Reddy","Jyothish Lal G","Premjith B","Jesin James"],"year":2026,"doi":"10.21437/Interspeech.2026-2368","isca_url":"https://www.isca-archive.org/interspeech_2026/lakshmi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lakshmi26_interspeech.pdf","session":"Queer and Trans Speech Science and Technology","topics":["speech-llm","paralinguistics","dataset"],"category":"resources-evaluation","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release","robustness-noise"],"institutions":["Amrita Vishwa Vidyapeetham","University of Auckland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lakshmi26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release","robustness-noise"],"institutions":["Amrita Vishwa Vidyapeetham","University of Auckland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2368","pdf":"https://www.isca-archive.org/interspeech_2026/lakshmi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lakshmi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lakshmi26_interspeech/markdown.md"},{"id":"lamba26_interspeech","title":"How Frequency Band Importance Affects Neural Network Predictions and Human Perception for Speech Quality Assessment","authors":["Ada Lamba","Donald S. Williamson"],"year":2026,"doi":"10.21437/Interspeech.2026-2382","isca_url":"https://www.isca-archive.org/interspeech_2026/lamba26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lamba26_interspeech.pdf","session":"Benchmarking Foundation Models","topics":["speech-enhancement","evaluation","paralinguistics"],"category":"resources-evaluation","institutions":["Ohio State University"],"funding":["National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lamba26_interspeech","category":"resources-evaluation","institutions":["Ohio State University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2382","pdf":"https://www.isca-archive.org/interspeech_2026/lamba26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lamba26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lamba26_interspeech/markdown.md"},{"id":"lameris26_interspeech","title":"Lost in Phonation: Voice Quality Variation as an Evaluation Dimension for Speech Foundation Models","authors":["Harm Lameris","Shree Harsha Bokkahalli Satish","Joakim Gustafson","Éva Székely"],"year":2026,"doi":"10.21437/Interspeech.2026-736","isca_url":"https://www.isca-archive.org/interspeech_2026/lameris26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lameris26_interspeech.pdf","session":"Audio & Speech Language Models: Evaluation, Representations, and Emerging Capabilities","topics":["speech-llm","paralinguistics","emotion-recognition"],"category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["KTH Royal Institute of Technology"],"funding":["Wallenberg AI, Autonomous Systems and Software Program","Knut and Alice Wallenberg Foundation","Swedish Research Council"],"code":{"url":"https://shreeharsha-bs.github.io/Lost-in-phonation/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lameris26_interspeech","category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["KTH Royal Institute of Technology"],"code":"https://shreeharsha-bs.github.io/Lost-in-phonation/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-736","pdf":"https://www.isca-archive.org/interspeech_2026/lameris26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lameris26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lameris26_interspeech/markdown.md"},{"id":"lameris26b_interspeech","title":"VoiceQualityGUI: A Tool for Word-Level Voice Quality Modifications","authors":["Harm Lameris","Alan Villamil","Nigel G. Ward"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/lameris26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lameris26b_interspeech.pdf","session":"Speech Synthesis, Voice Conversion and Audio Generation","topics":["voice-conversion","speech-synthesis","prosody"],"category":"tts","labels":["generative-model"],"institutions":["KTH Royal Institute of Technology","University of Texas at El Paso"],"funding":["Air Force Office of Scientific Research"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lameris26b_interspeech","category":"tts","labels":["generative-model"],"institutions":["KTH Royal Institute of Technology","University of Texas at El Paso"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/lameris26b_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/lameris26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lameris26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lameris26b_interspeech/markdown.md"},{"id":"lan26_interspeech","title":"SA-UAED: Joint Frame-Level Detection of Audio Events, Speaker Activities, and Speaker-Attributed Paralinguistic Events","authors":["Zekun Lan","Wangyou Zhang","Yanmin Qian"],"year":2026,"doi":"10.21437/Interspeech.2026-2486","isca_url":"https://www.isca-archive.org/interspeech_2026/lan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lan26_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["speaker-diarization","paralinguistics","speech-enhancement"],"category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["Shanghai Jiao Tong University","VUI Labs"],"funding":["China NSFC","SJTU Med-X (Medicine & Engineering) Translational Research Grant"],"code":{"url":"https://github.com/originallover/SA-UAED","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lan26_interspeech","category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["Shanghai Jiao Tong University","VUI Labs"],"code":"https://github.com/originallover/SA-UAED","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2486","pdf":"https://www.isca-archive.org/interspeech_2026/lan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lan26_interspeech/markdown.md"},{"id":"lanzendoerfer26_interspeech","title":"Evaluating Objective Speech Quality Metrics for Neural Audio Codecs","authors":["Luca A. Lanzendöerfer","Florian Grötschla","Roger Wattenhofer"],"year":2026,"doi":"10.21437/Interspeech.2026-1809","isca_url":"https://www.isca-archive.org/interspeech_2026/lanzendoerfer26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lanzendoerfer26_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["evaluation","speech-enhancement","self-supervised"],"category":"resources-evaluation","institutions":["ETH Zurich"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lanzendoerfer26_interspeech","category":"resources-evaluation","institutions":["ETH Zurich"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1809","pdf":"https://www.isca-archive.org/interspeech_2026/lanzendoerfer26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lanzendoerfer26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lanzendoerfer26_interspeech/markdown.md"},{"id":"lanzendoerfer26b_interspeech","title":"Speaker Separation via Audio Language Modeling","authors":["Luca A. Lanzendöerfer","Constantin Pinkl","Florian Grötschla","Roger Wattenhofer"],"year":2026,"doi":"10.21437/Interspeech.2026-2864","isca_url":"https://www.isca-archive.org/interspeech_2026/lanzendoerfer26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lanzendoerfer26b_interspeech.pdf","session":"Speaker Diarization 1","topics":["speaker-separation","speaker-diarization","self-supervised"],"category":"speaker","labels":["self-supervised","generative-model"],"institutions":["ETH Zurich"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lanzendoerfer26b_interspeech","category":"speaker","labels":["self-supervised","generative-model"],"institutions":["ETH Zurich"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2864","pdf":"https://www.isca-archive.org/interspeech_2026/lanzendoerfer26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lanzendoerfer26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lanzendoerfer26b_interspeech/markdown.md"},{"id":"laquatra26_interspeech","title":"Etiology-Aware Speech Language Models for Dysarthric Speech Recognition","authors":["Moreno La Quatra","Alkis Koudounas","Valerio Mario Salerno","Sabato Marco Siniscalchi"],"year":2026,"doi":"10.21437/Interspeech.2026-1773","isca_url":"https://www.isca-archive.org/interspeech_2026/laquatra26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/laquatra26_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 2","topics":["asr","speech-llm","low-resource"],"category":"asr","institutions":["Kore University of Enna","Sony Group Corporation","Universita degli Studi di Palermo"],"funding":["D.A.R.E. - Digital Lifelong Prevention"],"code":{"url":"https://github.com/MorenoLaQuatra/dysarthric-asr","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"laquatra26_interspeech","category":"asr","institutions":["Kore University of Enna","Sony Group Corporation","Universita degli Studi di Palermo"],"code":"https://github.com/MorenoLaQuatra/dysarthric-asr","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1773","pdf":"https://www.isca-archive.org/interspeech_2026/laquatra26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/laquatra26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/laquatra26_interspeech/markdown.md"},{"id":"laquatra26b_interspeech","title":"SSL-based Sequence Matching for Unsupervised Audio Retrieval","authors":["Moreno La Quatra","Alkis Koudounas","Sabato Marco Siniscalchi"],"year":2026,"doi":"10.21437/Interspeech.2026-2369","isca_url":"https://www.isca-archive.org/interspeech_2026/laquatra26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/laquatra26b_interspeech.pdf","session":"Information Extraction and Retrieval","topics":["audio-information-retrieval","self-supervised","spoken-language-understanding"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["Kore University of Enna","Politecnico di Torino","Università degli Studi di Palermo"],"code":{"url":"https://github.com/MorenoLaQuatra/SSL-AIR","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"laquatra26b_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["Kore University of Enna","Politecnico di Torino","Università degli Studi di Palermo"],"code":"https://github.com/MorenoLaQuatra/SSL-AIR","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2369","pdf":"https://www.isca-archive.org/interspeech_2026/laquatra26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/laquatra26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/laquatra26b_interspeech/markdown.md"},{"id":"lavechin26_interspeech","title":"BabAR: from phoneme recognition to developmental measures of young children's speech production","authors":["Marvin Lavechin","Elika Bergelson","Roger Levy"],"year":2026,"doi":"10.21437/Interspeech.2026-1132","isca_url":"https://www.isca-archive.org/interspeech_2026/lavechin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lavechin26_interspeech.pdf","session":"Clinical and Inclusive Speech Technology","topics":["child-speech","self-supervised","phoneme-recognition"],"category":"asr","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["Aix-Marseille University","CNRS","Harvard University","Massachusetts Institute of Technology"],"funding":["Simons Foundation International","National Institutes of Health"],"code":{"url":"https://github.com/MarvinLvn/BabAR","stars":17,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lavechin26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["Aix-Marseille University","CNRS","Harvard University","Massachusetts Institute of Technology"],"code":"https://github.com/MarvinLvn/BabAR","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1132","pdf":"https://www.isca-archive.org/interspeech_2026/lavechin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lavechin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lavechin26_interspeech/markdown.md"},{"id":"lay26_interspeech","title":"A Fast Solver for Interpolating Stochastic Differential Equation Diffusion Models for Speech Restoration","authors":["Bunlong Lay","Timo Gerkmann"],"year":2026,"doi":"10.21437/Interspeech.2026-2582","isca_url":"https://www.isca-archive.org/interspeech_2026/lay26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lay26_interspeech.pdf","session":"Speech Enhancement and Restoration","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["efficient-on-device","generative-model"],"institutions":["University of Hamburg","HITeC-Hamburg"],"funding":["Deutsche Forschungsgemeinschaft","Federal Ministry for Economic Affairs and Climate Action","Zentrales Innovationsprogramm Mittelstand","German Federal Ministry of Research, Technology and Space"],"code":{"url":"https://github.com/sp-uhh/fast_","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lay26_interspeech","category":"enhancement-separation","labels":["efficient-on-device","generative-model"],"institutions":["University of Hamburg","HITeC-Hamburg"],"code":"https://github.com/sp-uhh/fast_","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2582","pdf":"https://www.isca-archive.org/interspeech_2026/lay26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lay26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lay26_interspeech/markdown.md"},{"id":"le26_interspeech","title":"VerAno: Speaker Anonymization via Self-Supervised Tokenization and Conditional Flow Matching","authors":["Ngoc Hung Le","Thien-Phuc Doan","Thien An Nguyen","Kyujin Kim","Souhwan Jung"],"year":2026,"doi":"10.21437/Interspeech.2026-497","isca_url":"https://www.isca-archive.org/interspeech_2026/le26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/le26_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["speaker-anonymization","self-supervised","voice-conversion"],"category":"deepfake-security","labels":["self-supervised","generative-model"],"institutions":["Soongsil University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"le26_interspeech","category":"deepfake-security","labels":["self-supervised","generative-model"],"institutions":["Soongsil University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-497","pdf":"https://www.isca-archive.org/interspeech_2026/le26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/le26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/le26_interspeech/markdown.md"},{"id":"le26b_interspeech","title":"ViP-VL: Vietnamese Self-supervised Speech Pretraining Model with Vector-Quantization Learning","authors":["Khanh Le","Kiet Anh Hoang","Bao Nguyen","Duy Vo","Dung Vo","Thai Tran","Linh Pham","Khoa D Doan"],"year":2026,"doi":"10.21437/Interspeech.2026-1077","isca_url":"https://www.isca-archive.org/interspeech_2026/le26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/le26b_interspeech.pdf","session":"Multilingual & Low-Resource ASR","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["low-resource","self-supervised"],"institutions":["VinUniversity","UNEY"],"code":{"url":"https://github.com/khanld/chunkformer","stars":87,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"le26b_interspeech","category":"asr","labels":["low-resource","self-supervised"],"institutions":["VinUniversity","UNEY"],"code":"https://github.com/khanld/chunkformer","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1077","pdf":"https://www.isca-archive.org/interspeech_2026/le26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/le26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/le26b_interspeech/markdown.md"},{"id":"leal26_interspeech","title":"Tarsila-ASR: A Multi-Domain Test Suite for Benchmarking Brazilian Portuguese Speech Recognition","authors":["Sidney Leal","Ariadne Matos","Edresson Casanova","Frederico Gonçalves","Renato Moraes Silva","Arnaldo Cândido Jr","Sandra Aluísio"],"year":2026,"doi":"10.21437/Interspeech.2026-450","isca_url":"https://www.isca-archive.org/interspeech_2026/leal26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/leal26_interspeech.pdf","session":"Speech Benchmarks, Evaluation, and Resources","topics":["asr","dataset","evaluation"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Venturus","University of São Paulo","NVIDIA","São Paulo State University"],"funding":["Ministry of Science, Technology, and Innovations","PPI-SOFTEX","Softex"],"code":{"url":"https://github.com/nilc-nlp/tarsila-asr","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"leal26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Venturus","University of São Paulo","NVIDIA","São Paulo State University"],"code":"https://github.com/nilc-nlp/tarsila-asr","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-450","pdf":"https://www.isca-archive.org/interspeech_2026/leal26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/leal26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/leal26_interspeech/markdown.md"},{"id":"leal26b_interspeech","title":"Analyzing Longitudinal Vocal Changes During Cognitive Behavioral Therapy for Hikikomori Patients","authors":["Samara S. Leal","Stavros Ntalampiras","Antonio Trabacca","Marcella Bellani","Roberto Sassi"],"year":2026,"doi":"10.21437/Interspeech.2026-742","isca_url":"https://www.isca-archive.org/interspeech_2026/leal26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/leal26b_interspeech.pdf","session":"Clinically Useful Speech Representations 2","topics":["paralinguistics","speech-llm","self-supervised"],"category":"health-clinical","labels":["self-supervised"],"institutions":["University of Milan","Scientific Institute IRCCS E. Medea","University of Verona","Azienda Ospedaliera Universitaria Integrata"],"funding":["European Union"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"leal26b_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["University of Milan","Scientific Institute IRCCS E. Medea","University of Verona","Azienda Ospedaliera Universitaria Integrata"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-742","pdf":"https://www.isca-archive.org/interspeech_2026/leal26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/leal26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/leal26b_interspeech/markdown.md"},{"id":"lee26_interspeech","title":"AdaLTM: Adaptive Layer-wise Task Vector Merging for Categorical Speech Emotion Recognition with ASR Knowledge Integration","authors":["Chia-Yu Lee","Huang-Cheng Chou","Tzu-Quan Lin","Yuanchao Li","Ya-Tse Wu","Shrikanth Narayanan","Chi-Chun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-80","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26_interspeech.pdf","session":"Speech Emotion Recognition and Representation 3","topics":["speech-emotion-recognition","asr","self-supervised"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["National Tsing Hua University","University of Southern California","National Taiwan University","University of Edinburgh"],"funding":["NSTC, Taiwan","NSF","ODNI IARPA ARTS"],"code":{"url":"https://anonymous.4open.science/r/AdaLTM-62A2/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["National Tsing Hua University","University of Southern California","National Taiwan University","University of Edinburgh"],"code":"https://anonymous.4open.science/r/AdaLTM-62A2/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-80","pdf":"https://www.isca-archive.org/interspeech_2026/lee26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26_interspeech/markdown.md"},{"id":"lee26b_interspeech","title":"An Approach to Simultaneous Acquisition of Real-Time MRI Video, EEG, and Surface EMG for Articulatory, Brain, and Muscle Activity During Speech Production","authors":["Jihwan Lee","Parsa Razmara","Kevin Huang","Sean Foley","Aditya Kommineni","Haley Hsu","Woojae Jeong","Prakash Kumar","Xuan Shi","Yoonjeong Lee","Tiantian Feng","Takfarinas Medani","Ye Tian","Sudarsana Reddy Kadiri","Krishna Nayak","Dani Byrd","Louis Goldstein","Richard M. Leahy","Shrikanth Narayanan"],"year":2026,"doi":"10.21437/Interspeech.2026-140","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26b_interspeech.pdf","session":"Methods and Data for Vocal Tract Shape and Articulation Analysis","topics":["speech-llm","self-supervised","evaluation"],"category":"phonetics-linguistics","labels":["streaming-real-time"],"institutions":["University of Southern California"],"funding":["National Science Foundation","National Institute of Biomedical Imaging and Bioengineering","National Institutes of Health"],"code":{"url":"https://github.com/lee-jhwn/multimodal-speech-biosignals","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26b_interspeech","category":"phonetics-linguistics","labels":["streaming-real-time"],"institutions":["University of Southern California"],"code":"https://github.com/lee-jhwn/multimodal-speech-biosignals","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-140","pdf":"https://www.isca-archive.org/interspeech_2026/lee26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26b_interspeech/markdown.md"},{"id":"lee26c_interspeech","title":"NaturalFlow: Reducing Disruptive Pauses for Natural Speech Flow in Simultaneous Speech-to-Speech Translation","authors":["Dongwook Lee","Youngho Cho","Sangkwon Park","Heeseung Kim","Sungroh Yoon"],"year":2026,"doi":"10.21437/Interspeech.2026-540","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26c_interspeech.pdf","session":"Multilingual Speech 1","topics":["speech-translation","self-supervised","evaluation"],"category":"translation","labels":["multilingual","streaming-real-time"],"institutions":["Seoul National University","University of Seoul"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT","National Research Foundation of Korea","BK21 FOUR Program","Samsung Electronics"],"code":{"url":"https://naturalflows2st.github.io/naturalflow/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26c_interspeech","category":"translation","labels":["multilingual","streaming-real-time"],"institutions":["Seoul National University","University of Seoul"],"code":"https://naturalflows2st.github.io/naturalflow/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-540","pdf":"https://www.isca-archive.org/interspeech_2026/lee26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26c_interspeech/markdown.md"},{"id":"lee26d_interspeech","title":"SAM: A Mamba-2 State-Space Audio-Language Model","authors":["Taehan Lee","Jaehan Jung","Hyukjun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-639","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26d_interspeech.pdf","session":"Audio Foundation Models and Generation","topics":["speech-llm","self-supervised","multilingual"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["Sogang University"],"code":{"url":"https://github.com/sam-audio-language-model/sam","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26d_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["Sogang University"],"code":"https://github.com/sam-audio-language-model/sam","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-639","pdf":"https://www.isca-archive.org/interspeech_2026/lee26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26d_interspeech/markdown.md"},{"id":"lee26e_interspeech","title":"RAF: Relativistic Adversarial Feedback For Universal Speech Synthesis","authors":["Yongjoon Lee","Jung-Woo Choi"],"year":2026,"doi":"10.21437/Interspeech.2026-646","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26e_interspeech.pdf","session":"Text-to-Speech Synthesis","topics":["tts","voice-conversion","self-supervised"],"category":"tts","labels":["self-supervised","generative-model"],"institutions":["Korea Advanced Institute of Science and Technology"],"funding":["National Research Foundation of Korea","Ministry of Science and ICT of Korea government","Ministry of Education of Korea government","Industrial Technology Innovation R&D program of MOTIE/KEIT"],"code":{"url":"https://github.com/infected4098/Relativistic-Adversarial-Feedback","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26e_interspeech","category":"tts","labels":["self-supervised","generative-model"],"institutions":["Korea Advanced Institute of Science and Technology"],"code":"https://github.com/infected4098/Relativistic-Adversarial-Feedback","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-646","pdf":"https://www.isca-archive.org/interspeech_2026/lee26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26e_interspeech/markdown.md"},{"id":"lee26f_interspeech","title":"SEMamba++: A General Speech Restoration Framework Leveraging Global, Local, and Periodic Spectral Patterns","authors":["Yongjoon Lee","Jung-Woo Choi"],"year":2026,"doi":"10.21437/Interspeech.2026-665","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26f_interspeech.pdf","session":"Speech Enhancement and Restoration","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Korea Advanced Institute of Science and Technology"],"funding":["National Research Foundation of Korea","Ministry of Science and ICT of Korea","Ministry of Education of Korea","Ministry of Trade, Industry and Energy","Korea Evaluation Institute of Industrial Technology"],"code":{"url":"https://sites.google.com/view/semambapp","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26f_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Korea Advanced Institute of Science and Technology"],"code":"https://sites.google.com/view/semambapp","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-665","pdf":"https://www.isca-archive.org/interspeech_2026/lee26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26f_interspeech/markdown.md"},{"id":"lee26g_interspeech","title":"AccentDrift: Real-time Streaming Accent Conversion via Sparse Speech Tokenization","authors":["Sang-Hoon Lee","Heejin Choi","Joun Yeop Lee","Sangjun Park"],"year":2026,"doi":"10.21437/Interspeech.2026-710","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26g_interspeech.pdf","session":"Domain Adaptation & Accented ASR","topics":["speech-translation","voice-conversion","self-supervised"],"category":"tts","labels":["efficient-on-device","self-supervised","streaming-real-time","generative-model"],"institutions":["Ajou University","Samsung"],"code":{"url":"https://accentdrift.github.io/demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26g_interspeech","category":"tts","labels":["efficient-on-device","self-supervised","streaming-real-time","generative-model"],"institutions":["Ajou University","Samsung"],"code":"https://accentdrift.github.io/demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-710","pdf":"https://www.isca-archive.org/interspeech_2026/lee26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26g_interspeech/markdown.md"},{"id":"lee26h_interspeech","title":"UR-BERT: Scaling Text Encoders for Massively Multilingual TTS Through Universal Romanization and Speech Token Prediction","authors":["Sangmin Lee","Eek Gyun Ahn","Woongjib Choi","Hong-Goo Kang"],"year":2026,"doi":"10.21437/Interspeech.2026-909","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26h_interspeech.pdf","session":"Text Processing for Speech Synthesis","topics":["tts","multilingual","self-supervised"],"category":"tts","labels":["low-resource","multilingual","self-supervised"],"institutions":["Yonsei University"],"funding":["National Research Foundation of Korea"],"code":{"url":"https://github.com/sanghyang00/ur-bert","stars":10,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26h_interspeech","category":"tts","labels":["low-resource","multilingual","self-supervised"],"institutions":["Yonsei University"],"code":"https://github.com/sanghyang00/ur-bert","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-909","pdf":"https://www.isca-archive.org/interspeech_2026/lee26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26h_interspeech/markdown.md"},{"id":"lee26i_interspeech","title":"Designed Vocalizations Dataset: Sound-Designed Human and Animal Voices for Non-human Voice Conversion","authors":["Seolhee Lee","Minsu Kang","Yangsun Lee","Woosun Min","Choonghyeon Lee","Namhyun Cho"],"year":2026,"doi":"10.21437/Interspeech.2026-932","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26i_interspeech.pdf","session":"Long-Form Speech Synthesis","topics":["voice-conversion","dataset","evaluation"],"category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["NC AI","Sogang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea Creative Content Agency"],"code":{"url":"https://ncai-official.github.io/speech/publications/designed-vocalizations-dataset/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26i_interspeech","category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["NC AI","Sogang University"],"code":"https://ncai-official.github.io/speech/publications/designed-vocalizations-dataset/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-932","pdf":"https://www.isca-archive.org/interspeech_2026/lee26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26i_interspeech/markdown.md"},{"id":"lee26j_interspeech","title":"WAND: Windowed Attention and Knowledge Distillation for Efficient Autoregressive Text-to-Speech Models","authors":["Hanna Lee","Tan Dat Nguyen","Jaehoon Kang","Kyuhong Shim"],"year":2026,"doi":"10.21437/Interspeech.2026-943","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26j_interspeech.pdf","session":"Scaling and Zero-Shot Speech Synthesis","topics":["tts","speech-llm","self-supervised"],"category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Korea Advanced Institute of Science and Technology","Sungkyunkwan University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"https://onemeee.github.io/wand-tts/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26j_interspeech","category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Korea Advanced Institute of Science and Technology","Sungkyunkwan University"],"code":"https://onemeee.github.io/wand-tts/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-943","pdf":"https://www.isca-archive.org/interspeech_2026/lee26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26j_interspeech/markdown.md"},{"id":"lee26k_interspeech","title":"Spatial-Magnifier: Spatial upsampling for multichannel speech enhancement","authors":["Dongheon Lee","Ashutosh Pandey","Sanjeel Parekh","Daniel Wong","Jacob Donley","Buye Xu","Juan Azcarreta"],"year":2026,"doi":"10.21437/Interspeech.2026-1477","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26k_interspeech.pdf","session":"Multi-Channel, Beamforming and Spatial Speech Enhancement","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","generative-model"],"institutions":["Meta","Korea Advanced Institute of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26k_interspeech","category":"enhancement-separation","labels":["efficient-on-device","generative-model"],"institutions":["Meta","Korea Advanced Institute of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1477","pdf":"https://www.isca-archive.org/interspeech_2026/lee26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26k_interspeech/markdown.md"},{"id":"lee26l_interspeech","title":"How Speaker Normalization Procedures Influence the Computational Modelling of Non-native Vowel Perception: Implications for the L2LP model","authors":["Jooyoung Lee","Kakeru Yazawa","James Whang","Paola Escudero"],"year":2026,"doi":"10.21437/Interspeech.2026-1574","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26l_interspeech.pdf","session":"Model of Speech Perception","topics":["phonetics","evaluation","multilingual"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Western Sydney University","University of Tsukuba","Seoul National University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26l_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Western Sydney University","University of Tsukuba","Seoul National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1574","pdf":"https://www.isca-archive.org/interspeech_2026/lee26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26l_interspeech/markdown.md"},{"id":"lee26m_interspeech","title":"From Awareness to Adherence: Bridging the Context Gap in Spoken Dialogue Systems via Context-Aware Decoding","authors":["Che Hyun Lee","Heeseung Kim","Sungroh Yoon"],"year":2026,"doi":"10.21437/Interspeech.2026-1589","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26m_interspeech.pdf","session":"Spoken Dialogue Systems","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","institutions":["Seoul National University","University of Seoul"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea government","National Research Foundation of Korea","BK21 FOUR Program","Samsung Electronics Co., Ltd"],"code":{"url":"https://github.com/saga1214/AudioCAD","stars":5,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26m_interspeech","category":"speech-llm-dialogue","institutions":["Seoul National University","University of Seoul"],"code":"https://github.com/saga1214/AudioCAD","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1589","pdf":"https://www.isca-archive.org/interspeech_2026/lee26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26m_interspeech/markdown.md"},{"id":"lee26n_interspeech","title":"DroFiT: A Lightweight Band-Fused Frequency Attention Toward Real-Time UAV Speech Enhancement","authors":["Jeongmin Lee","Chanhong Jeon","Hyungjoo Seo","Kyuhong Shim","Taewook Kang"],"year":2026,"doi":"10.21437/Interspeech.2026-1620","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26n_interspeech.pdf","session":"Multi-Channel Processing and Specialized Acquisition (UAV, Radar, Hearables)","topics":["speech-enhancement","on-device","self-supervised"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time","robustness-noise"],"institutions":["Sungkyunkwan University","University of Illinois Urbana-Champaign"],"funding":["National Research Foundation","IITP","AI Semiconductor Innovation Research Center, Sungkyunkwan University","KIAT"],"code":{"url":"https://ml-sp.github.io/DroFiT/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26n_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time","robustness-noise"],"institutions":["Sungkyunkwan University","University of Illinois Urbana-Champaign"],"code":"https://ml-sp.github.io/DroFiT/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1620","pdf":"https://www.isca-archive.org/interspeech_2026/lee26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26n_interspeech/markdown.md"},{"id":"lee26o_interspeech","title":"A Sensitivity Analysis of Multi-Event Audio Grounding in Audio LLMs","authors":["Taehan Lee","Jaehan Jung","Hyukjun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1684","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26o_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26o_interspeech.pdf","session":"Acoustic Event Detection 4","topics":["speech-llm","evaluation","self-supervised"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["Sogang University"],"code":{"url":"https://github.com/alm-evaluation/multi-event","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26o_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["Sogang University"],"code":"https://github.com/alm-evaluation/multi-event","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1684","pdf":"https://www.isca-archive.org/interspeech_2026/lee26o_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26o_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26o_interspeech/markdown.md"},{"id":"lee26p_interspeech","title":"Cross-linguistic word-medial stop lenition: A Functional PCA approach","authors":["Seung Suk Lee","Morgan Sonderegger","Meghan Clayards"],"year":2026,"doi":"10.21437/Interspeech.2026-1786","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26p_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26p_interspeech.pdf","session":"Cross-Linguistic and L2 Phonetic Studies","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["McGill University"],"funding":["Canada Research Chair","NSERC","SSHRC"],"code":{"url":"https://doi.org/10.17605/OSF.IO/CE65H","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26p_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["McGill University"],"code":"https://doi.org/10.17605/OSF.IO/CE65H","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1786","pdf":"https://www.isca-archive.org/interspeech_2026/lee26p_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26p_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26p_interspeech/markdown.md"},{"id":"lee26q_interspeech","title":"PhonePrune: One-shot Phoneme-Aware Pruning for Large-scale ASR Models via Phoneme Set Generation and Calibration","authors":["Minsik Lee","Jihie Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-1787","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26q_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26q_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 1","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["efficient-on-device"],"institutions":["Dongguk University"],"funding":["Ministry of Science and ICT","Information Technology Research Center","Institute for Information & Communications Technology Planning & Evaluation","Artificial Intelligence Convergence Innovation Human Resources Development"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26q_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["Dongguk University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1787","pdf":"https://www.isca-archive.org/interspeech_2026/lee26q_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26q_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26q_interspeech/markdown.md"},{"id":"lee26r_interspeech","title":"GETS: Guiding EMG-to-Speech Synthesis via Silent Speech Recognition","authors":["Jiwon Lee","Jaejun Lee","Kyogu Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1938","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26r_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26r_interspeech.pdf","session":"Articulatory, EMG, and Visual Speech Generation","topics":["speech-enhancement","self-supervised","dataset"],"category":"tts","labels":["generative-model"],"institutions":["Seoul National University"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT","National IT Industry Promotion Agency"],"code":{"url":"https://jiwonlee-0218.github.io/GETS_demo-page/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26r_interspeech","category":"tts","labels":["generative-model"],"institutions":["Seoul National University"],"code":"https://jiwonlee-0218.github.io/GETS_demo-page/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1938","pdf":"https://www.isca-archive.org/interspeech_2026/lee26r_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26r_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26r_interspeech/markdown.md"},{"id":"lee26s_interspeech","title":"Comparing Self-Supervised and Domain-Invariant Features for Cross-Domain Voice Phishing Detection","authors":["Jeongmin Lee","Seung Yun","Minkyu Lee","Ran Han","Yoonkyu Woo","Jinxia Huang"],"year":2026,"doi":"10.21437/Interspeech.2026-1975","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26s_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26s_interspeech.pdf","session":"Spoofing and Deepfake Detection 1","topics":["speech-enhancement","self-supervised","on-device"],"category":"deepfake-security","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["Electronics and Telecommunications Research Institute","University of Science and Technology"],"funding":["Institute of Information and Communications Technology Planning and Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26s_interspeech","category":"deepfake-security","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["Electronics and Telecommunications Research Institute","University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1975","pdf":"https://www.isca-archive.org/interspeech_2026/lee26s_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26s_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26s_interspeech/markdown.md"},{"id":"lee26t_interspeech","title":"Progressive Alignment Objectives for Aligner-Encoder based ASR","authors":["Jaeyoung Lee","Masato Mimura","Takafumi Moriya"],"year":2026,"doi":"10.21437/Interspeech.2026-2132","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26t_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26t_interspeech.pdf","session":"New Training Methods for ASR","topics":["asr","self-supervised"],"category":"asr","institutions":["NTT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26t_interspeech","category":"asr","institutions":["NTT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2132","pdf":"https://www.isca-archive.org/interspeech_2026/lee26t_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26t_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26t_interspeech/markdown.md"},{"id":"lee26u_interspeech","title":"LLM-as-Joiner: Decoupling Alignment from Language Modeling in Label-synchronous ASR","authors":["Jaeyoung Lee","Masato Mimura","Ryo Magoshi","Tatsuya Kawahara"],"year":2026,"doi":"10.21437/Interspeech.2026-2149","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26u_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26u_interspeech.pdf","session":"New Training Methods for ASR","topics":["asr","speech-llm","multilingual"],"category":"asr","institutions":["NTT","Kyoto University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26u_interspeech","category":"asr","institutions":["NTT","Kyoto University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2149","pdf":"https://www.isca-archive.org/interspeech_2026/lee26u_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26u_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26u_interspeech/markdown.md"},{"id":"lee26v_interspeech","title":"Enhancing EMG-to-Speech via Silent-Voiced Representation Alignment","authors":["Jiwon Lee","Injune Hwang","Jaejun Lee","Eungbeom Kim","Dongyub Han","Kyogu Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-2931","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26v_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26v_interspeech.pdf","session":"Articulatory, EMG, and Visual Speech Generation","topics":["tts","speech-enhancement","dataset"],"category":"tts","labels":["generative-model"],"institutions":["Seoul National University"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT","National IT Industry Promotion Agency"],"code":{"url":"https://github.com/jiwonlee-0218/SVAETS","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26v_interspeech","category":"tts","labels":["generative-model"],"institutions":["Seoul National University"],"code":"https://github.com/jiwonlee-0218/SVAETS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2931","pdf":"https://www.isca-archive.org/interspeech_2026/lee26v_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26v_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26v_interspeech/markdown.md"},{"id":"lee26w_interspeech","title":"Diffusion Bridge Learning Between Overfitted and Underfitted Representations for speech emotion recognition","authors":["Shi-wook Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-2996","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26w_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26w_interspeech.pdf","session":"Speech Emotion Recognition and Representation 3","topics":["speech-emotion-recognition","self-supervised","multilingual"],"category":"paralinguistics-emotion","labels":["multilingual","self-supervised","generative-model"],"institutions":["National Institute of Advanced Industrial Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26w_interspeech","category":"paralinguistics-emotion","labels":["multilingual","self-supervised","generative-model"],"institutions":["National Institute of Advanced Industrial Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2996","pdf":"https://www.isca-archive.org/interspeech_2026/lee26w_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26w_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26w_interspeech/markdown.md"},{"id":"lee26x_interspeech","title":"AGENT: A Black-box Adversarial Attack Exposing the Achilles' Heel of SASV Systems","authors":["Yowon Lee","Seongkyu Han","Thien-Phuc Doan","Souhwan Jung"],"year":2026,"doi":"10.21437/Interspeech.2026-3207","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26x_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26x_interspeech.pdf","session":"Speaker Verification and Anti-Spoofing","topics":["speaker-verification","evaluation","self-supervised"],"category":"deepfake-security","institutions":["Soongsil University"],"funding":["Cyber Investigation Support Technology Development Program","Korea Institute of Police Technology","National Research Foundation of Korea"],"code":{"url":"https://github.com/2oil/AGENT.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26x_interspeech","category":"deepfake-security","institutions":["Soongsil University"],"code":"https://github.com/2oil/AGENT.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3207","pdf":"https://www.isca-archive.org/interspeech_2026/lee26x_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26x_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26x_interspeech/markdown.md"},{"id":"lee26y_interspeech","title":"Surgical-Robot Command Spotting: Safety-Aware Learning for Compositional Commands","authors":["Jaewon Lee","Sang-Beom Lee","Bum-Hwi Kim","Kwang-Yong Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-3372","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26y_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26y_interspeech.pdf","session":"Speech and Language Technologies in Healthcare","topics":["keyword-spotting","speech-enhancement","low-resource"],"category":"asr","labels":["efficient-on-device"],"institutions":["Electronics and Telecommunications Research Institute"],"funding":["Electronics and Telecommunications Research Institute","Commercialization Promotion Agency for R&D Outcomes","Korean government","Ministry of Science and ICT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26y_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["Electronics and Telecommunications Research Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3372","pdf":"https://www.isca-archive.org/interspeech_2026/lee26y_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26y_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26y_interspeech/markdown.md"},{"id":"lee26z_interspeech","title":"Trajectory Variance: An Unsupervised Measure of Developmental Vocal Plasticity in Birdsong","authors":["Kanghwi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-3557","isca_url":"https://www.isca-archive.org/interspeech_2026/lee26z_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lee26z_interspeech.pdf","session":"Acoustic Event Detection 4","topics":["paralinguistics","self-supervised","evaluation"],"category":"audio-understanding","institutions":["University of Zurich","ETH Zurich"],"funding":["Swiss National Science Foundation"],"code":{"url":"https://github.com/hwiora/trajectory_variance","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lee26z_interspeech","category":"audio-understanding","institutions":["University of Zurich","ETH Zurich"],"code":"https://github.com/hwiora/trajectory_variance","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3557","pdf":"https://www.isca-archive.org/interspeech_2026/lee26z_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26z_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lee26z_interspeech/markdown.md"},{"id":"lei26_interspeech","title":"ARCHES: An Agent-Based Refinement Cycle for Hierarchical Synthesis of Sound Effects for Variety Shows","authors":["Wentao Lei","Li Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-561","isca_url":"https://www.isca-archive.org/interspeech_2026/lei26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lei26_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["speech-enhancement","evaluation","speech-llm"],"category":"audio-understanding","labels":["generative-model"],"institutions":["Hong Kong University of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lei26_interspeech","category":"audio-understanding","labels":["generative-model"],"institutions":["Hong Kong University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-561","pdf":"https://www.isca-archive.org/interspeech_2026/lei26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lei26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lei26_interspeech/markdown.md"},{"id":"lemerle26_interspeech","title":"Low-Framerate Speech Tokenization via Two-Stage Latent Patch Modeling","authors":["Théodor Lemerle","Diego Torres","Téo Guichoux","Nicolas Obin","Axel Roebel"],"year":2026,"doi":"10.21437/Interspeech.2026-2863","isca_url":"https://www.isca-archive.org/interspeech_2026/lemerle26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lemerle26_interspeech.pdf","session":"Speech Synthesis: Speech Features, Codec and Representations","topics":["tts","speech-llm","self-supervised"],"category":"speech-coding","labels":["efficient-on-device","self-supervised"],"institutions":["IRCAM","Sorbonne Universite","CNRS"],"code":{"url":"https://github.com/theodorblackbird/z-codec","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lemerle26_interspeech","category":"speech-coding","labels":["efficient-on-device","self-supervised"],"institutions":["IRCAM","Sorbonne Universite","CNRS"],"code":"https://github.com/theodorblackbird/z-codec","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2863","pdf":"https://www.isca-archive.org/interspeech_2026/lemerle26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lemerle26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lemerle26_interspeech/markdown.md"},{"id":"lentz26_interspeech","title":"BeatGain - A Rhythmic Pattern Enhancement Algorithm for Music Listening with Cochlear Implants","authors":["Benjamin Lentz","Theresa Hartmann","Anil Nagathil","Ian Bruce","Rainer Martin"],"year":2026,"doi":"10.21437/Interspeech.2026-2713","isca_url":"https://www.isca-archive.org/interspeech_2026/lentz26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lentz26_interspeech.pdf","session":"Acoustic Signal Analysis and Generation","topics":["source-separation","speech-enhancement","evaluation"],"category":"health-clinical","institutions":["Ruhr-Universität Bochum","McMaster University"],"code":{"url":"https://www.ika.ruhr-uni-bochum.de/ika/demos/beatgain","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lentz26_interspeech","category":"health-clinical","institutions":["Ruhr-Universität Bochum","McMaster University"],"code":"https://www.ika.ruhr-uni-bochum.de/ika/demos/beatgain","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2713","pdf":"https://www.isca-archive.org/interspeech_2026/lentz26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lentz26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lentz26_interspeech/markdown.md"},{"id":"leoni26_interspeech","title":"Indigenising Speech Technology: Building a TTS Model for te Reo Māori","authors":["Gianna Leoni","Peter-Lucas Jones","Tūreiti Keith","Suzanne Duncan"],"year":2026,"doi":"10.21437/Interspeech.2026-1443","isca_url":"https://www.isca-archive.org/interspeech_2026/leoni26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/leoni26_interspeech.pdf","session":"Indigenous Voices in Speech Science and Technology","topics":["tts","low-resource","multilingual"],"category":"tts","labels":["low-resource","multilingual","generative-model"],"institutions":["Te Hiku Media"],"funding":["Te Puni Kokiri","Ministry for Business, Innovation and Employment"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"leoni26_interspeech","category":"tts","labels":["low-resource","multilingual","generative-model"],"institutions":["Te Hiku Media"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1443","pdf":"https://www.isca-archive.org/interspeech_2026/leoni26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/leoni26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/leoni26_interspeech/markdown.md"},{"id":"li26_interspeech","title":"XAI-Grounded Explanation Generation for Speech Deepfake Detection with Training-Free Multimodal Large Language Models","authors":["Yupei Li","Qiyang Sun","Xiaoliang Wu","Chenxi Wang","Berrak Sisman","Bjoern Schuller"],"year":2026,"doi":"10.21437/Interspeech.2026-161","isca_url":"https://www.isca-archive.org/interspeech_2026/li26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","institutions":["Imperial College London","Technical University of Munich","University of Southampton","Mohamed bin Zayed University of Artificial Intelligence","Johns Hopkins University"],"code":{"url":"https://github.com/glam-imperial/xai-grounded-speech-deepfake","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26_interspeech","category":"deepfake-security","institutions":["Imperial College London","Technical University of Munich","University of Southampton","Mohamed bin Zayed University of Artificial Intelligence","Johns Hopkins University"],"code":"https://github.com/glam-imperial/xai-grounded-speech-deepfake","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-161","pdf":"https://www.isca-archive.org/interspeech_2026/li26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26_interspeech/markdown.md"},{"id":"li26aa_interspeech","title":"Weakly Masked Residual Reliability Learning for Unsupervised Domain Adaptation in Speech Models","authors":["Yuan Li","Yonghe Wang","Zhenjie Gao","Feilong Bao","Xiaodong Yang","Bo Pang","Yandong Guo"],"year":2026,"doi":"10.21437/Interspeech.2026-1767","isca_url":"https://www.isca-archive.org/interspeech_2026/li26aa_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26aa_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 2","topics":["asr","speech-translation","self-supervised"],"category":"asr","labels":["robustness-noise"],"institutions":["Inner Mongolia University","National Computer Network Emergency Response Technical Team Coordination Center"],"funding":["National Natural Science Foundation of China","Science and Technology Major Project of Inner Mongolia","Natural Science Foundation of Inner Mongolia","Science and Technology Program of Inner Mongolia","Inner Mongolia Autonomous Region First-Class Discipline Research Special Project"],"code":{"url":"https://anonymous.4open.science/status/Speech-Model-Adaptation-8D45","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26aa_interspeech","category":"asr","labels":["robustness-noise"],"institutions":["Inner Mongolia University","National Computer Network Emergency Response Technical Team Coordination Center"],"code":"https://anonymous.4open.science/status/Speech-Model-Adaptation-8D45","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1767","pdf":"https://www.isca-archive.org/interspeech_2026/li26aa_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26aa_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26aa_interspeech/markdown.md"},{"id":"li26b_interspeech","title":"The Effect of Neck Skin Vibration on the Periauricular Acoustic Receiver","authors":["Ruoyan Li","Yuhao Sun","Fan Fan"],"year":2026,"doi":"10.21437/Interspeech.2026-170","isca_url":"https://www.isca-archive.org/interspeech_2026/li26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26b_interspeech.pdf","session":"Speech Production and Perception 1","topics":["speech-enhancement","paralinguistics","evaluation"],"category":"enhancement-separation","institutions":["Huawei"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26b_interspeech","category":"enhancement-separation","institutions":["Huawei"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-170","pdf":"https://www.isca-archive.org/interspeech_2026/li26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26b_interspeech/markdown.md"},{"id":"li26ba_interspeech","title":"Resonate: Reinforcing Text-to-Audio Generation via Online Feedback from Large Audio Language Models","authors":["Xiquan Li","Junxi Liu","Wenxi Chen","Haina Zhu","Ziyang Ma","Xie Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1823","isca_url":"https://www.isca-archive.org/interspeech_2026/li26ba_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26ba_interspeech.pdf","session":"Audio Foundation Models and Generation","topics":["tts","self-supervised","dataset"],"category":"tts","labels":["generative-model"],"institutions":["Shanghai Jiao Tong University","Shanghai Innovation Institute"],"funding":["Science and Technology Innovation (STI) 2030-Major Project","National Natural Science Foundation of China","Shanghai Municipal Science and Technology Major Project","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"https://github.com/xiquan-li/Resonate","stars":51,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26ba_interspeech","category":"tts","labels":["generative-model"],"institutions":["Shanghai Jiao Tong University","Shanghai Innovation Institute"],"code":"https://github.com/xiquan-li/Resonate","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1823","pdf":"https://www.isca-archive.org/interspeech_2026/li26ba_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ba_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ba_interspeech/markdown.md"},{"id":"li26c_interspeech","title":"MSR-Codec: A Low-Bitrate Multi-Stream Residual Codec for High-Fidelity Speech Generation with Information Disentanglement","authors":["Jingyu Li","Guangyan Zhang","Zhen Ye","Yiwen Guo"],"year":2026,"doi":"10.21437/Interspeech.2026-301","isca_url":"https://www.isca-archive.org/interspeech_2026/li26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26c_interspeech.pdf","session":"Neural Audio Codec Architectures","topics":["tts","speech-generation","self-supervised"],"category":"speech-coding","labels":["efficient-on-device","generative-model"],"institutions":["LIGHTSPEED","Hong Kong University of Science and Technology"],"code":{"url":"https://github.com/herbertLJY/MSRCodec","stars":14,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26c_interspeech","category":"speech-coding","labels":["efficient-on-device","generative-model"],"institutions":["LIGHTSPEED","Hong Kong University of Science and Technology"],"code":"https://github.com/herbertLJY/MSRCodec","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-301","pdf":"https://www.isca-archive.org/interspeech_2026/li26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26c_interspeech/markdown.md"},{"id":"li26ca_interspeech","title":"Self-supervised Speaker Verification with High-Confidence Pseudo-Label Selection and DINO-Style Self-Distillation Based on Pre-trained Models","authors":["Yishuang Li","Yi Yu","Wanli Dong","Weihao Gan"],"year":2026,"doi":"10.21437/Interspeech.2026-1965","isca_url":"https://www.isca-archive.org/interspeech_2026/li26ca_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26ca_interspeech.pdf","session":"Speaker Recognition and Verification","topics":["speaker-verification","self-supervised","speech-llm"],"category":"speaker","labels":["self-supervised"],"institutions":["Malanshan Audio and Video Laboratory"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26ca_interspeech","category":"speaker","labels":["self-supervised"],"institutions":["Malanshan Audio and Video Laboratory"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1965","pdf":"https://www.isca-archive.org/interspeech_2026/li26ca_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ca_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ca_interspeech/markdown.md"},{"id":"li26d_interspeech","title":"CAQA-Net: Continual Audio Quality Assessment Across Speech and Music Domains","authors":["Naiyuan Li","Xiaoxun Wu","Yuheng Huang","Diqun Yan"],"year":2026,"doi":"10.21437/Interspeech.2026-476","isca_url":"https://www.isca-archive.org/interspeech_2026/li26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26d_interspeech.pdf","session":"Evaluation, Benchmarking, and Reliability of Audio Systems","topics":["speech-enhancement","self-supervised","evaluation"],"category":"resources-evaluation","institutions":["Ningbo University","Ningbo University of Finance and Economics"],"funding":["National Natural Science Foundation of China","Zhejiang Provincial Collaborative Innovation Center for Digital Supply Chain and Artificial Intelligence of Bulk Commodities"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26d_interspeech","category":"resources-evaluation","institutions":["Ningbo University","Ningbo University of Finance and Economics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-476","pdf":"https://www.isca-archive.org/interspeech_2026/li26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26d_interspeech/markdown.md"},{"id":"li26da_interspeech","title":"Carrier-Aware Sound Zone Control for Parametric Array Loudspeakers","authors":["Mengtong Li","Tao Zhuang","Yu Sun","Shaozhe Li","Xuxiang Wu","Jia-Xin Zhong","Jing Lu"],"year":2026,"doi":"10.21437/Interspeech.2026-2170","isca_url":"https://www.isca-archive.org/interspeech_2026/li26da_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26da_interspeech.pdf","session":"Active Noise and Echo Control, Sound Zones and Packet-Loss Concealment","topics":["spatial-audio","sound-zone-control"],"category":"enhancement-separation","institutions":["Nanjing University","Samsung Electronics","Horizon Robotics"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26da_interspeech","category":"enhancement-separation","institutions":["Nanjing University","Samsung Electronics","Horizon Robotics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2170","pdf":"https://www.isca-archive.org/interspeech_2026/li26da_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26da_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26da_interspeech/markdown.md"},{"id":"li26e_interspeech","title":"Noisy Environment Adaptation of Neural Speech Codec via Focal Mask and Noise Feature Separation","authors":["Shaokai Li","Weiping Tu","Yuhong Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-512","isca_url":"https://www.isca-archive.org/interspeech_2026/li26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26e_interspeech.pdf","session":"Neural Speech Codecs: Low-Bitrate and Disentangled Coding","topics":["speech-enhancement","self-supervised","speech-coding"],"category":"speech-coding","labels":["robustness-noise"],"institutions":["Wuhan University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://github.com/shaokai1209/FocalSE","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26e_interspeech","category":"speech-coding","labels":["robustness-noise"],"institutions":["Wuhan University"],"code":"https://github.com/shaokai1209/FocalSE","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-512","pdf":"https://www.isca-archive.org/interspeech_2026/li26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26e_interspeech/markdown.md"},{"id":"li26ea_interspeech","title":"U2A-Net: Physically Motivated Ultrasound‑to‑Audio Neural Modeling for Parametric Array Loudspeakers","authors":["Mengtong Li","Yu Sun","Jia-Xin Zhong","Jing Lu"],"year":2026,"doi":"10.21437/Interspeech.2026-2390","isca_url":"https://www.isca-archive.org/interspeech_2026/li26ea_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26ea_interspeech.pdf","session":"Spatial Audio 2","topics":["speech-enhancement","evaluation"],"category":"enhancement-separation","institutions":["Nanjing University","Samsung Electronics","Horizon Robotics"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26ea_interspeech","category":"enhancement-separation","institutions":["Nanjing University","Samsung Electronics","Horizon Robotics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2390","pdf":"https://www.isca-archive.org/interspeech_2026/li26ea_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ea_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ea_interspeech/markdown.md"},{"id":"li26f_interspeech","title":"Perceptual Trade-offs Across Segmental and Suprasegmental Levels: Comparing Ganong Patterns in Mandarin Consonants and Lexical Tones","authors":["Jiaxin LI","Yi Weng","Yicheng Rong","Gang Peng"],"year":2026,"doi":"10.21437/Interspeech.2026-576","isca_url":"https://www.isca-archive.org/interspeech_2026/li26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26f_interspeech.pdf","session":"Perception and Appraisal of Prosody","topics":["phonetics","prosody","evaluation"],"category":"phonetics-linguistics","institutions":["Hong Kong Polytechnic University","Shanghai Jiao Tong University"],"funding":["Research Grants Council of the Hong Kong SAR","Shanghai Planning Project of Philosophy and Social Science"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26f_interspeech","category":"phonetics-linguistics","institutions":["Hong Kong Polytechnic University","Shanghai Jiao Tong University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-576","pdf":"https://www.isca-archive.org/interspeech_2026/li26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26f_interspeech/markdown.md"},{"id":"li26fa_interspeech","title":"Language-Invariant Multilingual Speaker Verification for the TidyVoice 2026 Challenge","authors":["Ze Li","Xiaoxiao Miao","Juan Liu","Ming Li"],"year":2026,"doi":"10.21437/Interspeech.2026-2437","isca_url":"https://www.isca-archive.org/interspeech_2026/li26fa_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26fa_interspeech.pdf","session":"Challenge - TidyVoice Challenge: Cross-Lingual Speaker Verification","topics":["speaker-verification","multilingual","self-supervised"],"category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Wuhan University","Chinese University of Hong Kong, Shenzhen","Duke Kunshan University"],"funding":["National Natural Science Foundation of China","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"https://github.com/ZXHY-82/LI-MSV-TidyVoice2026","stars":12,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26fa_interspeech","category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Wuhan University","Chinese University of Hong Kong, Shenzhen","Duke Kunshan University"],"code":"https://github.com/ZXHY-82/LI-MSV-TidyVoice2026","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2437","pdf":"https://www.isca-archive.org/interspeech_2026/li26fa_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26fa_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26fa_interspeech/markdown.md"},{"id":"li26g_interspeech","title":"Aleatoric Style Uncertainty Augmentation with GMM for Domain Generalization in Anti-spoofing","authors":["Jin Li","Man-Wai Mak","Johan Rohdin","Oldřich Plchot","Kong Aik Lee","Bo Wen","Yunfeng Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-582","isca_url":"https://www.isca-archive.org/interspeech_2026/li26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26g_interspeech.pdf","session":"Spoofing and Deepfake Detection 1","topics":["anti-spoofing","speaker-verification","self-supervised"],"category":"deepfake-security","labels":["robustness-noise"],"institutions":["Hong Kong Polytechnic University","Brno University of Technology","Shenzhen Zhuiyi Technology"],"funding":["Innovation and Technology Fund of the Hong Kong SAR","National Key R&D Program of China","Digital Europe Programme","Ministry of Education, Youth and Sports of the Czech Republic"],"code":{"url":"https://github.com/happyjin/ASU","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26g_interspeech","category":"deepfake-security","labels":["robustness-noise"],"institutions":["Hong Kong Polytechnic University","Brno University of Technology","Shenzhen Zhuiyi Technology"],"code":"https://github.com/happyjin/ASU","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-582","pdf":"https://www.isca-archive.org/interspeech_2026/li26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26g_interspeech/markdown.md"},{"id":"li26ga_interspeech","title":"Gated Multi-graph Fusion via Graph Attention Networks for Alzheimer’s Disease Detection","authors":["Jinyu Li","Xiao Wei","Bin Wen","Kai Li","Yuqin Lin","Xiaobao Wang","Longbiao Wang","Jianwu Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-2578","isca_url":"https://www.isca-archive.org/interspeech_2026/li26ga_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26ga_interspeech.pdf","session":"Pathological Speech Assessment 2","topics":["speech-llm","evaluation","health"],"category":"health-clinical","institutions":["Tianjin University","Chinese Academy of Sciences","Fuzhou University","Huiyan Technology"],"funding":["National Natural Science Foundation of China","National Talent Program"],"code":{"url":"https://github.com/opeacc/AD","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26ga_interspeech","category":"health-clinical","institutions":["Tianjin University","Chinese Academy of Sciences","Fuzhou University","Huiyan Technology"],"code":"https://github.com/opeacc/AD","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2578","pdf":"https://www.isca-archive.org/interspeech_2026/li26ga_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ga_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ga_interspeech/markdown.md"},{"id":"li26h_interspeech","title":"SRF-SVB: Style-Consistent Singing Voice Beautifying via Rectified Flow","authors":["Wenhui Li","Biao Dong","Liwei Hu","Jiqing Han","Yongjun He"],"year":2026,"doi":"10.21437/Interspeech.2026-615","isca_url":"https://www.isca-archive.org/interspeech_2026/li26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26h_interspeech.pdf","session":"Singing Voice and Music Generation","topics":["voice-conversion","tts","self-supervised"],"category":"tts","labels":["multilingual","generative-model"],"institutions":["Harbin Institute of Technology"],"code":{"url":"https://mrwho729.github.io/SRF-SVB/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26h_interspeech","category":"tts","labels":["multilingual","generative-model"],"institutions":["Harbin Institute of Technology"],"code":"https://mrwho729.github.io/SRF-SVB/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-615","pdf":"https://www.isca-archive.org/interspeech_2026/li26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26h_interspeech/markdown.md"},{"id":"li26ha_interspeech","title":"Phonetic evidence for contrastive voicing in Nakanamanga coronal plosives","authors":["Shubo Li"],"year":2026,"doi":"10.21437/Interspeech.2026-2832","isca_url":"https://www.isca-archive.org/interspeech_2026/li26ha_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26ha_interspeech.pdf","session":"Voice Quality Aspects of Speech","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["Australian National University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26ha_interspeech","category":"phonetics-linguistics","institutions":["Australian National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2832","pdf":"https://www.isca-archive.org/interspeech_2026/li26ha_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ha_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ha_interspeech/markdown.md"},{"id":"li26i_interspeech","title":"DAR-Boost: A Differentiable and Adaptive Raw Data Augmentation Framework for Robust Anti-Spoofing","authors":["Yingdong Li","Chengxin Chen","Nanli Zeng","Jianguo Hu","Kun Zeng"],"year":2026,"doi":"10.21437/Interspeech.2026-643","isca_url":"https://www.isca-archive.org/interspeech_2026/li26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26i_interspeech.pdf","session":"Spoofing and Deepfake Detection 3","topics":["self-supervised","evaluation","audio-deepfake"],"category":"deepfake-security","labels":["robustness-noise"],"institutions":["Sun Yat-sen University","China Mobile Internet Co., Ltd"],"funding":["National Key Research and Development Program of China"],"code":{"url":"https://github.com/lydsera/DAR-Boost","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26i_interspeech","category":"deepfake-security","labels":["robustness-noise"],"institutions":["Sun Yat-sen University","China Mobile Internet Co., Ltd"],"code":"https://github.com/lydsera/DAR-Boost","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-643","pdf":"https://www.isca-archive.org/interspeech_2026/li26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26i_interspeech/markdown.md"},{"id":"li26ia_interspeech","title":"HASS: Hierarchical Simulation of Logopenic Aphasic Speech for Scalable PPA Detection","authors":["Harrison Li","Kevin Wang","Cheol Jun Cho","Jiachen Lian","Rabab Rangwala","Chenxu Guo","Emma Yang","Lynn Kurteff","Zoe Ezzes","Willa Keegan-Rodewald","Jet Vonk","Siddarth Ramkrishnan","Giada Antonicelli","Zachary Miller","Marilu Gorno Tempini","Gopala Anumanchipalli"],"year":2026,"doi":"10.21437/Interspeech.2026-3080","isca_url":"https://www.isca-archive.org/interspeech_2026/li26ia_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26ia_interspeech.pdf","session":"Pathological Speech Assessment 2","topics":["speech-llm","health","evaluation"],"category":"health-clinical","labels":["low-resource","generative-model"],"institutions":["UC Berkeley","UCSF","Zhejiang University","Columbia University","Basque Center on Cognition, Brain and Language"],"code":{"url":"https://anonymous.4open.science/r/HASS-890D","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26ia_interspeech","category":"health-clinical","labels":["low-resource","generative-model"],"institutions":["UC Berkeley","UCSF","Zhejiang University","Columbia University","Basque Center on Cognition, Brain and Language"],"code":"https://anonymous.4open.science/r/HASS-890D","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3080","pdf":"https://www.isca-archive.org/interspeech_2026/li26ia_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ia_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ia_interspeech/markdown.md"},{"id":"li26j_interspeech","title":"Towards Paradigm-General Suicide Risk Detection via Speech LLM","authors":["Jialun Li","Weitao Jiang","Ziyun Cui","Yinan Duan","Diyang Qu","Chao Zhang","Runsen Chen","Chang Lei","Wen Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-666","isca_url":"https://www.isca-archive.org/interspeech_2026/li26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26j_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 1","topics":["speech-llm","paralinguistics","emotion-recognition"],"category":"health-clinical","institutions":["Shanghai Artificial Intelligence Laboratory","Tsinghua University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26j_interspeech","category":"health-clinical","institutions":["Shanghai Artificial Intelligence Laboratory","Tsinghua University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-666","pdf":"https://www.isca-archive.org/interspeech_2026/li26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26j_interspeech/markdown.md"},{"id":"li26ja_interspeech","title":"A Corpus-Based Study of Creaky Voice Production in English and Mandarin","authors":["Meixian Li","Yao Yao","Charles Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-3235","isca_url":"https://www.isca-archive.org/interspeech_2026/li26ja_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26ja_interspeech.pdf","session":"Voice Quality Aspects of Speech","topics":["paralinguistics","phonetics","dataset"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Hong Kong Polytechnic University","City University of Hong Kong"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26ja_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Hong Kong Polytechnic University","City University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3235","pdf":"https://www.isca-archive.org/interspeech_2026/li26ja_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ja_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ja_interspeech/markdown.md"},{"id":"li26k_interspeech","title":"POTSA: A Cross-Lingual Speech Alignment Framework for Speech-to-Text Translation","authors":["Xuanchen Li","Chenrui Cui","Tianrui Wang","Meng Ge","Zikang Huang","Yizhou Peng","Jin Li","Yuheng Lu","Yu Jiang","Nyima Tashi","Longbiao Wang","Jianwu Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-695","isca_url":"https://www.isca-archive.org/interspeech_2026/li26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26k_interspeech.pdf","session":"Multilingual Speech 2","topics":["speech-translation","multilingual","self-supervised"],"category":"translation","labels":["low-resource","multilingual","self-supervised"],"institutions":["Tianjin University","Nanyang Technological University","Huiyan Technology Company","Tibet University","Chinese Academy of Sciences"],"code":{"url":"https://github.com/Sslnon/POTSA","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26k_interspeech","category":"translation","labels":["low-resource","multilingual","self-supervised"],"institutions":["Tianjin University","Nanyang Technological University","Huiyan Technology Company","Tibet University","Chinese Academy of Sciences"],"code":"https://github.com/Sslnon/POTSA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-695","pdf":"https://www.isca-archive.org/interspeech_2026/li26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26k_interspeech/markdown.md"},{"id":"li26ka_interspeech","title":"Read What You Hear: Reference-Free Hypotheses Evaluation with Acoustic Discrepancy","authors":["Zhihan Li","Hankun Wang","Yiwei Guo","Bohan Li","Xie Chen","Kai Yu"],"year":2026,"doi":"10.21437/Interspeech.2026-3434","isca_url":"https://www.isca-archive.org/interspeech_2026/li26ka_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26ka_interspeech.pdf","session":"Spoken Language Processing: Evaluation and Metrics","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["generative-model"],"institutions":["Shanghai Jiao Tong University"],"funding":["China NSFC Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26ka_interspeech","category":"asr","labels":["generative-model"],"institutions":["Shanghai Jiao Tong University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3434","pdf":"https://www.isca-archive.org/interspeech_2026/li26ka_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ka_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26ka_interspeech/markdown.md"},{"id":"li26l_interspeech","title":"HFMSE: Harmonic-Guided Speech Enhancement with Flow Matching","authors":["Jizhen Li","Weiping Tu","Yuhong Yang","Xinhong Li"],"year":2026,"doi":"10.21437/Interspeech.2026-722","isca_url":"https://www.isca-archive.org/interspeech_2026/li26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26l_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","self-supervised","prosody"],"category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["Wuhan University"],"funding":["National Nature Science Foundation of China","Hubei Provincial Science and Technology Plan Project"],"code":{"url":"https://github.com/xxnhq/HFSE","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26l_interspeech","category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["Wuhan University"],"code":"https://github.com/xxnhq/HFSE","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-722","pdf":"https://www.isca-archive.org/interspeech_2026/li26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26l_interspeech/markdown.md"},{"id":"li26la_interspeech","title":"Spatially-Augmented Sequence-to-Sequence Neural Diarization for Meetings","authors":["Li Li","Ming Cheng","Juan Liu","Ming Li"],"year":2026,"doi":"10.21437/Interspeech.2026-3473","isca_url":"https://www.isca-archive.org/interspeech_2026/li26la_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26la_interspeech.pdf","session":"Speaker Diarization 1","topics":["speaker-diarization","multilingual","self-supervised"],"category":"speaker","institutions":["Wuhan University","Chinese University of Hong Kong, Shenzhen"],"funding":["National Natural Science Foundation of China","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26la_interspeech","category":"speaker","institutions":["Wuhan University","Chinese University of Hong Kong, Shenzhen"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3473","pdf":"https://www.isca-archive.org/interspeech_2026/li26la_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26la_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26la_interspeech/markdown.md"},{"id":"li26m_interspeech","title":"GenTSE: Enhancing Target Speaker Extraction via a Coarse-to-Fine Generative Language Model","authors":["Haoyang Li","Xuyi Zhuang","Azmat Adnan","Ye Ni","Wei Rao","Shreyas Gopal","Eng Siong Chng","Boon Siew Han","Yuanjin Zheng"],"year":2026,"doi":"10.21437/Interspeech.2026-893","isca_url":"https://www.isca-archive.org/interspeech_2026/li26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26m_interspeech.pdf","session":"Audio-Visual and Generative Target Speaker Extraction","topics":["target-speaker-extraction","self-supervised","speech-enhancement"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Nanyang Technological University","Southeast University","Schaeffler"],"funding":["RIE2025 Industry Alignment Fund - Industry Collaboration Projects","A*STAR","Schaeffler (Singapore) PTE. LTD","NTU Singapore","Schaeffler-NTU Corporate Lab: Intelligent Mechatronics Hub"],"code":{"url":"https://huggingface.co/yaoxunji/gen-se","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26m_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Nanyang Technological University","Southeast University","Schaeffler"],"code":"https://huggingface.co/yaoxunji/gen-se","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-893","pdf":"https://www.isca-archive.org/interspeech_2026/li26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26m_interspeech/markdown.md"},{"id":"li26n_interspeech","title":"Online Audio-Visual Target Speaker Extraction with Viseme-Guided Lightweight Visual Pretraining","authors":["Zixuan Li","Xueliang Zhang","Lei Miao","Zhipeng Yan","Ying Sun","Chong Zhu"],"year":2026,"doi":"10.21437/Interspeech.2026-948","isca_url":"https://www.isca-archive.org/interspeech_2026/li26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26n_interspeech.pdf","session":"Audio-Visual and Generative Target Speaker Extraction","topics":["source-separation","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Inner Mongolia University","Lenovo"],"funding":["Inner Mongolia Natural Science Foundation","Hohhot R&D Investment Incentive Program","Inner Mongolia Postgraduate Research Project","CCF-Lenovo Research Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26n_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Inner Mongolia University","Lenovo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-948","pdf":"https://www.isca-archive.org/interspeech_2026/li26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26n_interspeech/markdown.md"},{"id":"li26o_interspeech","title":"Audio-Cogito: Towards Deep Audio Reasoning in Large Audio Language Models","authors":["Longhao Li","Hongjie Chen","Zehan Li","Qihan Hu","Jian Kang","Jie Li","Lei Xie","Yongxiang Li"],"year":2026,"doi":"10.21437/Interspeech.2026-988","isca_url":"https://www.isca-archive.org/interspeech_2026/li26o_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26o_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","spoken-language-understanding","self-supervised"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Northwestern Polytechnical University","China Telecom Artificial Intelligence Technology (Beijing) Co., Ltd"],"code":{"url":"https://github.com/llh666521/Audio-Cogito","stars":6,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26o_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Northwestern Polytechnical University","China Telecom Artificial Intelligence Technology (Beijing) Co., Ltd"],"code":"https://github.com/llh666521/Audio-Cogito","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-988","pdf":"https://www.isca-archive.org/interspeech_2026/li26o_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26o_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26o_interspeech/markdown.md"},{"id":"li26p_interspeech","title":"Effects of Co-speech Gesture on the Acoustic Realization of Focus in Cantonese-speaking Children With and Without Autism Spectrum Disorder","authors":["Zhuoran Li","Si Chen","Yitian Hong","Bingxin Liu","Ho-Yi Ku","Jiayue Gao","Chun-Sing Wong","Angel Chan","Zhuoming Chen","Haoyan Ge","Bin Li","Li Sheng","Po-yi Tempo Tang","Ratree Wayland"],"year":2026,"doi":"10.21437/Interspeech.2026-1008","isca_url":"https://www.isca-archive.org/interspeech_2026/li26p_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26p_interspeech.pdf","session":"Child Speech and Health","topics":["paralinguistics","prosody","evaluation"],"category":"phonetics-linguistics","institutions":["Hong Kong Polytechnic University","Chinese University of Hong Kong","First Affiliated Hospital of Jinan University","Hong Kong Metropolitan University","City University of Hong Kong","University of Florida"],"funding":["General Research Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26p_interspeech","category":"phonetics-linguistics","institutions":["Hong Kong Polytechnic University","Chinese University of Hong Kong","First Affiliated Hospital of Jinan University","Hong Kong Metropolitan University","City University of Hong Kong","University of Florida"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1008","pdf":"https://www.isca-archive.org/interspeech_2026/li26p_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26p_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26p_interspeech/markdown.md"},{"id":"li26q_interspeech","title":"Few-shot Class-variable Incremental Audio Classification via Prototype Adaptation and Pseudo Class-variable Training","authors":["Yanxiong Li","Guoqing Chen","Qianqian Li","Sen Huang"],"year":2026,"doi":"10.21437/Interspeech.2026-1024","isca_url":"https://www.isca-archive.org/interspeech_2026/li26q_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26q_interspeech.pdf","session":"Acoustic Event Detection 3","topics":["speech-enhancement","self-supervised","audio-captioning"],"category":"audio-understanding","labels":["low-resource"],"institutions":["South China University of Technology"],"funding":["National Natural Science Foundation of China","China-Croatia Science and Technology Cooperation Committee"],"code":{"url":"https://github.com/cgq2971-afk/FCIAC","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26q_interspeech","category":"audio-understanding","labels":["low-resource"],"institutions":["South China University of Technology"],"code":"https://github.com/cgq2971-afk/FCIAC","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1024","pdf":"https://www.isca-archive.org/interspeech_2026/li26q_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26q_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26q_interspeech/markdown.md"},{"id":"li26r_interspeech","title":"INSPIRE: A Benchmark for Instruction-Aware Speech Retrieval","authors":["Chen-An Li","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1026","isca_url":"https://www.isca-archive.org/interspeech_2026/li26r_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26r_interspeech.pdf","session":"Benchmarking Foundation Models","topics":["speech-llm","evaluation","dataset"],"category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["National Taiwan University","NTU Artificial Intelligence Center of Research Excellence"],"funding":["National Science and Technology Council","Ministry of Education","Taiwan Centers of Excellence in Artificial Intelligence"],"code":{"url":"https://github.com/lca0503/INSPIRE","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26r_interspeech","category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["National Taiwan University","NTU Artificial Intelligence Center of Research Excellence"],"code":"https://github.com/lca0503/INSPIRE","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1026","pdf":"https://www.isca-archive.org/interspeech_2026/li26r_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26r_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26r_interspeech/markdown.md"},{"id":"li26s_interspeech","title":"Training-Free Intelligibility-Guided Observation Addition for Noisy ASR","authors":["Haoyang Li","Changsong Liu","Wei Rao","Hao Shi","Sakriani Sakti","Eng Siong Chng"],"year":2026,"doi":"10.21437/Interspeech.2026-1096","isca_url":"https://www.isca-archive.org/interspeech_2026/li26s_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26s_interspeech.pdf","session":"Robust ASR: Uncertainty and Confidence","topics":["asr","speech-enhancement"],"category":"asr","labels":["robustness-noise"],"institutions":["Nanyang Technological University","Nara Institute of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26s_interspeech","category":"asr","labels":["robustness-noise"],"institutions":["Nanyang Technological University","Nara Institute of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1096","pdf":"https://www.isca-archive.org/interspeech_2026/li26s_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26s_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26s_interspeech/markdown.md"},{"id":"li26t_interspeech","title":"Hearing Smiles in the Crowd: How Babble Noise Shapes Smiled Speech Perception","authors":["Rong Li","Esther Janse","Dirk Heylen","Khiet Truong"],"year":2026,"doi":"10.21437/Interspeech.2026-1178","isca_url":"https://www.isca-archive.org/interspeech_2026/li26t_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26t_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["paralinguistics","emotion-recognition","evaluation"],"category":"paralinguistics-emotion","labels":["robustness-noise"],"institutions":["University of Twente","Radboud University"],"funding":["European Union"],"code":{"url":"https://doi.org/10.5281/zenodo.20729215","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26t_interspeech","category":"paralinguistics-emotion","labels":["robustness-noise"],"institutions":["University of Twente","Radboud University"],"code":"https://doi.org/10.5281/zenodo.20729215","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1178","pdf":"https://www.isca-archive.org/interspeech_2026/li26t_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26t_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26t_interspeech/markdown.md"},{"id":"li26u_interspeech","title":"Geometric Second-Order Feature Correlation Learning for Self-Supervised Speech Emotion Recognition","authors":["Shuanglin Li","Ruxiao Qian","Siyang Song"],"year":2026,"doi":"10.21437/Interspeech.2026-1210","isca_url":"https://www.isca-archive.org/interspeech_2026/li26u_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26u_interspeech.pdf","session":"Speech Emotion Recognition and Representation 1","topics":["speech-emotion-recognition","self-supervised","paralinguistics"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Xiangjiang Laboratory","University of Exeter"],"code":{"url":"https://github.com/secret-code-source/SOC","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26u_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Xiangjiang Laboratory","University of Exeter"],"code":"https://github.com/secret-code-source/SOC","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1210","pdf":"https://www.isca-archive.org/interspeech_2026/li26u_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26u_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26u_interspeech/markdown.md"},{"id":"li26v_interspeech","title":"Tonal Contrasts in Different Vowel Contexts and Different Tonal Systems","authors":["Mingxing Li","Pauline Bolin Liu","Yufeng Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1215","isca_url":"https://www.isca-archive.org/interspeech_2026/li26v_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26v_interspeech.pdf","session":"Tones","topics":["phonetics","prosody","multilingual"],"category":"phonetics-linguistics","institutions":["Hong Kong Baptist University","Shandong University"],"funding":["Hong Kong University Grants Committee General Research Fund","Hong Kong Baptist University Faculty of Arts and Social Sciences Digital Humanities Pilot Grant"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26v_interspeech","category":"phonetics-linguistics","institutions":["Hong Kong Baptist University","Shandong University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1215","pdf":"https://www.isca-archive.org/interspeech_2026/li26v_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26v_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26v_interspeech/markdown.md"},{"id":"li26w_interspeech","title":"Zero-VC: Zero-Lookahead Streaming Voice Conversion via Speaker Anonymization","authors":["Yudong Li","Zihao Fang","Junwen Qiu","Ruihai Jing","Ruixiang Hang","Yingda Shen","Zhizheng Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-1340","isca_url":"https://www.isca-archive.org/interspeech_2026/li26w_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26w_interspeech.pdf","session":"Voice Conversion","topics":["voice-conversion","self-supervised","on-device"],"category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Chinese University of Hong Kong, Shenzhen","Shenzhen Loop Area Institute","Shenzhen Transsion Holdings Co., Ltd","Amphion Technology Co., Ltd"],"funding":["Internal Project Fund from Shenzhen Research Institute of Big Data","Program for Guangdong Introducing Innovative and Enterpreneurial Teams"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26w_interspeech","category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Chinese University of Hong Kong, Shenzhen","Shenzhen Loop Area Institute","Shenzhen Transsion Holdings Co., Ltd","Amphion Technology Co., Ltd"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1340","pdf":"https://www.isca-archive.org/interspeech_2026/li26w_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26w_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26w_interspeech/markdown.md"},{"id":"li26x_interspeech","title":"What Makes Synthetic Speech Sound Sarcastic? A Prosody-Controlled Perception Study","authors":["Zhu Li","Shekhar Nayak","Matt Coler"],"year":2026,"doi":"10.21437/Interspeech.2026-1487","isca_url":"https://www.isca-archive.org/interspeech_2026/li26x_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26x_interspeech.pdf","session":"Speaker Identity, States, and Traits in Paralinguistics","topics":["paralinguistics","speech-llm","evaluation"],"category":"paralinguistics-emotion","labels":["generative-model"],"institutions":["University of Groningen"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26x_interspeech","category":"paralinguistics-emotion","labels":["generative-model"],"institutions":["University of Groningen"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1487","pdf":"https://www.isca-archive.org/interspeech_2026/li26x_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26x_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26x_interspeech/markdown.md"},{"id":"li26y_interspeech","title":"KFC-KWS: Keyframe Fusion with CTC for User-Defined Keyword Spotting","authors":["Jin Li","Wenbin Jiang","Ji Hu"],"year":2026,"doi":"10.21437/Interspeech.2026-1586","isca_url":"https://www.isca-archive.org/interspeech_2026/li26y_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26y_interspeech.pdf","session":"Multi-Speaker Processing, Personalization, and Adaptation","topics":["keyword-spotting","self-supervised","multilingual"],"category":"asr","institutions":["Hangzhou Dianzi University"],"funding":["Yangtze River Delta Science and Technology Innovation Community Joint Research Project","Key R&D Program Project of Zhejiang Province"],"code":{"url":"https://github.com/gusrud1103/LibriPhrase.git","stars":40,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26y_interspeech","category":"asr","institutions":["Hangzhou Dianzi University"],"code":"https://github.com/gusrud1103/LibriPhrase.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1586","pdf":"https://www.isca-archive.org/interspeech_2026/li26y_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26y_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26y_interspeech/markdown.md"},{"id":"li26z_interspeech","title":"Phonetic evidence for contrastive length in Nakanamanga monophthongs","authors":["Shubo Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1597","isca_url":"https://www.isca-archive.org/interspeech_2026/li26z_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/li26z_interspeech.pdf","session":"Diphthongs and Monophthongs","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["Australian National University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"li26z_interspeech","category":"phonetics-linguistics","institutions":["Australian National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1597","pdf":"https://www.isca-archive.org/interspeech_2026/li26z_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/li26z_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/li26z_interspeech/markdown.md"},{"id":"liang26_interspeech","title":"DNSMOS-C: Improving End-to-end Speech Quality Models via Contrastive Learning","authors":["Xinyu Liang","Fredrik Cumlin","Victor Ungureanu","Chandan K. A. Reddy","Christian Schüldt","Saikat Chatterjee"],"year":2026,"doi":"10.21437/Interspeech.2026-342","isca_url":"https://www.isca-archive.org/interspeech_2026/liang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liang26_interspeech.pdf","session":"Speech and Audio Quality Assessment","topics":["speech-enhancement","evaluation","self-supervised"],"category":"resources-evaluation","institutions":["KTH Royal Institute of Technology","Google"],"funding":["Digital Futures Center","European Defence Fund","Wallenberg AI, Autonomous Systems and Software Program","Knut and Alice Wallenberg Foundation"],"code":{"url":"https://github.com/Hope-Liang/DNSMOS-C","stars":7,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liang26_interspeech","category":"resources-evaluation","institutions":["KTH Royal Institute of Technology","Google"],"code":"https://github.com/Hope-Liang/DNSMOS-C","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-342","pdf":"https://www.isca-archive.org/interspeech_2026/liang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liang26_interspeech/markdown.md"},{"id":"liang26b_interspeech","title":"FoleyImmersive: Decoupling What and Where for Video-to-First-Order Ambisonics","authors":["Liming Liang","Lingfeng Yang","Luo Chen","Chenxing Li","Yuexian Zou"],"year":2026,"doi":"10.21437/Interspeech.2026-531","isca_url":"https://www.isca-archive.org/interspeech_2026/liang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liang26b_interspeech.pdf","session":"Spatial Audio 2","topics":["spatial-audio","video-to-audio","self-supervised"],"category":"audio-understanding","labels":["generative-model"],"institutions":["Peking University","South China University of Technology","Tencent"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liang26b_interspeech","category":"audio-understanding","labels":["generative-model"],"institutions":["Peking University","South China University of Technology","Tencent"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-531","pdf":"https://www.isca-archive.org/interspeech_2026/liang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liang26b_interspeech/markdown.md"},{"id":"liang26c_interspeech","title":"Text-Independent Speaker Verification Using Discrete Audio Tokens","authors":["Zheng Liang","Junjie Li","Kong Aik Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1135","isca_url":"https://www.isca-archive.org/interspeech_2026/liang26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liang26c_interspeech.pdf","session":"Speaker Verification: Architectures, Losses, and LLMs","topics":["speaker-verification","self-supervised","speech-coding"],"category":"speaker","institutions":["Hong Kong Polytechnic University"],"funding":["Research Grants Council of the Hong Kong SAR","The Hong Kong Polytechnic University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liang26c_interspeech","category":"speaker","institutions":["Hong Kong Polytechnic University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1135","pdf":"https://www.isca-archive.org/interspeech_2026/liang26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liang26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liang26c_interspeech/markdown.md"},{"id":"liang26d_interspeech","title":"ContextCodec: Content-Focused Context Guidance for Ultra-Low Bitrate Speech Coding","authors":["Chengbin Liang","Wenqi Guo","Hao Cao","Zhijin Qin"],"year":2026,"doi":"10.21437/Interspeech.2026-3355","isca_url":"https://www.isca-archive.org/interspeech_2026/liang26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liang26d_interspeech.pdf","session":"Neural Speech Codecs: Low-Bitrate and Disentangled Coding","topics":["speech-coding","self-supervised","multilingual"],"category":"speech-coding","labels":["multilingual","self-supervised","generative-model"],"institutions":["Tsinghua University"],"funding":["National Key Research and Development Program of China","National Natural Science Foundation of China","Beijing Natural Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liang26d_interspeech","category":"speech-coding","labels":["multilingual","self-supervised","generative-model"],"institutions":["Tsinghua University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3355","pdf":"https://www.isca-archive.org/interspeech_2026/liang26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liang26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liang26d_interspeech/markdown.md"},{"id":"liao26_interspeech","title":"Role-Aware Semi-Supervised Domain Adaptation for Teacher-Student Speaker Diarization","authors":["Zhen Liao","Gaole Dai","Weiwei Jiang","Mengting Wang","Wei Xu"],"year":2026,"doi":"10.21437/Interspeech.2026-155","isca_url":"https://www.isca-archive.org/interspeech_2026/liao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liao26_interspeech.pdf","session":"Speaker Diarization 1","topics":["speaker-diarization","semi-supervised","dataset"],"category":"speaker","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Huazhong University of Science and Technology"],"funding":["National Key Research and Development Program of China"],"code":{"url":"https://github.com/lz-hust/TSSD","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liao26_interspeech","category":"speaker","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Huazhong University of Science and Technology"],"code":"https://github.com/lz-hust/TSSD","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-155","pdf":"https://www.isca-archive.org/interspeech_2026/liao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liao26_interspeech/markdown.md"},{"id":"liao26b_interspeech","title":"High-Precision Prosodic Boundary Anchors from Acoustic Cues under Weak Supervision","authors":["Hanyu Liao","Xiaoluan Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-209","isca_url":"https://www.isca-archive.org/interspeech_2026/liao26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liao26b_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["prosody","self-supervised","dataset"],"category":"phonetics-linguistics","institutions":["East China Normal University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liao26b_interspeech","category":"phonetics-linguistics","institutions":["East China Normal University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-209","pdf":"https://www.isca-archive.org/interspeech_2026/liao26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liao26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liao26b_interspeech/markdown.md"},{"id":"libera26_interspeech","title":"WavSLM: Single-Stream Speech Language Modeling via WavLM Distillation","authors":["Luca Della Libera","Cem Subakan","Mirco Ravanelli"],"year":2026,"doi":"10.21437/Interspeech.2026-2803","isca_url":"https://www.isca-archive.org/interspeech_2026/libera26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/libera26_interspeech.pdf","session":"Streaming Speech Synthesis","topics":["tts","speech-llm","self-supervised"],"category":"speech-llm-dialogue","labels":["self-supervised","streaming-real-time","generative-model"],"institutions":["Concordia University","Mila-Quebec AI Institute","Universite Laval"],"funding":["Natural Sciences and Engineering Research Council of Canada","Digital Research Alliance of Canada","Translated","Apple"],"code":{"url":"https://lucadellalib.github.io/wavslm-web/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"libera26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised","streaming-real-time","generative-model"],"institutions":["Concordia University","Mila-Quebec AI Institute","Universite Laval"],"code":"https://lucadellalib.github.io/wavslm-web/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2803","pdf":"https://www.isca-archive.org/interspeech_2026/libera26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/libera26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/libera26_interspeech/markdown.md"},{"id":"lietz26_interspeech","title":"Making Room for Speech Diversity: A 50 Year Retrospective of Speech Science and Technology through a Neurodivergent Lens","authors":["Rebecca Lietz","Jingjin Li","Peiyao Liu","Jennifer Chien","Norman Makoto Su","Shaomei Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-2962","isca_url":"https://www.isca-archive.org/interspeech_2026/lietz26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lietz26_interspeech.pdf","session":"Clinical and Inclusive Speech Technology","topics":["evaluation","dataset","speech-llm"],"category":"health-clinical","institutions":["University of California, Santa Cruz","AImpower.org","Stanford University"],"code":{"url":"https://doi.org/10.5281/zenodo.20754795","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lietz26_interspeech","category":"health-clinical","institutions":["University of California, Santa Cruz","AImpower.org","Stanford University"],"code":"https://doi.org/10.5281/zenodo.20754795","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2962","pdf":"https://www.isca-archive.org/interspeech_2026/lietz26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lietz26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lietz26_interspeech/markdown.md"},{"id":"lin26_interspeech","title":"Progressive Learnable Counterfactual Attention for Music Classification","authors":["Yi-Xing Lin","Wen-Li Wei","Jia-Ching Wang","Jen-Chun Lin"],"year":2026,"doi":"10.21437/Interspeech.2026-147","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["self-supervised","paralinguistics","evaluation"],"category":"audio-understanding","institutions":["Academia Sinica","National Central University"],"funding":["NSTC"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26_interspeech","category":"audio-understanding","institutions":["Academia Sinica","National Central University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-147","pdf":"https://www.isca-archive.org/interspeech_2026/lin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26_interspeech/markdown.md"},{"id":"lin26b_interspeech","title":"The Role of Context and Prosody in the Understanding of English Irony by Chinese L2 Learners","authors":["Yinan Lin","Shanpeng Li"],"year":2026,"doi":"10.21437/Interspeech.2026-739","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26b_interspeech.pdf","session":"Cross-Linguistic and L2 Phonetic Studies","topics":["paralinguistics","prosody","evaluation"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Nanjing University of Science and Technology"],"funding":["Ministry of Education Humanities and Social Sciences Research Youth Fund Project","Jiangsu Social Science Fund Youth Project","Research Project of Philosophy and Social Sciences in Higher Education of the Jiangsu Provincial Department of Education"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26b_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Nanjing University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-739","pdf":"https://www.isca-archive.org/interspeech_2026/lin26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26b_interspeech/markdown.md"},{"id":"lin26c_interspeech","title":"Hearing the Order: Investigating Position Bias in Large Audio-Language Models","authors":["Yu-Xiang Lin","Chen-An Li","Sheng-Lun Wei","Po-Chun Chen","Hsin-Hsi Chen","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1025","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26c_interspeech.pdf","session":"Audio Language Models: Reasoning, Reliability, and Multimodal Understanding","topics":["speech-llm","evaluation","dataset"],"category":"resources-evaluation","institutions":["National Taiwan University"],"funding":["Ministry of Education","Taiwan Centers of Excellence in Artificial Intelligence","NTU Artificial Intelligence Center of Research Excellence"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26c_interspeech","category":"resources-evaluation","institutions":["National Taiwan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1025","pdf":"https://www.isca-archive.org/interspeech_2026/lin26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26c_interspeech/markdown.md"},{"id":"lin26d_interspeech","title":"BridgeCodec: Mamba Enhanced Neural Audio Codec with Schrödinger Bridge at Low Bitrate","authors":["Zijian Lin","Jing Yang","Jinghao Luo","Zhuo Wang","Fan Fan","Zhiyong Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-1338","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26d_interspeech.pdf","session":"Neural Speech Codecs: Low-Bitrate and Disentangled Coding","topics":["speech-coding","self-supervised","low-resource"],"category":"speech-coding","labels":["generative-model"],"institutions":["Tsinghua University","Huawei"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://thuhcsi.github.io/interspeech2026-BridgeCodec/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26d_interspeech","category":"speech-coding","labels":["generative-model"],"institutions":["Tsinghua University","Huawei"],"code":"https://thuhcsi.github.io/interspeech2026-BridgeCodec/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1338","pdf":"https://www.isca-archive.org/interspeech_2026/lin26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26d_interspeech/markdown.md"},{"id":"lin26e_interspeech","title":"LOPA: Enhancing Spoken Language Assessment via Latent Ordinal Prototype Alignment","authors":["Hong-Yun Lin","Fu-An Chao","Bi-Cheng Yan","Berlin Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1346","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26e_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["speech-llm","evaluation","self-supervised"],"category":"applications-other","labels":["efficient-on-device","self-supervised"],"institutions":["National Taiwan Normal University"],"funding":["Language Training and Testing Center, Taiwan"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26e_interspeech","category":"applications-other","labels":["efficient-on-device","self-supervised"],"institutions":["National Taiwan Normal University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1346","pdf":"https://www.isca-archive.org/interspeech_2026/lin26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26e_interspeech/markdown.md"},{"id":"lin26f_interspeech","title":"Improving Streaming Speaker Diarization for LLM Based Multi-talker Speech Understanding","authors":["Ju Lin","Ruizhi Li","Ruizhe Huang","Jing Pan","Xuan Zhang","Zili Huang","Jing Zheng","Ming Sun","Florian Metze"],"year":2026,"doi":"10.21437/Interspeech.2026-1403","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26f_interspeech.pdf","session":"Multimodal Speech Processing and Speech LLM Systems","topics":["speech-recognition","speech-translation","speaker-diarization"],"category":"speaker","labels":["efficient-on-device","streaming-real-time"],"institutions":["Meta"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26f_interspeech","category":"speaker","labels":["efficient-on-device","streaming-real-time"],"institutions":["Meta"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1403","pdf":"https://www.isca-archive.org/interspeech_2026/lin26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26f_interspeech/markdown.md"},{"id":"lin26g_interspeech","title":"Silence is Golden: Mitigating Hallucinations in Large Audio-Language Models via Layer-Weighted Vector Steering","authors":["Tsung-En Lin","Kuan-Yi Lee","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1421","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26g_interspeech.pdf","session":"Reasoning with Speech/Audio Language Models","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","institutions":["National Taiwan University","ASUS"],"funding":["Ministry of Education","Taiwan Centers of Excellence in Artificial Intelligence","NTU Artificial Intelligence Center of Research Excellence"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26g_interspeech","category":"speech-llm-dialogue","institutions":["National Taiwan University","ASUS"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1421","pdf":"https://www.isca-archive.org/interspeech_2026/lin26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26g_interspeech/markdown.md"},{"id":"lin26h_interspeech","title":"Improving Cross-Dataset Speech Intelligibility Prediction for Hearing-Impaired Listeners with Few-Shot Adaptation","authors":["Guojian Lin","Xuefei Wang","Ryandhimas E. Zezario","Yu Tsao","Fei Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1567","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26h_interspeech.pdf","session":"Quality, Intelligibility and Evaluation of Speech and Codecs","topics":["speech-enhancement","paralinguistics","self-supervised"],"category":"resources-evaluation","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["Southern University of Science and Technology","Academia Sinica"],"funding":["National Key Research and Development Program of China","National Natural Science Foundation of China","Shenzhen Key Technology Program Funding","Center for Computational Science and Engineering at Southern University of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26h_interspeech","category":"resources-evaluation","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["Southern University of Science and Technology","Academia Sinica"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1567","pdf":"https://www.isca-archive.org/interspeech_2026/lin26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26h_interspeech/markdown.md"},{"id":"lin26i_interspeech","title":"RAISE: Resolving Ambiguity in Audio Understanding with Imagination and Selective Extraction","authors":["Yueqian Lin","Qinsi Wang","Yudong Liu","Hancheng Ye","Hai Li","Yiran Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1739","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26i_interspeech.pdf","session":"Acoustic Event Detection 4","topics":["speech-llm","source-separation","evaluation"],"category":"audio-understanding","labels":["generative-model"],"institutions":["Duke University"],"funding":["National Science Foundation","Army Research Office"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26i_interspeech","category":"audio-understanding","labels":["generative-model"],"institutions":["Duke University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1739","pdf":"https://www.isca-archive.org/interspeech_2026/lin26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26i_interspeech/markdown.md"},{"id":"lin26j_interspeech","title":"First-to-Spike: An Early-Exit Framework for Rapid and Energy-Efficient Spiking Neural Networks","authors":["Zheyuan Lin","Sirui Li","Zeyang Song","Zhiqi Zhang","Siqi Cai","Haizhou Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1858","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26j_interspeech.pdf","session":"ASR Under Real-World Constraints: Streaming, Adaptation, and Efficiency","topics":["keyword-spotting","self-supervised","health"],"category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Chinese University of Hong Kong","National University of Singapore","Harbin Institute of Technology","Tsinghua University"],"funding":["Program for Guangdong Introducing Innovative and Entrepreneurial Teams","Deutsche Forschungsgemeinschaft","National Natural Science Foundation of China","Shenzhen Stability Science Program","Shenzhen Key Lab of Multi-Modal Cognitive Computing","Guangdong Provincial Key Laboratory of Big Data Computing"],"code":{"url":"https://github.com/PatrickZLin/F2S","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26j_interspeech","category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Chinese University of Hong Kong","National University of Singapore","Harbin Institute of Technology","Tsinghua University"],"code":"https://github.com/PatrickZLin/F2S","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1858","pdf":"https://www.isca-archive.org/interspeech_2026/lin26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26j_interspeech/markdown.md"},{"id":"lin26k_interspeech","title":"BG-CRNN: Boundary-Guided Dynamic Attention for Sound Event Detection in Complex Scenarios","authors":["Zongmu Lin","Zhongxin Bai","Jisheng Bai","Zhenru Li","Ting Dang","Gongping Huang","Jingdong Chen","Jacob Benesty"],"year":2026,"doi":"10.21437/Interspeech.2026-2019","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26k_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["sound-event-detection","self-supervised","evaluation"],"category":"audio-understanding","labels":["robustness-noise"],"institutions":["Wuhan University","Harbin Engineering University","Xi'an University of Posts and Telecommunications","University of Melbourne","Northwestern Polytechnical University","University of Quebec"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26k_interspeech","category":"audio-understanding","labels":["robustness-noise"],"institutions":["Wuhan University","Harbin Engineering University","Xi'an University of Posts and Telecommunications","University of Melbourne","Northwestern Polytechnical University","University of Quebec"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2019","pdf":"https://www.isca-archive.org/interspeech_2026/lin26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26k_interspeech/markdown.md"},{"id":"lin26l_interspeech","title":"Decoding the Ear (DeEAR): A Framework for Objectifying Expressiveness from Human Preference Through Efficient Alignment","authors":["Zhiyu Lin","Jingwen Yang","Jiale Zhao","Meng Liu","Sunzhu Li","Zhengjun Yue","Benyou Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-2408","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26l_interspeech.pdf","session":"Speech Synthesis Evaluation 2","topics":["tts","speech-llm","evaluation"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Chinese University of Hong Kong","Li Auto","Shenzhen Loop Area Institution"],"funding":["Shenzhen Medical Academy of Research and Translation","Shenzhen Medical Research Fund","National Natural Science Foundation of China","CUHK-CUHK(SZ)-GDSTC Joint Collaboration Fund","Guangdong Provincial Key Laboratory of Mathematical Foundations for Artificial Intelligence","Ministry of Science and Technology of China"],"code":{"url":"https://freedomintelligence.github.io/ExpressiveSpeech/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26l_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Chinese University of Hong Kong","Li Auto","Shenzhen Loop Area Institution"],"code":"https://freedomintelligence.github.io/ExpressiveSpeech/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2408","pdf":"https://www.isca-archive.org/interspeech_2026/lin26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26l_interspeech/markdown.md"},{"id":"lin26m_interspeech","title":"Assessing True Generalisability of Audio-Visual Speech Recognisers","authors":["Zhaofeng Lin","Stavros Petridis","Maja Pantic","Naomi Harte"],"year":2026,"doi":"10.21437/Interspeech.2026-2583","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26m_interspeech.pdf","session":"Audio-Visual and Multimodal Perception","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["dataset-or-benchmark-release"],"institutions":["Trinity College Dublin","Imperial College London","NatWest AI Research"],"funding":["Research Ireland Centre for Research Training in Digitally-Enhanced Reality"],"code":{"url":"https://github.com/chaufanglin/mv2lrs3","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26m_interspeech","category":"asr","labels":["dataset-or-benchmark-release"],"institutions":["Trinity College Dublin","Imperial College London","NatWest AI Research"],"code":"https://github.com/chaufanglin/mv2lrs3","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2583","pdf":"https://www.isca-archive.org/interspeech_2026/lin26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26m_interspeech/markdown.md"},{"id":"lin26n_interspeech","title":"WQ-Fusion: Dynamic Gated Attention for Cross-Domain Audio Representation","authors":["Mingda Lin","Lei Ding","Xinyue Zhou","Tiantian Xiong","Hanchen Pei","Gongping Huang","Hao Zhang","Jingdong Chen","Jacob Benesty"],"year":2026,"doi":"10.21437/Interspeech.2026-3228","isca_url":"https://www.isca-archive.org/interspeech_2026/lin26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lin26n_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["self-supervised","multilingual","speech-llm"],"category":"speech-llm-dialogue","labels":["multilingual","self-supervised"],"institutions":["Wuhan University","Tencent","Northwestern Polytechnical University","Université du Québec"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://dataoceanai.github.io/Interspeech2026-Audio-Encoder-Challenge/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lin26n_interspeech","category":"speech-llm-dialogue","labels":["multilingual","self-supervised"],"institutions":["Wuhan University","Tencent","Northwestern Polytechnical University","Université du Québec"],"code":"https://dataoceanai.github.io/Interspeech2026-Audio-Encoder-Challenge/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3228","pdf":"https://www.isca-archive.org/interspeech_2026/lin26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lin26n_interspeech/markdown.md"},{"id":"ling26_interspeech","title":"TGTSE: Token-Guided Target Speaker Extraction with Visual Cue","authors":["Tongtao Ling","Shulin He","Zhong-Qiu Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1710","isca_url":"https://www.isca-archive.org/interspeech_2026/ling26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ling26_interspeech.pdf","session":"Target Speaker Extraction, Speech Separation and Audio Understanding","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","institutions":["Southern University of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ling26_interspeech","category":"enhancement-separation","institutions":["Southern University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1710","pdf":"https://www.isca-archive.org/interspeech_2026/ling26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ling26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ling26_interspeech/markdown.md"},{"id":"lipari26_interspeech","title":"Disentangling sociophonetic and physiological variation in /s/ acoustics across 12 languages","authors":["Massimo Lipari","Morgan Sonderegger","Meghan Clayards"],"year":2026,"doi":"10.21437/Interspeech.2026-2975","isca_url":"https://www.isca-archive.org/interspeech_2026/lipari26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lipari26_interspeech.pdf","session":"Cross-Linguistic and L2 Phonetic Studies","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["McGill University"],"funding":["Social Sciences and Humanities Research Council","Fonds de recherche du Quebec - Societe et culture","Canada Research Chairs","Natural Sciences and Engineering Research Council of Canada"],"code":{"url":"https://osf.io/m58e3/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lipari26_interspeech","category":"phonetics-linguistics","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["McGill University"],"code":"https://osf.io/m58e3/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2975","pdf":"https://www.isca-archive.org/interspeech_2026/lipari26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lipari26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lipari26_interspeech/markdown.md"},{"id":"liu26_interspeech","title":"CoSTA: Cognitive-State-Conditioned TTS Data Augmentation Using ASR Transcripts for Alzheimer’s Disease Detection","authors":["Yin-Long Liu","Yuanchao Li","Yiming Wang","Yue Li","Rui Feng","Jiaxin Chen","Shaobo Liu","Liu He","Yuang Chen","Jiahong Yuan","Zhen-Hua Ling"],"year":2026,"doi":"10.21437/Interspeech.2026-88","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 2","topics":["speech-llm","paralinguistics","dataset"],"category":"health-clinical","labels":["generative-model"],"institutions":["University of Science and Technology of China","University of Edinburgh"],"funding":["National Social Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26_interspeech","category":"health-clinical","labels":["generative-model"],"institutions":["University of Science and Technology of China","University of Edinburgh"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-88","pdf":"https://www.isca-archive.org/interspeech_2026/liu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26_interspeech/markdown.md"},{"id":"liu26b_interspeech","title":"LMPAN: A Lightweight Multi-Path Alignment Network for Joint Full-Duplex Acoustic Echo Cancellation and Noise Suppression","authors":["Chengwei Liu","Shaofei Xue","Haoyin Yan","Xiaotao Liang","Zheng Xue"],"year":2026,"doi":"10.21437/Interspeech.2026-191","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26b_interspeech.pdf","session":"Active Noise and Echo Control, Sound Zones and Packet-Loss Concealment","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time","robustness-noise"],"institutions":["Alibaba"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26b_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time","robustness-noise"],"institutions":["Alibaba"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-191","pdf":"https://www.isca-archive.org/interspeech_2026/liu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26b_interspeech/markdown.md"},{"id":"liu26c_interspeech","title":"StyleStream: Real-Time Zero-Shot Voice Style Conversion","authors":["Yisi Liu","Nicholas Lee","Gopala Anumanchipalli"],"year":2026,"doi":"10.21437/Interspeech.2026-404","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26c_interspeech.pdf","session":"Text-to-Speech Synthesis","topics":["speech-translation","voice-conversion","self-supervised"],"category":"tts","labels":["self-supervised","streaming-real-time","generative-model"],"institutions":["UC Berkeley"],"funding":["IARPA ARTS program","Meta AI","Robert E. & Beverly A. Brooks Endowed Chair in EECS, UC Berkeley"],"code":{"url":"https://berkeley-speech-group.github.io/StyleStream","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26c_interspeech","category":"tts","labels":["self-supervised","streaming-real-time","generative-model"],"institutions":["UC Berkeley"],"code":"https://berkeley-speech-group.github.io/StyleStream","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-404","pdf":"https://www.isca-archive.org/interspeech_2026/liu26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26c_interspeech/markdown.md"},{"id":"liu26d_interspeech","title":"Towards Array-Invariant Speech Enhancement via Geometry-Aware Dynamic Convolution","authors":["Zhenglong Liu","Wangyou Zhang","Chenda Li","Yanmin Qian"],"year":2026,"doi":"10.21437/Interspeech.2026-548","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26d_interspeech.pdf","session":"Multi-Channel, Beamforming and Spatial Speech Enhancement","topics":["speech-enhancement","self-supervised","multilingual"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Shanghai Jiao Tong University","VUI Labs"],"funding":["China NSFC","SJTU Med-X Translational Research Grant"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26d_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Shanghai Jiao Tong University","VUI Labs"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-548","pdf":"https://www.isca-archive.org/interspeech_2026/liu26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26d_interspeech/markdown.md"},{"id":"liu26e_interspeech","title":"CTC-TTS: LLM-Based Dual-Streaming Text-to-Speech with CTC Alignment","authors":["Hanwen Liu","Saierdaer Yusuyin","Hao Huang","Zhijian Ou"],"year":2026,"doi":"10.21437/Interspeech.2026-653","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26e_interspeech.pdf","session":"Streaming Speech Synthesis","topics":["tts","self-supervised","speech-llm"],"category":"tts","labels":["streaming-real-time","generative-model"],"institutions":["Xinjiang University","Tsinghua University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://github.com/thu-spmi/CTC-TTS","stars":22,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26e_interspeech","category":"tts","labels":["streaming-real-time","generative-model"],"institutions":["Xinjiang University","Tsinghua University"],"code":"https://github.com/thu-spmi/CTC-TTS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-653","pdf":"https://www.isca-archive.org/interspeech_2026/liu26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26e_interspeech/markdown.md"},{"id":"liu26f_interspeech","title":"A Semantic-Anchor-based Method for Open-Vocabulary Sound Event Detection","authors":["Jun Liu","Pengfei Cai","Yanfeng Shi","Qing Gu","Nan Jiang","Lirong Dai","Yan Song"],"year":2026,"doi":"10.21437/Interspeech.2026-731","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26f_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["speech-llm","self-supervised","multilingual"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["University of Science and Technology of China"],"funding":["Anhui Province Major Science and Technology Research Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26f_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["University of Science and Technology of China"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-731","pdf":"https://www.isca-archive.org/interspeech_2026/liu26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26f_interspeech/markdown.md"},{"id":"liu26g_interspeech","title":"Dual-Granularity Orthogonal Disentanglement for Generalizable Audio Deepfake Detection","authors":["Zhuodong Liu","Hugen Lv","Xiangyu Li","Chunhong Yuan"],"year":2026,"doi":"10.21437/Interspeech.2026-836","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26g_interspeech.pdf","session":"Speech Deepfake Detection: Robustness, Generalization, Attribution","topics":["audio-deepfake","speaker-verification","self-supervised"],"category":"deepfake-security","institutions":["Beijing Jiaotong University","Shanghai Jiao Tong University","ITMO University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26g_interspeech","category":"deepfake-security","institutions":["Beijing Jiaotong University","Shanghai Jiao Tong University","ITMO University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-836","pdf":"https://www.isca-archive.org/interspeech_2026/liu26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26g_interspeech/markdown.md"},{"id":"liu26h_interspeech","title":"Confidence-Gated Mean-Teacher Consistency Regularization for Low-Resource Multilingual ASR with Shared–Private Fusion-LoRA","authors":["Jie Liu","Liang He","Longwei Li","Qingyuan Ma","Xuejian Zhao"],"year":2026,"doi":"10.21437/Interspeech.2026-1183","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26h_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 2","topics":["asr","low-resource","multilingual"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Xinjiang University","Tsinghua University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26h_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Xinjiang University","Tsinghua University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1183","pdf":"https://www.isca-archive.org/interspeech_2026/liu26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26h_interspeech/markdown.md"},{"id":"liu26i_interspeech","title":"Prosodic Boundary-Aware Streaming Generation for LLM-Based TTS with Streaming Text Input","authors":["Changsong Liu","Tianrui Wang","Ye Ni","Yizhou Peng","Eng Siong Chng"],"year":2026,"doi":"10.21437/Interspeech.2026-1192","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26i_interspeech.pdf","session":"Streaming Speech Synthesis","topics":["tts","speech-llm","low-resource"],"category":"tts","labels":["streaming-real-time","generative-model"],"institutions":["Nanyang Technological University","Tianjin University","Southeast University"],"funding":["RIE2025 Industry Alignment Fund - Industry Collaboration Projects","A*STAR","Alibaba Group","NTU Singapore","Alibaba-NTU Global e-Sustainability CorpLab"],"code":{"url":"https://charlieliu331.github.io/Prosodic-Boundary-Aware-Streaming-Text-TTS/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26i_interspeech","category":"tts","labels":["streaming-real-time","generative-model"],"institutions":["Nanyang Technological University","Tianjin University","Southeast University"],"code":"https://charlieliu331.github.io/Prosodic-Boundary-Aware-Streaming-Text-TTS/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1192","pdf":"https://www.isca-archive.org/interspeech_2026/liu26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26i_interspeech/markdown.md"},{"id":"liu26j_interspeech","title":"Text-Annotated Noisy Speech as Supervision: A Dual-Learning Framework for Target-Domain Clean-Free Speech Enhancement","authors":["Xin Liu","Shulin He","Xueliang Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1259","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26j_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","asr","self-supervised"],"category":"enhancement-separation","labels":["low-resource","robustness-noise"],"institutions":["Inner Mongolia University","Southern University of Science and Technology"],"funding":["Inner Mongolia Natural Science Foundation","Hohhot R&D Investment Incentive Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26j_interspeech","category":"enhancement-separation","labels":["low-resource","robustness-noise"],"institutions":["Inner Mongolia University","Southern University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1259","pdf":"https://www.isca-archive.org/interspeech_2026/liu26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26j_interspeech/markdown.md"},{"id":"liu26k_interspeech","title":"HWB-plus: A Lightweight Speech Bandwidth Extension Method with Separate Modeling for Consonants and Vowels","authors":["Xin Liu","Xueliang Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1498","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26k_interspeech.pdf","session":"Dereverberation, Bandwidth Extension and Restoration","topics":["speech-enhancement","self-supervised","on-device"],"category":"speech-coding","labels":["efficient-on-device"],"institutions":["Inner Mongolia University"],"funding":["Inner Mongolia Natural Science Foundation","Hohhot R&D Investment Incentive Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26k_interspeech","category":"speech-coding","labels":["efficient-on-device"],"institutions":["Inner Mongolia University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1498","pdf":"https://www.isca-archive.org/interspeech_2026/liu26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26k_interspeech/markdown.md"},{"id":"liu26l_interspeech","title":"Decoupling Search and Evaluation: Efficient Beam Decoding for Language Model-Based Text-to-Speech Synthesis","authors":["Chenlin Liu","Jie Gao","Wei Zhou","Guangyan Zhang","Minghui Fang","Jiqing Han"],"year":2026,"doi":"10.21437/Interspeech.2026-1531","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26l_interspeech.pdf","session":"LLM Based Speech Synthesis","topics":["tts","speech-llm","self-supervised"],"category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Harbin Institute of Technology","Tsinghua University","Zhejiang University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26l_interspeech","category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Harbin Institute of Technology","Tsinghua University","Zhejiang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1531","pdf":"https://www.isca-archive.org/interspeech_2026/liu26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26l_interspeech/markdown.md"},{"id":"liu26m_interspeech","title":"MultiEmoVec: Learning Generalised Multimodal Emotion Representation by Momentum Contrast and Multi-task Reconstruction","authors":["Junchen Liu","Jesin James","Karan Nathwani","Michael Witbrock"],"year":2026,"doi":"10.21437/Interspeech.2026-1563","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26m_interspeech.pdf","session":"Multimodal Speech Processing and Speech LLM Systems","topics":["paralinguistics","emotion-recognition","self-supervised"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["University of Auckland","Indian Institute of Technology Jammu"],"code":{"url":"https://github.com/MaoriEnglish-Codeswitch/MultiEmoVec","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26m_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["University of Auckland","Indian Institute of Technology Jammu"],"code":"https://github.com/MaoriEnglish-Codeswitch/MultiEmoVec","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1563","pdf":"https://www.isca-archive.org/interspeech_2026/liu26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26m_interspeech/markdown.md"},{"id":"liu26n_interspeech","title":"Learning Contextualized Tonal Contours from F0: A Core-Auxiliary Branched Transformer for Mandarin Tone Recognition","authors":["Yi-Fen Liu","Xiang-Li Lu","Po-Yu Chiu"],"year":2026,"doi":"10.21437/Interspeech.2026-1747","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26n_interspeech.pdf","session":"Prosody, Pronunciation and Specialized Speech Processing","topics":["paralinguistics","low-resource","dataset"],"category":"phonetics-linguistics","institutions":["Feng Chia University"],"funding":["National Science and Technology Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26n_interspeech","category":"phonetics-linguistics","institutions":["Feng Chia University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1747","pdf":"https://www.isca-archive.org/interspeech_2026/liu26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26n_interspeech/markdown.md"},{"id":"liu26o_interspeech","title":"WhisperVC: Decoupled Cross-Domain Alignment and Speech Generation for Low-Resource Whisper-to-Normal Conversion","authors":["Dong Liu","Juan Liu","Wei Ju","Yao Tian","Ming Li"],"year":2026,"doi":"10.21437/Interspeech.2026-2002","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26o_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26o_interspeech.pdf","session":"Assistive Technologies 2","topics":["voice-conversion","speech-enhancement","self-supervised"],"category":"tts","labels":["self-supervised","generative-model"],"institutions":["Wuhan University","Chinese University of Hong Kong, Shenzhen","OPPO"],"funding":["National Natural Science Foundation of China","Yangtze River Delta Science and Technology Innovation Community Joint Research Project","OPPO"],"code":{"url":"https://demo-whispervc.github.io/demo-whispervc/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26o_interspeech","category":"tts","labels":["self-supervised","generative-model"],"institutions":["Wuhan University","Chinese University of Hong Kong, Shenzhen","OPPO"],"code":"https://demo-whispervc.github.io/demo-whispervc/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2002","pdf":"https://www.isca-archive.org/interspeech_2026/liu26o_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26o_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26o_interspeech/markdown.md"},{"id":"liu26p_interspeech","title":"audiobook-cc: Controllable Long-context Speech Generation for Multicast Audiobook","authors":["Min Liu","JingJing Yin","Xiang Zhang","JianHao Ye","Siyu Hao","Siwei Xia","Hongbin Zhou"],"year":2026,"doi":"10.21437/Interspeech.2026-2125","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26p_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26p_interspeech.pdf","session":"Long-Form Speech Synthesis","topics":["tts","speech-llm","multilingual"],"category":"tts","labels":["generative-model"],"institutions":["Shanghai Himalaya Technology Co., Ltd"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26p_interspeech","category":"tts","labels":["generative-model"],"institutions":["Shanghai Himalaya Technology Co., Ltd"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2125","pdf":"https://www.isca-archive.org/interspeech_2026/liu26p_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26p_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26p_interspeech/markdown.md"},{"id":"liu26q_interspeech","title":"Reducing Speaker Residual by Considering Pinhole Effect in Voice Anonymization","authors":["Zeyan Liu","Weili Jiang","Liping Chen","Kong Aik Lee","Boyu Zhao","Kai Gao","Zhen-Hua Ling"],"year":2026,"doi":"10.21437/Interspeech.2026-2346","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26q_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26q_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["speaker-verification","self-supervised","evaluation"],"category":"deepfake-security","institutions":["University of Science and Technology of China","Hong Kong Polytechnic University","Institute of Forensic Science, Ministry of Public Security"],"code":{"url":"https://anonymous.4open.science/r/Pinhole-loss-fine-tunning-4628","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26q_interspeech","category":"deepfake-security","institutions":["University of Science and Technology of China","Hong Kong Polytechnic University","Institute of Forensic Science, Ministry of Public Security"],"code":"https://anonymous.4open.science/r/Pinhole-loss-fine-tunning-4628","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2346","pdf":"https://www.isca-archive.org/interspeech_2026/liu26q_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26q_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26q_interspeech/markdown.md"},{"id":"liu26r_interspeech","title":"P-SED : Asymmetric Prototype Metric Learning for Weakly Supervised Speech Emotion Diarization","authors":["Yumeng Liu","Yukun Sun","Jian Peng","Lixu Sun","Nurmemet Yolwas"],"year":2026,"doi":"10.21437/Interspeech.2026-2388","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26r_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26r_interspeech.pdf","session":"Paralinguistics","topics":["speech-emotion-diarization","weakly-supervised","self-supervised"],"category":"paralinguistics-emotion","institutions":["Xinjiang University"],"code":{"url":"https://github.com/LiuYumeng-Lemon/P-SED","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26r_interspeech","category":"paralinguistics-emotion","institutions":["Xinjiang University"],"code":"https://github.com/LiuYumeng-Lemon/P-SED","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2388","pdf":"https://www.isca-archive.org/interspeech_2026/liu26r_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26r_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26r_interspeech/markdown.md"},{"id":"liu26s_interspeech","title":"Bridging the Distribution Gap in Real-World Far-Field Speech Enhancement via Lightweight Latent Representation Alignment","authors":["Biao Liu","Haoyuan Xie","Zengqiang Shang","Mou Wang","Xin Liu","Pengyuan Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-3478","isca_url":"https://www.isca-archive.org/interspeech_2026/liu26s_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liu26s_interspeech.pdf","session":"SE Architectures, Adaptation and Audio Front-Ends","topics":["speech-enhancement"],"category":"enhancement-separation","labels":["efficient-on-device","robustness-noise"],"institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences","OPPO"],"funding":["OPPO Research Fund","National Natural Science Foundation of China","CPSF Postdoctoral Fellowship"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liu26s_interspeech","category":"enhancement-separation","labels":["efficient-on-device","robustness-noise"],"institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences","OPPO"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3478","pdf":"https://www.isca-archive.org/interspeech_2026/liu26s_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26s_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liu26s_interspeech/markdown.md"},{"id":"liyanarachchi26_interspeech","title":"Paediatric-HGNN: A Hybrid Heterogeneous Graph Neural Network for Detecting Disfluency in Children’s Speech via Multiscale Acoustic Fusion","authors":["Rashini Liyanarachchi","Rachael Mackay","Alison Short","Aditya Joshi","Erik Meijering"],"year":2026,"doi":"10.21437/Interspeech.2026-1131","isca_url":"https://www.isca-archive.org/interspeech_2026/liyanarachchi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/liyanarachchi26_interspeech.pdf","session":"Child Speech and Health","topics":["speech-disorder","self-supervised","health"],"category":"health-clinical","institutions":["University of New South Wales","Resourced Music Therapy","Western Sydney University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"liyanarachchi26_interspeech","category":"health-clinical","institutions":["University of New South Wales","Resourced Music Therapy","Western Sydney University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1131","pdf":"https://www.isca-archive.org/interspeech_2026/liyanarachchi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/liyanarachchi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/liyanarachchi26_interspeech/markdown.md"},{"id":"lo26_interspeech","title":"A Novel Sentence Stress Detection Framework Leveraging Auxiliary Word-Stress Modeling and Loss Optimization","authors":["Tien-Hong Lo","Fong-Chun Tsai","Ting-An Hung","Yu-Hsuan Hsieh","Yao-Ting Sung","Berlin Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1494","isca_url":"https://www.isca-archive.org/interspeech_2026/lo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lo26_interspeech.pdf","session":"Prosody, Pronunciation and Specialized Speech Processing","topics":["asr","evaluation","self-supervised"],"category":"applications-other","labels":["self-supervised"],"institutions":["National Taiwan Normal University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lo26_interspeech","category":"applications-other","labels":["self-supervised"],"institutions":["National Taiwan Normal University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1494","pdf":"https://www.isca-archive.org/interspeech_2026/lo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lo26_interspeech/markdown.md"},{"id":"long26_interspeech","title":"Benchmarking Language Modeling for Lossless Compression of Full-Fidelity Audio","authors":["Phillip Long","Zachary Novack","Chris Donahue"],"year":2026,"doi":"10.21437/Interspeech.2026-1748","isca_url":"https://www.isca-archive.org/interspeech_2026/long26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/long26_interspeech.pdf","session":"Evaluation, Benchmarking, and Reliability of Audio Systems","topics":["self-supervised","speech-coding","dataset"],"category":"speech-coding","labels":["efficient-on-device","generative-model"],"institutions":["University of California, San Diego","Carnegie Mellon University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"long26_interspeech","category":"speech-coding","labels":["efficient-on-device","generative-model"],"institutions":["University of California, San Diego","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1748","pdf":"https://www.isca-archive.org/interspeech_2026/long26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/long26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/long26_interspeech/markdown.md"},{"id":"lopez26_interspeech","title":"Robustness Assessment of Large Audio Language Models in Multiple-choice Evaluation","authors":["Fernando López","Santosh Kesiraju","Jordi Luque"],"year":2026,"doi":"10.21437/Interspeech.2026-2503","isca_url":"https://www.isca-archive.org/interspeech_2026/lopez26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lopez26_interspeech.pdf","session":"Audio & Speech Language Models: Evaluation, Representations, and Emerging Capabilities","topics":["speech-llm","evaluation","spoken-language-understanding"],"category":"speech-llm-dialogue","institutions":["Telefonica","Universidad Autonoma de Madrid","Brno University of Technology"],"funding":["European Union's Horizon 2020","Ministry of Education, Youth and Sports of the Czech Republic"],"code":{"url":"https://github.com/ferugit/mcqa-lalms-robustness","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lopez26_interspeech","category":"speech-llm-dialogue","institutions":["Telefonica","Universidad Autonoma de Madrid","Brno University of Technology"],"code":"https://github.com/ferugit/mcqa-lalms-robustness","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2503","pdf":"https://www.isca-archive.org/interspeech_2026/lopez26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lopez26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lopez26_interspeech/markdown.md"},{"id":"lopez26b_interspeech","title":"S-DiverSe: Spanish Diverse Speech","authors":["Fernando López","Fernando Ibañez","Ana Martínez","Iván Alonso","Pablo Gómez","Santosh Kesiraju","Jordi Luque"],"year":2026,"doi":"10.21437/Interspeech.2026-2529","isca_url":"https://www.isca-archive.org/interspeech_2026/lopez26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lopez26b_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 1","topics":["asr","low-resource","evaluation"],"category":"asr","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Telefonica","Universidad Autonoma de Madrid","Brno University of Technology"],"funding":["European Union's Horizon 2020 RIA ELOQUENCE project","Ministry of Education, Youth and Sports of the Czech Republic","OP JAK project 'Linguistics, Artificial Intelligence and Language and Speech Technologies: from Research to Applications'"],"code":{"url":"https://github.com/ferugit/s-diverse","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lopez26b_interspeech","category":"asr","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Telefonica","Universidad Autonoma de Madrid","Brno University of Technology"],"code":"https://github.com/ferugit/s-diverse","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2529","pdf":"https://www.isca-archive.org/interspeech_2026/lopez26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lopez26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lopez26b_interspeech/markdown.md"},{"id":"louro26_interspeech","title":"‘I have to talk proper white ways’: Australian Aboriginal English Speakers’ Experiences with Voice Technologies","authors":["Celeste Rodríguez Louro","Glenys Dale Collard","Hope Narrier","Katrina Cox","Lily Hayward","Daniel Colbung","Ben Hutchinson"],"year":2026,"doi":"10.21437/Interspeech.2026-1595","isca_url":"https://www.isca-archive.org/interspeech_2026/louro26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/louro26_interspeech.pdf","session":"Indigenous Voices in Speech Science and Technology","topics":["asr","evaluation","multilingual"],"category":"asr","labels":["low-resource"],"institutions":["University of Western Australia","Google"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"louro26_interspeech","category":"asr","labels":["low-resource"],"institutions":["University of Western Australia","Google"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1595","pdf":"https://www.isca-archive.org/interspeech_2026/louro26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/louro26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/louro26_interspeech/markdown.md"},{"id":"loweimi26_interspeech","title":"To Be Multimodal or Not to Be: Query-Adaptive Audio-Visual Person Retrieval via Active Modality Detection","authors":["Erfan Loweimi","Mengjie Qian","Kate Knill","Guanfeng Wu","Chi-Ho Chan","Abbas Haider","Muhammad Awan","Josef Kittler","Hui Wang","Mark Gales"],"year":2026,"doi":"10.21437/Interspeech.2026-790","isca_url":"https://www.isca-archive.org/interspeech_2026/loweimi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/loweimi26_interspeech.pdf","session":"Information Extraction and Retrieval","topics":["speaker-verification","multimodal","dataset"],"category":"speaker","institutions":["University of Cambridge","Queen's University Belfast","University of Surrey","Cisco","Southwest Jiaotong University","Teesside University"],"funding":["Engineering and Physical Sciences Research Council","Cambridge University Press & Assessment"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"loweimi26_interspeech","category":"speaker","institutions":["University of Cambridge","Queen's University Belfast","University of Surrey","Cisco","Southwest Jiaotong University","Teesside University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-790","pdf":"https://www.isca-archive.org/interspeech_2026/loweimi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/loweimi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/loweimi26_interspeech/markdown.md"},{"id":"loweimi26b_interspeech","title":"Phonetic Error Analysis of Raw Waveform Acoustic Models","authors":["Erfan Loweimi","Zhengjun Yue","Andrea Carmantini","Zoran Cvetkovic","Steve Renals","Peter Bell"],"year":2026,"doi":"10.21437/Interspeech.2026-798","isca_url":"https://www.isca-archive.org/interspeech_2026/loweimi26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/loweimi26b_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["asr","phonetics","self-supervised"],"category":"asr","labels":["self-supervised"],"institutions":["University of Edinburgh","Cisco","SLAI","Chinese University of Hong Kong, Shenzhen","King's College London"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"loweimi26b_interspeech","category":"asr","labels":["self-supervised"],"institutions":["University of Edinburgh","Cisco","SLAI","Chinese University of Hong Kong, Shenzhen","King's College London"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-798","pdf":"https://www.isca-archive.org/interspeech_2026/loweimi26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/loweimi26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/loweimi26b_interspeech/markdown.md"},{"id":"lu26_interspeech","title":"PolyBench: Benchmarking LLM-based TTS Systems for Chinese Polyphone Disambiguation","authors":["Chunhui Lu","Rui Zhou","Feifan Chen","Liming Song","YoonChoon Hwang","Junkwang Oh","Gunu Jho"],"year":2026,"doi":"10.21437/Interspeech.2026-998","isca_url":"https://www.isca-archive.org/interspeech_2026/lu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lu26_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["tts","evaluation","low-resource"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Samsung","Samsung Electronics"],"code":{"url":"https://github.com/Chunhui-Lu/PolyBench","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lu26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Samsung","Samsung Electronics"],"code":"https://github.com/Chunhui-Lu/PolyBench","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-998","pdf":"https://www.isca-archive.org/interspeech_2026/lu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lu26_interspeech/markdown.md"},{"id":"lu26b_interspeech","title":"Alignment-Aware Continued Pre-training for Multilingual Speech Representation Learning","authors":["Xun Lu","Xuyang Wang","Fan Feng","Gaofeng Cheng","Pengyuan Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1185","isca_url":"https://www.isca-archive.org/interspeech_2026/lu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lu26b_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["multilingual","asr","self-supervised"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences"],"funding":["National Key Research and Development Program of China","Xinjiang Uygur Autonomous Region Key Research and Development Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lu26b_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1185","pdf":"https://www.isca-archive.org/interspeech_2026/lu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lu26b_interspeech/markdown.md"},{"id":"lu26c_interspeech","title":"Breaking Neutral Bias: Zero-Human-Annotation Fine-Grained Emotion Enrichment via Semantic Drift and Discriminative Re-ranking","authors":["Qihang Lu","Wenbing Yang","Bingsong Bai","Zihan Sun","Yueran Hou","Peilei Jia","Ya Li","Jun Gao","Yingming Gao"],"year":2026,"doi":"10.21437/Interspeech.2026-1925","isca_url":"https://www.isca-archive.org/interspeech_2026/lu26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lu26c_interspeech.pdf","session":"Reasoning with Speech/Audio Language Models","topics":["speech-llm","paralinguistics","emotion-recognition"],"category":"paralinguistics-emotion","institutions":["Beijing University of Posts and Telecommunications","Hello Group Inc"],"funding":["National Key R&D Program of China","National Natural Science Foundation of China","National Language Commission","National Social Science Fund of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lu26c_interspeech","category":"paralinguistics-emotion","institutions":["Beijing University of Posts and Telecommunications","Hello Group Inc"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1925","pdf":"https://www.isca-archive.org/interspeech_2026/lu26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lu26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lu26c_interspeech/markdown.md"},{"id":"lu26d_interspeech","title":"Speech-to-See: End-to-End Speech-Driven Open-Set Object Detection","authors":["Wenhuan Lu","Xinyue Song","Wenjun Ke","Zhizhi Yu","Wenhao Yang","Jianguo Wei"],"year":2026,"doi":"10.21437/Interspeech.2026-2183","isca_url":"https://www.isca-archive.org/interspeech_2026/lu26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lu26d_interspeech.pdf","session":"Audio-Visual Grounding, Synchronization & Video Understanding","topics":["speech-llm","self-supervised","multimodal"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["Tianjin University","PipeChina Institute of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lu26d_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["Tianjin University","PipeChina Institute of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2183","pdf":"https://www.isca-archive.org/interspeech_2026/lu26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lu26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lu26d_interspeech/markdown.md"},{"id":"lugo26_interspeech","title":"DiffVQE: Hybrid Diffusion Voice Quality Enhancement Under Acoustic Echo and Noise","authors":["Haljan Lugo","Ernst Seidel","Pejman Mowlaee","Ziyue Zhao","Tim Fingscheidt"],"year":2026,"doi":"10.21437/Interspeech.2026-2337","isca_url":"https://www.isca-archive.org/interspeech_2026/lugo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lugo26_interspeech.pdf","session":"Active Noise and Echo Control, Sound Zones and Packet-Loss Concealment","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["efficient-on-device","generative-model","robustness-noise"],"institutions":["Technische Universitat Braunschweig","GN Advanced Science"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lugo26_interspeech","category":"enhancement-separation","labels":["efficient-on-device","generative-model","robustness-noise"],"institutions":["Technische Universitat Braunschweig","GN Advanced Science"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2337","pdf":"https://www.isca-archive.org/interspeech_2026/lugo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lugo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lugo26_interspeech/markdown.md"},{"id":"luisi26_interspeech","title":"Smooth Formant Tracking with Differentiable Linear Prediction","authors":["Bryn Luisi","Lauri Juvela"],"year":2026,"doi":"10.21437/Interspeech.2026-1222","isca_url":"https://www.isca-archive.org/interspeech_2026/luisi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/luisi26_interspeech.pdf","session":"Audio signal analysis","topics":["phonetics","self-supervised","speech-enhancement"],"category":"phonetics-linguistics","institutions":["Aalto University"],"code":{"url":"https://github.com/brynluisi/allpole-formants.git","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"luisi26_interspeech","category":"phonetics-linguistics","institutions":["Aalto University"],"code":"https://github.com/brynluisi/allpole-formants.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1222","pdf":"https://www.isca-archive.org/interspeech_2026/luisi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/luisi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/luisi26_interspeech/markdown.md"},{"id":"luo26_interspeech","title":"FCPE: A Fast Context-based Pitch Estimation Model","authors":["Yuxin Luo","Ruoyi Zhang","Lu-Chuan Liu","Tianyu Li","Hangyu Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-500","isca_url":"https://www.isca-archive.org/interspeech_2026/luo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/luo26_interspeech.pdf","session":"Audio signal analysis","topics":["speech-enhancement","voice-conversion","self-supervised"],"category":"tts","labels":["efficient-on-device"],"institutions":["Fish Audio","University of Science and Technology of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"luo26_interspeech","category":"tts","labels":["efficient-on-device"],"institutions":["Fish Audio","University of Science and Technology of China"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-500","pdf":"https://www.isca-archive.org/interspeech_2026/luo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/luo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/luo26_interspeech/markdown.md"},{"id":"luo26b_interspeech","title":"Visually-Guided Spatial Audio Generation for 360° In-the-Wild Speech Scenes","authors":["Qingyu Luo","Peng Zhang","Wenwu Wang","Philip J.B. Jackson"],"year":2026,"doi":"10.21437/Interspeech.2026-2577","isca_url":"https://www.isca-archive.org/interspeech_2026/luo26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/luo26b_interspeech.pdf","session":"Spatial Audio 2","topics":["speech-enhancement","self-supervised","dataset"],"category":"enhancement-separation","labels":["dataset-or-benchmark-release"],"institutions":["University of Surrey"],"funding":["Bang & Olufsen A/S","AURIC"],"code":{"url":"https://spatial-audio-demo.github.io/demos/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"luo26b_interspeech","category":"enhancement-separation","labels":["dataset-or-benchmark-release"],"institutions":["University of Surrey"],"code":"https://spatial-audio-demo.github.io/demos/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2577","pdf":"https://www.isca-archive.org/interspeech_2026/luo26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/luo26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/luo26b_interspeech/markdown.md"},{"id":"ly26_interspeech","title":"TinyGiantALM: A Compact Audio-Language Model for Intent-Aware Reasoning under Resource Constraints","authors":["Vinh-Thuan Ly"],"year":2026,"doi":"10.21437/Interspeech.2026-491","isca_url":"https://www.isca-archive.org/interspeech_2026/ly26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ly26_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","self-supervised","low-resource"],"category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised"],"institutions":["Zalo AI","University of Science, VNU-HCM","Vietnam National University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ly26_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised"],"institutions":["Zalo AI","University of Science, VNU-HCM","Vietnam National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-491","pdf":"https://www.isca-archive.org/interspeech_2026/ly26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ly26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ly26_interspeech/markdown.md"},{"id":"lyu26_interspeech","title":"TriA Pipeline: A Large-Scale Automatic Audio Annotation Pipeline For Audio Classification In Specific Scenarios","authors":["Hong Lyu","Mingru Yang","Qianhua He","Yanxiong Li","Jinxin Huang","Zhengyu Pei"],"year":2026,"doi":"10.21437/Interspeech.2026-995","isca_url":"https://www.isca-archive.org/interspeech_2026/lyu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/lyu26_interspeech.pdf","session":"Acoustic Event Detection 1","topics":["speech-enhancement","dataset","self-supervised"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["South China University of Technology"],"funding":["National Natural Science Foundation of China","China-Croatia Science and Technology Cooperation Committee"],"code":{"url":"https://github.com/huanxian/TriA","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"lyu26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["South China University of Technology"],"code":"https://github.com/huanxian/TriA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-995","pdf":"https://www.isca-archive.org/interspeech_2026/lyu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/lyu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/lyu26_interspeech/markdown.md"},{"id":"ma26_interspeech","title":"The Interspeech 2026 Audio Reasoning Challenge: Evaluating Reasoning Process Quality for Audio Reasoning Models and Agents","authors":["Ziyang Ma","Ruiyang Xu","Yinghao Ma","Chao-Han Huck Yang","Bohan Li","Jaeyeon Kim","Jin Xu","Jinyu Li","Carlos Busso","Kai Yu","Eng Siong Chng","Xie Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-118","isca_url":"https://www.isca-archive.org/interspeech_2026/ma26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ma26_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","evaluation","self-supervised"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Shanghai Jiao Tong University","Nanyang Technological University","Queen Mary University of London","NVIDIA","Carnegie Mellon University","Alibaba Group","Microsoft Corporation"],"funding":["National Natural Science Foundation of China","Shanghai Municipal Science and Technology Major Project","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"https://github.com/ddlBoJack/MMAR","stars":222,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ma26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Shanghai Jiao Tong University","Nanyang Technological University","Queen Mary University of London","NVIDIA","Carnegie Mellon University","Alibaba Group","Microsoft Corporation"],"code":"https://github.com/ddlBoJack/MMAR","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-118","pdf":"https://www.isca-archive.org/interspeech_2026/ma26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ma26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ma26_interspeech/markdown.md"},{"id":"ma26b_interspeech","title":"More than a feeling: Expressive style influences cortical speech tracking in subjective cognitive decline","authors":["Matthew King-Hang Ma","Yun Feng","Cloris Pui-Hang Li","Manson Cheuk-Man Fong"],"year":2026,"doi":"10.21437/Interspeech.2026-527","isca_url":"https://www.isca-archive.org/interspeech_2026/ma26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ma26b_interspeech.pdf","session":"Neurophysiology of Speech","topics":["paralinguistics","health","phonetics"],"category":"health-clinical","institutions":["Hong Kong Polytechnic University"],"funding":["HKRGC Postdoctoral Fellowship Scheme"],"code":{"url":"https://doi.org/10.5281/zenodo.20748010","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ma26b_interspeech","category":"health-clinical","institutions":["Hong Kong Polytechnic University"],"code":"https://doi.org/10.5281/zenodo.20748010","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-527","pdf":"https://www.isca-archive.org/interspeech_2026/ma26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ma26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ma26b_interspeech/markdown.md"},{"id":"ma26c_interspeech","title":"MeanVC 2: Robust Low-Latency Streaming Zero-Shot Voice Conversion","authors":["Guobin Ma","Yuxuan Xia","Yuepeng Jiang","Dake Guo","Hanke Xie","Jingbin Hu","Yanbo Wang","Lei Xie","Pengcheng Zhu"],"year":2026,"doi":"10.21437/Interspeech.2026-1961","isca_url":"https://www.isca-archive.org/interspeech_2026/ma26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ma26c_interspeech.pdf","session":"Streaming Speech Synthesis","topics":["voice-conversion","streaming","self-supervised"],"category":"tts","labels":["efficient-on-device","self-supervised","streaming-real-time","generative-model"],"institutions":["Northwestern Polytechnical University","University of New South Wales","WeNet Open Source Community"],"code":{"url":"https://aslp-lab.github.io/MeanVC2/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ma26c_interspeech","category":"tts","labels":["efficient-on-device","self-supervised","streaming-real-time","generative-model"],"institutions":["Northwestern Polytechnical University","University of New South Wales","WeNet Open Source Community"],"code":"https://aslp-lab.github.io/MeanVC2/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1961","pdf":"https://www.isca-archive.org/interspeech_2026/ma26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ma26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ma26c_interspeech/markdown.md"},{"id":"ma26d_interspeech","title":"Retention-Preserving Gradient Projection with Entropy-Guided Token-Level Distillation for Rehearsal-Free Continual ASR","authors":["Seunghee Ma","Junseok Oh","Ji-Hwan Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-2309","isca_url":"https://www.isca-archive.org/interspeech_2026/ma26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ma26d_interspeech.pdf","session":"ASR Under Real-World Constraints: Streaming, Adaptation, and Efficiency","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["Sogang University","LOTTE INNOVATE"],"funding":["National Research Foundation of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ma26d_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["Sogang University","LOTTE INNOVATE"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2309","pdf":"https://www.isca-archive.org/interspeech_2026/ma26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ma26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ma26d_interspeech/markdown.md"},{"id":"madha26_interspeech","title":"DLLM-TTS: Block Discrete Diffusion Language Model for Text-to-Speech Synthesis","authors":["Wasim Madha","Nityanand Mathur","Hamees Sayed","Apoorv Singh","Sameer Khurana","Akshat Mandloi","Sudarshan Kamath"],"year":2026,"doi":"10.21437/Interspeech.2026-788","isca_url":"https://www.isca-archive.org/interspeech_2026/madha26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/madha26_interspeech.pdf","session":"LLM Based Speech Synthesis","topics":["tts","self-supervised","speech-llm"],"category":"tts","labels":["generative-model"],"institutions":["Smallest.ai"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"madha26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Smallest.ai"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-788","pdf":"https://www.isca-archive.org/interspeech_2026/madha26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/madha26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/madha26_interspeech/markdown.md"},{"id":"maghsoudi26_interspeech","title":"Relating the Neural Representations of Vocalized, Mimed, and Imagined Speech","authors":["Maryam Maghsoudi","Rupesh Chillale","Shihab A Shamma"],"year":2026,"doi":"10.21437/Interspeech.2026-2836","isca_url":"https://www.isca-archive.org/interspeech_2026/maghsoudi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/maghsoudi26_interspeech.pdf","session":"Neurophysiology of Speech","topics":["speech-decoding","brain-computer-interface","self-supervised"],"category":"applications-other","institutions":["University of Maryland"],"funding":["Airforce Office of Scientific Research","NIH"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"maghsoudi26_interspeech","category":"applications-other","institutions":["University of Maryland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2836","pdf":"https://www.isca-archive.org/interspeech_2026/maghsoudi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/maghsoudi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/maghsoudi26_interspeech/markdown.md"},{"id":"magoshi26_interspeech","title":"Refining Pseudo-Audio Prompts with Speech-Text Alignment for Text-Only Domain Adaptation in LLM-Based ASR","authors":["Ryo Magoshi","Takashi Maekaku","Yusuke Shinohara"],"year":2026,"doi":"10.21437/Interspeech.2026-977","isca_url":"https://www.isca-archive.org/interspeech_2026/magoshi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/magoshi26_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 1","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["multilingual"],"institutions":["Kyoto University","LY Corporation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"magoshi26_interspeech","category":"asr","labels":["multilingual"],"institutions":["Kyoto University","LY Corporation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-977","pdf":"https://www.isca-archive.org/interspeech_2026/magoshi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/magoshi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/magoshi26_interspeech/markdown.md"},{"id":"magoshi26b_interspeech","title":"Improving Zero-Shot Phonetic Classification through Language-Agnostic Articulatory Features","authors":["Ryo Magoshi","Jaeyoung Lee","Shinsuke Sakai","Tatsuya Kawahara"],"year":2026,"doi":"10.21437/Interspeech.2026-2246","isca_url":"https://www.isca-archive.org/interspeech_2026/magoshi26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/magoshi26b_interspeech.pdf","session":"Low-Resource & Endangered Language Speech Processing","topics":["self-supervised","phonetics","evaluation"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Kyoto University","NTT"],"funding":["JST NEXUS"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"magoshi26b_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Kyoto University","NTT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2246","pdf":"https://www.isca-archive.org/interspeech_2026/magoshi26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/magoshi26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/magoshi26b_interspeech/markdown.md"},{"id":"mahapatra26_interspeech","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","authors":["Aurosweta Mahapatra","Ismail Rasim Ulgen","Kong Aik Lee","Nicholas Andrews","Berrak Sisman"],"year":2026,"doi":"10.21437/Interspeech.2026-831","isca_url":"https://www.isca-archive.org/interspeech_2026/mahapatra26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mahapatra26_interspeech.pdf","session":"Spoofing and Deepfake Detection 1","topics":["audio-deepfake","self-supervised","prosody"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["Johns Hopkins University","Hong Kong Polytechnic University"],"funding":["National Science Foundation","Office of the Director of National Intelligence","Intelligence Advanced Research Projects Activity"],"code":{"url":"https://prosdd.github.io/ProSDD_website/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mahapatra26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["Johns Hopkins University","Hong Kong Polytechnic University"],"code":"https://prosdd.github.io/ProSDD_website/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-831","pdf":"https://www.isca-archive.org/interspeech_2026/mahapatra26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mahapatra26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mahapatra26_interspeech/markdown.md"},{"id":"mahmudi26_interspeech","title":"Easper: An Accessible ASR Pipeline for Language Documentation","authors":["Aso Mahmudi","Ting Dang","Ekaterina Vylomova","Nick Thieberger"],"year":2026,"doi":"10.21437/Interspeech.2026-2781","isca_url":"https://www.isca-archive.org/interspeech_2026/mahmudi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mahmudi26_interspeech.pdf","session":"Corpus Creation, Summarization and Understanding","topics":["asr","low-resource","self-supervised"],"category":"asr","labels":["low-resource","self-supervised"],"institutions":["University of Melbourne"],"funding":["Australian Research Council","Language Data Commons of Australia","Petascale Campus Initiative"],"code":{"url":"https://github.com/Aso-UniMelb/Easper","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mahmudi26_interspeech","category":"asr","labels":["low-resource","self-supervised"],"institutions":["University of Melbourne"],"code":"https://github.com/Aso-UniMelb/Easper","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2781","pdf":"https://www.isca-archive.org/interspeech_2026/mahmudi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mahmudi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mahmudi26_interspeech/markdown.md"},{"id":"makarov26_interspeech","title":"Repurposing a Speech Classifier for Guided Diffusion-Based Speech Generation","authors":["Rostislav Makarov","Timo Gerkmann"],"year":2026,"doi":"10.21437/Interspeech.2026-3448","isca_url":"https://www.isca-archive.org/interspeech_2026/makarov26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/makarov26_interspeech.pdf","session":"Acoustic Signal Analysis and Generation","topics":["tts","self-supervised","dataset"],"category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["University of Hamburg"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"https://sp-uhh.github.io/classifier-to-diffusion/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"makarov26_interspeech","category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["University of Hamburg"],"code":"https://sp-uhh.github.io/classifier-to-diffusion/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3448","pdf":"https://www.isca-archive.org/interspeech_2026/makarov26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/makarov26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/makarov26_interspeech/markdown.md"},{"id":"makishima26_interspeech","title":"Multi-Talker ASR Unaffected by Speaker Change Count","authors":["Naoki Makishima","Suzuka Yamada","Taiga Yamane","Mana Ihori","Tanaka Tomohiro","Satoshi Suzuki","Shota Orihashi","Ryo Masumura"],"year":2026,"doi":"10.21437/Interspeech.2026-1582","isca_url":"https://www.isca-archive.org/interspeech_2026/makishima26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/makishima26_interspeech.pdf","session":"Robust ASR: Hallucinations and Biases","topics":["asr","speaker-diarization","self-supervised"],"category":"asr","institutions":["NTT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"makishima26_interspeech","category":"asr","institutions":["NTT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1582","pdf":"https://www.isca-archive.org/interspeech_2026/makishima26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/makishima26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/makishima26_interspeech/markdown.md"},{"id":"makishima26b_interspeech","title":"Unified Audio-Visual Modeling to Recognize Which Face Spoke When and What in Scenarios with On- and Off-Screen Participants","authors":["Naoki Makishima","Suzuka Yamada","Taiga Yamane","Mana Ihori","Tanaka Tomohiro","Satoshi Suzuki","Shota Orihashi","Ryo Masumura"],"year":2026,"doi":"10.21437/Interspeech.2026-1592","isca_url":"https://www.isca-archive.org/interspeech_2026/makishima26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/makishima26b_interspeech.pdf","session":"Multimodal Speech Processing and Speech LLM Systems","topics":["audio-visual-speech-recognition","active-speaker-detection","speech-recognition"],"category":"asr","institutions":["NTT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"makishima26b_interspeech","category":"asr","institutions":["NTT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1592","pdf":"https://www.isca-archive.org/interspeech_2026/makishima26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/makishima26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/makishima26b_interspeech/markdown.md"},{"id":"mallik26_interspeech","title":"MAC-VAD: A Modality-Aligned Cross-Attentive Framework for Robust Voice Activity Detection","authors":["Bruhanth Mallik","Chintan Tundia","Kumud Tripathi","Shreyas Nagoor","Pankaj Wasnik"],"year":2026,"doi":"10.21437/Interspeech.2026-384","isca_url":"https://www.isca-archive.org/interspeech_2026/mallik26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mallik26_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["speech-enhancement","self-supervised","dataset"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["Sony"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mallik26_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["Sony"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-384","pdf":"https://www.isca-archive.org/interspeech_2026/mallik26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mallik26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mallik26_interspeech/markdown.md"},{"id":"manabe26_interspeech","title":"ProLAP: Probabilistic Language-Audio Pre-Training","authors":["Toranosuke Manabe","Yuchi Ishikawa","Hokuto Munakata","Yoshimitsu Aoki","Tatsuya Komatsu"],"year":2026,"doi":"10.21437/Interspeech.2026-845","isca_url":"https://www.isca-archive.org/interspeech_2026/manabe26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/manabe26_interspeech.pdf","session":"Acoustic Signal Analysis and Generation","topics":["self-supervised","multilingual","evaluation"],"category":"audio-understanding","labels":["multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["Keio University","LY Corporation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"manabe26_interspeech","category":"audio-understanding","labels":["multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["Keio University","LY Corporation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-845","pdf":"https://www.isca-archive.org/interspeech_2026/manabe26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/manabe26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/manabe26_interspeech/markdown.md"},{"id":"manamalage26_interspeech","title":"FedMPA: A Novel Privacy-Performance Optimization Approach for Multimodal Speech-Based Depression Detection","authors":["Dushanthi Madhushika Manamalage","Frederick Sundram","Partha S. Roop","Seyed Reza Shahamiri"],"year":2026,"doi":"10.21437/Interspeech.2026-586","isca_url":"https://www.isca-archive.org/interspeech_2026/manamalage26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/manamalage26_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 1","topics":["speech-llm","paralinguistics","evaluation"],"category":"health-clinical","institutions":["DeepNet Discovery Network","University of Auckland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"manamalage26_interspeech","category":"health-clinical","institutions":["DeepNet Discovery Network","University of Auckland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-586","pdf":"https://www.isca-archive.org/interspeech_2026/manamalage26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/manamalage26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/manamalage26_interspeech/markdown.md"},{"id":"mandel26_interspeech","title":"From A to B to A: Palindromic Zero-Shot Voice Conversion with Non-Parallel Data","authors":["Moshe Mandel","Shlomi E. Chazan"],"year":2026,"doi":"10.21437/Interspeech.2026-1663","isca_url":"https://www.isca-archive.org/interspeech_2026/mandel26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mandel26_interspeech.pdf","session":"Voice Conversion","topics":["voice-conversion","self-supervised","multilingual"],"category":"tts","labels":["self-supervised","generative-model"],"institutions":["OriginAI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mandel26_interspeech","category":"tts","labels":["self-supervised","generative-model"],"institutions":["OriginAI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1663","pdf":"https://www.isca-archive.org/interspeech_2026/mandel26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mandel26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mandel26_interspeech/markdown.md"},{"id":"manohar26_interspeech","title":"SCRIBE: Diagnostic Evaluation and Rich Transcription Models for Indic ASR","authors":["Kavya Manohar","Arghya Bhattacharya","Kush Juvekar","Kumarmanas Nethil"],"year":2026,"doi":"10.21437/Interspeech.2026-3436","isca_url":"https://www.isca-archive.org/interspeech_2026/manohar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/manohar26_interspeech.pdf","session":"Robust and Real-World ASR Systems","topics":["asr","evaluation","multilingual"],"category":"resources-evaluation","labels":["multilingual"],"institutions":["Adalat AI"],"code":{"url":"https://github.com/adalat-ai-tech/scribe-eval","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"manohar26_interspeech","category":"resources-evaluation","labels":["multilingual"],"institutions":["Adalat AI"],"code":"https://github.com/adalat-ai-tech/scribe-eval","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3436","pdf":"https://www.isca-archive.org/interspeech_2026/manohar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/manohar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/manohar26_interspeech/markdown.md"},{"id":"mao26_interspeech","title":"Neural Multichannel Distant Speaker Diarization and Source Separation with Beta Speaker Activity Prior","authors":["Sicheng Mao","Mathieu Fontaine","Anthony Larcher","Roland Badeau"],"year":2026,"doi":"10.21437/Interspeech.2026-1248","isca_url":"https://www.isca-archive.org/interspeech_2026/mao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mao26_interspeech.pdf","session":"Speaker Diarization 2","topics":["speaker-diarization","source-separation","self-supervised"],"category":"speaker","institutions":["Telecom Paris","Institut Polytechnique de Paris","Universite du Mans"],"funding":["ANR Project SAROUMANE"],"code":{"url":"https://github.com/alephpi/neural-fcasa","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mao26_interspeech","category":"speaker","institutions":["Telecom Paris","Institut Polytechnique de Paris","Universite du Mans"],"code":"https://github.com/alephpi/neural-fcasa","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1248","pdf":"https://www.isca-archive.org/interspeech_2026/mao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mao26_interspeech/markdown.md"},{"id":"mao26b_interspeech","title":"Age-Related Changes in Mandarin Lexical Tone Production: Acoustic Properties and Tonal Distinctiveness","authors":["Zhongxuan Mao","Yan Feng","Chenyu Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1702","isca_url":"https://www.isca-archive.org/interspeech_2026/mao26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mao26b_interspeech.pdf","session":"Gender- and Age-Related Speech Characteristics","topics":["asr","paralinguistics","evaluation"],"category":"phonetics-linguistics","institutions":["Nanjing University of Science and Technology","Johns Hopkins University"],"funding":["Ministry of Education of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mao26b_interspeech","category":"phonetics-linguistics","institutions":["Nanjing University of Science and Technology","Johns Hopkins University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1702","pdf":"https://www.isca-archive.org/interspeech_2026/mao26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mao26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mao26b_interspeech/markdown.md"},{"id":"marchenko26_interspeech","title":"TIMBRE: Layer-Wise Cross-Lingual Speech Emotion Recognition Across 49 Layers and 26 Corpora","authors":["Anatoly Marchenko"],"year":2026,"doi":"10.21437/Interspeech.2026-579","isca_url":"https://www.isca-archive.org/interspeech_2026/marchenko26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/marchenko26_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["paralinguistics","emotion-recognition","self-supervised"],"category":"paralinguistics-emotion","labels":["multilingual","self-supervised"],"code":{"url":"https://doi.org/10.5281/zenodo.20584918","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"marchenko26_interspeech","category":"paralinguistics-emotion","labels":["multilingual","self-supervised"],"code":"https://doi.org/10.5281/zenodo.20584918","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-579","pdf":"https://www.isca-archive.org/interspeech_2026/marchenko26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/marchenko26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/marchenko26_interspeech/markdown.md"},{"id":"marcinek26_interspeech","title":"Vocal Effort Modulation Strategies: A Cross-Corpus Taxonomy with Noise Robustness and ASR Implications","authors":["Lubos Marcinek","Jonas Beskow","Joakim Gustafson"],"year":2026,"doi":"10.21437/Interspeech.2026-2747","isca_url":"https://www.isca-archive.org/interspeech_2026/marcinek26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/marcinek26_interspeech.pdf","session":"Phonetic Aspects of TTS and ASR Systems","topics":["tts","asr","paralinguistics"],"category":"paralinguistics-emotion","labels":["robustness-noise"],"institutions":["KTH Royal Institute of Technology"],"funding":["WASP","Digital Futures"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"marcinek26_interspeech","category":"paralinguistics-emotion","labels":["robustness-noise"],"institutions":["KTH Royal Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2747","pdf":"https://www.isca-archive.org/interspeech_2026/marcinek26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/marcinek26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/marcinek26_interspeech/markdown.md"},{"id":"marcinek26b_interspeech","title":"Optimal Linguistic Complexity for Dialogue System Speech in Noise: Convergent Evidence from Automatic and Human Transcription","authors":["Lubos Marcinek","Jonas Beskow","Joakim Gustafson"],"year":2026,"doi":"10.21437/Interspeech.2026-2799","isca_url":"https://www.isca-archive.org/interspeech_2026/marcinek26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/marcinek26b_interspeech.pdf","session":"Spoken Dialogue Systems","topics":["asr","tts","spoken-language-understanding"],"category":"asr","labels":["robustness-noise"],"institutions":["KTH Royal Institute of Technology"],"funding":["PerCorSo","AAIS","WASP","Digital Futures"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"marcinek26b_interspeech","category":"asr","labels":["robustness-noise"],"institutions":["KTH Royal Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2799","pdf":"https://www.isca-archive.org/interspeech_2026/marcinek26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/marcinek26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/marcinek26b_interspeech/markdown.md"},{"id":"marew26_interspeech","title":"Constrained CTC decoding for Efficient Diacritic Restoration","authors":["Rufael Marew","Amr Keleg","Hanan Aldarmaki"],"year":2026,"doi":"10.21437/Interspeech.2026-3220","isca_url":"https://www.isca-archive.org/interspeech_2026/marew26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/marew26_interspeech.pdf","session":"Search Methods and Inference Algorithms","topics":["asr","tts","self-supervised"],"category":"asr","institutions":["Mohamed bin Zayed University of Artificial Intelligence"],"code":{"url":"https://github.com/rufaelfekadu/DiaCTC","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"marew26_interspeech","category":"asr","institutions":["Mohamed bin Zayed University of Artificial Intelligence"],"code":"https://github.com/rufaelfekadu/DiaCTC","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3220","pdf":"https://www.isca-archive.org/interspeech_2026/marew26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/marew26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/marew26_interspeech/markdown.md"},{"id":"martin26_interspeech","title":"Acoustic Biomarkers of Sleep Deprivation on French Read Speech: Interpretable and Frugal Modeling of Sleep Deprivation and Its Symptoms","authors":["Vincent P. Martin","Jean-Luc Rouas","Pierre Philip"],"year":2026,"doi":"10.21437/Interspeech.2026-777","isca_url":"https://www.isca-archive.org/interspeech_2026/martin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/martin26_interspeech.pdf","session":"Clinically Useful Speech Representations 2","topics":["paralinguistics","evaluation","low-resource"],"category":"health-clinical","labels":["efficient-on-device"],"institutions":["Université de Lorraine","CNRS","Inria","University of Mons","Université de Bordeaux","Bordeaux INP","CHU Bordeaux"],"funding":["Labex BRAIN","French National Research Agency"],"code":{"url":"https://github.com/vincentpmartin/Interspeech2026.SOMVOICE.classification","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"martin26_interspeech","category":"health-clinical","labels":["efficient-on-device"],"institutions":["Université de Lorraine","CNRS","Inria","University of Mons","Université de Bordeaux","Bordeaux INP","CHU Bordeaux"],"code":"https://github.com/vincentpmartin/Interspeech2026.SOMVOICE.classification","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-777","pdf":"https://www.isca-archive.org/interspeech_2026/martin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/martin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/martin26_interspeech/markdown.md"},{"id":"martinez26_interspeech","title":"findsylls: A Language‑Agnostic Toolkit for Syllable‑Level Speech Tokenization and Embedding","authors":["Héctor Javier Vázquez Martínez"],"year":2026,"doi":"10.21437/Interspeech.2026-820","isca_url":"https://www.isca-archive.org/interspeech_2026/martinez26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/martinez26_interspeech.pdf","session":"Speech signal analysis","topics":["speech-tokenization","self-supervised","low-resource"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Pennsylvania"],"code":{"url":"https://github.com/hjvm/findsylls","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"martinez26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Pennsylvania"],"code":"https://github.com/hjvm/findsylls","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-820","pdf":"https://www.isca-archive.org/interspeech_2026/martinez26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/martinez26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/martinez26_interspeech/markdown.md"},{"id":"marttila26_interspeech","title":"Differentiable Pitch Matching with Auditory Models","authors":["David Marttila","Joshua D. Reiss"],"year":2026,"doi":"10.21437/Interspeech.2026-2043","isca_url":"https://www.isca-archive.org/interspeech_2026/marttila26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/marttila26_interspeech.pdf","session":"Audio signal analysis","topics":["tts","speech-enhancement","self-supervised"],"category":"tts","institutions":["Queen Mary University of London"],"funding":["UK Research and Innovation"],"code":{"url":"https://github.com/google/carfac","stars":145,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"marttila26_interspeech","category":"tts","institutions":["Queen Mary University of London"],"code":"https://github.com/google/carfac","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2043","pdf":"https://www.isca-archive.org/interspeech_2026/marttila26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/marttila26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/marttila26_interspeech/markdown.md"},{"id":"maruyama26_interspeech","title":"Real-Time CARFAC/SAI-derived Pitchogram for Seeing and Correcting Pronunciation in Mandarin Chinese Tones","authors":["Yuka Maruyama","Jason Orlosky","Chris Lee","Flora Salim","Thad Starner","Benjamin Tag"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/maruyama26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/maruyama26_interspeech.pdf","session":"Speech and Language Learning Technologies","topics":["speech-enhancement","evaluation","low-resource"],"category":"applications-other","labels":["streaming-real-time"],"institutions":["University of New South Wales","Augusta University","Georgia Institute of Technology"],"code":{"url":"https://github.com/google/carfac","stars":145,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"maruyama26_interspeech","category":"applications-other","labels":["streaming-real-time"],"institutions":["University of New South Wales","Augusta University","Georgia Institute of Technology"],"code":"https://github.com/google/carfac","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/maruyama26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/maruyama26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/maruyama26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/maruyama26_interspeech/markdown.md"},{"id":"masmolla26_interspeech","title":"Improving streaming ASR with foundation models using emission policies","authors":["Gerard Mas Mollà","Albert Sanchis","Alfons Juan"],"year":2026,"doi":"10.21437/Interspeech.2026-3358","isca_url":"https://www.isca-archive.org/interspeech_2026/masmolla26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/masmolla26_interspeech.pdf","session":"Efficient Inference for ASR and Speech LMs","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["self-supervised","streaming-real-time"],"institutions":["Universitat Politecnica de Valencia"],"funding":["Government of Spain"],"code":{"url":"https://github.com/germol00/streaming_SFMs","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"masmolla26_interspeech","category":"asr","labels":["self-supervised","streaming-real-time"],"institutions":["Universitat Politecnica de Valencia"],"code":"https://github.com/germol00/streaming_SFMs","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3358","pdf":"https://www.isca-archive.org/interspeech_2026/masmolla26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/masmolla26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/masmolla26_interspeech/markdown.md"},{"id":"masuyama26_interspeech","title":"HRTF Personalization via Sim-to-Real Neural Field","authors":["Yoshiki Masuyama","Gordon Wichern","Christoph Boeddeker","Julius Richter","Takahiro Edo","Swapnil Bhosale","Jonathan Le Roux"],"year":2026,"doi":"10.21437/Interspeech.2026-1391","isca_url":"https://www.isca-archive.org/interspeech_2026/masuyama26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/masuyama26_interspeech.pdf","session":"Spatial Audio 1","topics":["spatial-audio","self-supervised","evaluation"],"category":"applications-other","institutions":["Mitsubishi Electric Research Laboratories","University of Surrey"],"code":{"url":"https://github.com/merlresearch/s2rnf","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"masuyama26_interspeech","category":"applications-other","institutions":["Mitsubishi Electric Research Laboratories","University of Surrey"],"code":"https://github.com/merlresearch/s2rnf","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1391","pdf":"https://www.isca-archive.org/interspeech_2026/masuyama26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/masuyama26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/masuyama26_interspeech/markdown.md"},{"id":"masztalski26_interspeech","title":"Samsone: A Family of Open Small Audio Language Models for On-Device Inference","authors":["Piotr Masztalski","Michał K. Grzeszczyk","Olaf Sikorski"],"year":2026,"doi":"10.21437/Interspeech.2026-763","isca_url":"https://www.isca-archive.org/interspeech_2026/masztalski26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/masztalski26_interspeech.pdf","session":"Audio Foundation Models and Generation","topics":["speech-llm","on-device","low-resource"],"category":"speech-llm-dialogue","labels":["efficient-on-device","streaming-real-time"],"institutions":["Samsung","AGH University of Krakow"],"code":{"url":"https://github.com/SamsungLabs/samsone","stars":28,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"masztalski26_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","streaming-real-time"],"institutions":["Samsung","AGH University of Krakow"],"code":"https://github.com/SamsungLabs/samsone","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-763","pdf":"https://www.isca-archive.org/interspeech_2026/masztalski26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/masztalski26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/masztalski26_interspeech/markdown.md"},{"id":"mathur26_interspeech","title":"How Do Instructions Shape Speech? Cross-Attention Attribution for Style-Captioned Text-to-Speech","authors":["Nityanand Mathur","Hamees Sayed","Wasim Madha","Apoorv Singh","Sameer Khurana","Akshat Mandloi","Sudarshan Kamath"],"year":2026,"doi":"10.21437/Interspeech.2026-2805","isca_url":"https://www.isca-archive.org/interspeech_2026/mathur26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mathur26_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["Smallest.ai"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mathur26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Smallest.ai"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2805","pdf":"https://www.isca-archive.org/interspeech_2026/mathur26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mathur26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mathur26_interspeech/markdown.md"},{"id":"maxwell26_interspeech","title":"Pronunciation and Intonation Structured Markup (PRISM): A Dataset for Australian English Pronunciation Feedback","authors":["Olga Maxwell","Uy Thinh Quang","Debbie Loakes","Adele Gregory","Robert Turnbull"],"year":2026,"doi":"10.21437/Interspeech.2026-2830","isca_url":"https://www.isca-archive.org/interspeech_2026/maxwell26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/maxwell26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["dataset","evaluation","prosody"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["University of Melbourne"],"funding":["Melbourne Data Analytics Program Collaboration Grant","Learning and Teaching Initiatives Grant"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"maxwell26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["University of Melbourne"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2830","pdf":"https://www.isca-archive.org/interspeech_2026/maxwell26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/maxwell26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/maxwell26_interspeech/markdown.md"},{"id":"mboungou26_interspeech","title":"Audio-visual Contrastive Alignment for Diffusion-based Visual-conditioned Speech Enhancement","authors":["Colombe Mboungou","Mostafa Sadeghi","Jean-Eudes Ayilo","Romain Serizel"],"year":2026,"doi":"10.21437/Interspeech.2026-766","isca_url":"https://www.isca-archive.org/interspeech_2026/mboungou26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mboungou26_interspeech.pdf","session":"Audio-Visual and Generative Target Speaker Extraction","topics":["speech-enhancement","self-supervised","multimodal"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Universite de Lorraine","CNRS","Inria","LORIA"],"code":{"url":"https://github.com/cexauce/AV-CA-DiffUSE","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mboungou26_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Universite de Lorraine","CNRS","Inria","LORIA"],"code":"https://github.com/cexauce/AV-CA-DiffUSE","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-766","pdf":"https://www.isca-archive.org/interspeech_2026/mboungou26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mboungou26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mboungou26_interspeech/markdown.md"},{"id":"mcauliffe26_interspeech","title":"Montreal Forced Aligner and the state of speech-to-text alignment in 2026","authors":["Michael McAuliffe","Kaylynn Gunter","Michael Wagner","Morgan Sonderegger"],"year":2026,"doi":"10.21437/Interspeech.2026-2734","isca_url":"https://www.isca-archive.org/interspeech_2026/mcauliffe26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mcauliffe26_interspeech.pdf","session":"Speech Representations and Alignment","topics":["asr","phonetics","dataset"],"category":"asr","labels":["multilingual"],"institutions":["University of Wisconsin-Madison","McGill University","University of Oregon"],"funding":["Social Sciences and Humanities Research Council","Fonds de recherche sur la societe et la culture","Canada Foundation for Innovation","Canada Research Chairs","National Institutes of Health"],"code":{"url":"https://github.com/MontrealCorpusTools/mfa-interspeech2026","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mcauliffe26_interspeech","category":"asr","labels":["multilingual"],"institutions":["University of Wisconsin-Madison","McGill University","University of Oregon"],"code":"https://github.com/MontrealCorpusTools/mfa-interspeech2026","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2734","pdf":"https://www.isca-archive.org/interspeech_2026/mcauliffe26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mcauliffe26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mcauliffe26_interspeech/markdown.md"},{"id":"mcghee26_interspeech","title":"Feature Design and Generative Modelling in Deep Articulatory Synthesis","authors":["Charles McGhee","Mark J.F. Gales","Kate Knill"],"year":2026,"doi":"10.21437/Interspeech.2026-694","isca_url":"https://www.isca-archive.org/interspeech_2026/mcghee26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mcghee26_interspeech.pdf","session":"Articulatory, EMG, and Visual Speech Generation","topics":["asr","tts","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["University of Cambridge"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mcghee26_interspeech","category":"tts","labels":["generative-model"],"institutions":["University of Cambridge"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-694","pdf":"https://www.isca-archive.org/interspeech_2026/mcghee26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mcghee26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mcghee26_interspeech/markdown.md"},{"id":"mcguire26_interspeech","title":"Lexical stress-conditioned spatiotemporal gestural coordination in L2 English","authors":["Paul McGuire","Michael Proctor","Feng-fan Hsieh","Yueh-chin Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-3116","isca_url":"https://www.isca-archive.org/interspeech_2026/mcguire26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mcguire26_interspeech.pdf","session":"Speech Production and Perception 2","topics":["phonetics","prosody","evaluation"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Macquarie University","National Tsing Hua University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mcguire26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Macquarie University","National Tsing Hua University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3116","pdf":"https://www.isca-archive.org/interspeech_2026/mcguire26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mcguire26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mcguire26_interspeech/markdown.md"},{"id":"mcintosh26_interspeech","title":"Speech Playground: An Interactive Tool for Speech Analysis and Comparison","authors":["Stephen McIntosh","Daisuke Saito","Nobuaki Minematsu"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/mcintosh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mcintosh26_interspeech.pdf","session":"Speech Analysis, Data Resources and Research Tools","topics":["evaluation","self-supervised","phonetics"],"category":"resources-evaluation","labels":["self-supervised"],"institutions":["University of Tokyo"],"code":{"url":"https://github.com/stephenmac7/mfa-service","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mcintosh26_interspeech","category":"resources-evaluation","labels":["self-supervised"],"institutions":["University of Tokyo"],"code":"https://github.com/stephenmac7/mfa-service","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/mcintosh26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/mcintosh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mcintosh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mcintosh26_interspeech/markdown.md"},{"id":"mehendale26_interspeech","title":"Indic DiarBench: A Multilingual Joint Diarization and ASR Benchmark for Indian Languages","authors":["Deovrat Mehendale","Aditya Mehndiratta","Dhruv Subhash Rathi","Kaushal Bhogale","Mitesh M Khapra"],"year":2026,"doi":"10.21437/Interspeech.2026-2484","isca_url":"https://www.isca-archive.org/interspeech_2026/mehendale26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mehendale26_interspeech.pdf","session":"Speaker Diarization 1","topics":["speech-llm","multilingual","evaluation"],"category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Sarvam AI","IIT Madras"],"code":{"url":"https://huggingface.co/datasets/sarvamai/indic-diarbench","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mehendale26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Sarvam AI","IIT Madras"],"code":"https://huggingface.co/datasets/sarvamai/indic-diarbench","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2484","pdf":"https://www.isca-archive.org/interspeech_2026/mehendale26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mehendale26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mehendale26_interspeech/markdown.md"},{"id":"mehlman26_interspeech","title":"Speech Entrainment in Multi-Party Conversations with a Digital Agent","authors":["Nicholas Mehlman","Kaitlin Zareno","Kleanthis Avramidis","Anfeng Xu","Shrikanth Narayanan"],"year":2026,"doi":"10.21437/Interspeech.2026-2851","isca_url":"https://www.isca-archive.org/interspeech_2026/mehlman26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mehlman26_interspeech.pdf","session":"Entrainment and Dialogue Coordination","topics":["paralinguistics","dataset","evaluation"],"category":"phonetics-linguistics","institutions":["University of Southern California"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mehlman26_interspeech","category":"phonetics-linguistics","institutions":["University of Southern California"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2851","pdf":"https://www.isca-archive.org/interspeech_2026/mehlman26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mehlman26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mehlman26_interspeech/markdown.md"},{"id":"mehralian26_interspeech","title":"Predict-Then-Adapt: Inferring Coordinates from Speech for Continuous Geo-Conditioned Dialectal ASR","authors":["Pouya Mehralian","Hugo Van hamme"],"year":2026,"doi":"10.21437/Interspeech.2026-3333","isca_url":"https://www.isca-archive.org/interspeech_2026/mehralian26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mehralian26_interspeech.pdf","session":"Language and Dialect Recognition","topics":["asr","low-resource","multilingual"],"category":"asr","institutions":["KU Leuven"],"funding":["Flemish Government"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mehralian26_interspeech","category":"asr","institutions":["KU Leuven"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3333","pdf":"https://www.isca-archive.org/interspeech_2026/mehralian26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mehralian26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mehralian26_interspeech/markdown.md"},{"id":"mejia26_interspeech","title":"Sound Reactor Mission: Gamified Misophonia Assessment to Bridge the Gap in Audiology and Hearing Care","authors":["Jorge Mejia","Jan-Willem Wasmann","Jane Gregory"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/mejia26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mejia26_interspeech.pdf","session":"Speech and Language Processing for Health and Accessibility","topics":["speech-enhancement","evaluation","paralinguistics"],"category":"health-clinical","labels":["robustness-noise"],"institutions":["National Acoustic Laboratories","Radboud University Medical Center","University of Oxford"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mejia26_interspeech","category":"health-clinical","labels":["robustness-noise"],"institutions":["National Acoustic Laboratories","Radboud University Medical Center","University of Oxford"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/mejia26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/mejia26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mejia26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mejia26_interspeech/markdown.md"},{"id":"meng26_interspeech","title":"A Federated Learning-Based Speaker Recognition Method with Dual Classification Heads","authors":["Ying Meng","Zhihua Fang","Liang He"],"year":2026,"doi":"10.21437/Interspeech.2026-25","isca_url":"https://www.isca-archive.org/interspeech_2026/meng26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/meng26_interspeech.pdf","session":"Speaker Recognition and Verification","topics":["speaker-verification","self-supervised"],"category":"speaker","institutions":["Xinjiang University","Xinjiang Multimodal Information Technology Engineering Research Center","Tsinghua University","AGIBOT"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"meng26_interspeech","category":"speaker","institutions":["Xinjiang University","Xinjiang Multimodal Information Technology Engineering Research Center","Tsinghua University","AGIBOT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-25","pdf":"https://www.isca-archive.org/interspeech_2026/meng26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26_interspeech/markdown.md"},{"id":"meng26b_interspeech","title":"Learning Global Key Knowledge for Federated Speaker Recognition via Fisher Information","authors":["Ying Meng","Zhihua Fang","Liang He"],"year":2026,"doi":"10.21437/Interspeech.2026-196","isca_url":"https://www.isca-archive.org/interspeech_2026/meng26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/meng26b_interspeech.pdf","session":"Speaker Recognition and Verification","topics":["speaker-verification","self-supervised"],"category":"speaker","institutions":["Xinjiang University","Tsinghua University","AGIBOT"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"meng26b_interspeech","category":"speaker","institutions":["Xinjiang University","Tsinghua University","AGIBOT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-196","pdf":"https://www.isca-archive.org/interspeech_2026/meng26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26b_interspeech/markdown.md"},{"id":"meng26c_interspeech","title":"BiEAR: A Human Auditory-Inspired Adaptive Binaural Front-end for Multi-Speaker Localisation and Distance Estimation","authors":["Hanyu Meng","Eliathamby Ambikairajah","Vidhyasaharan Sethu","Qiquan Zhang","Haizhou Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1618","isca_url":"https://www.isca-archive.org/interspeech_2026/meng26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/meng26c_interspeech.pdf","session":"Spatial Audio 1","topics":["speaker-diarization","self-supervised","paralinguistics"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["University of New South Wales","Alibaba Group","Chinese University of Hong Kong, Shenzhen"],"funding":["ARC Discovery Grant"],"code":{"url":"https://github.com/Hanyu-Meng/BiEAR","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"meng26c_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["University of New South Wales","Alibaba Group","Chinese University of Hong Kong, Shenzhen"],"code":"https://github.com/Hanyu-Meng/BiEAR","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1618","pdf":"https://www.isca-archive.org/interspeech_2026/meng26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26c_interspeech/markdown.md"},{"id":"meng26d_interspeech","title":"Neuromorphic Speech Enhancement with Dual-Branch Spiking Neural Networks","authors":["Taiyu Meng","Wenbin Jiang","Haoyi Zhang","Yuhan Zhou","Haoyi Yin"],"year":2026,"doi":"10.21437/Interspeech.2026-1797","isca_url":"https://www.isca-archive.org/interspeech_2026/meng26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/meng26d_interspeech.pdf","session":"SE Architectures, Adaptation and Audio Front-Ends","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","robustness-noise"],"institutions":["Hangzhou Dianzi University"],"funding":["Yangtze River Delta Science and Technology Innovation Community Joint Research","Zhejiang Provincial Natural Science Foundation of China","Zhejiang Provincial College Student Innovation and Entrepreneurship Training Program"],"code":{"url":"https://meng-taiyu.github.io/dpnet-demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"meng26d_interspeech","category":"enhancement-separation","labels":["efficient-on-device","robustness-noise"],"institutions":["Hangzhou Dianzi University"],"code":"https://meng-taiyu.github.io/dpnet-demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1797","pdf":"https://www.isca-archive.org/interspeech_2026/meng26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26d_interspeech/markdown.md"},{"id":"meng26e_interspeech","title":"Steps toward a wearable-informed model of real-world listening effort and fatigue among adults with hearing loss","authors":["David Meng","Marisa Poulos","Erin O'Neill","Qi Yang","Ivan Iotzov","Jorge Mejia"],"year":2026,"doi":"10.21437/Interspeech.2026-2022","isca_url":"https://www.isca-archive.org/interspeech_2026/meng26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/meng26e_interspeech.pdf","session":"Assistive Technologies 2","topics":["paralinguistics","emotion-recognition","evaluation"],"category":"health-clinical","institutions":["National Acoustic Laboratories","GN Store Nord"],"funding":["GN-NAL Research Alliance Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"meng26e_interspeech","category":"health-clinical","institutions":["National Acoustic Laboratories","GN Store Nord"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2022","pdf":"https://www.isca-archive.org/interspeech_2026/meng26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26e_interspeech/markdown.md"},{"id":"meng26f_interspeech","title":"Similarity as Evidence: An Explainable Siamese Framework for Snore Sound Classification","authors":["Boyang Meng","Mengkai Sun","Haojie Zhang","Kun Qian","Wei Xue","Bin Hu","Bjoern Schuller"],"year":2026,"doi":"10.21437/Interspeech.2026-2153","isca_url":"https://www.isca-archive.org/interspeech_2026/meng26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/meng26f_interspeech.pdf","session":"Audio Coding and Signal Analysis","topics":["paralinguistics","self-supervised","evaluation"],"category":"health-clinical","labels":["self-supervised"],"institutions":["Beijing Institute of Technology","Hong Kong University of Science and Technology","TUM University Hospital"],"funding":["National Key R&D Program of China","National Natural Science Foundation of China","Beijing Natural Science Foundation","Ministry of Science and Technology of the People's Republic of China","Teli Young Fellow Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"meng26f_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["Beijing Institute of Technology","Hong Kong University of Science and Technology","TUM University Hospital"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2153","pdf":"https://www.isca-archive.org/interspeech_2026/meng26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/meng26f_interspeech/markdown.md"},{"id":"metzger26_interspeech","title":"Scaling Human and G2P Supervision for Robust Phonetic Transcription","authors":["Alexander Metzger","Aruna Srivastava","Ruslan Mukhamedvaleev"],"year":2026,"doi":"10.21437/Interspeech.2026-3271","isca_url":"https://www.isca-archive.org/interspeech_2026/metzger26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/metzger26_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["asr","phonetics","low-resource"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["Koel Labs"],"funding":["Mozilla","Google"],"code":{"url":"https://github.com/KoelLabs/ML","stars":26,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"metzger26_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["Koel Labs"],"code":"https://github.com/KoelLabs/ML","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3271","pdf":"https://www.isca-archive.org/interspeech_2026/metzger26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/metzger26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/metzger26_interspeech/markdown.md"},{"id":"mi26_interspeech","title":"Learning Emotion-discriminative Representations for Zero-Shot Cross-Lingual Speech Emotion Recognition","authors":["Jinyi Mi","Ding Ma","Tomoki Toda"],"year":2026,"doi":"10.21437/Interspeech.2026-1170","isca_url":"https://www.isca-archive.org/interspeech_2026/mi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mi26_interspeech.pdf","session":"Multilingual and Cross-Lingual Paralinguistic Analysis and Processing","topics":["speech-emotion-recognition","self-supervised","multilingual"],"category":"paralinguistics-emotion","labels":["low-resource","multilingual"],"institutions":["Nagoya University"],"funding":["JST CREST","JSPS KAKENHI","JST SPRING"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mi26_interspeech","category":"paralinguistics-emotion","labels":["low-resource","multilingual"],"institutions":["Nagoya University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1170","pdf":"https://www.isca-archive.org/interspeech_2026/mi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mi26_interspeech/markdown.md"},{"id":"minematsu26_interspeech","title":"AURORA: A Web-based Authoring System for Bridging Aural-Oral Language Training and Communicative Practice","authors":["Nobuaki Minematsu"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/minematsu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/minematsu26_interspeech.pdf","session":"Speech and Language Learning Technologies","topics":["spoken-language-understanding","dataset","evaluation"],"category":"applications-other","institutions":["University of Tokyo"],"code":{"url":"https://bit.ly/4nbE6Jf","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"minematsu26_interspeech","category":"applications-other","institutions":["University of Tokyo"],"code":"https://bit.ly/4nbE6Jf","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/minematsu26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/minematsu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/minematsu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/minematsu26_interspeech/markdown.md"},{"id":"miniconi26_interspeech","title":"TDScore: Learning Synthetic Speech Quality Predictors from TTS Training Dynamics without Human annotation","authors":["Natacha Miniconi","Meysam Shamsi","Anthony Larcher"],"year":2026,"doi":"10.21437/Interspeech.2026-449","isca_url":"https://www.isca-archive.org/interspeech_2026/miniconi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/miniconi26_interspeech.pdf","session":"Speech Synthesis Evaluation 2","topics":["tts","evaluation","self-supervised"],"category":"resources-evaluation","institutions":["Le Mans Universite"],"funding":["European Union"],"code":{"url":"https://git-lium.univ-lemans.fr/jsalt2025/wp1/tts4all_eval","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"miniconi26_interspeech","category":"resources-evaluation","institutions":["Le Mans Universite"],"code":"https://git-lium.univ-lemans.fr/jsalt2025/wp1/tts4all_eval","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-449","pdf":"https://www.isca-archive.org/interspeech_2026/miniconi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/miniconi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/miniconi26_interspeech/markdown.md"},{"id":"miodonska26_interspeech","title":"Age-dependent acoustic changes of the frication noise in dental sibilants produced by typically developing Polish children between 5 and 8 years of age","authors":["Zuzanna Miodońska","Oliwia Skórzewska","Natalia Mocko","Magdalena Krycha","Michal Krecichwost"],"year":2026,"doi":"10.21437/Interspeech.2026-799","isca_url":"https://www.isca-archive.org/interspeech_2026/miodonska26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/miodonska26_interspeech.pdf","session":"Gender- and Age-Related Speech Characteristics","topics":["phonetics","paralinguistics","evaluation"],"category":"phonetics-linguistics","institutions":["Silesian University of Technology","University of Silesia in Katowice"],"funding":["National Science Centre, Poland","European Funds for Silesia","Just Transition Fund","National Centre for Research and Development"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"miodonska26_interspeech","category":"phonetics-linguistics","institutions":["Silesian University of Technology","University of Silesia in Katowice"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-799","pdf":"https://www.isca-archive.org/interspeech_2026/miodonska26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/miodonska26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/miodonska26_interspeech/markdown.md"},{"id":"mitra26_interspeech","title":"Adaptive Turn-Taking for Real-time Multi-Party Voice Agents","authors":["Soumyajit Mitra","Prabhat Pandey","Abhinav Jain","Shanmukha Sahith","K V Vijay Girish"],"year":2026,"doi":"10.21437/Interspeech.2026-2493","isca_url":"https://www.isca-archive.org/interspeech_2026/mitra26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mitra26_interspeech.pdf","session":"Turn-taking","topics":["speech-llm","multilingual","self-supervised"],"category":"speech-llm-dialogue","labels":["streaming-real-time","generative-model"],"institutions":["Amazon","IIT Kharagpur"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mitra26_interspeech","category":"speech-llm-dialogue","labels":["streaming-real-time","generative-model"],"institutions":["Amazon","IIT Kharagpur"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2493","pdf":"https://www.isca-archive.org/interspeech_2026/mitra26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mitra26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mitra26_interspeech/markdown.md"},{"id":"miura26_interspeech","title":"Profiling Speech Rate Abilities of Visually Impaired Screen Reader Users by Bayesian Item Response Theory","authors":["Takahiro Miura","Masatsugu Sakajiri","Masaki Matsuo","Keiichi Yasu","Junji Onishi","Ken-ichiro Yabu","Tohru Ifukube"],"year":2026,"doi":"10.21437/Interspeech.2026-3096","isca_url":"https://www.isca-archive.org/interspeech_2026/miura26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/miura26_interspeech.pdf","session":"Brain Studies and Speech","topics":["tts","accessibility","evaluation"],"category":"resources-evaluation","institutions":["National Institute of Advanced Industrial Science and Technology","Tsukuba University of Technology","University of Tokyo"],"funding":["Japan Science and Technology Agency","Japan Society for the Promotion of Science"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"miura26_interspeech","category":"resources-evaluation","institutions":["National Institute of Advanced Industrial Science and Technology","Tsukuba University of Technology","University of Tokyo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3096","pdf":"https://www.isca-archive.org/interspeech_2026/miura26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/miura26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/miura26_interspeech/markdown.md"},{"id":"miyahara26_interspeech","title":"Evaluating Zero-Shot Cross-Lingual Stuttering Detection Based on Self-Attention Weights of Temporal Acoustic Vector Sequence","authors":["Genzo Miyahara","Tsuneo Kato","Akihiro Tamura"],"year":2026,"doi":"10.21437/Interspeech.2026-1886","isca_url":"https://www.isca-archive.org/interspeech_2026/miyahara26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/miyahara26_interspeech.pdf","session":"Speech and Language Technologies in Healthcare","topics":["paralinguistics","self-supervised","multilingual"],"category":"health-clinical","labels":["multilingual","self-supervised"],"institutions":["Doshisha University"],"funding":["JST SPRING"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"miyahara26_interspeech","category":"health-clinical","labels":["multilingual","self-supervised"],"institutions":["Doshisha University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1886","pdf":"https://www.isca-archive.org/interspeech_2026/miyahara26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/miyahara26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/miyahara26_interspeech/markdown.md"},{"id":"mizumoto26_interspeech","title":"Does Translation-Enhanced Speech Encoder Pre-training Affect Speech LLMs?","authors":["Tomoya Mizumoto","Yusuke Fujita"],"year":2026,"doi":"10.21437/Interspeech.2026-3241","isca_url":"https://www.isca-archive.org/interspeech_2026/mizumoto26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mizumoto26_interspeech.pdf","session":"Multilingual Speech 2","topics":["speech-llm","speech-translation","self-supervised"],"category":"speech-llm-dialogue","labels":["multilingual","self-supervised"],"institutions":["SB Intuitions"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mizumoto26_interspeech","category":"speech-llm-dialogue","labels":["multilingual","self-supervised"],"institutions":["SB Intuitions"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3241","pdf":"https://www.isca-archive.org/interspeech_2026/mizumoto26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mizumoto26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mizumoto26_interspeech/markdown.md"},{"id":"mohapatra26_interspeech","title":"Influence of Vocal Tract Curvature on Speech Acoustics: A Three-Dimensional FEM Analysis","authors":["Debasish Ray Mohapatra","Sidney Fels"],"year":2026,"doi":"10.21437/Interspeech.2026-2325","isca_url":"https://www.isca-archive.org/interspeech_2026/mohapatra26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mohapatra26_interspeech.pdf","session":"Methods and Data for Vocal Tract Shape and Articulation Analysis","topics":["phonetics","evaluation","speech-production"],"category":"phonetics-linguistics","institutions":["University of British Columbia"],"funding":["Natural Sciences and Engineering Research Council of Canada","UBC Graduate and Postdoctoral Studies"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mohapatra26_interspeech","category":"phonetics-linguistics","institutions":["University of British Columbia"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2325","pdf":"https://www.isca-archive.org/interspeech_2026/mohapatra26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mohapatra26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mohapatra26_interspeech/markdown.md"},{"id":"mojarad26_interspeech","title":"Layer-wise Probing of wav2vec 2.0 and Whisper for Consonant Cluster Reduction in African American English","authors":["Hamid Mojarad","Kevin Tang"],"year":2026,"doi":"10.21437/Interspeech.2026-808","isca_url":"https://www.isca-archive.org/interspeech_2026/mojarad26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mojarad26_interspeech.pdf","session":"Pronunciation Diversity","topics":["asr","self-supervised","evaluation"],"category":"phonetics-linguistics","labels":["self-supervised"],"institutions":["Heinrich Heine University Dusseldorf","University of Florida"],"code":{"url":"https://doi.org/10.17605/OSF.IO/FE2D7","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mojarad26_interspeech","category":"phonetics-linguistics","labels":["self-supervised"],"institutions":["Heinrich Heine University Dusseldorf","University of Florida"],"code":"https://doi.org/10.17605/OSF.IO/FE2D7","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-808","pdf":"https://www.isca-archive.org/interspeech_2026/mojarad26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mojarad26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mojarad26_interspeech/markdown.md"},{"id":"mokgosi26_interspeech","title":"Tone-Conditioned Curriculum Learning for Low-Resource Bantu Speech Recognition","authors":["Kesego Mokgosi","Vukosi Marivate","Sitwala Mundia","Unarine Netshifhefhe","Tsholofelo Mogale","Thapelo Sindane"],"year":2026,"doi":"10.21437/Interspeech.2026-2905","isca_url":"https://www.isca-archive.org/interspeech_2026/mokgosi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mokgosi26_interspeech.pdf","session":"Indigenous Voices in Speech Science and Technology","topics":["asr","low-resource","multilingual"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Technological University Dublin","University of Pretoria","Lelapa AI"],"funding":["Gates Foundation","Meta","International Development Research Centre","Foreign, Commonwealth & Development Office","AI4D Africa Program","Research Ireland","ADAPT Research Ireland Centre for AI-Driven Digital Content Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mokgosi26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Technological University Dublin","University of Pretoria","Lelapa AI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2905","pdf":"https://www.isca-archive.org/interspeech_2026/mokgosi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mokgosi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mokgosi26_interspeech/markdown.md"},{"id":"mokshagundam26_interspeech","title":"Boundaryless Speech-to-Syllable Representations with Hierarchical CNN for Linguistically Inspired Automatic Stress Detection","authors":["Namrata Mokshagundam","Sai Harshitha Aluru","Chiranjeevi Yarra"],"year":2026,"doi":"10.21437/Interspeech.2026-3144","isca_url":"https://www.isca-archive.org/interspeech_2026/mokshagundam26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mokshagundam26_interspeech.pdf","session":"Audio signal analysis","topics":["paralinguistics","speech-recognition","evaluation"],"category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["International Institute of Information Technology Hyderabad"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mokshagundam26_interspeech","category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["International Institute of Information Technology Hyderabad"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3144","pdf":"https://www.isca-archive.org/interspeech_2026/mokshagundam26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mokshagundam26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mokshagundam26_interspeech/markdown.md"},{"id":"mondal26_interspeech","title":"Probing LoRA-to-LoRA Cross-Lingual Transfer for Unseen Low-Resource Conditions in Whisper-Based ASR","authors":["Hirak Mondal","Spandan Dey","Gopal Agrawal"],"year":2026,"doi":"10.21437/Interspeech.2026-1133","isca_url":"https://www.isca-archive.org/interspeech_2026/mondal26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mondal26_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["asr","low-resource","multilingual"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Samsung"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mondal26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1133","pdf":"https://www.isca-archive.org/interspeech_2026/mondal26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mondal26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mondal26_interspeech/markdown.md"},{"id":"mondal26b_interspeech","title":"Spontaneous Dialect-Aware Speech Corpus for Low-Resource Dakhini, A Southern Indo-Aryan Language: Methods, Challenges, and Insights","authors":["Anindita Mondal","Priyanka Kommagouni"],"year":2026,"doi":"10.21437/Interspeech.2026-2640","isca_url":"https://www.isca-archive.org/interspeech_2026/mondal26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mondal26b_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["asr","low-resource","dataset"],"category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["International Institute of Information Technology Hyderabad"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mondal26b_interspeech","category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["International Institute of Information Technology Hyderabad"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2640","pdf":"https://www.isca-archive.org/interspeech_2026/mondal26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mondal26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mondal26b_interspeech/markdown.md"},{"id":"monir26_interspeech","title":"Time–Frequency Weighted Losses for Phoneme Reconstruction in DNN-Based Speech Enhancement","authors":["Nasser-Eddine Monir","Paul Magron","Romain Serizel"],"year":2026,"doi":"10.21437/Interspeech.2026-3416","isca_url":"https://www.isca-archive.org/interspeech_2026/monir26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/monir26_interspeech.pdf","session":"SE Architectures, Adaptation and Audio Front-Ends","topics":["speech-enhancement","phonetics","evaluation"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Universite de Lorraine","CNRS","Inria","LORIA"],"funding":["French National Research Agency","REFINED project"],"code":{"url":"https://github.com/Nasseredd/fw-se-loss","stars":12,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"monir26_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Universite de Lorraine","CNRS","Inria","LORIA"],"code":"https://github.com/Nasseredd/fw-se-loss","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3416","pdf":"https://www.isca-archive.org/interspeech_2026/monir26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/monir26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/monir26_interspeech/markdown.md"},{"id":"moon26_interspeech","title":"SLICE: Speech Enhancement via Layer-wise Injection of Conditioning Embeddings","authors":["Seokhoon Moon","Kyudan Jung","Jaegul Choo"],"year":2026,"doi":"10.21437/Interspeech.2026-1715","isca_url":"https://www.isca-archive.org/interspeech_2026/moon26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/moon26_interspeech.pdf","session":"SE Architectures, Adaptation and Audio Front-Ends","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["KAIST"],"funding":["Institute for Information & Communications Technology Planning & Evaluation","Korea government (MSIT)","National Research Foundation of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"moon26_interspeech","category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["KAIST"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1715","pdf":"https://www.isca-archive.org/interspeech_2026/moon26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/moon26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/moon26_interspeech/markdown.md"},{"id":"moon26b_interspeech","title":"When Does Quality-Aware Multimodal Fusion Matter? A Leakage-Safe Diagnostic for Decision-Level Dependence","authors":["Jaden Moon","Arvind Pillai","Andrew Campbell"],"year":2026,"doi":"10.21437/Interspeech.2026-2989","isca_url":"https://www.isca-archive.org/interspeech_2026/moon26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/moon26b_interspeech.pdf","session":"Audio Language Models: Reasoning, Reliability, and Multimodal Understanding","topics":["paralinguistics","emotion-recognition","evaluation"],"category":"paralinguistics-emotion","institutions":["Dartmouth College"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"moon26b_interspeech","category":"paralinguistics-emotion","institutions":["Dartmouth College"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2989","pdf":"https://www.isca-archive.org/interspeech_2026/moon26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/moon26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/moon26b_interspeech/markdown.md"},{"id":"mori26_interspeech","title":"Evaluating Automatic Laughter Phone Annotation for Socially-Situated Laughter Synthesis","authors":["Hiroki Mori","Hiroto Ueda"],"year":2026,"doi":"10.21437/Interspeech.2026-2141","isca_url":"https://www.isca-archive.org/interspeech_2026/mori26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mori26_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["speech-enhancement","speech-llm","paralinguistics"],"category":"tts","labels":["generative-model"],"institutions":["Utsunomiya University"],"code":{"url":"https://www.speech-lab.org/hiroki/IS2026/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mori26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Utsunomiya University"],"code":"https://www.speech-lab.org/hiroki/IS2026/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2141","pdf":"https://www.isca-archive.org/interspeech_2026/mori26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mori26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mori26_interspeech/markdown.md"},{"id":"mosner26_interspeech","title":"Effectiveness of Language Variability Compensation in Speaker Verification","authors":["Ladislav Mošner","Sara Barahona","Sandro Cumani","Johan Rohdin","Jin Li","Oldřich Plchot"],"year":2026,"doi":"10.21437/Interspeech.2026-3367","isca_url":"https://www.isca-archive.org/interspeech_2026/mosner26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mosner26_interspeech.pdf","session":"Challenge - TidyVoice Challenge: Cross-Lingual Speaker Verification","topics":["speaker-verification","multilingual","self-supervised"],"category":"speaker","labels":["multilingual"],"institutions":["Brno University of Technology","Universidad Autonoma de Madrid","Politecnico di Torino","Hong Kong Polytechnic University"],"funding":["Ministry of Education, Youth and Sports of the Czech Republic","Digital Europe Programme","MICIU/AEI","Comunidad de Madrid"],"code":{"url":"https://github.com/wenet-e2e/wespeaker/blob/master/docs/pretrained.md","stars":1426,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mosner26_interspeech","category":"speaker","labels":["multilingual"],"institutions":["Brno University of Technology","Universidad Autonoma de Madrid","Politecnico di Torino","Hong Kong Polytechnic University"],"code":"https://github.com/wenet-e2e/wespeaker/blob/master/docs/pretrained.md","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3367","pdf":"https://www.isca-archive.org/interspeech_2026/mosner26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mosner26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mosner26_interspeech/markdown.md"},{"id":"mou26_interspeech","title":"DuraMark: Duration-Embedded Watermarking in LLM-based TTS","authors":["Zhenwei Mou","Weili Jiang","Liping Chen","Zhen-Hua Ling","Kong Aik Lee","Kai Gao","Boyu Zhao"],"year":2026,"doi":"10.21437/Interspeech.2026-2298","isca_url":"https://www.isca-archive.org/interspeech_2026/mou26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mou26_interspeech.pdf","session":"Audio Watermarking and Source Verification","topics":["tts","self-supervised","speech-llm"],"category":"deepfake-security","labels":["generative-model"],"institutions":["University of Science and Technology of China","Institute of Forensic Science, Ministry of Public Security","Hong Kong Polytechnic University"],"code":{"url":"https://muzw.github.io/duramark_demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mou26_interspeech","category":"deepfake-security","labels":["generative-model"],"institutions":["University of Science and Technology of China","Institute of Forensic Science, Ministry of Public Security","Hong Kong Polytechnic University"],"code":"https://muzw.github.io/duramark_demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2298","pdf":"https://www.isca-archive.org/interspeech_2026/mou26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mou26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mou26_interspeech/markdown.md"},{"id":"mou26b_interspeech","title":"Dynamic Prosody Prediction in LLM-based TTS for Improving Speaker Similarity","authors":["Zhenwei Mou","Liping Chen","Yajun Hu","Zhen-Hua Ling","Xin Fang","Jian-Qing Gao"],"year":2026,"doi":"10.21437/Interspeech.2026-2312","isca_url":"https://www.isca-archive.org/interspeech_2026/mou26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mou26b_interspeech.pdf","session":"Long-Form Speech Synthesis","topics":["tts","speech-llm","prosody"],"category":"tts","labels":["generative-model"],"institutions":["University of Science and Technology of China","iFLYTEK"],"code":{"url":"https://muzw.github.io/dynapros/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mou26b_interspeech","category":"tts","labels":["generative-model"],"institutions":["University of Science and Technology of China","iFLYTEK"],"code":"https://muzw.github.io/dynapros/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2312","pdf":"https://www.isca-archive.org/interspeech_2026/mou26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mou26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mou26b_interspeech/markdown.md"},{"id":"moumen26_interspeech","title":"Measuring the Redundancy of Decoder Layers in SpeechLLMs","authors":["Adel Moumen","Guangzhi Sun","Philip C Woodland"],"year":2026,"doi":"10.21437/Interspeech.2026-1873","isca_url":"https://www.isca-archive.org/interspeech_2026/moumen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/moumen26_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 2","topics":["speech-llm","asr","speech-translation"],"category":"asr","labels":["efficient-on-device"],"institutions":["University of Cambridge"],"code":{"url":"https://github.com/speechbrain/speechbrain","stars":11845,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"moumen26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["University of Cambridge"],"code":"https://github.com/speechbrain/speechbrain","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1873","pdf":"https://www.isca-archive.org/interspeech_2026/moumen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/moumen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/moumen26_interspeech/markdown.md"},{"id":"mousavi26_interspeech","title":"Investigating Faithfulness in Large Audio Language Models","authors":["Pooneh Mousavi","Lovenya Jain","Mirco Ravanelli","Cem Subakan"],"year":2026,"doi":"10.21437/Interspeech.2026-1533","isca_url":"https://www.isca-archive.org/interspeech_2026/mousavi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mousavi26_interspeech.pdf","session":"Audio Language Models","topics":["speech-llm","evaluation","self-supervised"],"category":"speech-llm-dialogue","institutions":["Concordia University","Mila - Quebec AI Institute","Universite Laval","Birla Institute of Technology and Science, Pilani"],"funding":["Natural Sciences and Engineering Research Council of Canada","Digital Research Alliance of Canada","Translated Imminent Program","Apple"],"code":{"url":"https://poonehmousavi.github.io/faithfulness/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mousavi26_interspeech","category":"speech-llm-dialogue","institutions":["Concordia University","Mila - Quebec AI Institute","Universite Laval","Birla Institute of Technology and Science, Pilani"],"code":"https://poonehmousavi.github.io/faithfulness/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1533","pdf":"https://www.isca-archive.org/interspeech_2026/mousavi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mousavi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mousavi26_interspeech/markdown.md"},{"id":"mousi26_interspeech","title":"Said Aloud, Read Different: Cross-Modal Instability in Multimodal Models","authors":["Basel Mousi","Fahim Dalvi","Shammur Absar Chowdhury","Firoj Alam","Nadir Durrani"],"year":2026,"doi":"10.21437/Interspeech.2026-1980","isca_url":"https://www.isca-archive.org/interspeech_2026/mousi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mousi26_interspeech.pdf","session":"Audio Language Models: Reasoning, Reliability, and Multimodal Understanding","topics":["speech-llm","multilingual","evaluation"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Qatar Computing Research Institute","Hamad Bin Khalifa University"],"code":{"url":"https://huggingface.co/datasets/QCRI/M2CQA-S","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mousi26_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Qatar Computing Research Institute","Hamad Bin Khalifa University"],"code":"https://huggingface.co/datasets/QCRI/M2CQA-S","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1980","pdf":"https://www.isca-archive.org/interspeech_2026/mousi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mousi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mousi26_interspeech/markdown.md"},{"id":"mukhituly26_interspeech","title":"A Unified Safety Subspace Exists in Speech Language Models","authors":["Nurdaulet Mukhituly","Muhammad Cendekia Airlangga","Rifo Ahmad Genadi","Nhi Hoai Doan","Amirbek Djanibekov","Samuel Munachiso Nwadike","Zangir Iklassov","Kentaro Inui"],"year":2026,"doi":"10.21437/Interspeech.2026-2797","isca_url":"https://www.isca-archive.org/interspeech_2026/mukhituly26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mukhituly26_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["MBZUAI","Tohoku University","RIKEN"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mukhituly26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["MBZUAI","Tohoku University","RIKEN"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2797","pdf":"https://www.isca-archive.org/interspeech_2026/mukhituly26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mukhituly26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mukhituly26_interspeech/markdown.md"},{"id":"munyampirwa26_interspeech","title":"Contextual Earnings-22: A Speech Recognition Benchmark with Custom Vocabulary in the Wild","authors":["Blaise Munyampirwa","Arda Ibis","Zach Nagengast","Brian Keene","Dylan Angus"],"year":2026,"doi":"10.21437/Interspeech.2026-1375","isca_url":"https://www.isca-archive.org/interspeech_2026/munyampirwa26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/munyampirwa26_interspeech.pdf","session":"Datasets","topics":["asr","evaluation","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Argmax","University of California, Los Angeles"],"code":{"url":"https://github.com/argmaxinc/OpenBench","stars":89,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"munyampirwa26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Argmax","University of California, Los Angeles"],"code":"https://github.com/argmaxinc/OpenBench","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1375","pdf":"https://www.isca-archive.org/interspeech_2026/munyampirwa26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/munyampirwa26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/munyampirwa26_interspeech/markdown.md"},{"id":"murata26_interspeech","title":"Exploring Pre-training Benefits on Phoneme Addition through Fine-tuning in Speech Synthesis","authors":["Masato Murata","Koichi Miyazaki","Tomoki Koriyama","Tomoki Toda"],"year":2026,"doi":"10.21437/Interspeech.2026-208","isca_url":"https://www.isca-archive.org/interspeech_2026/murata26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/murata26_interspeech.pdf","session":"Articulatory, EMG, and Visual Speech Generation","topics":["tts","multilingual","low-resource"],"category":"tts","labels":["generative-model"],"institutions":["CyberAgent","Nagoya University"],"funding":["JSPS KAKENHI","BRIDGE Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"murata26_interspeech","category":"tts","labels":["generative-model"],"institutions":["CyberAgent","Nagoya University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-208","pdf":"https://www.isca-archive.org/interspeech_2026/murata26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/murata26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/murata26_interspeech/markdown.md"},{"id":"murugaiyan26_interspeech","title":"WhiSSDapt: Adaptive Fusion of Whisper Layer Embeddings for Sentence Stress Detection","authors":["Someshwaran Murugaiyan","Jhansi Mallela","Chiranjeevi Yarra"],"year":2026,"doi":"10.21437/Interspeech.2026-3236","isca_url":"https://www.isca-archive.org/interspeech_2026/murugaiyan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/murugaiyan26_interspeech.pdf","session":"Prosody, Pronunciation and Specialized Speech Processing","topics":["paralinguistics","prosody","self-supervised"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Vellore Institute of Technology","International Institute of Information Technology Hyderabad"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"murugaiyan26_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Vellore Institute of Technology","International Institute of Information Technology Hyderabad"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3236","pdf":"https://www.isca-archive.org/interspeech_2026/murugaiyan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/murugaiyan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/murugaiyan26_interspeech/markdown.md"},{"id":"muthu26_interspeech","title":"DysfluentNet: Joint Stuttering Event Detection and Dysfluency-Aware Transcription via Hierarchical Self-Supervised Learning","authors":["Mohankumar Muthu","Sasikala E","Girirajan S"],"year":2026,"doi":"10.21437/Interspeech.2026-696","isca_url":"https://www.isca-archive.org/interspeech_2026/muthu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/muthu26_interspeech.pdf","session":"Self-supervised Speech Representation Learning","topics":["asr","stuttering-detection","self-supervised"],"category":"asr","labels":["self-supervised"],"institutions":["SRM Institute of Science and Technology"],"funding":["Department of Science and Technology"],"code":{"url":"https://github.com/mm0718-srmist/dysfluentnet","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"muthu26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["SRM Institute of Science and Technology"],"code":"https://github.com/mm0718-srmist/dysfluentnet","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-696","pdf":"https://www.isca-archive.org/interspeech_2026/muthu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/muthu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/muthu26_interspeech/markdown.md"},{"id":"mylvaganam26_interspeech","title":"Hybrid Continual Learning for Low-Resource Australian Aboriginal Language Identification","authors":["Pravina Mylvaganam","Ting Dang","Eliathamby Ambikairajah","Vidhyasaharan Sethu","Jingyao Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-1789","isca_url":"https://www.isca-archive.org/interspeech_2026/mylvaganam26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mylvaganam26_interspeech.pdf","session":"Speech and Language Representation","topics":["asr","low-resource","multilingual"],"category":"speaker","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of New South Wales","University of Melbourne","Massachusetts Institute of Technology"],"funding":["School of Electrical Engineering and Telecommunications at UNSW Sydney"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mylvaganam26_interspeech","category":"speaker","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of New South Wales","University of Melbourne","Massachusetts Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1789","pdf":"https://www.isca-archive.org/interspeech_2026/mylvaganam26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mylvaganam26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mylvaganam26_interspeech/markdown.md"},{"id":"mylvaganam26b_interspeech","title":"Which Languages Transfer Best to Warlpiri? A Similarity-Based Study for Low-Resource ASR","authors":["Pravina Mylvaganam","Eliathamby Ambikairajah","Ting Dang","Vidhyasaharan Sethu","Tünde Szalay"],"year":2026,"doi":"10.21437/Interspeech.2026-1837","isca_url":"https://www.isca-archive.org/interspeech_2026/mylvaganam26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/mylvaganam26b_interspeech.pdf","session":"Indigenous Voices in Speech Science and Technology","topics":["asr","low-resource","multilingual"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of New South Wales","University of Melbourne","University of Sydney"],"funding":["University of New South Wales"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"mylvaganam26b_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of New South Wales","University of Melbourne","University of Sydney"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1837","pdf":"https://www.isca-archive.org/interspeech_2026/mylvaganam26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/mylvaganam26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/mylvaganam26b_interspeech/markdown.md"},{"id":"myrgyyassov26_interspeech","title":"Automated Measurement of Geniohyoid Muscle Thickness During Speech Using Deep Learning and Ultrasound","authors":["Alisher Myrgyyassov","Bruce Xiao Wang","Yu Sun","Shuming Huang","Zhen Song","Min Ney Wong","Yongping Zheng"],"year":2026,"doi":"10.21437/Interspeech.2026-1664","isca_url":"https://www.isca-archive.org/interspeech_2026/myrgyyassov26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/myrgyyassov26_interspeech.pdf","session":"Speech Production and Perception 2","topics":["phonetics","evaluation","health"],"category":"phonetics-linguistics","institutions":["Hong Kong Polytechnic University"],"funding":["Young Innovative Researcher Award Scheme","Research Grants Council of Hong Kong"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"myrgyyassov26_interspeech","category":"phonetics-linguistics","institutions":["Hong Kong Polytechnic University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1664","pdf":"https://www.isca-archive.org/interspeech_2026/myrgyyassov26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/myrgyyassov26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/myrgyyassov26_interspeech/markdown.md"},{"id":"naini26_interspeech","title":"Comparative Reasoning: Making an Audio Language Model Better at Comparing Emotions","authors":["Abinay Reddy Naini","Jaeyeon Kim","Chao-Han Huck Yang","Shinji Watanabe","Carlos Busso"],"year":2026,"doi":"10.21437/Interspeech.2026-2935","isca_url":"https://www.isca-archive.org/interspeech_2026/naini26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/naini26_interspeech.pdf","session":"Multimodal Emotion Recognition","topics":["speech-emotion-recognition","speech-llm","self-supervised"],"category":"paralinguistics-emotion","labels":["low-resource","self-supervised"],"institutions":["Carnegie Mellon University","University of Texas at Dallas","NVIDIA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"naini26_interspeech","category":"paralinguistics-emotion","labels":["low-resource","self-supervised"],"institutions":["Carnegie Mellon University","University of Texas at Dallas","NVIDIA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2935","pdf":"https://www.isca-archive.org/interspeech_2026/naini26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/naini26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/naini26_interspeech/markdown.md"},{"id":"nakagome26_interspeech","title":"MixProLAP: Mixture-Induced Uncertainty Modeling for Probabilistic Language-Audio Pretraining","authors":["Yu Nakagome","Jaesong Lee","Soo-Whan Chung"],"year":2026,"doi":"10.21437/Interspeech.2026-360","isca_url":"https://www.isca-archive.org/interspeech_2026/nakagome26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nakagome26_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["self-supervised","multilingual","evaluation"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["LINE WORKS Corporation","NAVER Cloud Corporation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nakagome26_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["LINE WORKS Corporation","NAVER Cloud Corporation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-360","pdf":"https://www.isca-archive.org/interspeech_2026/nakagome26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nakagome26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nakagome26_interspeech/markdown.md"},{"id":"nan26_interspeech","title":"SpeechBench: A Unified Speech Annotation and Analysis Tool","authors":["Zheng Nan","Tharmakulasingam Sirojan","Mostafa Shahin","Tünde Szalay","Vidhyasaharan Sethu","Beena Ahmed"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/nan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nan26_interspeech.pdf","session":"Speech Analysis, Data Resources and Research Tools","topics":["dataset","asr","evaluation"],"category":"resources-evaluation","institutions":["UNSW","University of Sydney"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nan26_interspeech","category":"resources-evaluation","institutions":["UNSW","University of Sydney"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/nan26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/nan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nan26_interspeech/markdown.md"},{"id":"nanni26_interspeech","title":"Rethinking Consent Acquisition for Voice Synthesis: from Static to Dynamic Consent","authors":["Matilde Nanni"],"year":2026,"doi":"10.21437/Interspeech.2026-2379","isca_url":"https://www.isca-archive.org/interspeech_2026/nanni26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nanni26_interspeech.pdf","session":"Safeguards for Synthetic Speech: Ethical, Technical, and Legal Perspectives","topics":["evaluation"],"category":"deepfake-security","institutions":["University of Inland Norway"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nanni26_interspeech","category":"deepfake-security","institutions":["University of Inland Norway"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2379","pdf":"https://www.isca-archive.org/interspeech_2026/nanni26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nanni26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nanni26_interspeech/markdown.md"},{"id":"narasinghe26_interspeech","title":"Causal Redundancy in Speech Representations: The Hydra Effect and Limits of Sparse Disentanglement in WavLM","authors":["Patalee Narasinghe","Kasindu Bandara","Vilash Nawagamuwa","Janak Senevirathne","Uthayasanker Thayasivam"],"year":2026,"doi":"10.21437/Interspeech.2026-3316","isca_url":"https://www.isca-archive.org/interspeech_2026/narasinghe26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/narasinghe26_interspeech.pdf","session":"Self-supervised Speech Representation Learning","topics":["self-supervised","evaluation","phonetics"],"category":"applications-other","labels":["self-supervised"],"institutions":["University of Moratuwa"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"narasinghe26_interspeech","category":"applications-other","labels":["self-supervised"],"institutions":["University of Moratuwa"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3316","pdf":"https://www.isca-archive.org/interspeech_2026/narasinghe26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/narasinghe26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/narasinghe26_interspeech/markdown.md"},{"id":"nasrallah26_interspeech","title":"DECRA: Dynamic Emotion Control for Real-time Speech Anonymization","authors":["Ghady Nasrallah","Waris Quamer","Mu-Ruei Tseng","Ricardo Gutierrez-Osuna"],"year":2026,"doi":"10.21437/Interspeech.2026-2927","isca_url":"https://www.isca-archive.org/interspeech_2026/nasrallah26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nasrallah26_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["speech-anonymization","self-supervised","paralinguistics"],"category":"deepfake-security","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Texas A&M University"],"funding":["Intelligence Advanced Research Projects Activity","Department of Interior/Interior Business Center"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nasrallah26_interspeech","category":"deepfake-security","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Texas A&M University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2927","pdf":"https://www.isca-archive.org/interspeech_2026/nasrallah26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nasrallah26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nasrallah26_interspeech/markdown.md"},{"id":"nath26_interspeech","title":"An Acoustic Investigation of Mid Front Vowel Harmony in Assamese","authors":["Saurabh Nath","Rosey Billington","Danielle Barth"],"year":2026,"doi":"10.21437/Interspeech.2026-3034","isca_url":"https://www.isca-archive.org/interspeech_2026/nath26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nath26_interspeech.pdf","session":"Speaker-Specific, Forensic and Segmental Characteristics of Typical and Atypical Speech","topics":["phonetics","low-resource","multilingual"],"category":"phonetics-linguistics","institutions":["Australian National University"],"funding":["Australian Linguistic Society Research Grant","Bhati Family India Travel Grant","South Asian Research Institute Student Support Grant"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nath26_interspeech","category":"phonetics-linguistics","institutions":["Australian National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3034","pdf":"https://www.isca-archive.org/interspeech_2026/nath26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nath26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nath26_interspeech/markdown.md"},{"id":"naveriani26_interspeech","title":"Diffusion Language Models for Speech Recognition","authors":["Davyd Naveriani","Albert Zeyer","Ralf Schlüter","Hermann Ney"],"year":2026,"doi":"10.21437/Interspeech.2026-2070","isca_url":"https://www.isca-archive.org/interspeech_2026/naveriani26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/naveriani26_interspeech.pdf","session":"New Architecture and Analyses for ASR and Speech LMs","topics":["asr","self-supervised"],"category":"asr","labels":["generative-model"],"institutions":["RWTH Aachen University","AppTek"],"funding":["NeuroSys","Federal Ministry of Research, Technology and Space BMFTR","RESCALE","Federal Ministry for the Environment, Nature Conservation, Nuclear Safety and Consumer Protection","Federal Ministry of Education and Research"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"naveriani26_interspeech","category":"asr","labels":["generative-model"],"institutions":["RWTH Aachen University","AppTek"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2070","pdf":"https://www.isca-archive.org/interspeech_2026/naveriani26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/naveriani26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/naveriani26_interspeech/markdown.md"},{"id":"nema26_interspeech","title":"Post-ASR Proper Noun Grounding via Multi-View Phonetic and Semantic Retrieval","authors":["Pranshu Nema","Kush Shrivastava","Bhavik Vachhani","Rustom Lawyer"],"year":2026,"doi":"10.21437/Interspeech.2026-907","isca_url":"https://www.isca-archive.org/interspeech_2026/nema26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nema26_interspeech.pdf","session":"Robust ASR: Hallucinations and Biases","topics":["asr","evaluation","dataset"],"category":"asr","institutions":["Augnito"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nema26_interspeech","category":"asr","institutions":["Augnito"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-907","pdf":"https://www.isca-archive.org/interspeech_2026/nema26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nema26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nema26_interspeech/markdown.md"},{"id":"neumann26_interspeech","title":"Reducing Measurement Noise in Digital Speech Biomarkers: Interpretable Composite Index Scores for Longitudinal ALS Monitoring in Clinical Trials","authors":["Michael Neumann","Hardik Kothare","Diego Cadavid","Robert H. Scannevin","Anil Tarachandani","Ines Hoffmann","Tara Haley","Shane Raines","Vikram Ramanarayanan"],"year":2026,"doi":"10.21437/Interspeech.2026-2841","isca_url":"https://www.isca-archive.org/interspeech_2026/neumann26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/neumann26_interspeech.pdf","session":"Clinically Useful Speech Representations 2","topics":["paralinguistics","evaluation","health"],"category":"health-clinical","institutions":["Modality.AI","Verge Genomics","2b Analytics","University of California, San Francisco"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"neumann26_interspeech","category":"health-clinical","institutions":["Modality.AI","Verge Genomics","2b Analytics","University of California, San Francisco"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2841","pdf":"https://www.isca-archive.org/interspeech_2026/neumann26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/neumann26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/neumann26_interspeech/markdown.md"},{"id":"ngong26_interspeech","title":"DP-VOXLET: Provable Speaker Anonymization for Disentangled Speech Representations","authors":["Ivoline Ngong","Jack D'Iorio","Hailey Schoppe","Christopher Liberatore","Nichole Schimanski","Taisa Kushner","Joseph P. Near"],"year":2026,"doi":"10.21437/Interspeech.2026-2910","isca_url":"https://www.isca-archive.org/interspeech_2026/ngong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ngong26_interspeech.pdf","session":"Speaker Privacy and Anonymization","topics":["speaker-verification","voice-conversion","evaluation"],"category":"deepfake-security","institutions":["University of Vermont","Galois"],"funding":["Intelligence Advanced Research Projects Activity","Department of Interior/Interior Business Center"],"code":{"url":"https://github.com/uvm-plaid/dpvc","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ngong26_interspeech","category":"deepfake-security","institutions":["University of Vermont","Galois"],"code":"https://github.com/uvm-plaid/dpvc","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2910","pdf":"https://www.isca-archive.org/interspeech_2026/ngong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ngong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ngong26_interspeech/markdown.md"},{"id":"nguyen26_interspeech","title":"Direct Preference Optimization for English-Mandarin Code-Switching Speech Recognition in Audio LLMs","authors":["Trung Nguyen","Cheng Yi Lewis Won","Minh Duc Pham","Yingxu He","Shuo Sun","Ai Ti Aw"],"year":2026,"doi":"10.21437/Interspeech.2026-110","isca_url":"https://www.isca-archive.org/interspeech_2026/nguyen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nguyen26_interspeech.pdf","session":"Code-Switching ASR","topics":["asr","speech-llm","multilingual"],"category":"asr","labels":["multilingual"],"institutions":["Agency for Science, Technology and Research","Nanyang Technological University"],"funding":["National Research Foundation, Singapore"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nguyen26_interspeech","category":"asr","labels":["multilingual"],"institutions":["Agency for Science, Technology and Research","Nanyang Technological University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-110","pdf":"https://www.isca-archive.org/interspeech_2026/nguyen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26_interspeech/markdown.md"},{"id":"nguyen26b_interspeech","title":"Revisiting Active Speaker Detection: An In-the-Wild Benchmark for Generalization and Robustness","authors":["Le Thien Phuc Nguyen","Zhuoran Yu","Khoa Quang Nhat Cao","Yuwei Guo","Tu Ho Manh Pham","Tuan Tai Nguyen","Toan Ngo Duc Vo","Lucas Poon","Tuan Khai Nguyen","Soochahn Lee","Yong Jae Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-581","isca_url":"https://www.isca-archive.org/interspeech_2026/nguyen26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nguyen26b_interspeech.pdf","session":"LLMs and Conversational Interaction","topics":["speaker-diarization","evaluation","dataset"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["University of Wisconsin - Madison","Oregon State University","University of Sydney","Kookmin University"],"funding":["National Science Foundation","IBM","Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT"],"code":{"url":"https://github.com/plnguyen2908/UniTalk-ASD-code","stars":24,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nguyen26b_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["University of Wisconsin - Madison","Oregon State University","University of Sydney","Kookmin University"],"code":"https://github.com/plnguyen2908/UniTalk-ASD-code","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-581","pdf":"https://www.isca-archive.org/interspeech_2026/nguyen26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26b_interspeech/markdown.md"},{"id":"nguyen26c_interspeech","title":"MamTra: A Hybrid Mamba-Transformer Backbone for Speech Synthesis","authors":["Tan Dat Nguyen","Sangmin Bae","Joon Son Chung","Ji-Hoon Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-1031","isca_url":"https://www.isca-archive.org/interspeech_2026/nguyen26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nguyen26c_interspeech.pdf","session":"LLM Based Speech Synthesis","topics":["tts","speech-llm","self-supervised"],"category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Korea Advanced Institute of Science and Technology","Chung-Ang University"],"funding":["Ministry of Science and ICT, Korea"],"code":{"url":"https://mm.kaist.ac.kr/projects/mamtra/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nguyen26c_interspeech","category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Korea Advanced Institute of Science and Technology","Chung-Ang University"],"code":"https://mm.kaist.ac.kr/projects/mamtra/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1031","pdf":"https://www.isca-archive.org/interspeech_2026/nguyen26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26c_interspeech/markdown.md"},{"id":"nguyen26d_interspeech","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","authors":["Son Nguyen","Thanh Tran","Nghia Huynh","Son Hy","Van Nguyen"],"year":2026,"doi":"10.21437/Interspeech.2026-1043","isca_url":"https://www.isca-archive.org/interspeech_2026/nguyen26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nguyen26d_interspeech.pdf","session":"Text-to-Speech Synthesis","topics":["tts","self-supervised","low-resource"],"category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["FPT Software","University of Alabama at Birmingham"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nguyen26d_interspeech","category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["FPT Software","University of Alabama at Birmingham"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1043","pdf":"https://www.isca-archive.org/interspeech_2026/nguyen26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26d_interspeech/markdown.md"},{"id":"nguyen26e_interspeech","title":"Fair Cognitive Impairment Detection Through Unlearning","authors":["William Nguyen","Jiali Cheng","Hadi Amiri"],"year":2026,"doi":"10.21437/Interspeech.2026-1353","isca_url":"https://www.isca-archive.org/interspeech_2026/nguyen26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nguyen26e_interspeech.pdf","session":"Acoustic Event Detection 3","topics":["speech-llm","health","multilingual"],"category":"health-clinical","labels":["multilingual"],"institutions":["University of Massachusetts Lowell"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nguyen26e_interspeech","category":"health-clinical","labels":["multilingual"],"institutions":["University of Massachusetts Lowell"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1353","pdf":"https://www.isca-archive.org/interspeech_2026/nguyen26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26e_interspeech/markdown.md"},{"id":"nguyen26f_interspeech","title":"PiDA: Phonetically-Informed Data Augmentation for Robust Vietnamese Speech Translation","authors":["Giang Son Nguyen","Tung X. Nguyen","Hieu Minh Truong","Nhu Vo","Wray Buntine","Dung D. Le"],"year":2026,"doi":"10.21437/Interspeech.2026-1963","isca_url":"https://www.isca-archive.org/interspeech_2026/nguyen26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nguyen26f_interspeech.pdf","session":"Multilingual Speech 2","topics":["speech-translation","low-resource","multilingual"],"category":"translation","labels":["low-resource","multilingual"],"institutions":["VinUniversity","University of Technology Sydney","Monash University"],"funding":["VinUniversity","Vingroup Scholarship"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nguyen26f_interspeech","category":"translation","labels":["low-resource","multilingual"],"institutions":["VinUniversity","University of Technology Sydney","Monash University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1963","pdf":"https://www.isca-archive.org/interspeech_2026/nguyen26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26f_interspeech/markdown.md"},{"id":"nguyen26g_interspeech","title":"Domain-Aware Mispronunciation Detection and Diagnosis Using Language-Specific Statistical Graphs","authors":["Hanh Nguyen","Tuong Tu Huu","Huan Vu","Thien Van Luong","Tien Cuong Nguyen","Trang Thu Thi Nguyen"],"year":2026,"doi":"10.21437/Interspeech.2026-3239","isca_url":"https://www.isca-archive.org/interspeech_2026/nguyen26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nguyen26g_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["asr","self-supervised","multilingual"],"category":"applications-other","institutions":["Hanoi University of Science and Technology","VNPT Group","National Economics University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nguyen26g_interspeech","category":"applications-other","institutions":["Hanoi University of Science and Technology","VNPT Group","National Economics University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3239","pdf":"https://www.isca-archive.org/interspeech_2026/nguyen26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26g_interspeech/markdown.md"},{"id":"nguyen26h_interspeech","title":"What Does a Pathological Speech Assessment Model Know about Acoustic Features? A Case Study on Oral and Oropharyngeal Cancer Patients","authors":["Tuan Nguyen","Corinne Fredouille","Alain Ghio","Muriel Lalain","Virginie Woisard"],"year":2026,"doi":"10.21437/Interspeech.2026-3343","isca_url":"https://www.isca-archive.org/interspeech_2026/nguyen26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nguyen26h_interspeech.pdf","session":"Pathological Speech Assessment 4","topics":["paralinguistics","self-supervised","evaluation"],"category":"health-clinical","labels":["self-supervised"],"institutions":["Avignon University","Hopital Larrey","Aix Marseille University","CNRS","Universite Toulouse II Jean Jaures"],"funding":["Chair LIAvignon","French National Research Agency","OLINPIC"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nguyen26h_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["Avignon University","Hopital Larrey","Aix Marseille University","CNRS","Universite Toulouse II Jean Jaures"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3343","pdf":"https://www.isca-archive.org/interspeech_2026/nguyen26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26h_interspeech/markdown.md"},{"id":"nguyen26i_interspeech","title":"Contrastive Training with LLM-generated Near-Misses for Robust Code-Switching Speech Recognition","authors":["Tung X. Nguyen","Hieu Minh Truong","Giang Son Nguyen","Nhu Vo","Wray Buntine","Dung D. Le"],"year":2026,"doi":"10.21437/Interspeech.2026-3465","isca_url":"https://www.isca-archive.org/interspeech_2026/nguyen26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nguyen26i_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 2","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["multilingual"],"institutions":["VinUniversity","University of Technology Sydney","Monash University"],"funding":["Center for AI Research at VinUniversity","Vingroup Scholarship"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nguyen26i_interspeech","category":"asr","labels":["multilingual"],"institutions":["VinUniversity","University of Technology Sydney","Monash University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3465","pdf":"https://www.isca-archive.org/interspeech_2026/nguyen26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nguyen26i_interspeech/markdown.md"},{"id":"ni26_interspeech","title":"NV-Bench: Benchmark of Nonverbal Vocalization Synthesis for Expressive Text-to-Speech Generation","authors":["Qinke Ni","Huan Liao","Dekun Chen","Yuxiang Wang","Zhizheng Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-2211","isca_url":"https://www.isca-archive.org/interspeech_2026/ni26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ni26_interspeech.pdf","session":"Speech Synthesis Evaluation 1","topics":["tts","paralinguistics","dataset"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Chinese University of Hong Kong-Shenzhen","Shenzhen Loop Area Institute","Amphion Technology"],"funding":["Internal Project Fund from Shenzhen Research Institute of Big Data","Program for Guangdong Introducing Innovative and Entrepreneurial Teams"],"code":{"url":"https://charlesnii.github.io/nvbench.github.io","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ni26_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Chinese University of Hong Kong-Shenzhen","Shenzhen Loop Area Institute","Amphion Technology"],"code":"https://charlesnii.github.io/nvbench.github.io","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2211","pdf":"https://www.isca-archive.org/interspeech_2026/ni26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ni26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ni26_interspeech/markdown.md"},{"id":"ni26b_interspeech","title":"DTT-BSR+: A Generative-Regression Cascade for Music Source Restoration","authors":["Youran Ni","Shihong Tan","Yuzhu Wang","Gongping Huang"],"year":2026,"doi":"10.21437/Interspeech.2026-2291","isca_url":"https://www.isca-archive.org/interspeech_2026/ni26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ni26b_interspeech.pdf","session":"Source Separation 1","topics":["source-separation","speech-enhancement","self-supervised"],"category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["Wuhan University","Tampere University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ni26b_interspeech","category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["Wuhan University","Tampere University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2291","pdf":"https://www.isca-archive.org/interspeech_2026/ni26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ni26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ni26b_interspeech/markdown.md"},{"id":"nieto26_interspeech","title":"Dialect Bias in Speech Recognition Across 10 Spanish and French Varieties","authors":["Rodrigo Nieto","Maria Angelika-Nikita","Diane Sarkis","Diyi Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-458","isca_url":"https://www.isca-archive.org/interspeech_2026/nieto26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nieto26_interspeech.pdf","session":"Language and Dialect Recognition","topics":["asr","multilingual","evaluation"],"category":"asr","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Stanford University","Google"],"code":{"url":"https://doi.org/10.5281/zenodo.20575155","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nieto26_interspeech","category":"asr","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Stanford University","Google"],"code":"https://doi.org/10.5281/zenodo.20575155","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-458","pdf":"https://www.isca-archive.org/interspeech_2026/nieto26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nieto26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nieto26_interspeech/markdown.md"},{"id":"nihal26_interspeech","title":"Ecologically-Constrained Task Arithmetic for Multi-Taxa Bioacoustic Classifiers Without Shared Data","authors":["Ragib Amin Nihal","Benjamin Yen","Runwu Shi","Takeshi Ashizawa","Kazuhiro Nakadai"],"year":2026,"doi":"10.21437/Interspeech.2026-2629","isca_url":"https://www.isca-archive.org/interspeech_2026/nihal26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nihal26_interspeech.pdf","session":"Acoustic Event Detection 1","topics":["self-supervised","multilingual","dataset"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["Institute of Science Tokyo","RIKEN"],"funding":["Japan Society for the Promotion of Science","Research Organization of Information and Systems"],"code":{"url":"https://ragib-amin-nihal.github.io/BioAcousticArithmetic/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nihal26_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["Institute of Science Tokyo","RIKEN"],"code":"https://ragib-amin-nihal.github.io/BioAcousticArithmetic/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2629","pdf":"https://www.isca-archive.org/interspeech_2026/nihal26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nihal26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nihal26_interspeech/markdown.md"},{"id":"nitsu26_interspeech","title":"Pseudo-Spatially Conditioned TF-Locoformer with MHCA+FiLM Fusion for Single-Channel Speech Separation","authors":["Daichi Nitsu","Koichi Shinoda"],"year":2026,"doi":"10.21437/Interspeech.2026-2027","isca_url":"https://www.isca-archive.org/interspeech_2026/nitsu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nitsu26_interspeech.pdf","session":"Source Separation 2","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","institutions":["Institute of Science Tokyo"],"funding":["JSPS KAKENHI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nitsu26_interspeech","category":"enhancement-separation","institutions":["Institute of Science Tokyo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2027","pdf":"https://www.isca-archive.org/interspeech_2026/nitsu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nitsu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nitsu26_interspeech/markdown.md"},{"id":"niu26_interspeech","title":"Semantic-VAE: Semantic-Alignment Latent Representation for Better Speech Synthesis","authors":["Zhikang Niu","Shujie Hu","Jeongsoo Choi","Yushen Chen","Peining Chen","Pengcheng Zhu","Yunting Yang","Bowen Zhang","Jian Zhao","Chunhui Wang","Xie Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-533","isca_url":"https://www.isca-archive.org/interspeech_2026/niu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/niu26_interspeech.pdf","session":"Speech Synthesis: Speech Features, Codec and Representations","topics":["tts","self-supervised","speech-llm"],"category":"tts","labels":["generative-model"],"institutions":["Shanghai Jiao Tong University","Shanghai Innovation Institute","Chinese University of Hong Kong","KAIST","Geely"],"funding":["National Natural Science Foundation of China","Shanghai Municipal Science and Technology Major Project","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"https://zhikangniu.github.io/semantic-vae/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"niu26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Shanghai Jiao Tong University","Shanghai Innovation Institute","Chinese University of Hong Kong","KAIST","Geely"],"code":"https://zhikangniu.github.io/semantic-vae/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-533","pdf":"https://www.isca-archive.org/interspeech_2026/niu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/niu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/niu26_interspeech/markdown.md"},{"id":"niu26b_interspeech","title":"MCA-DCF-DS: An Adaptive Framework for Unified Diarization and Separation with Spatial Information","authors":["Shutong Niu","Ruo-Yu Wang","Gao-Bin Yang","Ya Jiang","Tian Gao","Jia Pan","Jun Du"],"year":2026,"doi":"10.21437/Interspeech.2026-1539","isca_url":"https://www.isca-archive.org/interspeech_2026/niu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/niu26b_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["speech-separation","speaker-diarization","multichannel"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["University of Science and Technology of China","iFLYTEK Research"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"niu26b_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["University of Science and Technology of China","iFLYTEK Research"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1539","pdf":"https://www.isca-archive.org/interspeech_2026/niu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/niu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/niu26b_interspeech/markdown.md"},{"id":"niu26c_interspeech","title":"Improving Stable Speech Synthesis Post-Training with ChatScorer and Margin-Based Preference Construction","authors":["Shihao Niu","Jianguo Wei","Wenhuan Lu","Xianghu Yue","Wei Li","Ming Zhou","Ming Cai","Luo Si"],"year":2026,"doi":"10.21437/Interspeech.2026-1881","isca_url":"https://www.isca-archive.org/interspeech_2026/niu26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/niu26c_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["Tianjin University","Banma Network Technology"],"code":{"url":"https://hajararabiu869.github.io/demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"niu26c_interspeech","category":"tts","labels":["generative-model"],"institutions":["Tianjin University","Banma Network Technology"],"code":"https://hajararabiu869.github.io/demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1881","pdf":"https://www.isca-archive.org/interspeech_2026/niu26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/niu26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/niu26c_interspeech/markdown.md"},{"id":"nkouanga26_interspeech","title":"Mitigating Speaker Leakage in Cascaded Multi-talker ASR with Diarization-based Transcript Correction","authors":["Hermann Yepdjio Nkouanga","Minwei Luo","Maggie Wigness","Suresh Singh"],"year":2026,"doi":"10.21437/Interspeech.2026-3191","isca_url":"https://www.isca-archive.org/interspeech_2026/nkouanga26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nkouanga26_interspeech.pdf","session":"Multi-Speaker Processing, Personalization, and Adaptation","topics":["asr","speaker-diarization","source-separation"],"category":"asr","institutions":["Portland State University","US Army Research Laboratory"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nkouanga26_interspeech","category":"asr","institutions":["Portland State University","US Army Research Laboratory"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3191","pdf":"https://www.isca-archive.org/interspeech_2026/nkouanga26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nkouanga26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nkouanga26_interspeech/markdown.md"},{"id":"nolasco26_interspeech","title":"Beyond task performance: Decoding bioacoustic embeddings with speech features","authors":["Ines Nolasco","Jules Cauzinille","Marius Miron","Gagan Narula","Milad Alizadeh","Emmanuel Fernandez","Matthieu Geist","Ellen Gilsenan-McMahon","Olivier Pietquin","Emmanuel Chemla","Sara Keen"],"year":2026,"doi":"10.21437/Interspeech.2026-2759","isca_url":"https://www.isca-archive.org/interspeech_2026/nolasco26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nolasco26_interspeech.pdf","session":"Acoustic Event Detection 1","topics":["self-supervised","evaluation","multilingual"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["Earth Species Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nolasco26_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["Earth Species Project"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2759","pdf":"https://www.isca-archive.org/interspeech_2026/nolasco26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nolasco26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nolasco26_interspeech/markdown.md"},{"id":"noronha26_interspeech","title":"Structured Prompting vs. Self-Training for Audio Reasoning Under Limited Data and Compute: Lessons from Interspeech Audio Reasoning Challenge 2026","authors":["Sujit Noronha","Steven Au","Kaushlendra Tripathi"],"year":2026,"doi":"10.21437/Interspeech.2026-2880","isca_url":"https://www.isca-archive.org/interspeech_2026/noronha26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/noronha26_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"noronha26_interspeech","category":"speech-llm-dialogue","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2880","pdf":"https://www.isca-archive.org/interspeech_2026/noronha26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/noronha26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/noronha26_interspeech/markdown.md"},{"id":"nozaki26_interspeech","title":"Semi-Supervised Joint Separation and Diarization for Multichannel Noisy Speech Mixtures","authors":["Yuto Nozaki","Kohei Saijo","Yoshiaki Bando","Masaki Onishi"],"year":2026,"doi":"10.21437/Interspeech.2026-3308","isca_url":"https://www.isca-archive.org/interspeech_2026/nozaki26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/nozaki26_interspeech.pdf","session":"Source Separation 2","topics":["speech-separation","speaker-diarization","self-supervised"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["National Institute of Advanced Industrial Science and Technology","Keio University","Waseda University"],"funding":["JST FOREST"],"code":{"url":"https://ybando.jp/projects/na_neural-fcasa/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"nozaki26_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["National Institute of Advanced Industrial Science and Technology","Keio University","Waseda University"],"code":"https://ybando.jp/projects/na_neural-fcasa/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3308","pdf":"https://www.isca-archive.org/interspeech_2026/nozaki26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/nozaki26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/nozaki26_interspeech/markdown.md"},{"id":"ogunremi26_interspeech","title":"Turning Speech Language Models into Multilingual Listeners","authors":["Tolúlọpẹ́ Ògúnrẹ̀mí","Dan Jurafsky","Christopher D. Manning","Ahnmet Üstün","Martijn Bartelds"],"year":2026,"doi":"10.21437/Interspeech.2026-2584","isca_url":"https://www.isca-archive.org/interspeech_2026/ogunremi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ogunremi26_interspeech.pdf","session":"Multilingual Speech 1","topics":["multilingual","speech-llm","self-supervised"],"category":"speech-llm-dialogue","labels":["multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["Stanford University","Cohere Labs","Cohere"],"funding":["Stanford Interdisciplinary Graduate Fellowship","Google"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ogunremi26_interspeech","category":"speech-llm-dialogue","labels":["multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["Stanford University","Cohere Labs","Cohere"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2584","pdf":"https://www.isca-archive.org/interspeech_2026/ogunremi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ogunremi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ogunremi26_interspeech/markdown.md"},{"id":"oh26_interspeech","title":"L-Proto: Language-Aware Episodic Prototypical Training for Multilingual Speaker Verification","authors":["Hyung-Seok Oh","Deok-Hyeon Cho","Seung-Bin Kim","Seong-Whan Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-410","isca_url":"https://www.isca-archive.org/interspeech_2026/oh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/oh26_interspeech.pdf","session":"Challenge - TidyVoice Challenge: Cross-Lingual Speaker Verification","topics":["speaker-verification","multilingual","self-supervised"],"category":"speaker","labels":["multilingual"],"institutions":["Korea University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT"],"code":{"url":"https://github.com/hs-oh-prml/L-Proto/","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"oh26_interspeech","category":"speaker","labels":["multilingual"],"institutions":["Korea University"],"code":"https://github.com/hs-oh-prml/L-Proto/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-410","pdf":"https://www.isca-archive.org/interspeech_2026/oh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/oh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/oh26_interspeech/markdown.md"},{"id":"ohmura26_interspeech","title":"LibriTTS-VI: A Public Corpus and Novel Methods for Efficient Voice Impression Control","authors":["Junki Ohmura","Yuki Ito","Emiru Tsunoo","Toshiyuki Sekiya","Toshiyuki Kumakura"],"year":2026,"doi":"10.21437/Interspeech.2026-2231","isca_url":"https://www.isca-archive.org/interspeech_2026/ohmura26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ohmura26_interspeech.pdf","session":"Controllable and Expressive Speech Synthesis","topics":["tts","voice-conversion","dataset"],"category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Sony Group Corporation"],"code":{"url":"https://github.com/sony/LibriTTS-VI","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ohmura26_interspeech","category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Sony Group Corporation"],"code":"https://github.com/sony/LibriTTS-VI","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2231","pdf":"https://www.isca-archive.org/interspeech_2026/ohmura26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ohmura26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ohmura26_interspeech/markdown.md"},{"id":"ojha26_interspeech","title":"Bridging Self-Supervised Learning and Speech Enhancement: A Wav2Vec2-Conditioned Framework","authors":["Shuubham Ojha","Carol Espy-Wilson"],"year":2026,"doi":"10.21437/Interspeech.2026-964","isca_url":"https://www.isca-archive.org/interspeech_2026/ojha26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ojha26_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","self-supervised","asr"],"category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["University of Maryland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ojha26_interspeech","category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["University of Maryland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-964","pdf":"https://www.isca-archive.org/interspeech_2026/ojha26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ojha26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ojha26_interspeech/markdown.md"},{"id":"ok26_interspeech","title":"Towards Privacy-Preserving ASR: Speaker-Level Machine Unlearning","authors":["Seaone Ok","Seungu Han","Eungbeom Kim","Kyogu Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-2458","isca_url":"https://www.isca-archive.org/interspeech_2026/ok26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ok26_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["asr","self-supervised","speaker-verification"],"category":"deepfake-security","institutions":["Seoul National University"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ok26_interspeech","category":"deepfake-security","institutions":["Seoul National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2458","pdf":"https://www.isca-archive.org/interspeech_2026/ok26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ok26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ok26_interspeech/markdown.md"},{"id":"okocha26_interspeech","title":"Reasoning Beyond Transcription: Audio Language Models on Child Stuttering Speech","authors":["Chibuzor Okocha","Christan Grant","Zoey Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-2909","isca_url":"https://www.isca-archive.org/interspeech_2026/okocha26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/okocha26_interspeech.pdf","session":"Child Speech and Health","topics":["speech-llm","paralinguistics","evaluation"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["University of Florida"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"okocha26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["University of Florida"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2909","pdf":"https://www.isca-archive.org/interspeech_2026/okocha26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/okocha26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/okocha26_interspeech/markdown.md"},{"id":"olev26_interspeech","title":"Multi-Source Evidence Fusion for Audio Question Answering","authors":["Aivo Olev","Tanel Alumäe"],"year":2026,"doi":"10.21437/Interspeech.2026-3297","isca_url":"https://www.isca-archive.org/interspeech_2026/olev26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/olev26_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","evaluation","spoken-language-understanding"],"category":"speech-llm-dialogue","institutions":["Tallinn University of Technology"],"funding":["Estonian Centre of Excellence in Artificial Intelligence (EXAI)","Estonian Ministry of Education and Research","National Program for Estonian Language Technology Program","Estonian Language Data Research Infrastructure (KeTA)"],"code":{"url":"https://github.com/aivo0/audio-reasoning-solution","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"olev26_interspeech","category":"speech-llm-dialogue","institutions":["Tallinn University of Technology"],"code":"https://github.com/aivo0/audio-reasoning-solution","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3297","pdf":"https://www.isca-archive.org/interspeech_2026/olev26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/olev26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/olev26_interspeech/markdown.md"},{"id":"olewale26_interspeech","title":"Vavanagi: a Community-run Platform for Documentation of the Hula Language in Papua New Guinea","authors":["Bri Olewale","Raphael Merx","Ekaterina Vylomova"],"year":2026,"doi":"10.21437/Interspeech.2026-815","isca_url":"https://www.isca-archive.org/interspeech_2026/olewale26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/olewale26_interspeech.pdf","session":"Pacific Voices: Speech Science and Technology for the Languages of the Pacific Ocean","topics":["low-resource","dataset","multilingual"],"category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Vula’a Kunenai Community","University of Melbourne"],"funding":["Australian Research Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"olewale26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Vula’a Kunenai Community","University of Melbourne"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-815","pdf":"https://www.isca-archive.org/interspeech_2026/olewale26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/olewale26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/olewale26_interspeech/markdown.md"},{"id":"oliveira26_interspeech","title":"DIALOG DeID: Role and Privacy Aware Transcription for Clinical Interviews Beyond WER","authors":["Guilherme C. Oliveira","Dominic Dwyer","Stephanie Fong","Leandro A. Passos","Dan Mo","Benjamin Dixon","Phillip Wolff","Scott W. Woods","Martha Shenton","Barnaby Nelson","Zongyuan Ge"],"year":2026,"doi":"10.21437/Interspeech.2026-1489","isca_url":"https://www.isca-archive.org/interspeech_2026/oliveira26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/oliveira26_interspeech.pdf","session":"Speech and Language Technologies in Healthcare","topics":["speech-llm","evaluation","health"],"category":"health-clinical","institutions":["Monash University","University of Melbourne","Sao Paulo State University","Emory University","Yale University","Harvard Medical School"],"funding":["Medical Research Future Fund","National Critical Research Infrastructure","National Health and Medical Research Council","FAPESP"],"code":{"url":"https://github.com/GuiCamargoX/dialog-deid","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"oliveira26_interspeech","category":"health-clinical","institutions":["Monash University","University of Melbourne","Sao Paulo State University","Emory University","Yale University","Harvard Medical School"],"code":"https://github.com/GuiCamargoX/dialog-deid","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1489","pdf":"https://www.isca-archive.org/interspeech_2026/oliveira26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/oliveira26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/oliveira26_interspeech/markdown.md"},{"id":"oliveira26b_interspeech","title":"Too Good to Be True: A Study on Modern Automatic Speech Recognition Systems for the Evaluation of Speech Enhancement","authors":["Danilo Oliveira","Tal Peer","Timo Gerkmann"],"year":2026,"doi":"10.21437/Interspeech.2026-2597","isca_url":"https://www.isca-archive.org/interspeech_2026/oliveira26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/oliveira26b_interspeech.pdf","session":"Quality, Intelligibility and Evaluation of Speech and Codecs","topics":["speech-enhancement","asr","evaluation"],"category":"resources-evaluation","labels":["robustness-noise"],"institutions":["University of Hamburg"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"oliveira26b_interspeech","category":"resources-evaluation","labels":["robustness-noise"],"institutions":["University of Hamburg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2597","pdf":"https://www.isca-archive.org/interspeech_2026/oliveira26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/oliveira26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/oliveira26b_interspeech/markdown.md"},{"id":"oliveira26c_interspeech","title":"The TinyExplorer Ecosystem: Open tools for studying infants’ auditory and visual experiences","authors":["Cátia M Oliveira","Teodor Y. Nikolov","Tamas Foldes","Charlotte Bocchetta","Hana D'Souza"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/oliveira26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/oliveira26c_interspeech.pdf","session":"Speech and Language Processing for Health and Accessibility","topics":["spoken-language-understanding","self-supervised","dataset"],"category":"resources-evaluation","institutions":["Cardiff University","University of Surrey"],"funding":["James S. McDonnell Foundation","UKRI Future Leaders Fellowship"],"code":{"url":"https://cardiffbabylab.github.io/tinyexplorer-detection-app/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"oliveira26c_interspeech","category":"resources-evaluation","institutions":["Cardiff University","University of Surrey"],"code":"https://cardiffbabylab.github.io/tinyexplorer-detection-app/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/oliveira26c_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/oliveira26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/oliveira26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/oliveira26c_interspeech/markdown.md"},{"id":"omidi26_interspeech","title":"Learning from Annotation Uncertainty: Entropy-Aware Curriculum for Speech Emotion Recognition","authors":["Zahra Omidi","John Hansen"],"year":2026,"doi":"10.21437/Interspeech.2026-2992","isca_url":"https://www.isca-archive.org/interspeech_2026/omidi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/omidi26_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["speech-emotion-recognition","self-supervised-learning","paralinguistics"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["University of Texas at Dallas"],"code":{"url":"https://github.com/zahraomidi/MSP-PODCAST","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"omidi26_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["University of Texas at Dallas"],"code":"https://github.com/zahraomidi/MSP-PODCAST","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2992","pdf":"https://www.isca-archive.org/interspeech_2026/omidi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/omidi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/omidi26_interspeech/markdown.md"},{"id":"onda26_interspeech","title":"Leveraging Soft Distributions of SSL-Derived Discrete Speech Tokens for Downstream Inference","authors":["Kentaro Onda","Satoru Fukayama","Daisuke Saito","Nobuaki Minematsu"],"year":2026,"doi":"10.21437/Interspeech.2026-1668","isca_url":"https://www.isca-archive.org/interspeech_2026/onda26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/onda26_interspeech.pdf","session":"Self-supervised Speech Representation Learning","topics":["asr","speech-synthesis","self-supervised"],"category":"asr","labels":["self-supervised"],"institutions":["University of Tokyo","National Institute of Advanced Industrial Science and Technology"],"funding":["AIST","Japan Science and Technology Agency"],"code":{"url":"https://ondatk68.github.io/onda-demo/projects/soft-token-inference/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"onda26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["University of Tokyo","National Institute of Advanced Industrial Science and Technology"],"code":"https://ondatk68.github.io/onda-demo/projects/soft-token-inference/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1668","pdf":"https://www.isca-archive.org/interspeech_2026/onda26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/onda26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/onda26_interspeech/markdown.md"},{"id":"ono26_interspeech","title":"Fast Multichannel Nonnegative Matrix Factorization with Directivity Regularization for DOA-Informed Speech Separation","authors":["Ryosuke Ono","Aditya Arie Nugraha","Yoshiaki Bando","Kazuyoshi Yoshii"],"year":2026,"doi":"10.21437/Interspeech.2026-2139","isca_url":"https://www.isca-archive.org/interspeech_2026/ono26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ono26_interspeech.pdf","session":"Spatial Audio 4","topics":["source-separation","speech-enhancement","self-supervised"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Kyoto University","RIKEN","National Institute of Advanced Industrial Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ono26_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Kyoto University","RIKEN","National Institute of Advanced Industrial Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2139","pdf":"https://www.isca-archive.org/interspeech_2026/ono26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ono26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ono26_interspeech/markdown.md"},{"id":"orepic26_interspeech","title":"SELFIX: An Interactive System for Natural Self-Voice Approximation","authors":["Pavo Orepic","Steven Moran","Volker Dellwo"],"year":2026,"doi":"10.21437/Interspeech.2026-1395","isca_url":"https://www.isca-archive.org/interspeech_2026/orepic26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/orepic26_interspeech.pdf","session":"Speech Production and Perception 1","topics":["speech-enhancement","voice-conversion","evaluation"],"category":"tts","institutions":["University of Zurich","University of Neuchâtel"],"code":{"url":"https://osf.io/6nxzw/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"orepic26_interspeech","category":"tts","institutions":["University of Zurich","University of Neuchâtel"],"code":"https://osf.io/6nxzw/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1395","pdf":"https://www.isca-archive.org/interspeech_2026/orepic26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/orepic26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/orepic26_interspeech/markdown.md"},{"id":"otani26_interspeech","title":"Speaker-Independent Speech Synthesis from Real-time MRI Articulatory Data","authors":["Yuto Otani","Shun Sawada","Hidefumi Ohmura","Kouichi Katsurada"],"year":2026,"doi":"10.21437/Interspeech.2026-3379","isca_url":"https://www.isca-archive.org/interspeech_2026/otani26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/otani26_interspeech.pdf","session":"Articulatory, EMG, and Visual Speech Generation","topics":["tts","self-supervised","multimodal"],"category":"tts","labels":["generative-model"],"institutions":["Tokyo University of Science","Nippon Institute of Technology"],"funding":["JSPS KAKENHI","JST SPRING"],"code":{"url":"https://github.com/y-otn/m2s-code","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"otani26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Tokyo University of Science","Nippon Institute of Technology"],"code":"https://github.com/y-otn/m2s-code","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3379","pdf":"https://www.isca-archive.org/interspeech_2026/otani26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/otani26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/otani26_interspeech/markdown.md"},{"id":"ou26_interspeech","title":"Positioning Effect of Syllable Shortening on Speech Segmentation: Evidence from Mandarin Listeners","authors":["Shu-Chen Ou"],"year":2026,"doi":"10.21437/Interspeech.2026-234","isca_url":"https://www.isca-archive.org/interspeech_2026/ou26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ou26_interspeech.pdf","session":"Perception and Appraisal of Prosody","topics":["phonetics","prosody","evaluation"],"category":"phonetics-linguistics","institutions":["National Sun Yat-sen University"],"funding":["National Science and Technology Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ou26_interspeech","category":"phonetics-linguistics","institutions":["National Sun Yat-sen University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-234","pdf":"https://www.isca-archive.org/interspeech_2026/ou26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ou26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ou26_interspeech/markdown.md"},{"id":"ozer26_interspeech","title":"A Training-Free Proactive Defense Against Partial Speech Manipulation via Self-Embedding Steganography","authors":["Yigitcan Özer","Zhe Zhang","Wanying Ge","Xin Wang","Junichi Yamagishi"],"year":2026,"doi":"10.21437/Interspeech.2026-1822","isca_url":"https://www.isca-archive.org/interspeech_2026/ozer26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ozer26_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","institutions":["National Institute of Informatics"],"funding":["JSPS MEXT KAKENHI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ozer26_interspeech","category":"deepfake-security","institutions":["National Institute of Informatics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1822","pdf":"https://www.isca-archive.org/interspeech_2026/ozer26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ozer26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ozer26_interspeech/markdown.md"},{"id":"ozkan26_interspeech","title":"Automatic pitch prediction from speech articulation: Where does the f0 information come from?","authors":["Beliz Ozkan","Jonas Michael","Thomas Hueber","Olivier Perrotin"],"year":2026,"doi":"10.21437/Interspeech.2026-1288","isca_url":"https://www.isca-archive.org/interspeech_2026/ozkan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ozkan26_interspeech.pdf","session":"Articulatory, EMG, and Visual Speech Generation","topics":["speech-production","prosody","self-supervised"],"category":"phonetics-linguistics","institutions":["Univ. Grenoble Alpes","CNRS","Grenoble INP"],"funding":["ANR SilentPitch","MIAI Cluster"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ozkan26_interspeech","category":"phonetics-linguistics","institutions":["Univ. Grenoble Alpes","CNRS","Grenoble INP"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1288","pdf":"https://www.isca-archive.org/interspeech_2026/ozkan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ozkan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ozkan26_interspeech/markdown.md"},{"id":"pagel26_interspeech","title":"What Happens When We Speak Together? Multidimensional Convergence in Face-to-Face Interaction","authors":["Lena Pagel","Doris Mücke","Simon Roessig"],"year":2026,"doi":"10.21437/Interspeech.2026-2444","isca_url":"https://www.isca-archive.org/interspeech_2026/pagel26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pagel26_interspeech.pdf","session":"Entrainment and Dialogue Coordination","topics":["paralinguistics","phonetics","prosody"],"category":"phonetics-linguistics","institutions":["University of Cologne"],"funding":["German Research Foundation","SFB1252 Prominence in Language","a.r.t.e.s. Graduate School for the Humanities Cologne"],"code":{"url":"https://osf.io/3uebk","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pagel26_interspeech","category":"phonetics-linguistics","institutions":["University of Cologne"],"code":"https://osf.io/3uebk","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2444","pdf":"https://www.isca-archive.org/interspeech_2026/pagel26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pagel26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pagel26_interspeech/markdown.md"},{"id":"pahwa26_interspeech","title":"Audio2Tool: Speak, Call, Act - A Dataset for Benchmarking Speech Tool Use","authors":["Ramit Pahwa","Apoorva Beedu","Parivesh Priye","Rutu Gandhi","Saloni Takawale","Aruna Baijal","Zengli Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-2857","isca_url":"https://www.isca-archive.org/interspeech_2026/pahwa26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pahwa26_interspeech.pdf","session":"Audio & Speech Language Models: Evaluation, Representations, and Emerging Capabilities","topics":["speech-llm","spoken-language-understanding","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Rivian and Volkswagen Group Technologies"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pahwa26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Rivian and Volkswagen Group Technologies"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2857","pdf":"https://www.isca-archive.org/interspeech_2026/pahwa26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pahwa26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pahwa26_interspeech/markdown.md"},{"id":"pal26_interspeech","title":"Two-stage semi-supervised learning with pseudo-labels: A case study on Northern Sámi ASR","authors":["Priyanshi Pal","Yaroslav Getman","Kristiina Ojala","Tamás Grósz","Mikko Kurimo"],"year":2026,"doi":"10.21437/Interspeech.2026-2497","isca_url":"https://www.isca-archive.org/interspeech_2026/pal26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pal26_interspeech.pdf","session":"Indigenous Voices in Speech Science and Technology","topics":["asr","low-resource","self-supervised"],"category":"asr","labels":["low-resource"],"institutions":["Aalto University","INESC-ID","Instituto Superior Técnico","Walton Institute","South East Technological University"],"funding":["Business Finland","Finnish Cultural Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pal26_interspeech","category":"asr","labels":["low-resource"],"institutions":["Aalto University","INESC-ID","Instituto Superior Técnico","Walton Institute","South East Technological University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2497","pdf":"https://www.isca-archive.org/interspeech_2026/pal26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pal26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pal26_interspeech/markdown.md"},{"id":"palka26_interspeech","title":"SphereVBx: Spherical Variational Bayes Clustering for Simplified EEND-VC Diarization","authors":["Petr Pálka","Jiangyu Han","Prachi Singh","Marc Delcroix","Naohiro Tawara","Lukáš Burget"],"year":2026,"doi":"10.21437/Interspeech.2026-2224","isca_url":"https://www.isca-archive.org/interspeech_2026/palka26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/palka26_interspeech.pdf","session":"Speaker Diarization 1","topics":["speaker-diarization","self-supervised","speech-llm"],"category":"speaker","institutions":["Brno University of Technology","NTT"],"funding":["European Union","Czech Ministry of Education, Youth and Sports"],"code":{"url":"https://github.com/BUTSpeechFIT/DiariZen","stars":547,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"palka26_interspeech","category":"speaker","institutions":["Brno University of Technology","NTT"],"code":"https://github.com/BUTSpeechFIT/DiariZen","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2224","pdf":"https://www.isca-archive.org/interspeech_2026/palka26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/palka26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/palka26_interspeech/markdown.md"},{"id":"pan26_interspeech","title":"Supervised Post-training of Speech Foundation Models for Robust Adaptation in Speech Deepfake Detection","authors":["Zihan Pan","Sailor Hardik","Jinyang Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-908","isca_url":"https://www.isca-archive.org/interspeech_2026/pan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pan26_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["speech-deepfake-detection","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised","robustness-noise"],"institutions":["Agency for Science, Technology and Research"],"code":{"url":"https://github.com/pandarialTJU/Mix-Frame-Post-Training","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pan26_interspeech","category":"deepfake-security","labels":["self-supervised","robustness-noise"],"institutions":["Agency for Science, Technology and Research"],"code":"https://github.com/pandarialTJU/Mix-Frame-Post-Training","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-908","pdf":"https://www.isca-archive.org/interspeech_2026/pan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pan26_interspeech/markdown.md"},{"id":"pandey26_interspeech","title":"Beyond Speaker Independence: Evaluating Cross-Lingual Acoustic-to-Articulatory Inversion Across Finnish and Russian","authors":["Ruchi Pandey","Tomi H. Kinnunen"],"year":2026,"doi":"10.21437/Interspeech.2026-1914","isca_url":"https://www.isca-archive.org/interspeech_2026/pandey26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pandey26_interspeech.pdf","session":"Cross-Linguistic and L2 Phonetic Studies","topics":["self-supervised","multilingual","evaluation"],"category":"phonetics-linguistics","labels":["multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["University of Eastern Finland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pandey26_interspeech","category":"phonetics-linguistics","labels":["multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["University of Eastern Finland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1914","pdf":"https://www.isca-archive.org/interspeech_2026/pandey26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pandey26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pandey26_interspeech/markdown.md"},{"id":"pandey26b_interspeech","title":"Evaluation of forced alignment of code-mixed speech: the case of Hindi-English","authors":["Ayushi Pandey","Pamir Gogoi","Kevin Tang"],"year":2026,"doi":"10.21437/Interspeech.2026-2179","isca_url":"https://www.isca-archive.org/interspeech_2026/pandey26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pandey26b_interspeech.pdf","session":"Multilingual Speech 2","topics":["speech-recognition","low-resource","multilingual"],"category":"asr","labels":["multilingual"],"institutions":["Karya","Heinrich Heine University Düsseldorf","University of Florida"],"code":{"url":"https://github.com/Ayushi113/mfa-hindi-code-mixed","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pandey26b_interspeech","category":"asr","labels":["multilingual"],"institutions":["Karya","Heinrich Heine University Düsseldorf","University of Florida"],"code":"https://github.com/Ayushi113/mfa-hindi-code-mixed","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2179","pdf":"https://www.isca-archive.org/interspeech_2026/pandey26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pandey26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pandey26b_interspeech/markdown.md"},{"id":"pang26_interspeech","title":"USDnet++: Distilling Signal Processing Based Dereverberation for Unsupervised Neural Speech Dereverberation","authors":["Ruizhe Pang","Shulin He","Jingqi Sun","Zhong-Qiu Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-2044","isca_url":"https://www.isca-archive.org/interspeech_2026/pang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pang26_interspeech.pdf","session":"Dereverberation, Bandwidth Extension and Restoration","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","institutions":["Southern University of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pang26_interspeech","category":"enhancement-separation","institutions":["Southern University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2044","pdf":"https://www.isca-archive.org/interspeech_2026/pang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pang26_interspeech/markdown.md"},{"id":"pang26b_interspeech","title":"ERM-MinMaxGAP: Benchmarking and Mitigating Gender Bias in Multilingual Multimodal Speech-LLM Emotion Recognition","authors":["Zi Haur Pang","Xiaoxue Gao","Tatsuya Kawahara","Nancy Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-3143","isca_url":"https://www.isca-archive.org/interspeech_2026/pang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pang26b_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 2","topics":["speech-llm","paralinguistic","evaluation"],"category":"paralinguistics-emotion","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Kyoto University","Agency for Science, Technology, and Research"],"funding":["Japan Science and Technology Agency","A*STAR","National Research Foundation, Singapore"],"code":{"url":"https://github.com/zihaurpang/ERM-MinMaxGAP","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pang26b_interspeech","category":"paralinguistics-emotion","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Kyoto University","Agency for Science, Technology, and Research"],"code":"https://github.com/zihaurpang/ERM-MinMaxGAP","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3143","pdf":"https://www.isca-archive.org/interspeech_2026/pang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pang26b_interspeech/markdown.md"},{"id":"pangsatabam26_interspeech","title":"Scalable Neural TTS for Latin-Script Low-Resource Languages of Manipur","authors":["Hoomexsun Pangsatabam","Khumanthem Chanchanbi","Kansham Tungran Maring","Yambem Jina Chanu"],"year":2026,"doi":"10.21437/Interspeech.2026-2304","isca_url":"https://www.isca-archive.org/interspeech_2026/pangsatabam26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pangsatabam26_interspeech.pdf","session":"Low-Resource Speech Synthesis","topics":["tts","low-resource","multilingual"],"category":"tts","labels":["low-resource","multilingual","dataset-or-benchmark-release","generative-model"],"institutions":["National Institute of Technology Manipur"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pangsatabam26_interspeech","category":"tts","labels":["low-resource","multilingual","dataset-or-benchmark-release","generative-model"],"institutions":["National Institute of Technology Manipur"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2304","pdf":"https://www.isca-archive.org/interspeech_2026/pangsatabam26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pangsatabam26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pangsatabam26_interspeech/markdown.md"},{"id":"papi26_interspeech","title":"Cross-Attention is Half Explanation in Speech-to-Text Models","authors":["Sara Papi","Dennis Fucci","Marco Gaido","Matteo Negri","Luisa Bentivogli"],"year":2026,"doi":"10.21437/Interspeech.2026-40","isca_url":"https://www.isca-archive.org/interspeech_2026/papi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/papi26_interspeech.pdf","session":"Speech Representations and Alignment","topics":["asr","speech-translation","self-supervised"],"category":"asr","institutions":["Fondazione Bruno Kessler"],"funding":["European Union"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"papi26_interspeech","category":"asr","institutions":["Fondazione Bruno Kessler"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-40","pdf":"https://www.isca-archive.org/interspeech_2026/papi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/papi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/papi26_interspeech/markdown.md"},{"id":"parikh26_interspeech","title":"A Finetuned SpeechLLM for Joint Multi-Granular L2 Assessment and Natural-Language Rationales","authors":["Aditya Kamlesh Parikh","Cristian Tejedor-Garcia","Catia Cucchiarini","Helmer Strik"],"year":2026,"doi":"10.21437/Interspeech.2026-2335","isca_url":"https://www.isca-archive.org/interspeech_2026/parikh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/parikh26_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["speech-llm","paralinguistics","evaluation"],"category":"applications-other","institutions":["Radboud University"],"funding":["Dutch Research Council","NGF AiNed Fellowship Grants"],"code":{"url":"https://github.com/Aditya3107/speechllm-l2-assessment","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"parikh26_interspeech","category":"applications-other","institutions":["Radboud University"],"code":"https://github.com/Aditya3107/speechllm-l2-assessment","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2335","pdf":"https://www.isca-archive.org/interspeech_2026/parikh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/parikh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/parikh26_interspeech/markdown.md"},{"id":"park26_interspeech","title":"Sleep Sound Event Detection Powered by Learnable Multi-Resolution Adaptive Line Enhancer","authors":["Chanwoo Park","Chanwoo Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-130","isca_url":"https://www.isca-archive.org/interspeech_2026/park26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26_interspeech.pdf","session":"Audio Coding and Signal Analysis","topics":["speech-enhancement","sound-event-detection","health"],"category":"audio-understanding","institutions":["Korea University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","National Research Foundation of Korea","Ministry of Science and ICT","Ministry of SMEs and Startups","Supreme Prosecutor's Office"],"code":{"url":"https://github.com/honeysleep/sed","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26_interspeech","category":"audio-understanding","institutions":["Korea University"],"code":"https://github.com/honeysleep/sed","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-130","pdf":"https://www.isca-archive.org/interspeech_2026/park26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26_interspeech/markdown.md"},{"id":"park26b_interspeech","title":"Accurate Source-Free Speech Classification via Meta-Learned Target-Centric Model Merging","authors":["Ka Hyun Park","Junghun Kim","U Kang"],"year":2026,"doi":"10.21437/Interspeech.2026-371","isca_url":"https://www.isca-archive.org/interspeech_2026/park26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26b_interspeech.pdf","session":"New Training Methods for ASR","topics":["emotion-recognition","self-supervised","multilingual"],"category":"paralinguistics-emotion","labels":["low-resource","self-supervised"],"institutions":["Seoul National University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea government (MSIT)","AI Star Fellowship Support Program","Global AI Frontier Lab","Artificial Intelligence Graduate School Program"],"code":{"url":"https://github.com/snudatalab/Mochee","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26b_interspeech","category":"paralinguistics-emotion","labels":["low-resource","self-supervised"],"institutions":["Seoul National University"],"code":"https://github.com/snudatalab/Mochee","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-371","pdf":"https://www.isca-archive.org/interspeech_2026/park26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26b_interspeech/markdown.md"},{"id":"park26c_interspeech","title":"LoRA-Tuned Large Language Models for Dementia Detection via Multi-View Speech-Derived Features","authors":["Jonghyeon Park","Olivier Jiyoun Jung","Myungwoo Oh"],"year":2026,"doi":"10.21437/Interspeech.2026-952","isca_url":"https://www.isca-archive.org/interspeech_2026/park26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26c_interspeech.pdf","session":"Pathological Speech Assessment 2","topics":["speech-llm","paralinguistics","evaluation"],"category":"health-clinical","institutions":["NAVER Cloud","Ewha Womans University"],"code":{"url":"https://github.com/vivivic/is26dementia","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26c_interspeech","category":"health-clinical","institutions":["NAVER Cloud","Ewha Womans University"],"code":"https://github.com/vivivic/is26dementia","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-952","pdf":"https://www.isca-archive.org/interspeech_2026/park26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26c_interspeech/markdown.md"},{"id":"park26d_interspeech","title":"Countering Neural Audio Codec Distortions in Watermarking with Adaptive Restoration","authors":["Sungho Park","Thien An Nguyen","Souhwan Jung"],"year":2026,"doi":"10.21437/Interspeech.2026-953","isca_url":"https://www.isca-archive.org/interspeech_2026/park26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26d_interspeech.pdf","session":"Audio Watermarking and Source Verification","topics":["speech-enhancement","self-supervised","evaluation"],"category":"deepfake-security","labels":["robustness-noise"],"institutions":["Soongsil University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea government","Korea Internet & Security Agency"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26d_interspeech","category":"deepfake-security","labels":["robustness-noise"],"institutions":["Soongsil University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-953","pdf":"https://www.isca-archive.org/interspeech_2026/park26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26d_interspeech/markdown.md"},{"id":"park26e_interspeech","title":"Pushing the Boundaries of Streaming Multi-Speaker ASR: A Systematic Study of Architectural Trade-offs","authors":["Taejin Park","Ivan Medennikov","Kunal Dhawan","Weiqing Wang","Jagadeesh Balam","Boris Ginsburg"],"year":2026,"doi":"10.21437/Interspeech.2026-2005","isca_url":"https://www.isca-archive.org/interspeech_2026/park26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26e_interspeech.pdf","session":"Multi-Talker ASR & Speaker Diarization","topics":["asr","speaker-diarization","self-supervised"],"category":"asr","labels":["streaming-real-time"],"institutions":["NVIDIA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26e_interspeech","category":"asr","labels":["streaming-real-time"],"institutions":["NVIDIA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2005","pdf":"https://www.isca-archive.org/interspeech_2026/park26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26e_interspeech/markdown.md"},{"id":"park26f_interspeech","title":"LLM-Based Multi-Reference Evaluation for Efficient and Robust Assessment of Phrase Break Annotations","authors":["Younghan Park","Hoyeon Lee","Hawon Jeong","Jong-Hwan Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-2225","isca_url":"https://www.isca-archive.org/interspeech_2026/park26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26f_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["tts","prosody","evaluation"],"category":"resources-evaluation","institutions":["NAVER Cloud","Yonsei University","KAIST"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26f_interspeech","category":"resources-evaluation","institutions":["NAVER Cloud","Yonsei University","KAIST"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2225","pdf":"https://www.isca-archive.org/interspeech_2026/park26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26f_interspeech/markdown.md"},{"id":"park26g_interspeech","title":"NaVo: Natural Voice Protection against Voice Cloning Attacks via Generative Universal Adversarial Audio","authors":["Seoyoung Park","Seungmin Kim","Sohee Park","Dain Kim","Thien An Nguyen","Thien-Phuc Doan","Souhwan Jung","Daeseon Choi"],"year":2026,"doi":"10.21437/Interspeech.2026-2944","isca_url":"https://www.isca-archive.org/interspeech_2026/park26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26g_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["speaker-verification","self-supervised","speech-enhancement"],"category":"deepfake-security","labels":["generative-model"],"institutions":["Soongsil University"],"funding":["Korea Internet & Security Agency","Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26g_interspeech","category":"deepfake-security","labels":["generative-model"],"institutions":["Soongsil University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2944","pdf":"https://www.isca-archive.org/interspeech_2026/park26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26g_interspeech/markdown.md"},{"id":"park26h_interspeech","title":"AnimeScore: A Preference-Based Dataset and Framework for Evaluating Anime-Like Speech Style","authors":["Joonyong Park","Jerry Li"],"year":2026,"doi":"10.21437/Interspeech.2026-3025","isca_url":"https://www.isca-archive.org/interspeech_2026/park26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26h_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["paralinguistics","evaluation","self-supervised"],"category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["Spellbrush"],"code":{"url":"https://github.com/sizigi/animescore","stars":12,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26h_interspeech","category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["Spellbrush"],"code":"https://github.com/sizigi/animescore","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3025","pdf":"https://www.isca-archive.org/interspeech_2026/park26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26h_interspeech/markdown.md"},{"id":"park26i_interspeech","title":"Word-level Emotional Intensity Control in TTS via Emotion Residual Vectors","authors":["Ji-Hyun Park","Nam-Seok Song","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-3079","isca_url":"https://www.isca-archive.org/interspeech_2026/park26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26i_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["tts","paralinguistics","self-supervised"],"category":"tts","labels":["self-supervised","generative-model"],"institutions":["Hanyang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT"],"code":{"url":"https://jjhh0210.github.io/EmoRes-tts-demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26i_interspeech","category":"tts","labels":["self-supervised","generative-model"],"institutions":["Hanyang University"],"code":"https://jjhh0210.github.io/EmoRes-tts-demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3079","pdf":"https://www.isca-archive.org/interspeech_2026/park26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26i_interspeech/markdown.md"},{"id":"park26j_interspeech","title":"From Masking to Merging: Rethinking SpecAugment for Efficient Audio Spectrogram Transformer","authors":["Minhee Park","Hyowon Ahn","Chanwoo Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-3273","isca_url":"https://www.isca-archive.org/interspeech_2026/park26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26j_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["asr","self-supervised","on-device"],"category":"audio-understanding","labels":["efficient-on-device"],"institutions":["Korea University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","National Research Foundation of Korea","Technology Development Program","Supreme Prosecutor's Office Research Grant"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26j_interspeech","category":"audio-understanding","labels":["efficient-on-device"],"institutions":["Korea University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3273","pdf":"https://www.isca-archive.org/interspeech_2026/park26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26j_interspeech/markdown.md"},{"id":"park26k_interspeech","title":"MeloDISinger: Melody-Aware & Duration-Preserving Singing Voice Editing with Audio Infilling","authors":["Yoonjeong Park","Jaekwon Im","Juhan Nam"],"year":2026,"doi":"10.21437/Interspeech.2026-3285","isca_url":"https://www.isca-archive.org/interspeech_2026/park26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26k_interspeech.pdf","session":"Singing Voice and Music Generation","topics":["tts","speech-editing","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["KAIST"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"https://cottonlove.github.io/MeloDISinger_demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26k_interspeech","category":"tts","labels":["generative-model"],"institutions":["KAIST"],"code":"https://cottonlove.github.io/MeloDISinger_demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3285","pdf":"https://www.isca-archive.org/interspeech_2026/park26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26k_interspeech/markdown.md"},{"id":"park26l_interspeech","title":"Prosody-Aware Speech Representations for Emotion Recognition under Pragmatic Ambiguity","authors":["Yeonwoo Park","Chioh Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-3486","isca_url":"https://www.isca-archive.org/interspeech_2026/park26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26l_interspeech.pdf","session":"Self-supervised Speech Representation Learning","topics":["speech-emotion-recognition","self-supervised","paralinguistics"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Hanyang University","Pusan National University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26l_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Hanyang University","Pusan National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3486","pdf":"https://www.isca-archive.org/interspeech_2026/park26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26l_interspeech/markdown.md"},{"id":"park26m_interspeech","title":"Listening to Motion in Space: Vision-Grounded Event-wise Video-to-Audio Generation and Rendering","authors":["Hyeonwoo Park","Dayeon Ku","Hong Kook Kim"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/park26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/park26m_interspeech.pdf","session":"Speech Synthesis, Voice Conversion and Audio Generation","topics":["speech-enhancement","self-supervised","evaluation"],"category":"audio-understanding","labels":["generative-model"],"institutions":["Gwangju Institute of Science and Technology","AunionAI"],"funding":["Technology Development Programs","Korea MSS","MOTIE","Science and Technology Opens the Future of the Region program","MSIT","Gwangju Metropolitan City"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"park26m_interspeech","category":"audio-understanding","labels":["generative-model"],"institutions":["Gwangju Institute of Science and Technology","AunionAI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/park26m_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/park26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/park26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/park26m_interspeech/markdown.md"},{"id":"parmonangan26_interspeech","title":"Audio-Visual Feature Reconstruction Pretraining for Noise-Robust Emotion Recognition","authors":["Ivan Halim Parmonangan","Tharindu Fernando","Simon Denman"],"year":2026,"doi":"10.21437/Interspeech.2026-605","isca_url":"https://www.isca-archive.org/interspeech_2026/parmonangan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/parmonangan26_interspeech.pdf","session":"Audio signal analysis","topics":["paralinguistics","emotion-recognition","self-supervised"],"category":"paralinguistics-emotion","labels":["self-supervised","robustness-noise"],"institutions":["Queensland University of Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"parmonangan26_interspeech","category":"paralinguistics-emotion","labels":["self-supervised","robustness-noise"],"institutions":["Queensland University of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-605","pdf":"https://www.isca-archive.org/interspeech_2026/parmonangan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/parmonangan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/parmonangan26_interspeech/markdown.md"},{"id":"pasha26_interspeech","title":"A Novel Transfer Learning Approach for Room Impulse Response Estimation and Speech Dereverberation Across Geometrically Diverse and Data-Scarce Environments","authors":["Shahab Pasha","Jiahong Zhao","Hualin Ren","Christian Ritz"],"year":2026,"doi":"10.21437/Interspeech.2026-31","isca_url":"https://www.isca-archive.org/interspeech_2026/pasha26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pasha26_interspeech.pdf","session":"Dereverberation, Bandwidth Extension and Restoration","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","labels":["low-resource","robustness-noise"],"institutions":["University of Western Australia","University of Southampton","University of Wollongong"],"code":{"url":"https://github.com/ShahabP/DeepRIRnet","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pasha26_interspeech","category":"enhancement-separation","labels":["low-resource","robustness-noise"],"institutions":["University of Western Australia","University of Southampton","University of Wollongong"],"code":"https://github.com/ShahabP/DeepRIRnet","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-31","pdf":"https://www.isca-archive.org/interspeech_2026/pasha26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pasha26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pasha26_interspeech/markdown.md"},{"id":"patman26_interspeech","title":"Assessing the effect of volitional and synthetic pitch raising in female speakers on automatic speaker recognition","authors":["Chloe Patman","Linda Gerlach","Anil Alexander","Finnian Kelly","Kirsty McDougall"],"year":2026,"doi":"10.21437/Interspeech.2026-444","isca_url":"https://www.isca-archive.org/interspeech_2026/patman26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/patman26_interspeech.pdf","session":"Phonetic Aspects of TTS and ASR Systems","topics":["speaker-verification","evaluation","paralinguistics"],"category":"speaker","institutions":["University of Cambridge","Oxford Wave Research"],"funding":["Harding Distinguished Postgraduate Scholarship"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"patman26_interspeech","category":"speaker","institutions":["University of Cambridge","Oxford Wave Research"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-444","pdf":"https://www.isca-archive.org/interspeech_2026/patman26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/patman26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/patman26_interspeech/markdown.md"},{"id":"paul26_interspeech","title":"PROGRESS: Coverage-guided RL to Train Search-augmented LLM Agent","authors":["Sudipta Paul","Vijay Srinivasan","Vivek Kulkarni","Aounon Kumar","Yashas Malur Saidutta","Wenbo Li","Srinivas Chappidi"],"year":2026,"doi":"10.21437/Interspeech.2026-2760","isca_url":"https://www.isca-archive.org/interspeech_2026/paul26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/paul26_interspeech.pdf","session":"Spoken Language Understanding","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","institutions":["Samsung Electronics"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"paul26_interspeech","category":"speech-llm-dialogue","institutions":["Samsung Electronics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2760","pdf":"https://www.isca-archive.org/interspeech_2026/paul26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/paul26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/paul26_interspeech/markdown.md"},{"id":"paver26_interspeech","title":"Acoustic correlates of voice quality settings: variation within and between individual speakers","authors":["Alice Paver","Kirsty McDougall"],"year":2026,"doi":"10.21437/Interspeech.2026-2053","isca_url":"https://www.isca-archive.org/interspeech_2026/paver26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/paver26_interspeech.pdf","session":"Voice Quality Aspects of Speech","topics":["paralinguistics","phonetics","speaker-verification"],"category":"phonetics-linguistics","institutions":["University of Cambridge"],"funding":["Arts and Humanities Research Council","Open-Oxford-Cambridge Doctoral Training Partnership"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"paver26_interspeech","category":"phonetics-linguistics","institutions":["University of Cambridge"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2053","pdf":"https://www.isca-archive.org/interspeech_2026/paver26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/paver26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/paver26_interspeech/markdown.md"},{"id":"payne26_interspeech","title":"Broad Focus Rise(-fall) Declaratives in Venetan: Investigating Typological Outliers in Italo-Romance Intonation","authors":["Elinor Payne","Angelo Dian"],"year":2026,"doi":"10.21437/Interspeech.2026-2691","isca_url":"https://www.isca-archive.org/interspeech_2026/payne26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/payne26_interspeech.pdf","session":"Prominence, Stress and Focus","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["University of Oxford"],"funding":["Gladys Krieble Delmas Foundation","Economic and Social Research Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"payne26_interspeech","category":"phonetics-linguistics","institutions":["University of Oxford"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2691","pdf":"https://www.isca-archive.org/interspeech_2026/payne26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/payne26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/payne26_interspeech/markdown.md"},{"id":"pedro26_interspeech","title":"How Bilingual Are SSL Speech Models? Cross-Lingual Probing of Articulatory Encoding with Finnish and Russian EMA","authors":["Ailín Pollio San Pedro","Tomi H. Kinnunen","Alexandre Nikolaev","Ruchi Pandey"],"year":2026,"doi":"10.21437/Interspeech.2026-1324","isca_url":"https://www.isca-archive.org/interspeech_2026/pedro26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pedro26_interspeech.pdf","session":"Speech Production and Perception 1","topics":["self-supervised","multilingual","phonetics"],"category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["University Grenoble Alpes","CNRS","Grenoble INP","University of Eastern Finland"],"funding":["THERADIA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pedro26_interspeech","category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["University Grenoble Alpes","CNRS","Grenoble INP","University of Eastern Finland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1324","pdf":"https://www.isca-archive.org/interspeech_2026/pedro26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pedro26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pedro26_interspeech/markdown.md"},{"id":"pekarekrosin26_interspeech","title":"MoDiCoL: A Modular Diagnostic Continual Learning Dataset for Robust Speech Recognition","authors":["Theresa Pekarek Rosin","Matthias Kerzel","Stefan Wermter"],"year":2026,"doi":"10.21437/Interspeech.2026-2111","isca_url":"https://www.isca-archive.org/interspeech_2026/pekarekrosin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pekarekrosin26_interspeech.pdf","session":"ASR Under Real-World Constraints: Streaming, Adaptation, and Efficiency","topics":["asr","self-supervised","dataset"],"category":"asr","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["University of Hamburg"],"funding":["Horizon Europe"],"code":{"url":"https://huggingface.co/datasets/TPekarekRosin/modicol","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pekarekrosin26_interspeech","category":"asr","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["University of Hamburg"],"code":"https://huggingface.co/datasets/TPekarekRosin/modicol","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2111","pdf":"https://www.isca-archive.org/interspeech_2026/pekarekrosin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pekarekrosin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pekarekrosin26_interspeech/markdown.md"},{"id":"peng26_interspeech","title":"Do Machines Listen Like Humans? A Temporal Benchmark for Phonological Competition in End-to-End ASR","authors":["Linkai Peng","Christian Brodbeck","Sahil Luthra","Kevin Brown","Jay Rueckl","Monty Escabi","David Gow","James S. Magnuson"],"year":2026,"doi":"10.21437/Interspeech.2026-401","isca_url":"https://www.isca-archive.org/interspeech_2026/peng26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/peng26_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["asr","evaluation","phonetics"],"category":"asr","labels":["streaming-real-time","dataset-or-benchmark-release"],"institutions":["University of Connecticut","McMaster University","Stony Brook University","Oregon State University","Massachusetts General Hospital","Basque Center on Cognition, Brain and Language","Ikerbasque"],"funding":["Spanish State Research Agency","National Science Foundation","National Institutes of Health"],"code":{"url":"https://comp-cogneuro-lang.github.io/listen-like-humans","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"peng26_interspeech","category":"asr","labels":["streaming-real-time","dataset-or-benchmark-release"],"institutions":["University of Connecticut","McMaster University","Stony Brook University","Oregon State University","Massachusetts General Hospital","Basque Center on Cognition, Brain and Language","Ikerbasque"],"code":"https://comp-cogneuro-lang.github.io/listen-like-humans","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-401","pdf":"https://www.isca-archive.org/interspeech_2026/peng26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26_interspeech/markdown.md"},{"id":"peng26b_interspeech","title":"TASU2: Controllable CTC Simulation for Alignment and Low-Resource Adaptation of Speech LLMs","authors":["Jing Peng","Chenghao Wang","Yi Yang","Lirong Qian","Junjie Li","Yu Xi","Shuai Wang","Kai Yu"],"year":2026,"doi":"10.21437/Interspeech.2026-866","isca_url":"https://www.isca-archive.org/interspeech_2026/peng26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/peng26b_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["asr","speech-llm","low-resource"],"category":"asr","labels":["low-resource","generative-model"],"institutions":["Shanghai Jiao Tong University","AISpeech Ltd","Nanjing University"],"funding":["China NSFC Projects","YangtzeRiver Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"peng26b_interspeech","category":"asr","labels":["low-resource","generative-model"],"institutions":["Shanghai Jiao Tong University","AISpeech Ltd","Nanjing University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-866","pdf":"https://www.isca-archive.org/interspeech_2026/peng26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26b_interspeech/markdown.md"},{"id":"peng26c_interspeech","title":"Probing the Layer-wise Geometry of Chinese Dialect Representations in Wav2Vec 2.0","authors":["Zhen Peng","Jiahong Yuan"],"year":2026,"doi":"10.21437/Interspeech.2026-975","isca_url":"https://www.isca-archive.org/interspeech_2026/peng26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/peng26c_interspeech.pdf","session":"Language and Dialect Recognition","topics":["self-supervised","multilingual","evaluation"],"category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["University of Science and Technology of China"],"funding":["USTC"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"peng26c_interspeech","category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["University of Science and Technology of China"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-975","pdf":"https://www.isca-archive.org/interspeech_2026/peng26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26c_interspeech/markdown.md"},{"id":"peng26d_interspeech","title":"MAC-SLU: Multi-Intent Automotive Cabin Spoken Language Understanding Benchmark","authors":["Yuezhang Peng","Chonghao Cai","Ziang Liu","Shuai Fan","Sheng Jiang","Hua Xu","Yuxin Liu","Sheng Wang","Qiguang Chen","Yao Li","Kele Xu","Kai Yu","Libo Qin","Xie Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1055","isca_url":"https://www.isca-archive.org/interspeech_2026/peng26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/peng26d_interspeech.pdf","session":"Spoken Language Understanding","topics":["speech-llm","spoken-language-understanding","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Shanghai Jiao Tong University","Central South University","AISpeech Co., Ltd","Shanghai Innovation Institute","Harbin Institute of Technology","Shanghai Aviation Electric Co., Ltd","National University of Defense Technology"],"funding":["National Natural Science Foundation of China","Shanghai Municipal Science and Technology Major Project","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"https://github.com/Gatsby-web/MAC_SLU","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"peng26d_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Shanghai Jiao Tong University","Central South University","AISpeech Co., Ltd","Shanghai Innovation Institute","Harbin Institute of Technology","Shanghai Aviation Electric Co., Ltd","National University of Defense Technology"],"code":"https://github.com/Gatsby-web/MAC_SLU","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1055","pdf":"https://www.isca-archive.org/interspeech_2026/peng26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26d_interspeech/markdown.md"},{"id":"peng26e_interspeech","title":"A Unified and Reproducible Experimentation Framework for Speech Understanding","authors":["Jing Peng","Junhao Du","Chenghao Wang","Hanqi Li","Yi Yang","Xiaoyu Gu","Guanyu Chen","Yixuan Wang","Haoyu Li","Zhangjie Zhao","Li Jiang","Haoran Wang","Wenming Tu","Yucheng Wang","Jiaqi Guo","Hui Zhang","Shuai Fan","Wenbin Jiang","Shuai Wang","Kai Yu"],"year":2026,"doi":"10.21437/Interspeech.2026-1225","isca_url":"https://www.isca-archive.org/interspeech_2026/peng26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/peng26e_interspeech.pdf","session":"Corpus Creation, Summarization and Understanding","topics":["speech-llm","asr","evaluation"],"category":"resources-evaluation","institutions":["Shanghai Jiao Tong University","AISpeech Ltd","ETH Zurich","Nanjing University","Hangzhou Dianzi University"],"funding":["China NSFC Projects","YangtzeRiver Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"https://sure-eval-framework.github.io/speechllm_series/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"peng26e_interspeech","category":"resources-evaluation","institutions":["Shanghai Jiao Tong University","AISpeech Ltd","ETH Zurich","Nanjing University","Hangzhou Dianzi University"],"code":"https://sure-eval-framework.github.io/speechllm_series/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1225","pdf":"https://www.isca-archive.org/interspeech_2026/peng26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26e_interspeech/markdown.md"},{"id":"peng26f_interspeech","title":"Cross-Lingual Speaker Verification with Self-Supervised Pre-Trained Models","authors":["Jinghan Peng","Yu Zheng","Weiqiang Wang","Jian Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-1799","isca_url":"https://www.isca-archive.org/interspeech_2026/peng26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/peng26f_interspeech.pdf","session":"Challenge - TidyVoice Challenge: Cross-Lingual Speaker Verification","topics":["speaker-verification","multilingual","self-supervised"],"category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Ant Group"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"peng26f_interspeech","category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Ant Group"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1799","pdf":"https://www.isca-archive.org/interspeech_2026/peng26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26f_interspeech/markdown.md"},{"id":"peng26g_interspeech","title":"Cross-modal Consistency Guidance for Robust Emotion Control in Auto-Regressive TTS Models","authors":["Yizhou Peng","Yukun Ma","Chong Zhang","Yi-Wen Chao","Chongjia Ni","Bin Ma","Eng Siong Chng"],"year":2026,"doi":"10.21437/Interspeech.2026-1986","isca_url":"https://www.isca-archive.org/interspeech_2026/peng26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/peng26g_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["tts","speech-llm","emotion-recognition"],"category":"tts","labels":["generative-model"],"institutions":["Nanyang Technological University","Alibaba","Alibaba-NTU Global e-Sustainability CorpLab"],"funding":["Agency for Science, Technology and Research","Alibaba Group","Nanyang Technological University"],"code":{"url":"https://pengyizhou.github.io/Emotional_tts_demo","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"peng26g_interspeech","category":"tts","labels":["generative-model"],"institutions":["Nanyang Technological University","Alibaba","Alibaba-NTU Global e-Sustainability CorpLab"],"code":"https://pengyizhou.github.io/Emotional_tts_demo","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1986","pdf":"https://www.isca-archive.org/interspeech_2026/peng26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26g_interspeech/markdown.md"},{"id":"peng26h_interspeech","title":"Discrete vs. Continuous: A Comprehensive Study of Unified Audio Understanding in LALMs","authors":["Jing Peng","Zichao Nie","Zhisheng Zhang","Jingran Xie","Zhiyong Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-2074","isca_url":"https://www.isca-archive.org/interspeech_2026/peng26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/peng26h_interspeech.pdf","session":"Audio & Speech Language Models: Evaluation, Representations, and Emerging Capabilities","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Tsinghua University"],"funding":["National Natural Science Foundation of China","Shenzhen Science and Technology Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"peng26h_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Tsinghua University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2074","pdf":"https://www.isca-archive.org/interspeech_2026/peng26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/peng26h_interspeech/markdown.md"},{"id":"penney26_interspeech","title":"Achieving voicelessness in coda stop contexts: Insights from combined electroglottography and laryngoscopy","authors":["Joshua Penney","Jae Hyun Kim","Dijana Dragicevich","Prue Gourley"],"year":2026,"doi":"10.21437/Interspeech.2026-2610","isca_url":"https://www.isca-archive.org/interspeech_2026/penney26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/penney26_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["phonetics","evaluation","dataset"],"category":"phonetics-linguistics","institutions":["Macquarie University"],"funding":["Early Career Researcher Enabling Scheme"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"penney26_interspeech","category":"phonetics-linguistics","institutions":["Macquarie University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2610","pdf":"https://www.isca-archive.org/interspeech_2026/penney26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/penney26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/penney26_interspeech/markdown.md"},{"id":"perezgonzalezdemartos26_interspeech","title":"Not Quite My Tempo: Voice Activity-aware Speech Synthesis for Lip-Synchronous Dubbing","authors":["Alejandro Pérez-González-de-Martos","Florian Lux","Angelina Elizarova","Milana Shkhanukova","Andreas Kellner","Mattia Antonino Di Gangi"],"year":2026,"doi":"10.21437/Interspeech.2026-1407","isca_url":"https://www.isca-archive.org/interspeech_2026/perezgonzalezdemartos26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/perezgonzalezdemartos26_interspeech.pdf","session":"Articulatory, EMG, and Visual Speech Generation","topics":["tts","speech-translation","multilingual"],"category":"tts","labels":["generative-model"],"institutions":["AppTek"],"code":{"url":"https://alexdemartos.github.io/NQMT_IS26","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"perezgonzalezdemartos26_interspeech","category":"tts","labels":["generative-model"],"institutions":["AppTek"],"code":"https://alexdemartos.github.io/NQMT_IS26","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1407","pdf":"https://www.isca-archive.org/interspeech_2026/perezgonzalezdemartos26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/perezgonzalezdemartos26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/perezgonzalezdemartos26_interspeech/markdown.md"},{"id":"pham26_interspeech","title":"VieSpeaker: A Large-Scale Vietnamese Speaker Recognition Dataset Beyond Visual Dependency","authors":["Viet Hoang Pham","Tran Trung Nguyen","Bao Thu Ho","Phuong Tuan Dat","Trang Thu Thi Nguyen"],"year":2026,"doi":"10.21437/Interspeech.2026-3449","isca_url":"https://www.isca-archive.org/interspeech_2026/pham26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pham26_interspeech.pdf","session":"Speaker Recognition and Verification","topics":["speaker-verification","speaker-diarization","low-resource"],"category":"speaker","labels":["dataset-or-benchmark-release"],"institutions":["Hanoi University of Science and Technology"],"funding":["Ministry of Education and Training of Vietnam"],"code":{"url":"https://huggingface.co/datasets/hustep-lab/VieSpeaker-Dataset","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pham26_interspeech","category":"speaker","labels":["dataset-or-benchmark-release"],"institutions":["Hanoi University of Science and Technology"],"code":"https://huggingface.co/datasets/hustep-lab/VieSpeaker-Dataset","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3449","pdf":"https://www.isca-archive.org/interspeech_2026/pham26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pham26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pham26_interspeech/markdown.md"},{"id":"phukan26_interspeech","title":"Bridging the Age Gap: Towards Detecting Neural Audio Codec Synthesized Elderly Speech Deepfake","authors":["Orchid Chetia Phukan","Girish","Mohd Mujtaba Akhtar","Chi-Chun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-2283","isca_url":"https://www.isca-archive.org/interspeech_2026/phukan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/phukan26_interspeech.pdf","session":"Speech Deepfake Detection, Attribution and Characterization","topics":["speech-deepfake","self-supervised","multilingual"],"category":"deepfake-security","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["National Tsing Hua University","University of Petroleum and Energy Studies","Veer Bahadur Singh Purvanchal University"],"funding":["National Science and Technology Council"],"code":{"url":"https://helixometry.github.io/ElderlyCodecFake/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"phukan26_interspeech","category":"deepfake-security","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["National Tsing Hua University","University of Petroleum and Energy Studies","Veer Bahadur Singh Purvanchal University"],"code":"https://helixometry.github.io/ElderlyCodecFake/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2283","pdf":"https://www.isca-archive.org/interspeech_2026/phukan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/phukan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/phukan26_interspeech/markdown.md"},{"id":"piao26_interspeech","title":"Unlocking In-Context Learning in Audio-Language Models from Decentralized Medical Audio","authors":["Ran Piao","Tsai-Ning Wang","Martijn den Dekker","Linda Moonen","Hareld Kemps","Yuan Lu","Aaqib Saeed"],"year":2026,"doi":"10.21437/Interspeech.2026-430","isca_url":"https://www.isca-archive.org/interspeech_2026/piao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/piao26_interspeech.pdf","session":"Medical Dialogue and Conversational Understanding","topics":["speech-llm","multilingual","self-supervised"],"category":"speech-llm-dialogue","labels":["low-resource","self-supervised"],"institutions":["Eindhoven University of Technology","Erasmus MC","Rijnstate Hospital","Maxima MC Hospital"],"funding":["NWO","Google.org","Google Cloud Research Credits program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"piao26_interspeech","category":"speech-llm-dialogue","labels":["low-resource","self-supervised"],"institutions":["Eindhoven University of Technology","Erasmus MC","Rijnstate Hospital","Maxima MC Hospital"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-430","pdf":"https://www.isca-archive.org/interspeech_2026/piao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/piao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/piao26_interspeech/markdown.md"},{"id":"pieniazek26_interspeech","title":"Detection of Incorrect Place of Articulation in Polish Sibilants Using Convolutional Autoencoders","authors":["Wojciech Pieniążek","Oliwia Skórzewska","Maria Filipek","Zuzanna Miodońska"],"year":2026,"doi":"10.21437/Interspeech.2026-2624","isca_url":"https://www.isca-archive.org/interspeech_2026/pieniazek26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pieniazek26_interspeech.pdf","session":"Child Speech and Health","topics":["paralinguistics","evaluation","low-resource"],"category":"health-clinical","labels":["low-resource"],"institutions":["Silesian University of Technology"],"funding":["National Science Centre, Poland","European Union","Ministry of Science and Higher Education, Poland","National Centre for Research and Development"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pieniazek26_interspeech","category":"health-clinical","labels":["low-resource"],"institutions":["Silesian University of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2624","pdf":"https://www.isca-archive.org/interspeech_2026/pieniazek26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pieniazek26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pieniazek26_interspeech/markdown.md"},{"id":"pine26_interspeech","title":"Two Lessons Learned from the SGILE project: Efficient Building and Evaluation of TTS Voices","authors":["Aidan Pine","Korin Richmond"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/pine26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pine26_interspeech.pdf","session":"Speech Synthesis, Voice Conversion and Audio Generation","topics":["tts","low-resource","evaluation"],"category":"tts","labels":["low-resource","generative-model"],"institutions":["National Research Council","University of Edinburgh"],"funding":["National Research Council of Canada","UK ESRC Impact Acceleration Award"],"code":{"url":"https://github.com/EveryVoiceTTS/EveryVoice","stars":45,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pine26_interspeech","category":"tts","labels":["low-resource","generative-model"],"institutions":["National Research Council","University of Edinburgh"],"code":"https://github.com/EveryVoiceTTS/EveryVoice","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/pine26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/pine26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pine26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pine26_interspeech/markdown.md"},{"id":"piyadasa26_interspeech","title":"Morphoacoustic Modeling of a Dynamic 3D Vocal Tract Using MRI-Constrained Deformations and FEM Acoustics","authors":["Tharinda Piyadasa","Michael Proctor","Tünde Szalay","Joan Glaunès","Amelia Gully","Kirrie Ballard","Emily Kiff","Naeim Sanaei","Sheryl Foster","David Waddington","Craig Jin"],"year":2026,"doi":"10.21437/Interspeech.2026-1890","isca_url":"https://www.isca-archive.org/interspeech_2026/piyadasa26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/piyadasa26_interspeech.pdf","session":"Methods and Data for Vocal Tract Shape and Articulation Analysis","topics":["phonetics","evaluation","dataset"],"category":"phonetics-linguistics","institutions":["University of Sydney","Macquarie University","Universite Paris Cite","University of York","Westmead Hospital"],"funding":["Australian Research Council","National Health and Medical Research Council"],"code":{"url":"https://github.com/TharindaDilshan/vocal-tract-acoustics-comsol-fem","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"piyadasa26_interspeech","category":"phonetics-linguistics","institutions":["University of Sydney","Macquarie University","Universite Paris Cite","University of York","Westmead Hospital"],"code":"https://github.com/TharindaDilshan/vocal-tract-acoustics-comsol-fem","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1890","pdf":"https://www.isca-archive.org/interspeech_2026/piyadasa26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/piyadasa26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/piyadasa26_interspeech/markdown.md"},{"id":"pizarro26_interspeech","title":"Lightweight Detection and Model Attribution of Synthetic Speech via Residual Statistical Fingerprints","authors":["Matías Pizarro","Mike Laszkiewicz","Dorothea Kolossa","Asja Fischer"],"year":2026,"doi":"10.21437/Interspeech.2026-1361","isca_url":"https://www.isca-archive.org/interspeech_2026/pizarro26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pizarro26_interspeech.pdf","session":"Spoofing, Deepfake Detection and Watermarking","topics":["audio-deepfake","speaker-verification","self-supervised"],"category":"deepfake-security","institutions":["Ruhr University Bochum","LKA NRW","Technische Universität Berlin"],"funding":["Deutsche Forschungsgemeinschaft","Ministry of Culture and Science of North Rhine-Westphalia"],"code":{"url":"https://github.com/matiuste/RSF","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pizarro26_interspeech","category":"deepfake-security","institutions":["Ruhr University Bochum","LKA NRW","Technische Universität Berlin"],"code":"https://github.com/matiuste/RSF","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1361","pdf":"https://www.isca-archive.org/interspeech_2026/pizarro26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pizarro26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pizarro26_interspeech/markdown.md"},{"id":"ploujnikov26_interspeech","title":"HybridCodec: Modeling Discrete and Continuous Representations For Efficient Speech Language Models","authors":["Artem Ploujnikov","Francesco Verdini","Samir Sadok","Mirco Ravanelli"],"year":2026,"doi":"10.21437/Interspeech.2026-2784","isca_url":"https://www.isca-archive.org/interspeech_2026/ploujnikov26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ploujnikov26_interspeech.pdf","session":"Audio Language Models","topics":["tts","asr","self-supervised"],"category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised","generative-model"],"institutions":["Mila, Quebec AI Institute","Concordia University","Sapienza University of Rome","Inria","Universite Grenoble Alpes CNRS"],"funding":["NSERC","Digital Research Alliance of Canada","Translated","Apple","VisaSpeech Inria Associated Team"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ploujnikov26_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised","generative-model"],"institutions":["Mila, Quebec AI Institute","Concordia University","Sapienza University of Rome","Inria","Universite Grenoble Alpes CNRS"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2784","pdf":"https://www.isca-archive.org/interspeech_2026/ploujnikov26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ploujnikov26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ploujnikov26_interspeech/markdown.md"},{"id":"pludra26_interspeech","title":"Automatic Assessment of L2 Speech Intelligibility: Segmental Error Ranking","authors":["Agnieszka Pludra","Izabela Krysińska","Matuesz Jekiel"],"year":2026,"doi":"10.21437/Interspeech.2026-2522","isca_url":"https://www.isca-archive.org/interspeech_2026/pludra26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pludra26_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["speech-enhancement","evaluation","low-resource"],"category":"applications-other","institutions":["Pearson Central Europe","Adam Mickiewicz University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pludra26_interspeech","category":"applications-other","institutions":["Pearson Central Europe","Adam Mickiewicz University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2522","pdf":"https://www.isca-archive.org/interspeech_2026/pludra26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pludra26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pludra26_interspeech/markdown.md"},{"id":"pokel26_interspeech","title":"Data-Efficient ASR Personalization for Non-Normative Speech Using an Uncertainty-Based Phoneme Difficulty Score for Guided Sampling","authors":["Niclas Pokel","Pehuén Moure","Roman Böehringer","Yingqiang Gao"],"year":2026,"doi":"10.21437/Interspeech.2026-776","isca_url":"https://www.isca-archive.org/interspeech_2026/pokel26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pokel26_interspeech.pdf","session":"Assistive Technologies 1","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Zurich","ETH Zurich","Technical University of Munich"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pokel26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Zurich","ETH Zurich","Technical University of Munich"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-776","pdf":"https://www.isca-archive.org/interspeech_2026/pokel26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pokel26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pokel26_interspeech/markdown.md"},{"id":"polak26_interspeech","title":"Better Late Than Never: Meta-Evaluation of Latency Metrics for Simultaneous Speech-to-Text Translation","authors":["Peter Polák","Sara Papi","Luisa Bentivogli","Ondřej Bojar"],"year":2026,"doi":"10.21437/Interspeech.2026-575","isca_url":"https://www.isca-archive.org/interspeech_2026/polak26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/polak26_interspeech.pdf","session":"Multilingual Speech 1","topics":["speech-translation","evaluation","asr"],"category":"translation","labels":["multilingual","streaming-real-time"],"institutions":["Charles University","Fondazione Bruno Kessler","AppTek"],"funding":["Project OP JAK Mezisektorová spolupráce","European Union's Horizon research and innovation programme"],"code":{"url":"https://github.com/pe-trik/OmniSTEval","stars":11,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"polak26_interspeech","category":"translation","labels":["multilingual","streaming-real-time"],"institutions":["Charles University","Fondazione Bruno Kessler","AppTek"],"code":"https://github.com/pe-trik/OmniSTEval","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-575","pdf":"https://www.isca-archive.org/interspeech_2026/polak26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/polak26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/polak26_interspeech/markdown.md"},{"id":"poli26_interspeech","title":"DiscoPhon: Benchmarking the Unsupervised Discovery of Phoneme Inventories With Discrete Speech Units","authors":["Maxime Poli","Manel Khentout","Angelo Ortiz Tandazo","Ewan Dunbar","Emmanuel Chemla","Emmanuel Dupoux"],"year":2026,"doi":"10.21437/Interspeech.2026-2791","isca_url":"https://www.isca-archive.org/interspeech_2026/poli26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/poli26_interspeech.pdf","session":"Speech Benchmarks, Evaluation, and Resources","topics":["self-supervised","multilingual","evaluation"],"category":"resources-evaluation","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["École Normale Supérieure","École des Hautes Études en Sciences Sociales","Centre National de la Recherche Scientifique","Université PSL","University of Toronto"],"funding":["Agence Nationale de la Recherche","Agence de l’Innovation de Défense","European Research Council"],"code":{"url":"https://benchmarks.cognitive-ml.fr/discophon","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"poli26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release"],"institutions":["École Normale Supérieure","École des Hautes Études en Sciences Sociales","Centre National de la Recherche Scientifique","Université PSL","University of Toronto"],"code":"https://benchmarks.cognitive-ml.fr/discophon","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2791","pdf":"https://www.isca-archive.org/interspeech_2026/poli26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/poli26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/poli26_interspeech/markdown.md"},{"id":"polle26_interspeech","title":"Synthetic Speech, Real Signal: Paralinguistic Preservation and Cross-Lingual Augmentation via Voice Cloning","authors":["Roseline Polle","Owen Parsons","George Fairs","Luis Miguel San Martin Fernandez","Cole Looney","Xiaoliang Wu","Alexandra Livia Georgescu","Stefano Goria"],"year":2026,"doi":"10.21437/Interspeech.2026-2993","isca_url":"https://www.isca-archive.org/interspeech_2026/polle26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/polle26_interspeech.pdf","session":"Multilingual and Cross-Lingual Paralinguistic Analysis and Processing","topics":["paralinguistics","emotion-recognition","low-resource"],"category":"paralinguistics-emotion","labels":["low-resource","multilingual","generative-model"],"institutions":["thymia","University of Edinburgh","University of Southampton"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"polle26_interspeech","category":"paralinguistics-emotion","labels":["low-resource","multilingual","generative-model"],"institutions":["thymia","University of Edinburgh","University of Southampton"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2993","pdf":"https://www.isca-archive.org/interspeech_2026/polle26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/polle26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/polle26_interspeech/markdown.md"},{"id":"polok26_interspeech","title":"Mind the Gap: Impact of Synthetic Conversational Data on Multi-Talker ASR and Speaker Diarization","authors":["Alexander Polok","Ivan Medennikov","Honza Černocký","Shinji Watanabe","Lukáš Burget","Samuele Cornell"],"year":2026,"doi":"10.21437/Interspeech.2026-443","isca_url":"https://www.isca-archive.org/interspeech_2026/polok26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/polok26_interspeech.pdf","session":"Multi-Talker ASR & Speaker Diarization","topics":["asr","speaker-diarization","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Brno University of Technology","Carnegie Mellon University","NVIDIA"],"funding":["Ministry of Education, Youth and Sports of the Czech Republic","Brno Ph.D. Talent Scholarship Programme"],"code":{"url":"https://github.com/popcornell/FastMSS","stars":44,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"polok26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Brno University of Technology","Carnegie Mellon University","NVIDIA"],"code":"https://github.com/popcornell/FastMSS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-443","pdf":"https://www.isca-archive.org/interspeech_2026/polok26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/polok26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/polok26_interspeech/markdown.md"},{"id":"polok26b_interspeech","title":"Grounding Spoken LLMs in Multi-Speaker Audio via Diarization Conditioning","authors":["Alexander Polok","Samuele Cornell","Sathvik Udupa","Honza Černocký","Shinji Watanabe","Lukáš Burget"],"year":2026,"doi":"10.21437/Interspeech.2026-445","isca_url":"https://www.isca-archive.org/interspeech_2026/polok26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/polok26b_interspeech.pdf","session":"Multi-Talker ASR & Speaker Diarization","topics":["speech-recognition","speech-llm","spoken-language-understanding"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Brno University of Technology","Carnegie Mellon University"],"funding":["Ministry of Education, Youth and Sports of the Czech Republic","Brno Ph.D. Talent Scholarship Programme","e-INFRA CZ"],"code":{"url":"https://github.com/BUTSpeechFIT/Dixtral","stars":17,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"polok26b_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Brno University of Technology","Carnegie Mellon University"],"code":"https://github.com/BUTSpeechFIT/Dixtral","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-445","pdf":"https://www.isca-archive.org/interspeech_2026/polok26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/polok26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/polok26b_interspeech/markdown.md"},{"id":"poncelet26_interspeech","title":"Speech Encoder Fusion for LLM-based Automatic Speech Recognition","authors":["Jakob Poncelet","Hugo Van hamme"],"year":2026,"doi":"10.21437/Interspeech.2026-1039","isca_url":"https://www.isca-archive.org/interspeech_2026/poncelet26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/poncelet26_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 2","topics":["asr","speech-llm","speaker-diarization"],"category":"asr","labels":["self-supervised"],"institutions":["KU Leuven"],"funding":["Research Foundation Flanders","Flemish Government","Flanders AI Research Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"poncelet26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["KU Leuven"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1039","pdf":"https://www.isca-archive.org/interspeech_2026/poncelet26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/poncelet26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/poncelet26_interspeech/markdown.md"},{"id":"poncelet26b_interspeech","title":"Towards Deep Contextual Reasoning from Broad Descriptions for ASR with Speech-LLM via Metadata-Driven Reasoning Chains","authors":["Jakob Poncelet","Hugo Van hamme"],"year":2026,"doi":"10.21437/Interspeech.2026-1041","isca_url":"https://www.isca-archive.org/interspeech_2026/poncelet26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/poncelet26b_interspeech.pdf","session":"Reasoning with Speech/Audio Language Models","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["self-supervised"],"institutions":["KU Leuven"],"funding":["Research Foundation Flanders","Flemish Government","Flanders AI Research Program"],"code":{"url":"https://huggingface.co/datasets/kul-speech-lab/contextual-reasoning-speechllm","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"poncelet26b_interspeech","category":"asr","labels":["self-supervised"],"institutions":["KU Leuven"],"code":"https://huggingface.co/datasets/kul-speech-lab/contextual-reasoning-speechllm","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1041","pdf":"https://www.isca-archive.org/interspeech_2026/poncelet26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/poncelet26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/poncelet26b_interspeech/markdown.md"},{"id":"popescu26_interspeech","title":"lisero: An interactive practice app for learning Romanian Sign Language","authors":["Anisia Popescu","Serban Din","Marinela Axinte"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/popescu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/popescu26_interspeech.pdf","session":"Speech and Language Learning Technologies","topics":["low-resource","multilingual","dataset"],"category":"applications-other","labels":["low-resource"],"institutions":["Université Paris 8","Solid Technologies","Babeș-Bolyai University","CODA - Farmecul Tăcerii Foundation"],"funding":["Fundatia CODA - Farmecul Tacerii"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"popescu26_interspeech","category":"applications-other","labels":["low-resource"],"institutions":["Université Paris 8","Solid Technologies","Babeș-Bolyai University","CODA - Farmecul Tăcerii Foundation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/popescu26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/popescu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/popescu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/popescu26_interspeech/markdown.md"},{"id":"porupski26_interspeech","title":"Umm... With Transformers? Insights from Filled Pause Use across Four Slavic Parliaments","authors":["Ivan Porupski","Branimir Dropuljić","Nikola Ljubešić"],"year":2026,"doi":"10.21437/Interspeech.2026-3262","isca_url":"https://www.isca-archive.org/interspeech_2026/porupski26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/porupski26_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["paralinguistics","self-supervised","evaluation"],"category":"paralinguistics-emotion","labels":["multilingual"],"institutions":["Jožef Stefan Institute","TransUnion","University of Zagreb","University of Ljubljana","Institute of Contemporary History"],"funding":["ARIS Slovenian Research and Innovation Agency"],"code":{"url":"https://clarinsi.github.io/parlaspeech/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"porupski26_interspeech","category":"paralinguistics-emotion","labels":["multilingual"],"institutions":["Jožef Stefan Institute","TransUnion","University of Zagreb","University of Ljubljana","Institute of Contemporary History"],"code":"https://clarinsi.github.io/parlaspeech/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3262","pdf":"https://www.isca-archive.org/interspeech_2026/porupski26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/porupski26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/porupski26_interspeech/markdown.md"},{"id":"powelldavies26_interspeech","title":"An investigation of post-stop breathiness in Australian English","authors":["Thomas Powell-Davies","Rosey Billington"],"year":2026,"doi":"10.21437/Interspeech.2026-1555","isca_url":"https://www.isca-archive.org/interspeech_2026/powelldavies26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/powelldavies26_interspeech.pdf","session":"Voice Quality Aspects of Speech","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["Australian National University"],"funding":["Australian Government Research Training Program","HDR funding from the School of Literature, Language and Linguistics at the Australian National University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"powelldavies26_interspeech","category":"phonetics-linguistics","institutions":["Australian National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1555","pdf":"https://www.isca-archive.org/interspeech_2026/powelldavies26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/powelldavies26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/powelldavies26_interspeech/markdown.md"},{"id":"pratiwi26_interspeech","title":"Adaptation to Room Acoustics in Understanding Vocoded Speech: A Comparison Between Listeners With Varying Immersion Age","authors":["Epri Pratiwi","C. T. Justine Hui","Yusuke Hioka"],"year":2026,"doi":"10.21437/Interspeech.2026-113","isca_url":"https://www.isca-archive.org/interspeech_2026/pratiwi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/pratiwi26_interspeech.pdf","session":"Assistive Technologies 1","topics":["speech-enhancement","paralinguistics","evaluation"],"category":"health-clinical","labels":["robustness-noise"],"institutions":["University of Auckland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"pratiwi26_interspeech","category":"health-clinical","labels":["robustness-noise"],"institutions":["University of Auckland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-113","pdf":"https://www.isca-archive.org/interspeech_2026/pratiwi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/pratiwi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/pratiwi26_interspeech/markdown.md"},{"id":"puggaardrode26_interspeech","title":"Automatic identification of the onset of creaky voice according to F0 instability","authors":["Rasmus Puggaard-Rode","Joshua Penney"],"year":2026,"doi":"10.21437/Interspeech.2026-1430","isca_url":"https://www.isca-archive.org/interspeech_2026/puggaardrode26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/puggaardrode26_interspeech.pdf","session":"Audio signal analysis","topics":["phonetics","paralinguistics","evaluation"],"category":"phonetics-linguistics","institutions":["University of Oxford","Macquarie University"],"code":{"url":"https://osf.io/C8627","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"puggaardrode26_interspeech","category":"phonetics-linguistics","institutions":["University of Oxford","Macquarie University"],"code":"https://osf.io/C8627","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1430","pdf":"https://www.isca-archive.org/interspeech_2026/puggaardrode26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/puggaardrode26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/puggaardrode26_interspeech/markdown.md"},{"id":"purser26_interspeech","title":"Variation and change in dynamicity of Australian English diphthongs in Sydney","authors":["Benjamin Purser"],"year":2026,"doi":"10.21437/Interspeech.2026-3206","isca_url":"https://www.isca-archive.org/interspeech_2026/purser26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/purser26_interspeech.pdf","session":"Diphthongs and Monophthongs","topics":["phonetics","evaluation","dataset"],"category":"phonetics-linguistics","institutions":["Australian National University"],"funding":["Australian Government Research Training Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"purser26_interspeech","category":"phonetics-linguistics","institutions":["Australian National University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3206","pdf":"https://www.isca-archive.org/interspeech_2026/purser26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/purser26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/purser26_interspeech/markdown.md"},{"id":"qi26_interspeech","title":"MuVAP: Multimodal Multiparty Voice Activity Projection for Turn-taking Prediction in the wild","authors":["Haotian Qi","Gabriel Skantze"],"year":2026,"doi":"10.21437/Interspeech.2026-1381","isca_url":"https://www.isca-archive.org/interspeech_2026/qi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/qi26_interspeech.pdf","session":"LLMs and Conversational Interaction","topics":["speech-llm","self-supervised","multilingual"],"category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["KTH Royal Institute of Technology"],"funding":["Wallenberg AI, Autonomous Systems and Software Program (WASP)","Knut and Alice Wallenberg Foundation","Swedish Research Council"],"code":{"url":"https://github.com/Haotian-Qi/MuVAP","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"qi26_interspeech","category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["KTH Royal Institute of Technology"],"code":"https://github.com/Haotian-Qi/MuVAP","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1381","pdf":"https://www.isca-archive.org/interspeech_2026/qi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/qi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/qi26_interspeech/markdown.md"},{"id":"qin26_interspeech","title":"DGS-MLDG: Domain Gradient Surgery Guided Meta-Learning for Domain Generalization in Speech Deepfake Detection","authors":["Siqing Qin","Kong Aik Lee","Youzhi Tu","Eng Siong Chng","Man-Wai Mak"],"year":2026,"doi":"10.21437/Interspeech.2026-1042","isca_url":"https://www.isca-archive.org/interspeech_2026/qin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/qin26_interspeech.pdf","session":"Speech Deepfake Detection, Attribution and Characterization","topics":["speech-deepfake-detection","self-supervised","speaker-verification"],"category":"deepfake-security","labels":["robustness-noise"],"institutions":["Hong Kong Polytechnic University","Nanyang Technological University"],"funding":["Innovation and Technology Fund of the Hong Kong SAR","National Key R&D Program of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"qin26_interspeech","category":"deepfake-security","labels":["robustness-noise"],"institutions":["Hong Kong Polytechnic University","Nanyang Technological University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1042","pdf":"https://www.isca-archive.org/interspeech_2026/qin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/qin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/qin26_interspeech/markdown.md"},{"id":"qin26b_interspeech","title":"Domain-Adaptive Dual-Gating Mixture of Experts for Generalizable Speech Deepfake Detection","authors":["Siqing Qin","Zhe Li","Kong Aik Lee","Man-Wai Mak"],"year":2026,"doi":"10.21437/Interspeech.2026-1778","isca_url":"https://www.isca-archive.org/interspeech_2026/qin26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/qin26b_interspeech.pdf","session":"Speech Deepfake Detection, Attribution and Characterization","topics":["speech-deepfake-detection","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised","robustness-noise"],"institutions":["Hong Kong Polytechnic University","University of Hong Kong"],"funding":["Innovation and Technology Fund of the Hong Kong SAR","National Key R&D Program of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"qin26b_interspeech","category":"deepfake-security","labels":["self-supervised","robustness-noise"],"institutions":["Hong Kong Polytechnic University","University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1778","pdf":"https://www.isca-archive.org/interspeech_2026/qin26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/qin26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/qin26b_interspeech/markdown.md"},{"id":"qiu26_interspeech","title":"Mixture of Spectral Experts for Audio Deepfake Detection","authors":["Yaxuan Qiu","Zhe Li","Mieradilijiang Maimaiti","Zunwang Ke","Wushour Silamu"],"year":2026,"doi":"10.21437/Interspeech.2026-661","isca_url":"https://www.isca-archive.org/interspeech_2026/qiu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/qiu26_interspeech.pdf","session":"Speech Deepfake Detection: Robustness, Generalization, Attribution","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["Xinjiang University","University of Hong Kong"],"funding":["National Natural Science Foundation of China","Xinjiang \"Tianchi Talent\" Recruitment and Introduction Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"qiu26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["Xinjiang University","University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-661","pdf":"https://www.isca-archive.org/interspeech_2026/qiu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/qiu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/qiu26_interspeech/markdown.md"},{"id":"quamer26_interspeech","title":"Privacy and quality trade-off in real-time speaker anonymization via editing of age and sex attributes","authors":["Waris Quamer","Ricardo Gutierrez-Osuna"],"year":2026,"doi":"10.21437/Interspeech.2026-2741","isca_url":"https://www.isca-archive.org/interspeech_2026/quamer26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/quamer26_interspeech.pdf","session":"Voice Editing","topics":["voice-conversion","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["Texas A&M University"],"funding":["Intelligence Advanced Research Projects Activity","Department of Interior/Interior Business Center"],"code":{"url":"https://anonymousis23.github.io/demos/pca-voice-editing/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"quamer26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["Texas A&M University"],"code":"https://anonymousis23.github.io/demos/pca-voice-editing/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2741","pdf":"https://www.isca-archive.org/interspeech_2026/quamer26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/quamer26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/quamer26_interspeech/markdown.md"},{"id":"rachumallu26_interspeech","title":"QuadVAD: Fine-Grained Speech Detection with a Compact Architecture","authors":["Ramakrishna Chaitanya Rachumallu","Nivedita Chennupati","Ankit Gupta","Balaji Padmanaban","ParvathiPriyanka Bolla","Harish Rajamani","Naveen Ambati"],"year":2026,"doi":"10.21437/Interspeech.2026-2009","isca_url":"https://www.isca-archive.org/interspeech_2026/rachumallu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rachumallu26_interspeech.pdf","session":"Acoustic Event Detection 3","topics":["speech-enhancement","keyword-spotting","on-device"],"category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Meeami Technologies"],"code":{"url":"https://github.com/Ramakrishna-Chaitanya/QuadVAD-Interspeech2026","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rachumallu26_interspeech","category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Meeami Technologies"],"code":"https://github.com/Ramakrishna-Chaitanya/QuadVAD-Interspeech2026","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2009","pdf":"https://www.isca-archive.org/interspeech_2026/rachumallu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rachumallu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rachumallu26_interspeech/markdown.md"},{"id":"rackauckas26_interspeech","title":"AdaptLingo: A Speech-to-Speech English Practice System with Fluency-Adaptive Responses","authors":["Zackary Rackauckas"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/rackauckas26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rackauckas26_interspeech.pdf","session":"Speech and Language Learning Technologies","topics":["spoken-language-understanding","tts","evaluation"],"category":"speech-llm-dialogue","institutions":["RoleGaku","Columbia University"],"code":{"url":"https://github.com/zackrack/AdaptLingo","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rackauckas26_interspeech","category":"speech-llm-dialogue","institutions":["RoleGaku","Columbia University"],"code":"https://github.com/zackrack/AdaptLingo","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/rackauckas26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/rackauckas26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rackauckas26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rackauckas26_interspeech/markdown.md"},{"id":"rackauckas26b_interspeech","title":"A Speech-First Character Interface for Stylized Japanese Dialogue Practice","authors":["Zackary Rackauckas"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/rackauckas26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rackauckas26b_interspeech.pdf","session":"Speech and Language Learning Technologies","topics":["tts","spoken-language-understanding","speech-llm"],"category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["RoleGaku"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rackauckas26b_interspeech","category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["RoleGaku"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/rackauckas26b_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/rackauckas26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rackauckas26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rackauckas26b_interspeech/markdown.md"},{"id":"rafat26_interspeech","title":"Dynamic Block-Online Streaming ASR for Low-Resource Agglutinative Code-Switching Speech with Morphology-Aware Evaluation","authors":["Kazi Rafat","Afifa Imran","Md. Ismail Hossain","Md. Romzan Ali","Fuad Rahman","Sifat Momen","Shafin Rahman","Nabeel Mohammed"],"year":2026,"doi":"10.21437/Interspeech.2026-3334","isca_url":"https://www.isca-archive.org/interspeech_2026/rafat26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rafat26_interspeech.pdf","session":"Code-Switching ASR","topics":["asr","low-resource","multilingual"],"category":"asr","labels":["low-resource","multilingual","streaming-real-time"],"institutions":["North South University","Apurba Technologies"],"code":{"url":"https://github.com/Dynamic-ASR","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rafat26_interspeech","category":"asr","labels":["low-resource","multilingual","streaming-real-time"],"institutions":["North South University","Apurba Technologies"],"code":"https://github.com/Dynamic-ASR","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3334","pdf":"https://www.isca-archive.org/interspeech_2026/rafat26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rafat26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rafat26_interspeech/markdown.md"},{"id":"rahman26_interspeech","title":"Pashto Common Voice: Building the First Open Speech Corpus for a 60-Million-Speaker Low-Resource Language","authors":["Hanif Rahman","Shafeeq ur Rehman"],"year":2026,"doi":"10.21437/Interspeech.2026-1432","isca_url":"https://www.isca-archive.org/interspeech_2026/rahman26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rahman26_interspeech.pdf","session":"Indigenous Voices in Speech Science and Technology","topics":["asr","low-resource","dataset"],"category":"resources-evaluation","labels":["low-resource","self-supervised","dataset-or-benchmark-release"],"institutions":["Pashto DAO"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rahman26_interspeech","category":"resources-evaluation","labels":["low-resource","self-supervised","dataset-or-benchmark-release"],"institutions":["Pashto DAO"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1432","pdf":"https://www.isca-archive.org/interspeech_2026/rahman26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rahman26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rahman26_interspeech/markdown.md"},{"id":"rahman26b_interspeech","title":"Voice Privacy from an Attribute-based Perspective","authors":["Mehtab Ur Rahman","Martha Larson","Cristian Tejedor-Garcia"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/rahman26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rahman26b_interspeech.pdf","session":"Speaker Privacy and Anonymization","topics":["speech-anonymization","speaker-verification","paralinguistics"],"category":"deepfake-security","institutions":["Radboud University"],"funding":["Dutch Research Council","NGF AiNed Fellowship Grants"],"code":{"url":"https://github.com/Mehtab9/Voice-Privacy-from-an-Attribute-based-Perspective","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rahman26b_interspeech","category":"deepfake-security","institutions":["Radboud University"],"code":"https://github.com/Mehtab9/Voice-Privacy-from-an-Attribute-based-Perspective","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/rahman26b_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/rahman26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rahman26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rahman26b_interspeech/markdown.md"},{"id":"rajaa26_interspeech","title":"DualTurn: Learning Turn-Taking from Dual-Channel Generative Speech Pretraining","authors":["Shangeth Rajaa"],"year":2026,"doi":"10.21437/Interspeech.2026-2424","isca_url":"https://www.isca-archive.org/interspeech_2026/rajaa26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rajaa26_interspeech.pdf","session":"Turn-taking","topics":["speech-llm","spoken-language-understanding","self-supervised"],"category":"speech-llm-dialogue","labels":["self-supervised","generative-model"],"institutions":["Anyreach AI"],"code":{"url":"https://github.com/anyreachai/dualturn","stars":11,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rajaa26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised","generative-model"],"institutions":["Anyreach AI"],"code":"https://github.com/anyreachai/dualturn","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2424","pdf":"https://www.isca-archive.org/interspeech_2026/rajaa26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rajaa26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rajaa26_interspeech/markdown.md"},{"id":"rakotoarivony26_interspeech","title":"Evolution Strategy-Based Calibration for Low-Bit Quantization of Speech Models","authors":["Lucas RAKOTOARIVONY"],"year":2026,"doi":"10.21437/Interspeech.2026-119","isca_url":"https://www.isca-archive.org/interspeech_2026/rakotoarivony26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rakotoarivony26_interspeech.pdf","session":"Acoustic Signal Analysis and Generation","topics":["self-supervised","on-device","evaluation"],"category":"applications-other","labels":["efficient-on-device","self-supervised"],"institutions":["Thales"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rakotoarivony26_interspeech","category":"applications-other","labels":["efficient-on-device","self-supervised"],"institutions":["Thales"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-119","pdf":"https://www.isca-archive.org/interspeech_2026/rakotoarivony26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rakotoarivony26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rakotoarivony26_interspeech/markdown.md"},{"id":"ramapuram26_interspeech","title":"Scaling Properties of Continuous Diffusion Spoken Language Models","authors":["Jason Ramapuram","Eeshan Gunesh Dhekane","Amitis Shidani","Dan Busbridge","Bogdan Mazoure","Zijin Gu","Russ Webb","Tatiana Likhomanenko","Navdeep Jaitly"],"year":2026,"doi":"10.21437/Interspeech.2026-2980","isca_url":"https://www.isca-archive.org/interspeech_2026/ramapuram26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ramapuram26_interspeech.pdf","session":"Generative Audio and Music","topics":["speech-llm","self-supervised","tts"],"category":"speech-llm-dialogue","labels":["self-supervised","generative-model"],"institutions":["Apple"],"code":{"url":"https://github.com/apple/ml-diffuslm","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ramapuram26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised","generative-model"],"institutions":["Apple"],"code":"https://github.com/apple/ml-diffuslm","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2980","pdf":"https://www.isca-archive.org/interspeech_2026/ramapuram26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ramapuram26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ramapuram26_interspeech/markdown.md"},{"id":"ramesh26_interspeech","title":"Duration-Aware Soft Targets for Text-Independent Supervised Phone Segmentation","authors":["Raghavan Ramesh","Meka Nani","Sri Rama Murty Kodukula"],"year":2026,"doi":"10.21437/Interspeech.2026-2568","isca_url":"https://www.isca-archive.org/interspeech_2026/ramesh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ramesh26_interspeech.pdf","session":"Speech signal analysis","topics":["phonetics","evaluation","speech-segmentation"],"category":"asr","institutions":["Indian Institute of Technology Hyderabad"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ramesh26_interspeech","category":"asr","institutions":["Indian Institute of Technology Hyderabad"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2568","pdf":"https://www.isca-archive.org/interspeech_2026/ramesh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ramesh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ramesh26_interspeech/markdown.md"},{"id":"ramonda26_interspeech","title":"Readability Does Not Predict Speech Recognition Errors: Contrasting Human and Machine Perception.","authors":["Baptiste Ramonda","Laurianne Sitbon","Julien Pinquier"],"year":2026,"doi":"10.21437/Interspeech.2026-2439","isca_url":"https://www.isca-archive.org/interspeech_2026/ramonda26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ramonda26_interspeech.pdf","session":"New Architecture and Analyses for ASR and Speech LMs","topics":["asr","evaluation","self-supervised"],"category":"asr","institutions":["IRIT","CNRS","Université de Toulouse","Queensland University of Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ramonda26_interspeech","category":"asr","institutions":["IRIT","CNRS","Université de Toulouse","Queensland University of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2439","pdf":"https://www.isca-archive.org/interspeech_2026/ramonda26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ramonda26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ramonda26_interspeech/markdown.md"},{"id":"rao26_interspeech","title":"A Causal Reference-Enhanced Keep-Speech Active Noise Control Method","authors":["Li Rao","Xiaobin Rong","Yu Sun","Yiming He","Kai Chen","Haishan Zou","Jing Lu"],"year":2026,"doi":"10.21437/Interspeech.2026-1608","isca_url":"https://www.isca-archive.org/interspeech_2026/rao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rao26_interspeech.pdf","session":"Active Noise and Echo Control, Sound Zones and Packet-Loss Concealment","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["streaming-real-time"],"institutions":["Nanjing University","Nanjing Institute of Advanced Artificial Intelligence","Samsung Electronics"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://github.com/RaoLi666/RSE_KSANC.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rao26_interspeech","category":"enhancement-separation","labels":["streaming-real-time"],"institutions":["Nanjing University","Nanjing Institute of Advanced Artificial Intelligence","Samsung Electronics"],"code":"https://github.com/RaoLi666/RSE_KSANC.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1608","pdf":"https://www.isca-archive.org/interspeech_2026/rao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rao26_interspeech/markdown.md"},{"id":"rathnayake26_interspeech","title":"Pā‑Kakare: The First Emotional Speech Database for Te Reo Māori","authors":["Himashi Rathnayake","Jesin James","Sally Akevai Nicholas","Gianna Leoni","C. I. Watson","Peter J Keegan"],"year":2026,"doi":"10.21437/Interspeech.2026-1543","isca_url":"https://www.isca-archive.org/interspeech_2026/rathnayake26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rathnayake26_interspeech.pdf","session":"Pacific Voices: Speech Science and Technology for the Languages of the Pacific Ocean","topics":["paralinguistics","emotion-recognition","dataset"],"category":"paralinguistics-emotion","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["University of Auckland","Te Hiku Media"],"funding":["Science for Technological Innovation National Science Challenge","Ministry of Business, Innovation and Employment","Te Hiku Media","University of Auckland"],"code":{"url":"https://speechresearch.auckland.ac.nz/maori-emotions","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rathnayake26_interspeech","category":"paralinguistics-emotion","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["University of Auckland","Te Hiku Media"],"code":"https://speechresearch.auckland.ac.nz/maori-emotions","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1543","pdf":"https://www.isca-archive.org/interspeech_2026/rathnayake26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rathnayake26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rathnayake26_interspeech/markdown.md"},{"id":"rathore26_interspeech","title":"SᴜTRA: Structurally-Unified Tokenization with Root Awareness","authors":["Vaibhav Rathore","Siddhant Gole","Dadhichi Telwadkar","Rooshil Bhatia","Maulik Ruparel","Siddharth Surekha","Neha Bhargava"],"year":2026,"doi":"10.21437/Interspeech.2026-291","isca_url":"https://www.isca-archive.org/interspeech_2026/rathore26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rathore26_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 2","topics":["multilingual","evaluation","dataset"],"category":"asr","labels":["multilingual"],"institutions":["Motilal Oswal Financial Services","Indian Institute of Technology Bombay"],"funding":["Motilal Oswal Financial Services"],"code":{"url":"https://mo-vaibhavr-43300.github.io/SuTRA/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rathore26_interspeech","category":"asr","labels":["multilingual"],"institutions":["Motilal Oswal Financial Services","Indian Institute of Technology Bombay"],"code":"https://mo-vaibhavr-43300.github.io/SuTRA/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-291","pdf":"https://www.isca-archive.org/interspeech_2026/rathore26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rathore26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rathore26_interspeech/markdown.md"},{"id":"rautenberg26_interspeech","title":"Hierarchical Conditional Continuous Normalizing Flows for Creaky Voice Editing under Speaker Identity Preservation","authors":["Frederik Rautenberg","Fritz Seebauer","Petra Wagner","Reinhold Haeb-Umbach"],"year":2026,"doi":"10.21437/Interspeech.2026-1341","isca_url":"https://www.isca-archive.org/interspeech_2026/rautenberg26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rautenberg26_interspeech.pdf","session":"Flow Matching for Speech Synthesis","topics":["voice-conversion","self-supervised","paralinguistics"],"category":"tts","labels":["generative-model"],"institutions":["Paderborn University","Bielefeld University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rautenberg26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Paderborn University","Bielefeld University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1341","pdf":"https://www.isca-archive.org/interspeech_2026/rautenberg26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rautenberg26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rautenberg26_interspeech/markdown.md"},{"id":"ravi26_interspeech","title":"Rank-Distance Based Confidence Estimation for ASR","authors":["Nagarathna Ravi","Madduri Aiswarya Lakshmi","Ragesh M","Rajalakshmi Elangovan"],"year":2026,"doi":"10.21437/Interspeech.2026-1355","isca_url":"https://www.isca-archive.org/interspeech_2026/ravi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ravi26_interspeech.pdf","session":"Robust ASR: Uncertainty and Confidence","topics":["asr","evaluation","self-supervised"],"category":"asr","labels":["multilingual","robustness-noise"],"institutions":["CSIR Fourth Paradigm Institute"],"funding":["Council of Scientific and Industrial Research","Anusandhan National Research Foundation"],"code":{"url":"https://github.com/Nagarathna-R/2026_RanDiS_Interspeech","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ravi26_interspeech","category":"asr","labels":["multilingual","robustness-noise"],"institutions":["CSIR Fourth Paradigm Institute"],"code":"https://github.com/Nagarathna-R/2026_RanDiS_Interspeech","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1355","pdf":"https://www.isca-archive.org/interspeech_2026/ravi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ravi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ravi26_interspeech/markdown.md"},{"id":"raybarman26_interspeech","title":"Towards a Phonology-Informed Evaluation of Multilingual TTS","authors":["Sneha Ray Barman","Neeraj Kumar Sharma","Shakuntala Mahanta"],"year":2026,"doi":"10.21437/Interspeech.2026-3311","isca_url":"https://www.isca-archive.org/interspeech_2026/raybarman26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/raybarman26_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["tts","phonetics","evaluation"],"category":"tts","labels":["multilingual"],"institutions":["Indian Institute of Technology Guwahati"],"code":{"url":"https://github.com/snehagitrep/TTSEvalVH_interspeech2026.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"raybarman26_interspeech","category":"tts","labels":["multilingual"],"institutions":["Indian Institute of Technology Guwahati"],"code":"https://github.com/snehagitrep/TTSEvalVH_interspeech2026.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3311","pdf":"https://www.isca-archive.org/interspeech_2026/raybarman26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/raybarman26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/raybarman26_interspeech/markdown.md"},{"id":"reitsema26_interspeech","title":"Returning the Turn: Do Backchannels Pattern More Like Turn-Holds or Turn-Changes Given Preceding Syntactic Completion and Boundary Tones?","authors":["Ariëlle Reitsema","Matthijs Westera","Yiya Chen","Johanneke Caspers"],"year":2026,"doi":"10.21437/Interspeech.2026-3530","isca_url":"https://www.isca-archive.org/interspeech_2026/reitsema26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/reitsema26_interspeech.pdf","session":"Perception and Appraisal of Prosody","topics":["prosody","evaluation","phonetics"],"category":"phonetics-linguistics","institutions":["Leiden University"],"funding":["Dutch Research Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"reitsema26_interspeech","category":"phonetics-linguistics","institutions":["Leiden University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3530","pdf":"https://www.isca-archive.org/interspeech_2026/reitsema26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/reitsema26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/reitsema26_interspeech/markdown.md"},{"id":"ren26_interspeech","title":"MOS-Bias: From Hidden Gender Bias to Gender-Aware Speech Quality Assessment","authors":["Wenze Ren","Yi-Cheng Lin","Wen-Chin Huang","Erica Cooper","Ryandhimas Zezario","Hsin-Min Wang","Hung-yi Lee","Yu Tsao"],"year":2026,"doi":"10.21437/Interspeech.2026-67","isca_url":"https://www.isca-archive.org/interspeech_2026/ren26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ren26_interspeech.pdf","session":"Speech Synthesis Evaluation 2","topics":["speech-enhancement","evaluation","self-supervised"],"category":"resources-evaluation","institutions":["National Taiwan University","Nagoya University","National Institute of Information and Communications Technology","Academia Sinica"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ren26_interspeech","category":"resources-evaluation","institutions":["National Taiwan University","Nagoya University","National Institute of Information and Communications Technology","Academia Sinica"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-67","pdf":"https://www.isca-archive.org/interspeech_2026/ren26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26_interspeech/markdown.md"},{"id":"ren26b_interspeech","title":"CoSTALA: Compositional Spatio-Temporal Audio-Language Alignment via Multi-Grain Hierarchical Contrastive Learning","authors":["Peiwei Ren","Jinbo Hu","Fang Kang","Shan Liang","Yin Cao"],"year":2026,"doi":"10.21437/Interspeech.2026-1110","isca_url":"https://www.isca-archive.org/interspeech_2026/ren26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ren26b_interspeech.pdf","session":"Spatial Audio 3","topics":["speech-llm","self-supervised","dataset"],"category":"audio-understanding","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["Xi'an Jiaotong-Liverpool University","Xiaomi","University of Oulu","Chinese Academy of Sciences"],"funding":["Xi'an Jiaotong-Liverpool University"],"code":{"url":"https://github.com/Cell778/CoSTALA26.git","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ren26b_interspeech","category":"audio-understanding","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["Xi'an Jiaotong-Liverpool University","Xiaomi","University of Oulu","Chinese Academy of Sciences"],"code":"https://github.com/Cell778/CoSTALA26.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1110","pdf":"https://www.isca-archive.org/interspeech_2026/ren26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26b_interspeech/markdown.md"},{"id":"ren26c_interspeech","title":"Adapting Audio Large Language Models for Speaker Verification","authors":["Yiming Ren","Xuenan Xu","Shuai Wang","Chao Zhang","Baoxiang Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1117","isca_url":"https://www.isca-archive.org/interspeech_2026/ren26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ren26c_interspeech.pdf","session":"Speaker Verification: Architectures, Losses, and LLMs","topics":["speaker-verification","speech-llm","self-supervised"],"category":"speaker","labels":["self-supervised"],"institutions":["Shanghai Artificial Intelligence Laboratory","Nanjing University","Tsinghua University"],"funding":["Shanghai Artificial Intelligence Laboratory"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ren26c_interspeech","category":"speaker","labels":["self-supervised"],"institutions":["Shanghai Artificial Intelligence Laboratory","Nanjing University","Tsinghua University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1117","pdf":"https://www.isca-archive.org/interspeech_2026/ren26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26c_interspeech/markdown.md"},{"id":"ren26d_interspeech","title":"AuDirector: A Self-Reflective Closed-Loop Framework for Immersive Audio Storytelling","authors":["Yiming Ren","Ziyang Zhang","Wen Wu","Baoxiang Li","Chao Zhang","Xuenan Xu"],"year":2026,"doi":"10.21437/Interspeech.2026-1180","isca_url":"https://www.isca-archive.org/interspeech_2026/ren26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ren26d_interspeech.pdf","session":"Long-Form Speech Synthesis","topics":["speech-llm","tts","audio-captioning"],"category":"tts","labels":["generative-model"],"institutions":["Shanghai Artificial Intelligence Laboratory","Tsinghua University"],"funding":["Shanghai Artificial Intelligence Laboratory"],"code":{"url":"https://github.com/Riddae/AuDirector","stars":13,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ren26d_interspeech","category":"tts","labels":["generative-model"],"institutions":["Shanghai Artificial Intelligence Laboratory","Tsinghua University"],"code":"https://github.com/Riddae/AuDirector","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1180","pdf":"https://www.isca-archive.org/interspeech_2026/ren26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26d_interspeech/markdown.md"},{"id":"ren26e_interspeech","title":"Edit Content, Preserve Acoustics: Imperceptible Text-Based Speech Editing via Self-Consistency Rewards","authors":["Yong Ren","Jiangyan Yi","Jianhua Tao","Tao Wang","Le Xu","Zhengqi Wen"],"year":2026,"doi":"10.21437/Interspeech.2026-1186","isca_url":"https://www.isca-archive.org/interspeech_2026/ren26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ren26e_interspeech.pdf","session":"Voice Editing","topics":["speech-editing","self-supervised","speech-llm"],"category":"tts","labels":["self-supervised","generative-model"],"institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences","Tsinghua University"],"funding":["National Natural Science Foundation of China","China Postdoctoral Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ren26e_interspeech","category":"tts","labels":["self-supervised","generative-model"],"institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences","Tsinghua University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1186","pdf":"https://www.isca-archive.org/interspeech_2026/ren26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26e_interspeech/markdown.md"},{"id":"ren26f_interspeech","title":"Can Hands Articulate? Kinematic, Acoustic, and Perceptual Analyses of Vowel Production via External Resonators in Kaxi","authors":["Jinyang Ren","Bingliang Zhao","Jiangping Kong","Xiyu Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-1807","isca_url":"https://www.isca-archive.org/interspeech_2026/ren26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ren26f_interspeech.pdf","session":"Perception and Appraisal of Prosody","topics":["phonetics","evaluation","speech-enhancement"],"category":"phonetics-linguistics","institutions":["Peking University"],"funding":["National Social Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ren26f_interspeech","category":"phonetics-linguistics","institutions":["Peking University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1807","pdf":"https://www.isca-archive.org/interspeech_2026/ren26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26f_interspeech/markdown.md"},{"id":"ren26g_interspeech","title":"Unified Gradient Projection: Language-Balanced Continual Learning for Multilingual Low-Resource ASR","authors":["Ziang Ren","Guodong Lin","Yuchen Ai","Kaize Tan","Wei-Qiang Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1915","isca_url":"https://www.isca-archive.org/interspeech_2026/ren26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ren26g_interspeech.pdf","session":"Multilingual & Low-Resource ASR","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Tsinghua University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ren26g_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Tsinghua University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1915","pdf":"https://www.isca-archive.org/interspeech_2026/ren26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ren26g_interspeech/markdown.md"},{"id":"ries26_interspeech","title":"On Entrainment in Semi-Spontaneous Multilingual Parliamentary Speech","authors":["Jennifer Jane Ries","Debasmita Bhattacharya","Julia Hirschberg"],"year":2026,"doi":"10.21437/Interspeech.2026-2492","isca_url":"https://www.isca-archive.org/interspeech_2026/ries26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ries26_interspeech.pdf","session":"Entrainment and Dialogue Coordination","topics":["multilingual","paralinguistics","dataset"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Columbia University"],"funding":["National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ries26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Columbia University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2492","pdf":"https://www.isca-archive.org/interspeech_2026/ries26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ries26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ries26_interspeech/markdown.md"},{"id":"risso26_interspeech","title":"OnDA: On-device Channel Pruning for Efficient Personalized Keyword Spotting","authors":["Matteo Risso","Alessio Burrello","Daniele Jahier Pagliari"],"year":2026,"doi":"10.21437/Interspeech.2026-1253","isca_url":"https://www.isca-archive.org/interspeech_2026/risso26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/risso26_interspeech.pdf","session":"Efficient Inference for ASR and Speech LMs","topics":["keyword-spotting","on-device","self-supervised"],"category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Politecnico di Torino"],"funding":["NEURAL research project","Fondazione Compagnia di San Paolo"],"code":{"url":"https://github.com/eml-eda/onda","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"risso26_interspeech","category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Politecnico di Torino"],"code":"https://github.com/eml-eda/onda","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1253","pdf":"https://www.isca-archive.org/interspeech_2026/risso26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/risso26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/risso26_interspeech/markdown.md"},{"id":"romerodiaz26_interspeech","title":"Listening or Reading? Evaluating Speech Awareness in Chain-of-Thought Speech-to-Text Translation","authors":["Jacobo Romero-Díaz","Gerard I. Gállego","Oriol Pareras","Federico Costa","Javier Hernando","Cristina España-Bonet"],"year":2026,"doi":"10.21437/Interspeech.2026-800","isca_url":"https://www.isca-archive.org/interspeech_2026/romerodiaz26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/romerodiaz26_interspeech.pdf","session":"Translation","topics":["speech-translation","speech-llm","self-supervised"],"category":"translation","labels":["robustness-noise"],"institutions":["Barcelona Supercomputing Center","Universitat Politecnica de Catalunya","DFKI"],"funding":["Ministerio para la Transformacion Digital y de la Funcion Publica","Plan de Recuperacion, Transformacion y Resiliencia","European Union","MICIU/AEI","Red.es"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"romerodiaz26_interspeech","category":"translation","labels":["robustness-noise"],"institutions":["Barcelona Supercomputing Center","Universitat Politecnica de Catalunya","DFKI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-800","pdf":"https://www.isca-archive.org/interspeech_2026/romerodiaz26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/romerodiaz26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/romerodiaz26_interspeech/markdown.md"},{"id":"rong26_interspeech","title":"StuPASE: Towards Low-Hallucination Studio-Quality Generative Speech Enhancement","authors":["Xiaobin Rong","Jun Gao","Zheng Wang","Mansur Yesilbursa","Kamil Wojcicki","Jing Lu"],"year":2026,"doi":"10.21437/Interspeech.2026-837","isca_url":"https://www.isca-archive.org/interspeech_2026/rong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rong26_interspeech.pdf","session":"Language-Model and Codec-Token Speech Enhancement","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Nanjing University","Cisco Systems","Horizon Robotics"],"funding":["National Natural Science Foundation of China","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rong26_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Nanjing University","Cisco Systems","Horizon Robotics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-837","pdf":"https://www.isca-archive.org/interspeech_2026/rong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rong26_interspeech/markdown.md"},{"id":"rong26b_interspeech","title":"Beyond Symmetric Interaction: Capability-Aware Asymmetric Multi-Agent Collaboration for Audio Deep Reasoning","authors":["Yan Rong","Jinting Wang","Tianxin Xie","Xiang He","Chenxing Li","Dong Yu","Li Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-2273","isca_url":"https://www.isca-archive.org/interspeech_2026/rong26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rong26b_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","institutions":["Hong Kong University of Science and Technology","Tencent"],"funding":["National Natural Science Foundation of China","Guangdong Basic and Applied Basic Research Foundation","Tencent AI Lab Rhino-Bird Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rong26b_interspeech","category":"speech-llm-dialogue","institutions":["Hong Kong University of Science and Technology","Tencent"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2273","pdf":"https://www.isca-archive.org/interspeech_2026/rong26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rong26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rong26b_interspeech/markdown.md"},{"id":"ross26_interspeech","title":"Revisiting the NZE front vowel shift: evidence from New Zealand's largest and most linguistically diverse city","authors":["Brooke Ross","C. I. Watson","Elaine Ballard","Miriam Meyerhoff"],"year":2026,"doi":"10.21437/Interspeech.2026-1526","isca_url":"https://www.isca-archive.org/interspeech_2026/ross26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ross26_interspeech.pdf","session":"Speaker-Specific, Forensic and Segmental Characteristics of Typical and Atypical Speech","topics":["phonetics","dataset","evaluation"],"category":"phonetics-linguistics","institutions":["University of Auckland","University of Oxford"],"funding":["Marsden Fund Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ross26_interspeech","category":"phonetics-linguistics","institutions":["University of Auckland","University of Oxford"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1526","pdf":"https://www.isca-archive.org/interspeech_2026/ross26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ross26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ross26_interspeech/markdown.md"},{"id":"ross26b_interspeech","title":"Sexualised Synthetic Personas Encode and Amplify Gendered Power Asymmetries through Voice","authors":["Alice Ross","Ariadna Sanchez","Elin Kanhov","Catherine Lai","Éva Székely"],"year":2026,"doi":"10.21437/Interspeech.2026-2411","isca_url":"https://www.isca-archive.org/interspeech_2026/ross26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ross26b_interspeech.pdf","session":"Queer and Trans Speech Science and Technology","topics":["tts","paralinguistics","evaluation"],"category":"tts","institutions":["University of Edinburgh","KTH Royal Institute of Technology"],"funding":["UK Research and Innovation","Swedish Research Council"],"code":{"url":"https://ariadnasc.github.io/synth-personas","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ross26b_interspeech","category":"tts","institutions":["University of Edinburgh","KTH Royal Institute of Technology"],"code":"https://ariadnasc.github.io/synth-personas","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2411","pdf":"https://www.isca-archive.org/interspeech_2026/ross26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ross26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ross26b_interspeech/markdown.md"},{"id":"rubenchik26_interspeech","title":"Latent Flow Matching Based Speech Separation Using Speaker Diarization","authors":["Boris Rubenchik","Sharon Gannot","Ethan Fetaya"],"year":2026,"doi":"10.21437/Interspeech.2026-1401","isca_url":"https://www.isca-archive.org/interspeech_2026/rubenchik26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rubenchik26_interspeech.pdf","session":"Target Speaker Extraction, Speech Separation and Audio Understanding","topics":["speech-separation","self-supervised","speaker-diarization"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Bar-Ilan University"],"funding":["Israel Science Foundation","German Research Foundation","ISF-DFG Joint Research Program","AUDIENCE: Audio-Visual Analysis and Separation","Council of Higher Education, Israel"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rubenchik26_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Bar-Ilan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1401","pdf":"https://www.isca-archive.org/interspeech_2026/rubenchik26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rubenchik26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rubenchik26_interspeech/markdown.md"},{"id":"rust26_interspeech","title":"Dynamic Time Warping Reveals Prosodic Alignment in Caregiver–Child Interactions across Languages","authors":["Olivier Rüst","Sabine Stoll"],"year":2026,"doi":"10.21437/Interspeech.2026-2356","isca_url":"https://www.isca-archive.org/interspeech_2026/rust26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/rust26_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["prosody","phonetics","dataset"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["University of Zurich"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"rust26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["University of Zurich"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2356","pdf":"https://www.isca-archive.org/interspeech_2026/rust26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/rust26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/rust26_interspeech/markdown.md"},{"id":"ryu26_interspeech","title":"Modality Importance is Not Static: Temporal Dynamics via Gating in Multimodal Emotion Recognition","authors":["Jiyeon Ryu","SeongHun Noh","Jin-Hyuk Hong","Woojin Kang"],"year":2026,"doi":"10.21437/Interspeech.2026-1399","isca_url":"https://www.isca-archive.org/interspeech_2026/ryu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ryu26_interspeech.pdf","session":"Multimodal Emotion Recognition","topics":["speech-llm","paralinguistic","emotion-recognition"],"category":"paralinguistics-emotion","institutions":["Gwangju Institute of Science and Technology"],"funding":["Ministry of Trade, Industry and Energy","Korea Institute for Advancement of Technology","Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ryu26_interspeech","category":"paralinguistics-emotion","institutions":["Gwangju Institute of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1399","pdf":"https://www.isca-archive.org/interspeech_2026/ryu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ryu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ryu26_interspeech/markdown.md"},{"id":"ryu26b_interspeech","title":"Segment-level Tree Search for Long Meeting Document Summarization","authors":["Sangwon Ryu","Heejin Do","Jun Seo","Daehui Kim","Yunsu Kim","Gary Geunbae Lee","Jungseul Ok"],"year":2026,"doi":"10.21437/Interspeech.2026-3011","isca_url":"https://www.isca-archive.org/interspeech_2026/ryu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ryu26b_interspeech.pdf","session":"Corpus Creation, Summarization and Understanding","topics":["speech-llm","spoken-language-understanding","evaluation"],"category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["POSTECH","ETH Zurich","LILT"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea Creative Content Agency","Ministry of Science and ICT","Ministry of Culture, Sports and Tourism"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ryu26b_interspeech","category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["POSTECH","ETH Zurich","LILT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3011","pdf":"https://www.isca-archive.org/interspeech_2026/ryu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ryu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ryu26b_interspeech/markdown.md"},{"id":"ryu26c_interspeech","title":"SPOT-TSE: Spatial Point-Guided Target Speech Extraction","authors":["Taewon Ryu","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-3266","isca_url":"https://www.isca-archive.org/interspeech_2026/ryu26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ryu26c_interspeech.pdf","session":"Source Separation 2","topics":["target-speech-extraction","speech-enhancement","on-device"],"category":"enhancement-separation","institutions":["Hanyang University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ryu26c_interspeech","category":"enhancement-separation","institutions":["Hanyang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3266","pdf":"https://www.isca-archive.org/interspeech_2026/ryu26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ryu26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ryu26c_interspeech/markdown.md"},{"id":"s26_interspeech","title":"Gender Bias in ASR: A Controlled Study of Gender Composition Across Training Paradigms","authors":["Seshan S","Murali Kadambi","Amartya Veer","Prasanta Kumar Ghosh"],"year":2026,"doi":"10.21437/Interspeech.2026-3047","isca_url":"https://www.isca-archive.org/interspeech_2026/s26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/s26_interspeech.pdf","session":"Robust ASR: Hallucinations and Biases","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["self-supervised"],"institutions":["Indian Institute of Science"],"funding":["Defence Research and Development Organisation"],"code":{"url":"https://asr.iitm.ac.in/models","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"s26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Indian Institute of Science"],"code":"https://asr.iitm.ac.in/models","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3047","pdf":"https://www.isca-archive.org/interspeech_2026/s26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/s26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/s26_interspeech/markdown.md"},{"id":"sadok26_interspeech","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","authors":["Samir Sadok","Xavier Alameda-Pineda"],"year":2026,"doi":"10.21437/Interspeech.2026-733","isca_url":"https://www.isca-archive.org/interspeech_2026/sadok26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sadok26_interspeech.pdf","session":"Speech Representations and Alignment","topics":["self-supervised","asr","speaker-verification"],"category":"asr","labels":["self-supervised"],"institutions":["Inria","Universite Grenoble Alpes","CNRS"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sadok26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Inria","Universite Grenoble Alpes","CNRS"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-733","pdf":"https://www.isca-archive.org/interspeech_2026/sadok26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sadok26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sadok26_interspeech/markdown.md"},{"id":"saif26_interspeech","title":"BELLA: Efficient Bilevel Learning with LoRA for Multilingual ASR","authors":["A F M Saif","Xiaodong Cui","Brian Kingsbury","Tianyi Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-2771","isca_url":"https://www.isca-archive.org/interspeech_2026/saif26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/saif26_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["multilingual","efficient-on-device","self-supervised"],"institutions":["Rensselaer Polytechnic Institute","IBM","Cornell Tech"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"saif26_interspeech","category":"asr","labels":["multilingual","efficient-on-device","self-supervised"],"institutions":["Rensselaer Polytechnic Institute","IBM","Cornell Tech"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2771","pdf":"https://www.isca-archive.org/interspeech_2026/saif26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/saif26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/saif26_interspeech/markdown.md"},{"id":"saini26_interspeech","title":"Listening Like a Judge: A Music-Aware Framework for Automatic Singing Performance Evaluation","authors":["Neelam Saini","Sourav Ghosh"],"year":2026,"doi":"10.21437/Interspeech.2026-912","isca_url":"https://www.isca-archive.org/interspeech_2026/saini26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/saini26_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["speech-recognition","paralinguistics","multilingual"],"category":"resources-evaluation","institutions":["Samsung"],"code":{"url":"https://neelam472.github.io/MusicJudge/Supp.pdf","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"saini26_interspeech","category":"resources-evaluation","institutions":["Samsung"],"code":"https://neelam472.github.io/MusicJudge/Supp.pdf","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-912","pdf":"https://www.isca-archive.org/interspeech_2026/saini26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/saini26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/saini26_interspeech/markdown.md"},{"id":"sanchez26_interspeech","title":"An Evaluation Framework for Text-to-Speech Voice Reconstruction","authors":["Ariadna Sanchez","Christoph Minixhofer","Korin Richmond","Ondřej Klejch","Peter Bell","Simon King"],"year":2026,"doi":"10.21437/Interspeech.2026-2600","isca_url":"https://www.isca-archive.org/interspeech_2026/sanchez26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sanchez26_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["tts","evaluation","speaker-verification"],"category":"resources-evaluation","institutions":["University of Edinburgh"],"funding":["UKRI Centre for Doctoral Training in Natural Language Processing","UKRI"],"code":{"url":"https://minixc.github.io/sap/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sanchez26_interspeech","category":"resources-evaluation","institutions":["University of Edinburgh"],"code":"https://minixc.github.io/sap/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2600","pdf":"https://www.isca-archive.org/interspeech_2026/sanchez26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sanchez26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sanchez26_interspeech/markdown.md"},{"id":"sanchez26b_interspeech","title":"From Lab to Laptop: Validating 3D Speech Kinematics with MediaPipe Face Mesh","authors":["Victoria Sanchez","Oliver Roesler","Michael Neumann","David Pautler","Brian Richburg","Karen Chenausky","Yana Yunusova","Vikram Ramanarayanan","Jordan Green"],"year":2026,"doi":"10.21437/Interspeech.2026-3022","isca_url":"https://www.isca-archive.org/interspeech_2026/sanchez26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sanchez26b_interspeech.pdf","session":"Multimodal and Non-Speech Healthcare Applications","topics":["paralinguistics","evaluation","health"],"category":"phonetics-linguistics","institutions":["Harvard University","Massachusetts General Hospital Institute of Health Professions","Modality.AI","University of California San Francisco","Sunnybrook Research Institute","University of Toronto"],"funding":["National Institute on Deafness and Other Communication Disorders","Autism Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sanchez26b_interspeech","category":"phonetics-linguistics","institutions":["Harvard University","Massachusetts General Hospital Institute of Health Professions","Modality.AI","University of California San Francisco","Sunnybrook Research Institute","University of Toronto"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3022","pdf":"https://www.isca-archive.org/interspeech_2026/sanchez26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sanchez26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sanchez26b_interspeech/markdown.md"},{"id":"sanjotra26_interspeech","title":"Investigating the Relationship between Objective AI-driven Metrics and Subjective MOS for In-the-Wild Speech","authors":["Jasmer Sanjotra","Nagendra Kumar","Shekhar Nayak"],"year":2026,"doi":"10.21437/Interspeech.2026-2203","isca_url":"https://www.isca-archive.org/interspeech_2026/sanjotra26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sanjotra26_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["tts","evaluation","self-supervised"],"category":"resources-evaluation","labels":["self-supervised"],"institutions":["Indian Institute of Technology Indore","University of Groningen"],"funding":["IEEE Signal Processing Society Signal Processing Mentorship Academy"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sanjotra26_interspeech","category":"resources-evaluation","labels":["self-supervised"],"institutions":["Indian Institute of Technology Indore","University of Groningen"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2203","pdf":"https://www.isca-archive.org/interspeech_2026/sanjotra26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sanjotra26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sanjotra26_interspeech/markdown.md"},{"id":"sannigrahi26_interspeech","title":"AdaTS: Adaptive Token Sampling for Efficient Speech Language Models","authors":["Sonal Sannigrahi","Giuseppe Attanasio","André F. T. Martins"],"year":2026,"doi":"10.21437/Interspeech.2026-2753","isca_url":"https://www.isca-archive.org/interspeech_2026/sannigrahi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sannigrahi26_interspeech.pdf","session":"Efficient Inference for ASR and Speech LMs","topics":["asr","speech-translation","spoken-language-understanding"],"category":"speech-llm-dialogue","labels":["efficient-on-device"],"institutions":["Universidade de Lisboa","Instituto de Telecomunicacoes","TransPerfect"],"funding":["Portuguese Recovery and Resilience Plan","Center for ResponsibleAI","DECOLLAGE","ERC","FCT","MECI","EU funds","Instituto de Telecomunicacoes"],"code":{"url":"https://github.com/sonalsannigrahi/AdaTS","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sannigrahi26_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device"],"institutions":["Universidade de Lisboa","Instituto de Telecomunicacoes","TransPerfect"],"code":"https://github.com/sonalsannigrahi/AdaTS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2753","pdf":"https://www.isca-archive.org/interspeech_2026/sannigrahi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sannigrahi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sannigrahi26_interspeech/markdown.md"},{"id":"saon26_interspeech","title":"Self-Speculative Decoding for LLM-based ASR with CTC Encoder Drafts","authors":["George Saon","Samuel Thomas","Takashi Fukuda","Tohru Nagano","Avihu Dekel","Luis Lastras"],"year":2026,"doi":"10.21437/Interspeech.2026-2680","isca_url":"https://www.isca-archive.org/interspeech_2026/saon26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/saon26_interspeech.pdf","session":"Search Methods and Inference Algorithms","topics":["asr","speech-llm","multilingual"],"category":"asr","labels":["efficient-on-device"],"institutions":["IBM"],"code":{"url":"https://ibm.biz/˜5pwn29DW4","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"saon26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["IBM"],"code":"https://ibm.biz/˜5pwn29DW4","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2680","pdf":"https://www.isca-archive.org/interspeech_2026/saon26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/saon26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/saon26_interspeech/markdown.md"},{"id":"sapkota26_interspeech","title":"IACC-HuBERT: Intelligibility-Aware Channel Conditioning of HuBERT Frontend for Dysarthric Speech Conformer ASR","authors":["Paban Sapkota","Hemant Kumar Kathania"],"year":2026,"doi":"10.21437/Interspeech.2026-2375","isca_url":"https://www.isca-archive.org/interspeech_2026/sapkota26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sapkota26_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["low-resource","self-supervised"],"institutions":["National Institute of Technology Sikkim"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sapkota26_interspeech","category":"asr","labels":["low-resource","self-supervised"],"institutions":["National Institute of Technology Sikkim"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2375","pdf":"https://www.isca-archive.org/interspeech_2026/sapkota26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sapkota26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sapkota26_interspeech/markdown.md"},{"id":"sara26_interspeech","title":"Light-weight Pronunciation Assessment via Discrete Speech Token Surprisal","authors":["Syeda Faiza Ahmed Sara","Shammur Absar Chowdhury"],"year":2026,"doi":"10.21437/Interspeech.2026-1153","isca_url":"https://www.isca-archive.org/interspeech_2026/sara26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sara26_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["speech-enhancement","self-supervised","evaluation"],"category":"applications-other","labels":["self-supervised"],"institutions":["Qatar Computing Research Institute"],"funding":["HBKU flagship research grant"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sara26_interspeech","category":"applications-other","labels":["self-supervised"],"institutions":["Qatar Computing Research Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1153","pdf":"https://www.isca-archive.org/interspeech_2026/sara26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sara26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sara26_interspeech/markdown.md"},{"id":"sato26_interspeech","title":"Latency Controllable Speech Enhancement","authors":["Hiroshi Sato","Takafumi Moriya","Tsubasa Ochiai","Marc Delcroix"],"year":2026,"doi":"10.21437/Interspeech.2026-2997","isca_url":"https://www.isca-archive.org/interspeech_2026/sato26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sato26_interspeech.pdf","session":"Real-Time, Low-Latency and Edge Speech Enhancement","topics":["speech-enhancement","streaming","low-resource"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["NTT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sato26_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["NTT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2997","pdf":"https://www.isca-archive.org/interspeech_2026/sato26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sato26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sato26_interspeech/markdown.md"},{"id":"schade26_interspeech","title":"Naming Heroes and Villains – The Influence of Phonaesthetics","authors":["Leonie Schade","Daniel Duran","Florian Kankowski","Joana Cholin","Petra Wagner","Christine Mooshammer"],"year":2026,"doi":"10.21437/Interspeech.2026-2247","isca_url":"https://www.isca-archive.org/interspeech_2026/schade26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/schade26_interspeech.pdf","session":"Perception and Appraisal of Prosody","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["Bielefeld University","Humboldt-Universität zu Berlin"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"https://osf.io/gtajk","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"schade26_interspeech","category":"phonetics-linguistics","institutions":["Bielefeld University","Humboldt-Universität zu Berlin"],"code":"https://osf.io/gtajk","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2247","pdf":"https://www.isca-archive.org/interspeech_2026/schade26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/schade26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/schade26_interspeech/markdown.md"},{"id":"scharff26_interspeech","title":"Gradient phonetic detail is less detrimental to word segmentation in infant-directed speech","authors":["Gabriel Scharff","Megha Sundara"],"year":2026,"doi":"10.21437/Interspeech.2026-2920","isca_url":"https://www.isca-archive.org/interspeech_2026/scharff26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/scharff26_interspeech.pdf","session":"Modeling L1 Acquisition","topics":["spoken-language-understanding","phonetics","dataset"],"category":"phonetics-linguistics","institutions":["University of California Los Angeles"],"funding":["National Science Foundation"],"code":{"url":"https://doi.org/10.5281/zenodo.20767608","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"scharff26_interspeech","category":"phonetics-linguistics","institutions":["University of California Los Angeles"],"code":"https://doi.org/10.5281/zenodo.20767608","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2920","pdf":"https://www.isca-archive.org/interspeech_2026/scharff26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/scharff26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/scharff26_interspeech/markdown.md"},{"id":"schlicher26_interspeech","title":"Daily Affect Inference from Longitudinal Speech-based Journals: A Comparison of Acoustic and Linguistic Models","authors":["Michelle D Schlicher","Andreas Triantafyllopoulos","Nadine N Schmitt","Johanna Löchner","Bjoern Schuller"],"year":2026,"doi":"10.21437/Interspeech.2026-2383","isca_url":"https://www.isca-archive.org/interspeech_2026/schlicher26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/schlicher26_interspeech.pdf","session":"Multimodal Emotion Recognition","topics":["paralinguistics","speech-llm","evaluation"],"category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["Technical University of Munich","University Hospital of Tubingen","University of Erlangen-Nuremberg","Imperial College London"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"schlicher26_interspeech","category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["Technical University of Munich","University Hospital of Tubingen","University of Erlangen-Nuremberg","Imperial College London"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2383","pdf":"https://www.isca-archive.org/interspeech_2026/schlicher26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/schlicher26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/schlicher26_interspeech/markdown.md"},{"id":"schlotterbeck26_interspeech","title":"Content–Speaker Trade-offs in Continued Self-Supervised Pre-Training Across SSL Paradigms for Multilingual Speech","authors":["Danner Schlotterbeck","Alessandro Huaman","Juan Gomez"],"year":2026,"doi":"10.21437/Interspeech.2026-2946","isca_url":"https://www.isca-archive.org/interspeech_2026/schlotterbeck26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/schlotterbeck26_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["self-supervised","multilingual","asr"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["Factored AI"],"funding":["Factored AI"],"code":{"url":"https://github.com/dannersm/ups-continuous-pretraining","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"schlotterbeck26_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["Factored AI"],"code":"https://github.com/dannersm/ups-continuous-pretraining","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2946","pdf":"https://www.isca-archive.org/interspeech_2026/schlotterbeck26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/schlotterbeck26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/schlotterbeck26_interspeech/markdown.md"},{"id":"sedaghati26_interspeech","title":"VoxWatermark: A Large-Scale Benchmark for Audio Watermark Detection under Perturbations","authors":["Farnaz Sedaghati","Yuxi Wang","Zicheng Weng","Wei Rao"],"year":2026,"doi":"10.21437/Interspeech.2026-1771","isca_url":"https://www.isca-archive.org/interspeech_2026/sedaghati26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sedaghati26_interspeech.pdf","session":"Audio Watermarking and Source Verification","topics":["audio-deepfake","evaluation","self-supervised"],"category":"deepfake-security","labels":["multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["University of Tehran","Nanyang Technological University"],"code":{"url":"https://github.com/wailywang/VoxWatermark","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sedaghati26_interspeech","category":"deepfake-security","labels":["multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["University of Tehran","Nanyang Technological University"],"code":"https://github.com/wailywang/VoxWatermark","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1771","pdf":"https://www.isca-archive.org/interspeech_2026/sedaghati26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sedaghati26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sedaghati26_interspeech/markdown.md"},{"id":"seebauer26_interspeech","title":"Application context in speech synthesis evaluation: A problem and a solution","authors":["Fritz Seebauer","Markus Rothgänger","Sven Wachsmuth","Petra Wagner"],"year":2026,"doi":"10.21437/Interspeech.2026-747","isca_url":"https://www.isca-archive.org/interspeech_2026/seebauer26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/seebauer26_interspeech.pdf","session":"Speech Synthesis Evaluation 1","topics":["tts","evaluation","spoken-language-understanding"],"category":"resources-evaluation","institutions":["Bielefeld University"],"funding":["Ministry of Culture and Science of the State of North Rhine-Westphalia","Netzwerke 2021","SAIL: SustAInable Life-cycle of Intelligent Socio-Technical Systems"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"seebauer26_interspeech","category":"resources-evaluation","institutions":["Bielefeld University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-747","pdf":"https://www.isca-archive.org/interspeech_2026/seebauer26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/seebauer26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/seebauer26_interspeech/markdown.md"},{"id":"seki26_interspeech","title":"Improving DF-Conformer using Hydra for high-fidelity generative speech enhancement on discrete codec token","authors":["Shogo Seki","Shaoxiang Dang","Li Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1833","isca_url":"https://www.isca-archive.org/interspeech_2026/seki26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/seki26_interspeech.pdf","session":"Language-Model and Codec-Token Speech Enhancement","topics":["speech-enhancement","self-supervised","tts"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["CyberAgent"],"code":{"url":"https://github.com/goombalab/hydra","stars":177,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"seki26_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["CyberAgent"],"code":"https://github.com/goombalab/hydra","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1833","pdf":"https://www.isca-archive.org/interspeech_2026/seki26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/seki26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/seki26_interspeech/markdown.md"},{"id":"seo26_interspeech","title":"When Multiple Script Matters: Evaluating ASR in Clinical Settings","authors":["Jean Seo","Minkyu Kim","Jeonguk Lee","Jisoo Jung","Wooseok Han","Eunho Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-1126","isca_url":"https://www.isca-archive.org/interspeech_2026/seo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/seo26_interspeech.pdf","session":"Spoken Language Processing: Evaluation and Metrics","topics":["asr","multilingual","dataset"],"category":"asr","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["AITRICS","University of Copenhagen","KAIST"],"code":{"url":"https://github.com/aitrics-ronaldo/Interspeech_MultiClin","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"seo26_interspeech","category":"asr","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["AITRICS","University of Copenhagen","KAIST"],"code":"https://github.com/aitrics-ronaldo/Interspeech_MultiClin","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1126","pdf":"https://www.isca-archive.org/interspeech_2026/seo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/seo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/seo26_interspeech/markdown.md"},{"id":"seo26b_interspeech","title":"Hard Positive-targeted Training for Robust Audio Deepfake Detection under Neural Codec Processing","authors":["Jiwon Seo","Inho Kim","Seongkyu Han","Thien-Phuc Doan","Souhwan Jung"],"year":2026,"doi":"10.21437/Interspeech.2026-2167","isca_url":"https://www.isca-archive.org/interspeech_2026/seo26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/seo26b_interspeech.pdf","session":"Speech Deepfake Detection: Robustness, Generalization, Attribution","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised","robustness-noise"],"institutions":["Soongsil University"],"funding":["Ministry of Science and ICT","Institute of Information & Communications Technology Planning & Evaluation","Korea Institute of Police Technology","Korean National Police Agency"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"seo26b_interspeech","category":"deepfake-security","labels":["self-supervised","robustness-noise"],"institutions":["Soongsil University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2167","pdf":"https://www.isca-archive.org/interspeech_2026/seo26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/seo26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/seo26b_interspeech/markdown.md"},{"id":"sepanta26_interspeech","title":"From Game-Based Annotation to Representation Probing: Cross-Validated Prosodic Speech and Privacy Implications","authors":["Sia Vosh Sepanta","Roberto Zamparelli","Alessio Brutti"],"year":2026,"doi":"10.21437/Interspeech.2026-2459","isca_url":"https://www.isca-archive.org/interspeech_2026/sepanta26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sepanta26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["speech-llm","emotion-recognition","dataset"],"category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["Fondazione Bruno Kessler","University of Trento"],"funding":["European Union"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sepanta26_interspeech","category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["Fondazione Bruno Kessler","University of Trento"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2459","pdf":"https://www.isca-archive.org/interspeech_2026/sepanta26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sepanta26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sepanta26_interspeech/markdown.md"},{"id":"seth26_interspeech","title":"Audio Hallucination Attacks: Probing the Reliability of Large Audio Language Models","authors":["Ashish Seth","Sonal Kumar","Ramaneswaran Selvakuma","Nishit Anand","Utkarsh Tyagi","Prem Seetharaman","Ramani Duraiswami","Dinesh Manocha"],"year":2026,"doi":"10.21437/Interspeech.2026-2448","isca_url":"https://www.isca-archive.org/interspeech_2026/seth26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/seth26_interspeech.pdf","session":"Speech and Audio Quality Assessment","topics":["speech-llm","evaluation","self-supervised"],"category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release"],"institutions":["University of Maryland, College Park","Adobe Research"],"code":{"url":"https://cs20s030.github.io/AHA-website/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"seth26_interspeech","category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release"],"institutions":["University of Maryland, College Park","Adobe Research"],"code":"https://cs20s030.github.io/AHA-website/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2448","pdf":"https://www.isca-archive.org/interspeech_2026/seth26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/seth26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/seth26_interspeech/markdown.md"},{"id":"shah26_interspeech","title":"SingFox: A Multi-Lingual Singfake Detection Corpus","authors":["Arth J. Shah","Devanshi K. Trivedi","Himanshi U. Borad","Hemant A. Patil"],"year":2026,"doi":"10.21437/Interspeech.2026-2573","isca_url":"https://www.isca-archive.org/interspeech_2026/shah26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shah26_interspeech.pdf","session":"Singing Voice and Music Generation","topics":["speech-enhancement","evaluation","dataset"],"category":"deepfake-security","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Dhirubhai Ambani University","Sarvajanik College of Engineering and Technology"],"code":{"url":"https://github.com/Arth-Shah/SingFox","stars":5,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shah26_interspeech","category":"deepfake-security","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Dhirubhai Ambani University","Sarvajanik College of Engineering and Technology"],"code":"https://github.com/Arth-Shah/SingFox","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2573","pdf":"https://www.isca-archive.org/interspeech_2026/shah26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shah26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shah26_interspeech/markdown.md"},{"id":"shahamiri26_interspeech","title":"BetterSpeak: An Atypical Speech to Typical Speech Platform for Dysarthric Speakers","authors":["Seyed Reza Shahamiri","Zihan Zhong","Qianli Wang","Satwinder Singh"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/shahamiri26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shahamiri26_interspeech.pdf","session":"Speech and Language Processing for Health and Accessibility","topics":["asr","tts","speech-llm"],"category":"asr","institutions":["University of Auckland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shahamiri26_interspeech","category":"asr","institutions":["University of Auckland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/shahamiri26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/shahamiri26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shahamiri26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shahamiri26_interspeech/markdown.md"},{"id":"shahin26_interspeech","title":"SayCheck: Gamified Speech Practice and Attribute-Based Speech Analysis for Children","authors":["Mostafa Shahin","Kirrie Ballard","Beena Ahmed"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/shahin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shahin26_interspeech.pdf","session":"Speech and Language Learning Technologies","topics":["health","evaluation","self-supervised"],"category":"health-clinical","institutions":["University of New South Wales","University of Sydney"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shahin26_interspeech","category":"health-clinical","institutions":["University of New South Wales","University of Sydney"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/shahin26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/shahin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shahin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shahin26_interspeech/markdown.md"},{"id":"shang26_interspeech","title":"Seed-Enh: Generative Speech Enhancement in Decoupled Semantic and Timbre Spaces","authors":["Zengqiang Shang","Biao Liu","Yu Zhao","Pengyuan Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-200","isca_url":"https://www.isca-archive.org/interspeech_2026/shang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shang26_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","voice-conversion","self-supervised"],"category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Institute of Acoustics","University of Chinese Academy of Sciences"],"funding":["National Natural Science Foundation of China","CPSF Postdoctoral Fellowship"],"code":{"url":"https://github.com/shangqwe123/Seed-Enh","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shang26_interspeech","category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Institute of Acoustics","University of Chinese Academy of Sciences"],"code":"https://github.com/shangqwe123/Seed-Enh","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-200","pdf":"https://www.isca-archive.org/interspeech_2026/shang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shang26_interspeech/markdown.md"},{"id":"shangguan26_interspeech","title":"Dual-LoRA: Parameter-Efficient Adversarial Disentanglement for Cross-Lingual Speaker Verification","authors":["Qituan Shangguan","Junhao Du","Kunyang Peng","Feng Xue","Hui Zhang","Xinsheng Wang","Kai Yu","Shuai Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-2274","isca_url":"https://www.isca-archive.org/interspeech_2026/shangguan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shangguan26_interspeech.pdf","session":"Challenge - TidyVoice Challenge: Cross-Lingual Speaker Verification","topics":["speaker-verification","multilingual","self-supervised"],"category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Nanjing University","Shanghai Jiao Tong University","AISpeech","Soul AI Lab"],"funding":["National Natural Science Foundation of China","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shangguan26_interspeech","category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Nanjing University","Shanghai Jiao Tong University","AISpeech","Soul AI Lab"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2274","pdf":"https://www.isca-archive.org/interspeech_2026/shangguan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shangguan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shangguan26_interspeech/markdown.md"},{"id":"shankar26_interspeech","title":"GC-LoRA: Gated Convolutional LoRA for Parameter-Efficient Acoustic Adaptation","authors":["Natarajan Balaji Shankar","Zilai Wang","Kaiyuan Zhang","Mohan Shi","Abeer Alwan"],"year":2026,"doi":"10.21437/Interspeech.2026-822","isca_url":"https://www.isca-archive.org/interspeech_2026/shankar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shankar26_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 1","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["efficient-on-device","self-supervised","robustness-noise"],"institutions":["University of California, Los Angeles"],"funding":["National Science Foundation","Institute of Education Sciences, U.S. Department of Education"],"code":{"url":"https://github.com/balaji1312/gc_lora","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shankar26_interspeech","category":"asr","labels":["efficient-on-device","self-supervised","robustness-noise"],"institutions":["University of California, Los Angeles"],"code":"https://github.com/balaji1312/gc_lora","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-822","pdf":"https://www.isca-archive.org/interspeech_2026/shankar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shankar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shankar26_interspeech/markdown.md"},{"id":"shao26_interspeech","title":"Phoneme-Aware Mamba Watermark: An Active Defense System Against Purified Speech Deepfakes","authors":["Yanda Shao","Mengke Zhang","Zhixin Lin","Tianyi Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-2105","isca_url":"https://www.isca-archive.org/interspeech_2026/shao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shao26_interspeech.pdf","session":"Audio Watermarking and Source Verification","topics":["speech-watermarking","self-supervised","evaluation"],"category":"deepfake-security","labels":["streaming-real-time"],"institutions":["Beijing University of Posts and Telecommunications","Beijing Institute of Technology"],"code":{"url":"https://github.com/Silence-ai423/phoneme-aware-mamba-watermark","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shao26_interspeech","category":"deepfake-security","labels":["streaming-real-time"],"institutions":["Beijing University of Posts and Telecommunications","Beijing Institute of Technology"],"code":"https://github.com/Silence-ai423/phoneme-aware-mamba-watermark","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2105","pdf":"https://www.isca-archive.org/interspeech_2026/shao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shao26_interspeech/markdown.md"},{"id":"sharma26_interspeech","title":"Robust Language Identification Using Semi-positive Contrastive Learning","authors":["Shubham Sharma","Padmanabhan Rajan"],"year":2026,"doi":"10.21437/Interspeech.2026-2502","isca_url":"https://www.isca-archive.org/interspeech_2026/sharma26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sharma26_interspeech.pdf","session":"Language and Dialect Recognition","topics":["asr","low-resource","multilingual"],"category":"speaker","labels":["low-resource","multilingual","robustness-noise"],"institutions":["Indian Institute of Technology Mandi"],"funding":["Ministry of Electronics and Information Technology"],"code":{"url":"https://github.com/HeisenBug-07/Interspeech2026_SpCL","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sharma26_interspeech","category":"speaker","labels":["low-resource","multilingual","robustness-noise"],"institutions":["Indian Institute of Technology Mandi"],"code":"https://github.com/HeisenBug-07/Interspeech2026_SpCL","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2502","pdf":"https://www.isca-archive.org/interspeech_2026/sharma26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sharma26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sharma26_interspeech/markdown.md"},{"id":"sharma26b_interspeech","title":"VoxENES 2026: Benchmarking Generalization of Speech Spoofing Detectors Against LLM-Era TTS and Voice Conversion","authors":["Aastha Sharma","Guangjing Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-2712","isca_url":"https://www.isca-archive.org/interspeech_2026/sharma26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sharma26b_interspeech.pdf","session":"Spoofing and Deepfake Detection 3","topics":["audio-deepfake","evaluation","dataset"],"category":"deepfake-security","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["University of South Florida"],"code":{"url":"https://www.kaggle.com/datasets/interspeech2712/voxenes-2026","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sharma26b_interspeech","category":"deepfake-security","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["University of South Florida"],"code":"https://www.kaggle.com/datasets/interspeech2712/voxenes-2026","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2712","pdf":"https://www.isca-archive.org/interspeech_2026/sharma26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sharma26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sharma26b_interspeech/markdown.md"},{"id":"sharma26c_interspeech","title":"LavaSR: Fast and Flexible Audio Bandwidth Extension via Vocos","authors":["Yatharth Sharma"],"year":2026,"doi":"10.21437/Interspeech.2026-2839","isca_url":"https://www.isca-archive.org/interspeech_2026/sharma26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sharma26c_interspeech.pdf","session":"Dereverberation, Bandwidth Extension and Restoration","topics":["speech-enhancement","self-supervised"],"category":"speech-coding","labels":["efficient-on-device","generative-model"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sharma26c_interspeech","category":"speech-coding","labels":["efficient-on-device","generative-model"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2839","pdf":"https://www.isca-archive.org/interspeech_2026/sharma26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sharma26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sharma26c_interspeech/markdown.md"},{"id":"sharon26_interspeech","title":"Less can be More: What Aspects of Speech Drive End-of-Turn Detection","authors":["Rini Sharon","Manickavela A","Kadri Hacioglu","Andreas Stolcke"],"year":2026,"doi":"10.21437/Interspeech.2026-1705","isca_url":"https://www.isca-archive.org/interspeech_2026/sharon26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sharon26_interspeech.pdf","session":"Turn-taking","topics":["asr","prosody","speech-llm"],"category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["Uniphore"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sharon26_interspeech","category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["Uniphore"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1705","pdf":"https://www.isca-archive.org/interspeech_2026/sharon26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sharon26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sharon26_interspeech/markdown.md"},{"id":"shea26_interspeech","title":"Masculinity and Sexual Orientation as Predictors of f0 Variation in the Speech of Australian English Speaking Men","authors":["Timothy Shea","Hannah White","Joshua Penney","Anita Szakay","Felicity Cox"],"year":2026,"doi":"10.21437/Interspeech.2026-1906","isca_url":"https://www.isca-archive.org/interspeech_2026/shea26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shea26_interspeech.pdf","session":"Queer and Trans Speech Science and Technology","topics":["paralinguistics","phonetics","evaluation"],"category":"phonetics-linguistics","institutions":["Macquarie University","LMU Munich"],"funding":["Australian Research Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shea26_interspeech","category":"phonetics-linguistics","institutions":["Macquarie University","LMU Munich"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1906","pdf":"https://www.isca-archive.org/interspeech_2026/shea26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shea26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shea26_interspeech/markdown.md"},{"id":"shen26_interspeech","title":"CoDeTT: A Context-Aware Decision Benchmark for Turn-Taking Evaluation","authors":["Huan Shen","Yingao Wang","Shangkun Huang","Wei Zou","Yunzhang Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-974","isca_url":"https://www.isca-archive.org/interspeech_2026/shen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shen26_interspeech.pdf","session":"Turn-taking","topics":["speech-llm","evaluation","multilingual"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Bairong"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shen26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Bairong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-974","pdf":"https://www.isca-archive.org/interspeech_2026/shen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shen26_interspeech/markdown.md"},{"id":"shen26b_interspeech","title":"LaS-LCA: Layer-Selected Latent Cross-Attention Adapters and Margin-Mixup for Robust Cross-Lingual Speaker Verification","authors":["Xu Shen","Yihao Zhao","Xinwei Wu","Hao Liang","Yujie Zhu","Yujin Wang","Wei Liu","Gongping Huang","Shoji Makino"],"year":2026,"doi":"10.21437/Interspeech.2026-1255","isca_url":"https://www.isca-archive.org/interspeech_2026/shen26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shen26b_interspeech.pdf","session":"Challenge - TidyVoice Challenge: Cross-Lingual Speaker Verification","topics":["speaker-verification","multilingual","self-supervised"],"category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Waseda University","Wuhan University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shen26b_interspeech","category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Waseda University","Wuhan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1255","pdf":"https://www.isca-archive.org/interspeech_2026/shen26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shen26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shen26b_interspeech/markdown.md"},{"id":"shen26c_interspeech","title":"Parallel Time-Band Mixing with Learned Observation-Adding for Robust ASR Front-Ends","authors":["Xingyu Shen","Runze Wang","Wei-Ping Zhu","Benoit Champagne"],"year":2026,"doi":"10.21437/Interspeech.2026-1972","isca_url":"https://www.isca-archive.org/interspeech_2026/shen26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shen26c_interspeech.pdf","session":"SE Architectures, Adaptation and Audio Front-Ends","topics":["asr","speech-enhancement","self-supervised"],"category":"enhancement-separation","labels":["efficient-on-device","self-supervised","robustness-noise"],"institutions":["Concordia University","McGill University","Shenzhen University of Advanced Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shen26c_interspeech","category":"enhancement-separation","labels":["efficient-on-device","self-supervised","robustness-noise"],"institutions":["Concordia University","McGill University","Shenzhen University of Advanced Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1972","pdf":"https://www.isca-archive.org/interspeech_2026/shen26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shen26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shen26c_interspeech/markdown.md"},{"id":"shen26d_interspeech","title":"Iterate to Differentiate: Enhancing Discriminability and Reliability in Zero-Shot TTS Evaluation","authors":["Shengfan Shen","Di Wu","Xingchen Song","Dinghao Zhou","Liumeng Xue","Meng Meng","Jian Luan","Shuai Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-2414","isca_url":"https://www.isca-archive.org/interspeech_2026/shen26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shen26d_interspeech.pdf","session":"Text-to-Speech Synthesis","topics":["tts","evaluation","self-supervised"],"category":"resources-evaluation","labels":["generative-model"],"institutions":["Nanjing University","Xiaomi"],"funding":["National Natural Science Foundation of China","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"https://ssf-1103.github.io/I2D-Bench/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shen26d_interspeech","category":"resources-evaluation","labels":["generative-model"],"institutions":["Nanjing University","Xiaomi"],"code":"https://ssf-1103.github.io/I2D-Bench/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2414","pdf":"https://www.isca-archive.org/interspeech_2026/shen26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shen26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shen26d_interspeech/markdown.md"},{"id":"shen26e_interspeech","title":"Adaptive Hard-Pair Sampling via Curriculum Learning for Speech Separation","authors":["Pengjie Shen","Xueliang Zhang","Zhong-Qiu Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-3139","isca_url":"https://www.isca-archive.org/interspeech_2026/shen26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shen26e_interspeech.pdf","session":"Source Separation 1","topics":["speech-separation","self-supervised"],"category":"enhancement-separation","institutions":["Inner Mongolia University","Southern University of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shen26e_interspeech","category":"enhancement-separation","institutions":["Inner Mongolia University","Southern University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3139","pdf":"https://www.isca-archive.org/interspeech_2026/shen26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shen26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shen26e_interspeech/markdown.md"},{"id":"sheppard26_interspeech","title":"Queer inclusion in speech datasets: An audit and taxonomy of practical tensions","authors":["Brooklyn Sheppard","Anaelia Ovalle","Adina Williams","Levent Sagun"],"year":2026,"doi":"10.21437/Interspeech.2026-2904","isca_url":"https://www.isca-archive.org/interspeech_2026/sheppard26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sheppard26_interspeech.pdf","session":"Queer and Trans Speech Science and Technology","topics":["dataset","evaluation","low-resource"],"category":"resources-evaluation","labels":["low-resource"],"institutions":["University of Calgary","Meta"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sheppard26_interspeech","category":"resources-evaluation","labels":["low-resource"],"institutions":["University of Calgary","Meta"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2904","pdf":"https://www.isca-archive.org/interspeech_2026/sheppard26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sheppard26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sheppard26_interspeech/markdown.md"},{"id":"sheppard26b_interspeech","title":"Towards participatory speech dataset curation: A queer case study and conceptual framework","authors":["Brooklyn Sheppard","Anaelia Ovalle","Adina Williams","Levent Sagun"],"year":2026,"doi":"10.21437/Interspeech.2026-2908","isca_url":"https://www.isca-archive.org/interspeech_2026/sheppard26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sheppard26b_interspeech.pdf","session":"Queer and Trans Speech Science and Technology","topics":["self-supervised","evaluation","dataset"],"category":"resources-evaluation","institutions":["University of Calgary","Meta"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sheppard26b_interspeech","category":"resources-evaluation","institutions":["University of Calgary","Meta"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2908","pdf":"https://www.isca-archive.org/interspeech_2026/sheppard26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sheppard26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sheppard26b_interspeech/markdown.md"},{"id":"sheth26_interspeech","title":"Deriving Benchmarking Datasets from Long-Form Recordings: Challenges and Opportunities","authors":["Kaveri K. Sheth","Lawrence Borst","Tarek Kunze","Marvin Lavechin","Okko Räsänen","Sho Tsuji","Loann Peurey","Alix Bourree","Alejandrina Cristia"],"year":2026,"doi":"10.21437/Interspeech.2026-2363","isca_url":"https://www.isca-archive.org/interspeech_2026/sheth26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sheth26_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["speech-enhancement","low-resource","multilingual"],"category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["PSL University","CNRS","EHESS","ENS","Universite Aix-Marseille","Tampere University"],"funding":["Agence Nationale de la Recherche","PSL","J. S. McDonnell Foundation","European Research Council","Simons Foundation International"],"code":{"url":"https://github.com/LAAC-LSCP/benchmarking-dataset-factory","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sheth26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release","robustness-noise"],"institutions":["PSL University","CNRS","EHESS","ENS","Universite Aix-Marseille","Tampere University"],"code":"https://github.com/LAAC-LSCP/benchmarking-dataset-factory","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2363","pdf":"https://www.isca-archive.org/interspeech_2026/sheth26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sheth26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sheth26_interspeech/markdown.md"},{"id":"sheth26b_interspeech","title":"From Academic Tool to Community Infrastructure: A Call for Indigenous Partnership in Speech Data Governance","authors":["Kaveri K. Sheth","Sebastien Christian"],"year":2026,"doi":"10.21437/Interspeech.2026-2489","isca_url":"https://www.isca-archive.org/interspeech_2026/sheth26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sheth26b_interspeech.pdf","session":"Indigenous Voices in Speech Science and Technology","topics":["dataset","low-resource","multilingual"],"category":"resources-evaluation","labels":["low-resource","multilingual"],"institutions":["ENS","EHESS","CNRS","PSL University","UPF"],"funding":["Agence Nationale pour la Recherche","European Research Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sheth26b_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual"],"institutions":["ENS","EHESS","CNRS","PSL University","UPF"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2489","pdf":"https://www.isca-archive.org/interspeech_2026/sheth26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sheth26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sheth26b_interspeech/markdown.md"},{"id":"sheth26c_interspeech","title":"ELSI: An Interface for Standardizing Child-Centered Datasets, Applying Machine Learning Models, and Extracting Metrics","authors":["Kaveri K. Sheth","Loann Peurey","Sho Tsuji","Alejandrina Cristia"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/sheth26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sheth26c_interspeech.pdf","session":"Speech and Language Processing for Health and Accessibility","topics":["speaker-diarization","dataset","low-resource"],"category":"resources-evaluation","institutions":["CNRS","EHESS","ENS","PSL University"],"funding":["Agence Nationale de la Recherche","PSL","J. S. McDonnell Foundation","European Research Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sheth26c_interspeech","category":"resources-evaluation","institutions":["CNRS","EHESS","ENS","PSL University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/sheth26c_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/sheth26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sheth26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sheth26c_interspeech/markdown.md"},{"id":"shi26_interspeech","title":"Distilling LLM Semantic Priors into Encoder-Only Multi-Talker ASR with Talker-Count Routing","authors":["Hao Shi","Yusuke Fujita","Roman Koshkin","Mengjie Zhao","Yuan Gao","Lianbo Liu","Yui Sudo"],"year":2026,"doi":"10.21437/Interspeech.2026-612","isca_url":"https://www.isca-archive.org/interspeech_2026/shi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shi26_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["efficient-on-device","self-supervised"],"institutions":["SB Intuitions"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shi26_interspeech","category":"asr","labels":["efficient-on-device","self-supervised"],"institutions":["SB Intuitions"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-612","pdf":"https://www.isca-archive.org/interspeech_2026/shi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26_interspeech/markdown.md"},{"id":"shi26b_interspeech","title":"Towards Fine-Grained Temporal Perception: Post-Training Large Audio-Language Models with Audio-Side Time Prompt","authors":["Yanfeng Shi","Pengfei Cai","Jun Liu","Qing Gu","Nan Jiang","Lirong Dai","Ian McLoughlin","Yan Song"],"year":2026,"doi":"10.21437/Interspeech.2026-745","isca_url":"https://www.isca-archive.org/interspeech_2026/shi26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shi26b_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["University of Science and Technology of China","Singapore Institute of Technology"],"funding":["Anhui Province Major Science and Technology Research Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shi26b_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["University of Science and Technology of China","Singapore Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-745","pdf":"https://www.isca-archive.org/interspeech_2026/shi26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26b_interspeech/markdown.md"},{"id":"shi26c_interspeech","title":"Entropy-Aware Domain-Routed Mixture-of-Experts Speech-LLM Framework: A Case Study of Multi-Domain Child-Adult ASR","authors":["Mohan Shi","Kaiyuan Zhang","Zilai Wang","Natarajan Balaji Shankar","Eray Eren","Abeer Alwan"],"year":2026,"doi":"10.21437/Interspeech.2026-877","isca_url":"https://www.isca-archive.org/interspeech_2026/shi26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shi26c_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["asr","speech-llm","low-resource"],"category":"asr","institutions":["University of California, Los Angeles"],"funding":["National Science Foundation","Institute of Education Sciences","U.S. Department of Education"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shi26c_interspeech","category":"asr","institutions":["University of California, Los Angeles"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-877","pdf":"https://www.isca-archive.org/interspeech_2026/shi26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26c_interspeech/markdown.md"},{"id":"shi26d_interspeech","title":"Emo-BPO: Emotion Bidirectional Preference Optimization for Diffusion-based Emotional TTS","authors":["Jiacheng Shi","Hongfei Du","Xinyuan Song","Y. Alicia Hong","Yanfu Zhang","Ye Gao"],"year":2026,"doi":"10.21437/Interspeech.2026-1613","isca_url":"https://www.isca-archive.org/interspeech_2026/shi26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shi26d_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["tts","self-supervised","emotion-recognition"],"category":"tts","labels":["generative-model"],"institutions":["College of William & Mary","Emory University","George Mason University"],"code":{"url":"https://jiachengqaq.github.io/emo-bpo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shi26d_interspeech","category":"tts","labels":["generative-model"],"institutions":["College of William & Mary","Emory University","George Mason University"],"code":"https://jiachengqaq.github.io/emo-bpo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1613","pdf":"https://www.isca-archive.org/interspeech_2026/shi26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26d_interspeech/markdown.md"},{"id":"shi26e_interspeech","title":"Leveraging Modality-Specific Label Distributions for Enhanced Multimodal Emotion Recognition","authors":["Xiaohan Shi","Xingfeng Li","Tomoki Toda"],"year":2026,"doi":"10.21437/Interspeech.2026-2076","isca_url":"https://www.isca-archive.org/interspeech_2026/shi26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shi26e_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["speech-llm","paralinguistics","emotion-recognition"],"category":"paralinguistics-emotion","institutions":["Nagoya University","City University of Macau"],"funding":["JST CREST","JSPS KAKENHI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shi26e_interspeech","category":"paralinguistics-emotion","institutions":["Nagoya University","City University of Macau"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2076","pdf":"https://www.isca-archive.org/interspeech_2026/shi26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26e_interspeech/markdown.md"},{"id":"shi26f_interspeech","title":"EffVOC: Low-Delay Efficient Speech Waveform Reconstruction from Spectral Representations Without Phase","authors":["Renzheng Shi","Simon Welker","Timo Gerkmann","Tim Fingscheidt"],"year":2026,"doi":"10.21437/Interspeech.2026-2407","isca_url":"https://www.isca-archive.org/interspeech_2026/shi26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shi26f_interspeech.pdf","session":"Dereverberation, Bandwidth Extension and Restoration","topics":["tts","speech-enhancement","on-device"],"category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Technische Universitat Braunschweig","Universitat Hamburg"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shi26f_interspeech","category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Technische Universitat Braunschweig","Universitat Hamburg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2407","pdf":"https://www.isca-archive.org/interspeech_2026/shi26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26f_interspeech/markdown.md"},{"id":"shi26g_interspeech","title":"Speech Codec Probing from Semantic and Phonetic Perspectives","authors":["Xuan Shi","Chang Zeng","Tiantian Feng","Shih-Heng Wang","Jianbo Ma","Shrikanth Narayanan"],"year":2026,"doi":"10.21437/Interspeech.2026-3135","isca_url":"https://www.isca-archive.org/interspeech_2026/shi26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shi26g_interspeech.pdf","session":"Audio segmentation","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-coding","labels":["self-supervised"],"institutions":["University of Southern California","Dolby Laboratories"],"funding":["National Science Foundation","IARPA ARTS","Dolby"],"code":{"url":"https://github.com/Alexuan/codec_probing_release","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shi26g_interspeech","category":"speech-coding","labels":["self-supervised"],"institutions":["University of Southern California","Dolby Laboratories"],"code":"https://github.com/Alexuan/codec_probing_release","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3135","pdf":"https://www.isca-archive.org/interspeech_2026/shi26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shi26g_interspeech/markdown.md"},{"id":"shields26_interspeech","title":"A preliminary exploration of stop-vowel coarticulation in Māori","authors":["Isabella Shields","Hiraia Haami-Wells","C. T. Justine Hui","Peter J Keegan","C. I. Watson"],"year":2026,"doi":"10.21437/Interspeech.2026-1456","isca_url":"https://www.isca-archive.org/interspeech_2026/shields26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shields26_interspeech.pdf","session":"Pacific Voices: Speech Science and Technology for the Languages of the Pacific Ocean","topics":["phonetics","prosody","low-resource"],"category":"phonetics-linguistics","labels":["low-resource"],"institutions":["University of Auckland"],"funding":["Royal Society of New Zealand Marsden Fast-Start Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shields26_interspeech","category":"phonetics-linguistics","labels":["low-resource"],"institutions":["University of Auckland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1456","pdf":"https://www.isca-archive.org/interspeech_2026/shields26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shields26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shields26_interspeech/markdown.md"},{"id":"shigabeev26_interspeech","title":"Dialogs: a studio-quality expressive conversational Russian speech corpus for dialog assistants","authors":["Ilya Shigabeev","Ilia Latyshev"],"year":2026,"doi":"10.21437/Interspeech.2026-809","isca_url":"https://www.isca-archive.org/interspeech_2026/shigabeev26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shigabeev26_interspeech.pdf","session":"Datasets","topics":["tts","dataset","multilingual"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Langswap"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shigabeev26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Langswap"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-809","pdf":"https://www.isca-archive.org/interspeech_2026/shigabeev26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shigabeev26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shigabeev26_interspeech/markdown.md"},{"id":"shim26_interspeech","title":"Should Robots Sound more like Machines than like Humans? User Expectations Affect the Perception of Prosody in TTS Voices","authors":["Ha Eun Shim","Paige Tuttösí","Olivia Yung","Ivan Fong","Sara Ng","Angelica Lim","Yue Wang","H. Henny Yeung"],"year":2026,"doi":"10.21437/Interspeech.2026-3075","isca_url":"https://www.isca-archive.org/interspeech_2026/shim26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shim26_interspeech.pdf","session":"Prominence, Stress and Focus","topics":["tts","prosody","evaluation"],"category":"tts","institutions":["Simon Fraser University","University of Massachusetts Amherst","Enchanted Tools","CNRS","Universite Paris Cite"],"code":{"url":"https://osf.io/8ka5y","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shim26_interspeech","category":"tts","institutions":["Simon Fraser University","University of Massachusetts Amherst","Enchanted Tools","CNRS","Universite Paris Cite"],"code":"https://osf.io/8ka5y","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3075","pdf":"https://www.isca-archive.org/interspeech_2026/shim26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shim26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shim26_interspeech/markdown.md"},{"id":"shimizu26_interspeech","title":"MeanFlow-TSE: One-Step Generative Target Speaker Extraction with Mean Flow","authors":["Riki Shimizu","Xilin Jiang","Nima Mesgarani"],"year":2026,"doi":"10.21437/Interspeech.2026-109","isca_url":"https://www.isca-archive.org/interspeech_2026/shimizu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shimizu26_interspeech.pdf","session":"Audio-Visual and Generative Target Speaker Extraction","topics":["target-speaker-extraction","speech-enhancement","self-supervised"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Columbia University"],"funding":["National Institutes of Health","Marie-Josée and Henry R. Kravis"],"code":{"url":"https://github.com/rikishimizu/MeanFlow-TSE","stars":32,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shimizu26_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Columbia University"],"code":"https://github.com/rikishimizu/MeanFlow-TSE","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-109","pdf":"https://www.isca-archive.org/interspeech_2026/shimizu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shimizu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shimizu26_interspeech/markdown.md"},{"id":"shimizu26b_interspeech","title":"Auditory Contrast Network for Text-Free Prominence Detection","authors":["Kosuke Shimizu","Keiichi Zempo"],"year":2026,"doi":"10.21437/Interspeech.2026-250","isca_url":"https://www.isca-archive.org/interspeech_2026/shimizu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shimizu26b_interspeech.pdf","session":"Prominence, Stress and Focus","topics":["prosody","self-supervised","on-device"],"category":"phonetics-linguistics","labels":["efficient-on-device","streaming-real-time"],"institutions":["University of Tsukuba"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shimizu26b_interspeech","category":"phonetics-linguistics","labels":["efficient-on-device","streaming-real-time"],"institutions":["University of Tsukuba"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-250","pdf":"https://www.isca-archive.org/interspeech_2026/shimizu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shimizu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shimizu26b_interspeech/markdown.md"},{"id":"shin26_interspeech","title":"Breaking Shortcut Learning for Cross-Trial EEG-Guided Target Speech Extraction via Two-Stage Training","authors":["Wonchul Shin","Inyong Choi","Kyogu Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-595","isca_url":"https://www.isca-archive.org/interspeech_2026/shin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shin26_interspeech.pdf","session":"Clinical and Inclusive Speech Technology","topics":["target-speech-extraction","self-supervised","evaluation"],"category":"enhancement-separation","institutions":["Seoul National University","University of Iowa"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation"],"code":{"url":"https://github.com/argaaw/TRUST-TSE","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shin26_interspeech","category":"enhancement-separation","institutions":["Seoul National University","University of Iowa"],"code":"https://github.com/argaaw/TRUST-TSE","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-595","pdf":"https://www.isca-archive.org/interspeech_2026/shin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shin26_interspeech/markdown.md"},{"id":"shin26b_interspeech","title":"EchoLoc: Audio-Aware Object Grounding via Joint Heatmap and Box-Level Localization","authors":["Junghwa Shin","Yeonwoo Kim","HyungJune Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-2320","isca_url":"https://www.isca-archive.org/interspeech_2026/shin26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shin26b_interspeech.pdf","session":"Audio-Visual Grounding, Synchronization & Video Understanding","topics":["audio-captioning","self-supervised","evaluation"],"category":"audio-understanding","institutions":["Ewha Womans University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shin26b_interspeech","category":"audio-understanding","institutions":["Ewha Womans University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2320","pdf":"https://www.isca-archive.org/interspeech_2026/shin26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shin26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shin26b_interspeech/markdown.md"},{"id":"shinayama26_interspeech","title":"Upcycling Pretrained Transformers into Mixture-of-Experts for Multilingual Speech Recognition","authors":["Kentaro Shinayama","Kohei Matsuura","Jaeyoung Lee","Masato Mimura"],"year":2026,"doi":"10.21437/Interspeech.2026-1630","isca_url":"https://www.isca-archive.org/interspeech_2026/shinayama26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shinayama26_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 1","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["NTT"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shinayama26_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["NTT"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1630","pdf":"https://www.isca-archive.org/interspeech_2026/shinayama26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shinayama26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shinayama26_interspeech/markdown.md"},{"id":"shridhar26_interspeech","title":"Lung-SRAD: Spectral-Aware Regularized Audio DASS with Dual-Axis Patch-Mix Contrastive Learning for Respiratory Sound Classification","authors":["Hemansh Shridhar","Miika Toikkanen","June-Woo Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-550","isca_url":"https://www.isca-archive.org/interspeech_2026/shridhar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/shridhar26_interspeech.pdf","session":"Acoustic Event Detection 1","topics":["paralinguistics","self-supervised","low-resource"],"category":"health-clinical","labels":["self-supervised"],"institutions":["MODULABS","Wonkwang University"],"funding":["Regional Innovation System & Education program","National Research Foundation of Korea"],"code":{"url":"https://github.com/RSC-Toolkit/Lung-SRAD","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"shridhar26_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["MODULABS","Wonkwang University"],"code":"https://github.com/RSC-Toolkit/Lung-SRAD","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-550","pdf":"https://www.isca-archive.org/interspeech_2026/shridhar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/shridhar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/shridhar26_interspeech/markdown.md"},{"id":"si26_interspeech","title":"Explicit Context-Driven Neural Acoustic Modeling for High-Fidelity RIR Generation","authors":["Chen Si","Qianyi Wu","Chaitanya Amballa","Romit Roy Choudhury"],"year":2026,"doi":"10.21437/Interspeech.2026-513","isca_url":"https://www.isca-archive.org/interspeech_2026/si26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/si26_interspeech.pdf","session":"Speech Enhancement and Restoration","topics":["evaluation","self-supervised"],"category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["University of California San Diego","Monash University","University of Illinois Urbana-Champaign"],"code":{"url":"https://chen-si-cs.github.io/projects/MiNAF/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"si26_interspeech","category":"enhancement-separation","labels":["generative-model","robustness-noise"],"institutions":["University of California San Diego","Monash University","University of Illinois Urbana-Champaign"],"code":"https://chen-si-cs.github.io/projects/MiNAF/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-513","pdf":"https://www.isca-archive.org/interspeech_2026/si26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/si26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/si26_interspeech/markdown.md"},{"id":"si26b_interspeech","title":"Cross Domain Few-Shot Class-Incremental Audio Classification Via Adversarial Contrastive Learning","authors":["Yongjie Si","Yanxiong Li","Sen Huang","Beibei Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-1250","isca_url":"https://www.isca-archive.org/interspeech_2026/si26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/si26b_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["audio-classification","self-supervised","few-shot-learning"],"category":"audio-understanding","labels":["low-resource","robustness-noise"],"institutions":["South China University of Technology"],"funding":["National Natural Science Foundation of China","China-Croatia Science and Technology Cooperation Committee"],"code":{"url":"https://github.com/YongjieSi/ACL","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"si26b_interspeech","category":"audio-understanding","labels":["low-resource","robustness-noise"],"institutions":["South China University of Technology"],"code":"https://github.com/YongjieSi/ACL","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1250","pdf":"https://www.isca-archive.org/interspeech_2026/si26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/si26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/si26b_interspeech/markdown.md"},{"id":"silva26_interspeech","title":"NeuroMultiSpEx: Neuro-Guided Target Speaker Extraction for Multi-Speaker Scenarios","authors":["Dashanka De Silva","Saurav Pahuja","Siqi Cai","Tanja Schultz","Haizhou Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1120","isca_url":"https://www.isca-archive.org/interspeech_2026/silva26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/silva26_interspeech.pdf","session":"Target Speaker Extraction, Speech Separation and Audio Understanding","topics":["source-separation","speech-enhancement","self-supervised"],"category":"enhancement-separation","institutions":["University of Bremen","Chinese University of Hong Kong, Shenzhen","Harbin Institute of Technology"],"funding":["Deutsche Forschungsgemeinschaft","Program for Guangdong Introducing Innovative and Entrepreneurial Teams"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"silva26_interspeech","category":"enhancement-separation","institutions":["University of Bremen","Chinese University of Hong Kong, Shenzhen","Harbin Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1120","pdf":"https://www.isca-archive.org/interspeech_2026/silva26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/silva26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/silva26_interspeech/markdown.md"},{"id":"silverman26_interspeech","title":"Learning Self-Supervised Spatial Representations via Soft Acoustic Contrastive Alignment","authors":["Yotam Silverman","Bracha Laufer-Goldshtein"],"year":2026,"doi":"10.21437/Interspeech.2026-641","isca_url":"https://www.isca-archive.org/interspeech_2026/silverman26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/silverman26_interspeech.pdf","session":"Spatial Audio 4","topics":["self-supervised","spatial-audio","speech-enhancement"],"category":"enhancement-separation","labels":["self-supervised"],"institutions":["Tel Aviv University"],"funding":["Israel Science Foundation","Israeli Ministry of Innovation, Science and Technology"],"code":{"url":"https://github.com/Yotamsil/SAC_Alignment","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"silverman26_interspeech","category":"enhancement-separation","labels":["self-supervised"],"institutions":["Tel Aviv University"],"code":"https://github.com/Yotamsil/SAC_Alignment","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-641","pdf":"https://www.isca-archive.org/interspeech_2026/silverman26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/silverman26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/silverman26_interspeech/markdown.md"},{"id":"simic26_interspeech","title":"Adaptive AVSR: Integrating Speaker and Environmental Embeddings for Robust Audio-Visual Speech Recognition","authors":["Christopher Simic","Korbinian Riedhammer","Tobias Bocklet"],"year":2026,"doi":"10.21437/Interspeech.2026-2081","isca_url":"https://www.isca-archive.org/interspeech_2026/simic26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/simic26_interspeech.pdf","session":"Robust Audio-Visual Speech Recognition","topics":["asr","self-supervised","speaker-verification"],"category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["Technische Hochschule Nuernberg"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"simic26_interspeech","category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["Technische Hochschule Nuernberg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2081","pdf":"https://www.isca-archive.org/interspeech_2026/simic26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/simic26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/simic26_interspeech/markdown.md"},{"id":"singh26_interspeech","title":"Low-Burden Data Augmentation for Dysarthric ASR via Zero-Shot Voice Cloning","authors":["Satwinder Singh","Qianli Wang","Zihan Zhong","Clarion Mendes","Mark Hasegawa-Johnson","Waleed Abdulla","Seyed Reza Shahamiri"],"year":2026,"doi":"10.21437/Interspeech.2026-1501","isca_url":"https://www.isca-archive.org/interspeech_2026/singh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/singh26_interspeech.pdf","session":"Beyond Speech Technologies in Healthcare","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["low-resource","generative-model"],"institutions":["DeepNet Discovery Network","University of Auckland","University of Illinois Urbana-Champaign"],"code":{"url":"https://github.com/boson-ai/higgs-audio","stars":8364,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"singh26_interspeech","category":"asr","labels":["low-resource","generative-model"],"institutions":["DeepNet Discovery Network","University of Auckland","University of Illinois Urbana-Champaign"],"code":"https://github.com/boson-ai/higgs-audio","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1501","pdf":"https://www.isca-archive.org/interspeech_2026/singh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/singh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/singh26_interspeech/markdown.md"},{"id":"singh26b_interspeech","title":"CHUCKLE - When Humans Teach AI to Learn Emotions the Easy Way","authors":["Ankush Pratap Singh","Houwei Cao","Yong Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-1591","isca_url":"https://www.isca-archive.org/interspeech_2026/singh26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/singh26b_interspeech.pdf","session":"Paralinguistics","topics":["speech-llm","self-supervised","paralinguistics"],"category":"paralinguistics-emotion","institutions":["New York Institute of Technology","New York University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"singh26b_interspeech","category":"paralinguistics-emotion","institutions":["New York Institute of Technology","New York University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1591","pdf":"https://www.isca-archive.org/interspeech_2026/singh26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/singh26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/singh26b_interspeech/markdown.md"},{"id":"singh26c_interspeech","title":"FlowEdit: Associative Memory for Lifelong Pronunciation Adaptation in Flow-Matching TTS","authors":["Harshit Singh","Ayush Pratap Singh","Nityanand Mathur"],"year":2026,"doi":"10.21437/Interspeech.2026-2764","isca_url":"https://www.isca-archive.org/interspeech_2026/singh26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/singh26c_interspeech.pdf","session":"Voice Editing","topics":["tts","self-supervised","multilingual"],"category":"tts","labels":["generative-model"],"institutions":["University of Maryland","TU Darmstadt","Smallest AI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"singh26c_interspeech","category":"tts","labels":["generative-model"],"institutions":["University of Maryland","TU Darmstadt","Smallest AI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2764","pdf":"https://www.isca-archive.org/interspeech_2026/singh26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/singh26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/singh26c_interspeech/markdown.md"},{"id":"singh26d_interspeech","title":"Selective Capability Unlearning in End-to-End Spoken Language Understanding","authors":["Akanksha Singh","Vinod Kumar Kurmi"],"year":2026,"doi":"10.21437/Interspeech.2026-3349","isca_url":"https://www.isca-archive.org/interspeech_2026/singh26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/singh26d_interspeech.pdf","session":"Spoken Language Understanding","topics":["spoken-language-understanding","self-supervised","evaluation"],"category":"speech-llm-dialogue","institutions":["Indian Institute of Science Education and Research Bhopal"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"singh26d_interspeech","category":"speech-llm-dialogue","institutions":["Indian Institute of Science Education and Research Bhopal"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3349","pdf":"https://www.isca-archive.org/interspeech_2026/singh26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/singh26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/singh26d_interspeech/markdown.md"},{"id":"singh26e_interspeech","title":"ProSarc: Prosody-Aware Sarcasm Recognition Framework via Temporal Prosodic Incongruity","authors":["Prathamjyot Singh","Ashima Sood","Sahil Sharma","Jasmeet Singh"],"year":2026,"doi":"10.21437/Interspeech.2026-3451","isca_url":"https://www.isca-archive.org/interspeech_2026/singh26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/singh26e_interspeech.pdf","session":"Emotion, Prosody, and Articulation","topics":["paralinguistics","self-supervised","evaluation"],"category":"paralinguistics-emotion","institutions":["Thapar Institute of Engineering and Technology","Ulster University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"singh26e_interspeech","category":"paralinguistics-emotion","institutions":["Thapar Institute of Engineering and Technology","Ulster University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3451","pdf":"https://www.isca-archive.org/interspeech_2026/singh26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/singh26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/singh26e_interspeech/markdown.md"},{"id":"sinha26_interspeech","title":"Collection and Curation of a Spontaneous Multilingual Speech Corpus for Low-Resource Himalayan Languages","authors":["Abhijit Sinha","Subham Kutum","Udara Laxman Kumar","Paban Sapkota","Hemant Kumar Kathania"],"year":2026,"doi":"10.21437/Interspeech.2026-2634","isca_url":"https://www.isca-archive.org/interspeech_2026/sinha26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sinha26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["low-resource","multilingual","dataset"],"category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["National Institute of Technology Sikkim"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sinha26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["National Institute of Technology Sikkim"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2634","pdf":"https://www.isca-archive.org/interspeech_2026/sinha26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sinha26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sinha26_interspeech/markdown.md"},{"id":"sinha26b_interspeech","title":"Error Diversity and Performance Variability in Zero-Shot Children's Speech Recognition","authors":["Abhijit Sinha","Hemant Kumar Kathania","Paban Sapkota","Mikko Kurimo"],"year":2026,"doi":"10.21437/Interspeech.2026-2666","isca_url":"https://www.isca-archive.org/interspeech_2026/sinha26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sinha26b_interspeech.pdf","session":"Robust ASR: Hallucinations and Biases","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["NIT Sikkim","Aalto University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sinha26b_interspeech","category":"asr","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["NIT Sikkim","Aalto University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2666","pdf":"https://www.isca-archive.org/interspeech_2026/sinha26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sinha26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sinha26b_interspeech/markdown.md"},{"id":"sirbu26_interspeech","title":"Inverse Text Normalization in Romanian: A Comparative Study of Rule-Based, Neural, and Large Language Model Approaches","authors":["Oana Sirbu","Alexandra Diaconu","Sergiu Nisioi","Bogdan Alexe"],"year":2026,"doi":"10.21437/Interspeech.2026-3454","isca_url":"https://www.isca-archive.org/interspeech_2026/sirbu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sirbu26_interspeech.pdf","session":"Multilingual Speech 2","topics":["asr","dataset","evaluation"],"category":"asr","labels":["dataset-or-benchmark-release"],"institutions":["University of Bucharest","Romanian Academy"],"funding":["Romanian Hub for Artificial Intelligence - HRIA","Smart Growth, Digitization and Financial Instruments Program","CNCS - UEFISCDI"],"code":{"url":"https://github.com/RoITN","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sirbu26_interspeech","category":"asr","labels":["dataset-or-benchmark-release"],"institutions":["University of Bucharest","Romanian Academy"],"code":"https://github.com/RoITN","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3454","pdf":"https://www.isca-archive.org/interspeech_2026/sirbu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sirbu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sirbu26_interspeech/markdown.md"},{"id":"sirigiraju26_interspeech","title":"ALFreeD: Teacher-Guided Few-Shot Pronunciation Assessment via Segmentation-Free Deviation Modeling","authors":["Meenakshi Sirigiraju","Nevin K Mathew","Reni K Cherian","Chiranjeevi Yarra"],"year":2026,"doi":"10.21437/Interspeech.2026-3247","isca_url":"https://www.isca-archive.org/interspeech_2026/sirigiraju26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sirigiraju26_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["speech-enhancement","self-supervised","evaluation"],"category":"applications-other","labels":["low-resource","self-supervised"],"institutions":["International Institute of Information Technology Hyderabad","Saintgits College of Engineering"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sirigiraju26_interspeech","category":"applications-other","labels":["low-resource","self-supervised"],"institutions":["International Institute of Information Technology Hyderabad","Saintgits College of Engineering"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3247","pdf":"https://www.isca-archive.org/interspeech_2026/sirigiraju26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sirigiraju26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sirigiraju26_interspeech/markdown.md"},{"id":"skorzewska26_interspeech","title":"Informativity of high-frequency bands on the place of articulation shift in retroflex sibilants produced by children","authors":["Oliwia Skórzewska","Maria Filipek","Wojciech Pieniążek","Zuzanna Miodońska"],"year":2026,"doi":"10.21437/Interspeech.2026-2606","isca_url":"https://www.isca-archive.org/interspeech_2026/skorzewska26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/skorzewska26_interspeech.pdf","session":"Modeling L1 Acquisition","topics":["paralinguistics","phonetics","evaluation"],"category":"phonetics-linguistics","institutions":["Silesian University of Technology"],"funding":["National Science Centre, Poland","European Union","Ministry of Science and Higher Education, Poland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"skorzewska26_interspeech","category":"phonetics-linguistics","institutions":["Silesian University of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2606","pdf":"https://www.isca-archive.org/interspeech_2026/skorzewska26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/skorzewska26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/skorzewska26_interspeech/markdown.md"},{"id":"so26_interspeech","title":"Toward Open-Set Speaker Attribute Prediction with Keyword-Appended LLM Embeddings","authors":["Byoungjun So","Jaejun Lee","Kyogu Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-3203","isca_url":"https://www.isca-archive.org/interspeech_2026/so26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/so26_interspeech.pdf","session":"Speaker Verification: Advances in Speaker Embeddings","topics":["speaker-verification","self-supervised","speech-llm"],"category":"speaker","institutions":["Seoul National University"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation","Advanced GPU Utilization Support Program"],"code":{"url":"https://github.com/jaejunL/vove","stars":10,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"so26_interspeech","category":"speaker","institutions":["Seoul National University"],"code":"https://github.com/jaejunL/vove","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3203","pdf":"https://www.isca-archive.org/interspeech_2026/so26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/so26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/so26_interspeech/markdown.md"},{"id":"sohn26_interspeech","title":"DTM-Codec: Dynamic Token Masking for VFR Speech Coding with Efficient Boundary Selection","authors":["Hoyeol Sohn","Juhan Nam"],"year":2026,"doi":"10.21437/Interspeech.2026-2984","isca_url":"https://www.isca-archive.org/interspeech_2026/sohn26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sohn26_interspeech.pdf","session":"Audio Coding and Signal Analysis","topics":["speech-coding","self-supervised","low-resource"],"category":"speech-coding","institutions":["KAIST"],"funding":["National Research Foundation of Korea"],"code":{"url":"https://github.com/hoyso48/DTM-Codec","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sohn26_interspeech","category":"speech-coding","institutions":["KAIST"],"code":"https://github.com/hoyso48/DTM-Codec","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2984","pdf":"https://www.isca-archive.org/interspeech_2026/sohn26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sohn26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sohn26_interspeech/markdown.md"},{"id":"someki26_interspeech","title":"ESPnet3: Infrastructure for Scalable Speech and Audio Research in the Foundation Model Era","authors":["Masao Someki","Alexander Polok","Carlos Carvalho","Chyi-Jiunn Lin","Da-Hee Yang","Jiatong Shi","Jinchuan Tian","Nelson Enrique Yalta Soplin","Samuele Cornell","Siddhant Arora","Francisco Teixeira","Wei Wang","William Chen","Alberto Abad","Chenda Li","Shinji Watanabe","Wangyou Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-2698","isca_url":"https://www.isca-archive.org/interspeech_2026/someki26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/someki26_interspeech.pdf","session":"Robust and Real-World ASR Systems","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["efficient-on-device"],"institutions":["Carnegie Mellon University","Brno University of Technology","Instituto Superior Técnico","Hanyang University","Hitachi Astemo","Shanghai Jiao Tong University"],"funding":["Advanced Cyberinfrastructure Coordination Ecosystem: Services & Support","National Science Foundation","Ministry of Education, Youth and Sports of the Czech Republic"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"someki26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["Carnegie Mellon University","Brno University of Technology","Instituto Superior Técnico","Hanyang University","Hitachi Astemo","Shanghai Jiao Tong University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2698","pdf":"https://www.isca-archive.org/interspeech_2026/someki26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/someki26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/someki26_interspeech/markdown.md"},{"id":"son26_interspeech","title":"Teacher-Agnostic Temporal Knowledge Distillation for Resource-Efficient Sound Event Detection","authors":["Gihun Son","Pil Moo Byun","Joon-Hyuk Chang"],"year":2026,"doi":"10.21437/Interspeech.2026-855","isca_url":"https://www.isca-archive.org/interspeech_2026/son26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/son26_interspeech.pdf","session":"Acoustic Event Detection 4","topics":["sound-event-detection","self-supervised","on-device"],"category":"audio-understanding","labels":["efficient-on-device"],"institutions":["Hanyang University"],"funding":["National Research Foundation of Korea"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"son26_interspeech","category":"audio-understanding","labels":["efficient-on-device"],"institutions":["Hanyang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-855","pdf":"https://www.isca-archive.org/interspeech_2026/son26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/son26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/son26_interspeech/markdown.md"},{"id":"song26_interspeech","title":"CFLOW-VC: An unsupervised cycle training strategy based on normalizing flows for Voice Conversion","authors":["FeiBao Song"],"year":2026,"doi":"10.21437/Interspeech.2026-48","isca_url":"https://www.isca-archive.org/interspeech_2026/song26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/song26_interspeech.pdf","session":"Voice Conversion","topics":["voice-conversion","self-supervised","speech-enhancement"],"category":"tts","labels":["generative-model"],"institutions":["Anhui University"],"code":{"url":"https://bigdan12.github.io/CFLOW_VC_demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"song26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Anhui University"],"code":"https://bigdan12.github.io/CFLOW_VC_demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-48","pdf":"https://www.isca-archive.org/interspeech_2026/song26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/song26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/song26_interspeech/markdown.md"},{"id":"song26b_interspeech","title":"MSMC: Multi-Scale Masked Convolution network for Robust Speech Emotion Recognition","authors":["Haoyu Song","Ian McLoughlin","Xiaoxiao Miao","Aik Beng Ng","Simon See","Timothy Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-951","isca_url":"https://www.isca-archive.org/interspeech_2026/song26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/song26b_interspeech.pdf","session":"Speech Emotion Recognition and Representation 1","topics":["speech-emotion-recognition","self-supervised","on-device"],"category":"paralinguistics-emotion","labels":["efficient-on-device"],"institutions":["Singapore Institute of Technology","Duke Kunshan University","NVIDIA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"song26b_interspeech","category":"paralinguistics-emotion","labels":["efficient-on-device"],"institutions":["Singapore Institute of Technology","Duke Kunshan University","NVIDIA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-951","pdf":"https://www.isca-archive.org/interspeech_2026/song26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/song26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/song26b_interspeech/markdown.md"},{"id":"song26c_interspeech","title":"Segment-wise Embedding based Graph Attention Network for Effective Speech Emotion Recognition","authors":["Haoyu Song","Ian McLoughlin","Yan Song","Lirong Dai"],"year":2026,"doi":"10.21437/Interspeech.2026-969","isca_url":"https://www.isca-archive.org/interspeech_2026/song26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/song26c_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["speech-emotion-recognition","self-supervised","paralinguistics"],"category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Singapore Institute of Technology","University of Science and Technology of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"song26c_interspeech","category":"paralinguistics-emotion","labels":["self-supervised"],"institutions":["Singapore Institute of Technology","University of Science and Technology of China"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-969","pdf":"https://www.isca-archive.org/interspeech_2026/song26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/song26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/song26c_interspeech/markdown.md"},{"id":"song26d_interspeech","title":"ARTT: Augmented Reverberant-Target Training for Unsupervised Monaural Speech Dereverberation","authors":["Siqi Song","Fulin Wu","Zhong-Qiu Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1857","isca_url":"https://www.isca-archive.org/interspeech_2026/song26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/song26d_interspeech.pdf","session":"SE Architectures, Adaptation and Audio Front-Ends","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Southern University of Science and Technology"],"code":{"url":"https://arttdemo.github.io/artt_demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"song26d_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Southern University of Science and Technology"],"code":"https://arttdemo.github.io/artt_demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1857","pdf":"https://www.isca-archive.org/interspeech_2026/song26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/song26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/song26d_interspeech/markdown.md"},{"id":"song26e_interspeech","title":"Speaker-Filtered Heterogeneous Graph Network: Toward Privacy-Preserving Multimodal Emotion Recognition","authors":["Heying Song","Jing Han","Yandi Zheng","Zixing Zhang","Ziping Zhao","Bjoern Schuller"],"year":2026,"doi":"10.21437/Interspeech.2026-2282","isca_url":"https://www.isca-archive.org/interspeech_2026/song26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/song26e_interspeech.pdf","session":"Speaker Identity, States, and Traits in Paralinguistics","topics":["emotion-recognition","speech-llm","evaluation"],"category":"paralinguistics-emotion","institutions":["Tianjin Normal University","Hunan University","Yuelushan Center for Industrial Innovation","Technische Universitat Munchen","Imperial College London"],"funding":["Project of Yuelushan Center for Industrial Innovation","National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"song26e_interspeech","category":"paralinguistics-emotion","institutions":["Tianjin Normal University","Hunan University","Yuelushan Center for Industrial Innovation","Technische Universitat Munchen","Imperial College London"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2282","pdf":"https://www.isca-archive.org/interspeech_2026/song26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/song26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/song26e_interspeech/markdown.md"},{"id":"song26f_interspeech","title":"Evaluating and Preserving Lexical Stress in English-to-Chinese Speech-to-Speech Translation","authors":["Yuchen Song","Xi Chen","Mingze Li","Satoshi Nakamura"],"year":2026,"doi":"10.21437/Interspeech.2026-2321","isca_url":"https://www.isca-archive.org/interspeech_2026/song26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/song26f_interspeech.pdf","session":"Translation","topics":["speech-translation","prosody","tts"],"category":"translation","labels":["multilingual","dataset-or-benchmark-release","generative-model"],"institutions":["Chinese University of Hong Kong, Shenzhen","Shenzhen Loop Area Institute"],"funding":["National Natural Science Foundation of China","Program for Guangdong Introducing Innovative and Entrepreneurial Teams"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"song26f_interspeech","category":"translation","labels":["multilingual","dataset-or-benchmark-release","generative-model"],"institutions":["Chinese University of Hong Kong, Shenzhen","Shenzhen Loop Area Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2321","pdf":"https://www.isca-archive.org/interspeech_2026/song26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/song26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/song26f_interspeech/markdown.md"},{"id":"song26g_interspeech","title":"Real-Time Speech Enhancement on Edge Devices Guided by Harmonic and Voice-Activity Cues Utilizing Skin-Attachable Accelerometer","authors":["Yonghun Song","Yeongmin Kim","Yunsik Kim","Yeeun Kim","Yoonyoung Chung"],"year":2026,"doi":"10.21437/Interspeech.2026-3119","isca_url":"https://www.isca-archive.org/interspeech_2026/song26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/song26g_interspeech.pdf","session":"Real-Time, Low-Latency and Edge Speech Enhancement","topics":["speech-enhancement","on-device","multimodal"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time","robustness-noise"],"institutions":["Pohang University of Science and Technology","Intus"],"funding":["National Research Foundation","Institute of Information & Communications Technology Planning & Evaluation","High-Performance Computing Support Project","Regional Innovation System & Education project"],"code":{"url":"https://github.com/yhsong06/LAU-NetV2","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"song26g_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time","robustness-noise"],"institutions":["Pohang University of Science and Technology","Intus"],"code":"https://github.com/yhsong06/LAU-NetV2","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3119","pdf":"https://www.isca-archive.org/interspeech_2026/song26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/song26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/song26g_interspeech/markdown.md"},{"id":"song26h_interspeech","title":"MER-Live: An Interactive Browser Demo of Prosody-Driven Multimodal Emotion Recognition","authors":["Haoyu Song","Xiaoxiao Miao","Pai Chet Ng","Timothy Liu","Aik Beng Ng","Ian McLoughlin"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/song26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/song26h_interspeech.pdf","session":"Speech Analysis, Data Resources and Research Tools","topics":["speech-emotion-recognition","self-supervised","on-device"],"category":"paralinguistics-emotion","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Singapore Institute of Technology","Duke Kunshan University","NVIDIA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"song26h_interspeech","category":"paralinguistics-emotion","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Singapore Institute of Technology","Duke Kunshan University","NVIDIA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/song26h_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/song26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/song26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/song26h_interspeech/markdown.md"},{"id":"sonkar26_interspeech","title":"Tongue2Speech: Real-Time Speech Synthesis from Tongue Ultrasound Videos via Spatiotemporal Transformers","authors":["Yash Sonkar","Yasaswi Kilaru","Sudheera Yelimeli","Neil Shah","Vineet Gandhi"],"year":2026,"doi":"10.21437/Interspeech.2026-3023","isca_url":"https://www.isca-archive.org/interspeech_2026/sonkar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sonkar26_interspeech.pdf","session":"Beyond Speech Technologies in Healthcare","topics":["speech-synthesis","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["International Institute of Information Technology Hyderabad","TCS Research"],"funding":["Anusandhan National Research Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sonkar26_interspeech","category":"tts","labels":["generative-model"],"institutions":["International Institute of Information Technology Hyderabad","TCS Research"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3023","pdf":"https://www.isca-archive.org/interspeech_2026/sonkar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sonkar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sonkar26_interspeech/markdown.md"},{"id":"souganidis26_interspeech","title":"Discriminating Proficiency Levels in L2 Speech: A Comparative Study of Self-Supervised Models in Basque","authors":["Christoforos Souganidis","Aitor Bellanco","Andoni Sudupe","Inma Hernáez","Ibon Saratxaga","Eva Navas"],"year":2026,"doi":"10.21437/Interspeech.2026-1948","isca_url":"https://www.isca-archive.org/interspeech_2026/souganidis26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/souganidis26_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["spoken-language-understanding","self-supervised","low-resource"],"category":"applications-other","labels":["low-resource","self-supervised"],"institutions":["University of the Basque Country UPV/EHU"],"funding":["Department of Culture and Language Policy of the Basque Government"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"souganidis26_interspeech","category":"applications-other","labels":["low-resource","self-supervised"],"institutions":["University of the Basque Country UPV/EHU"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1948","pdf":"https://www.isca-archive.org/interspeech_2026/souganidis26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/souganidis26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/souganidis26_interspeech/markdown.md"},{"id":"spang26_interspeech","title":"GADVOX: The German Anxiety and Depression Voice Examination Dataset","authors":["Robert P. Spang","Wafaa Wardah","Hritik Sauw","Ole Möller-Nilsson","Etleva Gjoni","Stefan Brandenburg","Maria Kreußlein","Lana Mohr","Sebastian Möller"],"year":2026,"doi":"10.21437/Interspeech.2026-3523","isca_url":"https://www.isca-archive.org/interspeech_2026/spang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/spang26_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 1","topics":["paralinguistics","dataset","multilingual"],"category":"health-clinical","labels":["dataset-or-benchmark-release"],"institutions":["Bauhaus-Universitat Weimar","Technische Universitat Berlin","German Research Center for Artificial Intelligence","Senseven Health GmbH","University of Technology Chemnitz"],"funding":["German Federal Ministry of Education and Research"],"code":{"url":"https://osf.io/k4z2v/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"spang26_interspeech","category":"health-clinical","labels":["dataset-or-benchmark-release"],"institutions":["Bauhaus-Universitat Weimar","Technische Universitat Berlin","German Research Center for Artificial Intelligence","Senseven Health GmbH","University of Technology Chemnitz"],"code":"https://osf.io/k4z2v/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3523","pdf":"https://www.isca-archive.org/interspeech_2026/spang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/spang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/spang26_interspeech/markdown.md"},{"id":"spiesberger26_interspeech","title":"Predicting Menstrual Cycle Phases from Speech: A Paralinguistic Approach","authors":["Anika A. Spiesberger","Andreas Triantafyllopoulos","Melanie Weirich","Bjoern Schuller"],"year":2026,"doi":"10.21437/Interspeech.2026-1878","isca_url":"https://www.isca-archive.org/interspeech_2026/spiesberger26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/spiesberger26_interspeech.pdf","session":"Paralinguistics","topics":["paralinguistics","self-supervised","health"],"category":"health-clinical","institutions":["Technical University of Munich","Munich Center for Machine Learning","Friedrich-Schiller-University Jena","Imperial College London","Munich Data Science Institute"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"spiesberger26_interspeech","category":"health-clinical","institutions":["Technical University of Munich","Munich Center for Machine Learning","Friedrich-Schiller-University Jena","Imperial College London","Munich Data Science Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1878","pdf":"https://www.isca-archive.org/interspeech_2026/spiesberger26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/spiesberger26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/spiesberger26_interspeech/markdown.md"},{"id":"srirag26_interspeech","title":"TriageSim: A Conversational Emergency Triage Simulation Framework from Structured Electronic Health Records","authors":["Dipankar Srirag","Quoc Dung Nguyen","Aditya Joshi","Padmanesan Narasimhan","Salil Kanhere"],"year":2026,"doi":"10.21437/Interspeech.2026-819","isca_url":"https://www.isca-archive.org/interspeech_2026/srirag26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/srirag26_interspeech.pdf","session":"Speech and Language Technologies in Healthcare","topics":["speech-llm","spoken-language-understanding","dataset"],"category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["University of New South Wales"],"funding":["NHMRC Ideas Grant"],"code":{"url":"https://github.com/dipankarsrirag/triage-sim.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"srirag26_interspeech","category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["University of New South Wales"],"code":"https://github.com/dipankarsrirag/triage-sim.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-819","pdf":"https://www.isca-archive.org/interspeech_2026/srirag26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/srirag26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/srirag26_interspeech/markdown.md"},{"id":"srivastav26_interspeech","title":"Open ASR Leaderboard: Towards Reproducible and Transparent Multilingual and Long-Form Speech Recognition Evaluation","authors":["Vaibhav Srivastav","Steven Zheng","Eric Bezzam","Eustache Le Bihan","Nithin Rao Koluguri","Piotr Żelasko","Somshubra Majumdar","Adel Moumen","Sanchit Gandhi"],"year":2026,"doi":"10.21437/Interspeech.2026-1902","isca_url":"https://www.isca-archive.org/interspeech_2026/srivastav26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/srivastav26_interspeech.pdf","session":"Speech Benchmarks, Evaluation, and Resources","topics":["asr","multilingual","evaluation"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Hugging Face","NVIDIA","University of Cambridge","Mistral AI","OpenAI"],"code":{"url":"https://github.com/huggingface/open_asr_leaderboard","stars":260,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"srivastav26_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Hugging Face","NVIDIA","University of Cambridge","Mistral AI","OpenAI"],"code":"https://github.com/huggingface/open_asr_leaderboard","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1902","pdf":"https://www.isca-archive.org/interspeech_2026/srivastav26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/srivastav26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/srivastav26_interspeech/markdown.md"},{"id":"stanek26_interspeech","title":"What Do Deepfake Speech Detectors Actually Hear?","authors":["Vojtěch Staněk","Veronika Jirmusová","Anton Firc","Kamil Malinka","Jakub Reš","Martin Perešíni"],"year":2026,"doi":"10.21437/Interspeech.2026-123","isca_url":"https://www.isca-archive.org/interspeech_2026/stanek26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/stanek26_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["Brno University of Technology"],"funding":["Brno University of Technology internal project"],"code":{"url":"https://github.com/Security-FIT/IG_for_SSL_detectors","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"stanek26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["Brno University of Technology"],"code":"https://github.com/Security-FIT/IG_for_SSL_detectors","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-123","pdf":"https://www.isca-archive.org/interspeech_2026/stanek26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/stanek26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/stanek26_interspeech/markdown.md"},{"id":"stanek26b_interspeech","title":"Ethical and Technical Limits of Deepfake Speech Datasets","authors":["Vojtěch Staněk","Eva Trnovská","Kamil Malinka","Anton Firc"],"year":2026,"doi":"10.21437/Interspeech.2026-124","isca_url":"https://www.isca-archive.org/interspeech_2026/stanek26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/stanek26b_interspeech.pdf","session":"Safeguards for Synthetic Speech: Ethical, Technical, and Legal Perspectives","topics":["audio-deepfake","evaluation","dataset"],"category":"resources-evaluation","institutions":["Brno University of Technology"],"funding":["Brno University of Technology internal project FIT-S-26-9011","Ministry of Education, Youth and Sports of the Czech Republic"],"code":{"url":"https://security-fit.github.io/deepfake_speech_datasets_app/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"stanek26b_interspeech","category":"resources-evaluation","institutions":["Brno University of Technology"],"code":"https://security-fit.github.io/deepfake_speech_datasets_app/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-124","pdf":"https://www.isca-archive.org/interspeech_2026/stanek26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/stanek26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/stanek26b_interspeech/markdown.md"},{"id":"stanek26c_interspeech","title":"RAT: Reference-Augmented Training for ASV Anti-Spoofing","authors":["Vojtěch Staněk","Anton Firc","Jakub Reš","Kamil Malinka"],"year":2026,"doi":"10.21437/Interspeech.2026-132","isca_url":"https://www.isca-archive.org/interspeech_2026/stanek26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/stanek26c_interspeech.pdf","session":"Speaker Verification and Anti-Spoofing","topics":["asr","self-supervised","evaluation"],"category":"speaker","institutions":["Brno University of Technology"],"funding":["Brno University of Technology","Ministry of Education, Youth and Sports of the Czech Republic"],"code":{"url":"https://github.com/Security-FIT/RAT","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"stanek26c_interspeech","category":"speaker","institutions":["Brno University of Technology"],"code":"https://github.com/Security-FIT/RAT","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-132","pdf":"https://www.isca-archive.org/interspeech_2026/stanek26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/stanek26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/stanek26c_interspeech/markdown.md"},{"id":"stanley26_interspeech","title":"Beta Rebound as a Neural Signature for Speech Movement: Preliminary Evidence Using Magnetoencephalography","authors":["Keerthana Stanley","Jun Wang","Paul Ferrari"],"year":2026,"doi":"10.21437/Interspeech.2026-2995","isca_url":"https://www.isca-archive.org/interspeech_2026/stanley26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/stanley26_interspeech.pdf","session":"Brain Studies and Speech","topics":["paralinguistics","speech-llm","health"],"category":"phonetics-linguistics","institutions":["University of Texas at Austin","Helen DeVos Children's Hospital","Corewell Health","Michigan State University"],"funding":["University of Texas System Brain Initiative","National Institutes of Health"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"stanley26_interspeech","category":"phonetics-linguistics","institutions":["University of Texas at Austin","Helen DeVos Children's Hospital","Corewell Health","Michigan State University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2995","pdf":"https://www.isca-archive.org/interspeech_2026/stanley26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/stanley26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/stanley26_interspeech/markdown.md"},{"id":"stasak26_interspeech","title":"CalliOpeNLP: A Standalone Digital Health Voice Data Collection Research Tool","authors":["Brian Stasak","Rebecca Li","Antonia Chacon","Rebecca Black","Cate Madill"],"year":2026,"doi":"10.21437/Interspeech.2026-1783","isca_url":"https://www.isca-archive.org/interspeech_2026/stasak26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/stasak26_interspeech.pdf","session":"Challenges in Speech Data Collection, Curation, and Annotation","topics":["dataset","evaluation","speech-enhancement"],"category":"resources-evaluation","institutions":["University of Sydney"],"code":{"url":"https://github.com/DrBrianStasak/CalliOpeNLP/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"stasak26_interspeech","category":"resources-evaluation","institutions":["University of Sydney"],"code":"https://github.com/DrBrianStasak/CalliOpeNLP/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1783","pdf":"https://www.isca-archive.org/interspeech_2026/stasak26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/stasak26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/stasak26_interspeech/markdown.md"},{"id":"stasica26_interspeech","title":"Revisiting Emotion-Based Triage: Evidence from French Emergency Call Data","authors":["Elio Stasica","Clément Joly","Amandine Lecomte","Vincent P. Martin","Romain Serizel","Emmanuel Vincent","Tahar Chouihed"],"year":2026,"doi":"10.21437/Interspeech.2026-1265","isca_url":"https://www.isca-archive.org/interspeech_2026/stasica26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/stasica26_interspeech.pdf","session":"Speech and Language Technologies in Healthcare","topics":["paralinguistics","emotion-recognition","evaluation"],"category":"health-clinical","institutions":["University of Lorraine","CNRS","Inria","CHRU-Nancy","INSERM"],"funding":["Grand Est ENACT AI Cluster"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"stasica26_interspeech","category":"health-clinical","institutions":["University of Lorraine","CNRS","Inria","CHRU-Nancy","INSERM"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1265","pdf":"https://www.isca-archive.org/interspeech_2026/stasica26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/stasica26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/stasica26_interspeech/markdown.md"},{"id":"stergaard26_interspeech","title":"Don't Listen to Me: A Lightweight, Low-Latency Model for Own-Voice Cancellation in Far-Field Speech Enhancement","authors":["Mads Østergaard","Alexander Neergaard Zahid","Karl Ulbæk","Andreas Bagge","Kenny Falkær Olsen","Rasmus Lindrup"],"year":2026,"doi":"10.21437/Interspeech.2026-3430","isca_url":"https://www.isca-archive.org/interspeech_2026/stergaard26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/stergaard26_interspeech.pdf","session":"Source Separation 2","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["WS Audiology","Technical University of Denmark","Verth"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"stergaard26_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["WS Audiology","Technical University of Denmark","Verth"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3430","pdf":"https://www.isca-archive.org/interspeech_2026/stergaard26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/stergaard26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/stergaard26_interspeech/markdown.md"},{"id":"su26_interspeech","title":"Robust LLM-based Audio-Visual Speech Recognition with Sparse Modality Alignment and Visual Unit-Guided Refinement","authors":["Fei Su","Cancan Li","Juan Liu","Wei Ju","Hongbin Suo","Ming Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1277","isca_url":"https://www.isca-archive.org/interspeech_2026/su26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/su26_interspeech.pdf","session":"Robust Audio-Visual Speech Recognition","topics":["asr","speech-llm","multilingual"],"category":"asr","labels":["robustness-noise"],"institutions":["Wuhan University","Chinese University of Hong Kong, Shenzhen","OPPO"],"funding":["National Natural Science Foundation of China","Yangtze River Delta Science and Technology Innovation Community Joint Research Project","OPPO"],"code":{"url":"https://github.com/yakumo72/AVUR-LLM","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"su26_interspeech","category":"asr","labels":["robustness-noise"],"institutions":["Wuhan University","Chinese University of Hong Kong, Shenzhen","OPPO"],"code":"https://github.com/yakumo72/AVUR-LLM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1277","pdf":"https://www.isca-archive.org/interspeech_2026/su26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/su26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/su26_interspeech/markdown.md"},{"id":"subedi26_interspeech","title":"BiMamba2 Masked Discrete-Unit Prediction for Multilingual Speech Representation for Unsupervised Speech in the Wild Challenge","authors":["Prakriti Subedi","Howard Prioleau","Saurav Aryal"],"year":2026,"doi":"10.21437/Interspeech.2026-2966","isca_url":"https://www.isca-archive.org/interspeech_2026/subedi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/subedi26_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["self-supervised","multilingual","speaker-verification"],"category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Howard University"],"funding":["Office of Naval Research","Department of the Navy","NIH Common Fund","Amazon Research Award"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"subedi26_interspeech","category":"speaker","labels":["multilingual","self-supervised"],"institutions":["Howard University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2966","pdf":"https://www.isca-archive.org/interspeech_2026/subedi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/subedi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/subedi26_interspeech/markdown.md"},{"id":"sultana26_interspeech","title":"A Fine-Grained Acoustically-Aware Pre-training Encoder for Speech Quality Assessment","authors":["Subrina Sultana","Donald S. Williamson"],"year":2026,"doi":"10.21437/Interspeech.2026-1607","isca_url":"https://www.isca-archive.org/interspeech_2026/sultana26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sultana26_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["speech-enhancement","self-supervised","evaluation"],"category":"resources-evaluation","labels":["efficient-on-device","self-supervised","robustness-noise"],"institutions":["Ohio State University"],"funding":["National Science Foundation","Ohio Supercomputer Center"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sultana26_interspeech","category":"resources-evaluation","labels":["efficient-on-device","self-supervised","robustness-noise"],"institutions":["Ohio State University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1607","pdf":"https://www.isca-archive.org/interspeech_2026/sultana26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sultana26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sultana26_interspeech/markdown.md"},{"id":"sulun26_interspeech","title":"Lightweight Emotion Recognition with Disjoint Modality Fusion","authors":["Serkan Sulun"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/sulun26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sulun26_interspeech.pdf","session":"Speech Analysis, Data Resources and Research Tools","topics":["speech-emotion-recognition","multimodal-fusion","on-device"],"category":"paralinguistics-emotion","labels":["efficient-on-device"],"institutions":["INESC TEC","University of Porto"],"code":{"url":"https://sulun.org/lightdmf","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sulun26_interspeech","category":"paralinguistics-emotion","labels":["efficient-on-device"],"institutions":["INESC TEC","University of Porto"],"code":"https://sulun.org/lightdmf","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/sulun26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/sulun26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sulun26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sulun26_interspeech/markdown.md"},{"id":"sun26_interspeech","title":"Prosodic ABX: A Language-Agnostic Method for Measuring Prosodic Contrast in Speech Representations","authors":["Haitong Sun","Stephen McIntosh","Kwanghee Choi","Eunjung Yeo","Daisuke Saito","Nobuaki Minematsu"],"year":2026,"doi":"10.21437/Interspeech.2026-478","isca_url":"https://www.isca-archive.org/interspeech_2026/sun26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sun26_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["prosody","self-supervised","evaluation"],"category":"resources-evaluation","labels":["multilingual","self-supervised"],"institutions":["University of Tokyo","University of Texas at Austin"],"code":{"url":"https://github.com/stephenmac7/prosodic-abx","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sun26_interspeech","category":"resources-evaluation","labels":["multilingual","self-supervised"],"institutions":["University of Tokyo","University of Texas at Austin"],"code":"https://github.com/stephenmac7/prosodic-abx","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-478","pdf":"https://www.isca-archive.org/interspeech_2026/sun26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26_interspeech/markdown.md"},{"id":"sun26b_interspeech","title":"PAN-Mask: Pathology-Aware Neurological Masking with End-to-End Learnable Weights for Neurological Disorder Detection from Speech","authors":["Qi Sun","Junhao Fan","Boao Jing","Ziqi Chen","Guodong Lin","Wei-Qiang Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1006","isca_url":"https://www.isca-archive.org/interspeech_2026/sun26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sun26b_interspeech.pdf","session":"Pathological Speech Assessment 2","topics":["self-supervised","health","paralinguistics"],"category":"health-clinical","labels":["multilingual","self-supervised"],"institutions":["Tsinghua University","Georgetown University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sun26b_interspeech","category":"health-clinical","labels":["multilingual","self-supervised"],"institutions":["Tsinghua University","Georgetown University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1006","pdf":"https://www.isca-archive.org/interspeech_2026/sun26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26b_interspeech/markdown.md"},{"id":"sun26c_interspeech","title":"Non-linear Effects of Semantic Relevance on Word Duration in Spontaneous Speech","authors":["Kun Sun","Rong Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1062","isca_url":"https://www.isca-archive.org/interspeech_2026/sun26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sun26c_interspeech.pdf","session":"Modeling L1 Acquisition","topics":["prosody","phonetics","evaluation"],"category":"phonetics-linguistics","institutions":["Tongji University","University of Tubingen"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sun26c_interspeech","category":"phonetics-linguistics","institutions":["Tongji University","University of Tubingen"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1062","pdf":"https://www.isca-archive.org/interspeech_2026/sun26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26c_interspeech/markdown.md"},{"id":"sun26d_interspeech","title":"Automated Gradient-Driven Parameter Sharing for Low-Resource Multilingual Speech-to-Text Translation","authors":["Ruiyan Sun","Satoshi Nakamura"],"year":2026,"doi":"10.21437/Interspeech.2026-1292","isca_url":"https://www.isca-archive.org/interspeech_2026/sun26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sun26d_interspeech.pdf","session":"Multilingual Speech 2","topics":["speech-translation","multilingual","low-resource"],"category":"translation","labels":["low-resource","multilingual"],"institutions":["Chinese University of Hong Kong"],"funding":["National Natural Science Foundation of China","Program for Guangdong Introducing Innovative and Entrepreneurial Teams"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sun26d_interspeech","category":"translation","labels":["low-resource","multilingual"],"institutions":["Chinese University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1292","pdf":"https://www.isca-archive.org/interspeech_2026/sun26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26d_interspeech/markdown.md"},{"id":"sun26e_interspeech","title":"Label Correction Enhanced Dual-Stream Multiple Instance Learning for Weakly-Supervised Depression Detection in Speech","authors":["Yanfei Sun","Yuanyuan Zhou","Xinzhou Xu","Jin Qi","Feiyi Xu","Zhao Ren","Bjoern Schuller"],"year":2026,"doi":"10.21437/Interspeech.2026-1716","isca_url":"https://www.isca-archive.org/interspeech_2026/sun26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sun26e_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 2","topics":["paralinguistics","emotion-recognition","self-supervised"],"category":"health-clinical","institutions":["Nanjing University of Posts and Telecommunications","Wuxi University","Graz University of Technology","University of Bremen","Technische Universitat Munchen","Imperial College London"],"funding":["National Natural Science Foundation of China","Primary Research & Development Plan of Jiangsu Province","Humanities and Social Science Foundation of China Ministry of Education","China Postdoctoral Science Foundation","DFG"],"code":{"url":"https://github.com/zhou123122/SLLC","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sun26e_interspeech","category":"health-clinical","institutions":["Nanjing University of Posts and Telecommunications","Wuxi University","Graz University of Technology","University of Bremen","Technische Universitat Munchen","Imperial College London"],"code":"https://github.com/zhou123122/SLLC","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1716","pdf":"https://www.isca-archive.org/interspeech_2026/sun26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26e_interspeech/markdown.md"},{"id":"sun26f_interspeech","title":"Decoding the Trade-off: A Large-Scale Analysis of Latency and Stability in LLM-based Speech Translation Cascades","authors":["Shinyoung Sun","Taehoon Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-1821","isca_url":"https://www.isca-archive.org/interspeech_2026/sun26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sun26f_interspeech.pdf","session":"Robust and Real-World ASR Systems","topics":["speech-translation","asr","evaluation"],"category":"translation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Sogang University"],"funding":["Institute for Information & Communications Technology Planning & Evaluation","Ministry of Science and ICT, Republic of Korea","Ministry of Culture, Sports and Tourism, Republic of Korea","Korea Creative Content Agency","National Research Foundation of Korea","Sogang University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sun26f_interspeech","category":"translation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Sogang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1821","pdf":"https://www.isca-archive.org/interspeech_2026/sun26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26f_interspeech/markdown.md"},{"id":"sun26g_interspeech","title":"ADD-DINO: A Two-Stage Self-Distillation Framework for Audio Deepfake Detection","authors":["Zhaorui Sun","Yihao Chen","Qiao Chen","Minqiang Xu","Sian Fang","Lin Liu","Jianbo Zhan","Yan Song","Guoping Hu","Lirong Dai"],"year":2026,"doi":"10.21437/Interspeech.2026-1847","isca_url":"https://www.isca-archive.org/interspeech_2026/sun26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sun26g_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["audio-deepfake","self-supervised","speaker-verification"],"category":"deepfake-security","labels":["low-resource","self-supervised"],"institutions":["Hefei iFly Digital Technology Co. Ltd","University of Science and Technology of China","Xinjiang University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sun26g_interspeech","category":"deepfake-security","labels":["low-resource","self-supervised"],"institutions":["Hefei iFly Digital Technology Co. Ltd","University of Science and Technology of China","Xinjiang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1847","pdf":"https://www.isca-archive.org/interspeech_2026/sun26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26g_interspeech/markdown.md"},{"id":"sun26h_interspeech","title":"Activation Steering for Accent Adaptation in Large Audio Language Models","authors":["Jinuo Sun","Yang Xiao","Sung Kyun Chung","Qiuchi Hu","Gongping Huang","Eun-Jung Holden","Ting Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-2166","isca_url":"https://www.isca-archive.org/interspeech_2026/sun26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sun26h_interspeech.pdf","session":"Domain Adaptation & Accented ASR","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["University of Melbourne","Wuhan University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sun26h_interspeech","category":"asr","labels":["low-resource","self-supervised","robustness-noise"],"institutions":["University of Melbourne","Wuhan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2166","pdf":"https://www.isca-archive.org/interspeech_2026/sun26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26h_interspeech/markdown.md"},{"id":"sun26i_interspeech","title":"Moot-Court: Training-Free Dialectical Reasoning for Depression Detection","authors":["Yuqing Sun","Jian Zhao","Haoxun Li","Leyuan Qu","Taihao Li"],"year":2026,"doi":"10.21437/Interspeech.2026-3099","isca_url":"https://www.isca-archive.org/interspeech_2026/sun26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sun26i_interspeech.pdf","session":"Pathological Speech Assessment 1","topics":["paralinguistics","speech-llm","evaluation"],"category":"health-clinical","institutions":["University of Chinese Academy of Sciences"],"funding":["National Natural Science Foundation of China","Key Scientific Research Program of Hangzhou","Natural Science Foundation of Hangzhou","Scientific Research Starting Foundation of Hangzhou Institute for Advanced Study","Zhejiang Provincial Natural Science Foundation of China","Key R&D Program of Zhejiang"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sun26i_interspeech","category":"health-clinical","institutions":["University of Chinese Academy of Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3099","pdf":"https://www.isca-archive.org/interspeech_2026/sun26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26i_interspeech/markdown.md"},{"id":"sun26j_interspeech","title":"MSU-Bench: Towards Understanding the Conversational Multi-Speaker Scenarios","authors":["Zhaokai Sun","Shuai Wang","Zhennan Lin","Chengyou Wang","Dehui Gao","Yuang Cao","Chunjiang He","Pan Zhou","Lei Xie"],"year":2026,"doi":"10.21437/Interspeech.2026-3344","isca_url":"https://www.isca-archive.org/interspeech_2026/sun26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sun26j_interspeech.pdf","session":"Spoken Language Understanding","topics":["speech-llm","speaker-diarization","spoken-language-understanding"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Northwestern Polytechnical University","Nanjing University","Shenzhen Loop Area Institute","Li Auto"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sun26j_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Northwestern Polytechnical University","Nanjing University","Shenzhen Loop Area Institute","Li Auto"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3344","pdf":"https://www.isca-archive.org/interspeech_2026/sun26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sun26j_interspeech/markdown.md"},{"id":"sunder26_interspeech","title":"Parameter-Efficient Adaptation of Speech-Aware LLMs for Timestamp Prediction","authors":["Vishal Sunder","Samuel Thomas","Xulin Fan","Brian Kingsbury","George Saon","Avihu Dekel","Luis Lastras"],"year":2026,"doi":"10.21437/Interspeech.2026-2441","isca_url":"https://www.isca-archive.org/interspeech_2026/sunder26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sunder26_interspeech.pdf","session":"Multimodal Speech Processing and Speech LLM Systems","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["self-supervised"],"institutions":["IBM","University of Illinois Urbana-Champaign"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sunder26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["IBM","University of Illinois Urbana-Champaign"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2441","pdf":"https://www.isca-archive.org/interspeech_2026/sunder26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sunder26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sunder26_interspeech/markdown.md"},{"id":"sung26_interspeech","title":"fMRI Decoding of Speech Conditions Across Brain Regions of Interest for Neural Evaluation of Speech Enhancement","authors":["Ching-Chih Sung","Francis Pingfan Chien","Li-wei Chen","Berrak Sisman","Borching Su","Yu-Te Wang","Yu Tsao"],"year":2026,"doi":"10.21437/Interspeech.2026-1947","isca_url":"https://www.isca-archive.org/interspeech_2026/sung26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/sung26_interspeech.pdf","session":"Brain Studies and Speech","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","labels":["self-supervised","robustness-noise"],"institutions":["National Taiwan University","Academia Sinica","Johns Hopkins University","National Yang Ming Chiao Tung University"],"code":{"url":"https://github.com/JohnSung0501/fMRI-Decoding","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"sung26_interspeech","category":"enhancement-separation","labels":["self-supervised","robustness-noise"],"institutions":["National Taiwan University","Academia Sinica","Johns Hopkins University","National Yang Ming Chiao Tung University"],"code":"https://github.com/JohnSung0501/fMRI-Decoding","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1947","pdf":"https://www.isca-archive.org/interspeech_2026/sung26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/sung26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/sung26_interspeech/markdown.md"},{"id":"susac26_interspeech","title":"Stuttering Classification and Segmentation with Attention-Based Multiple Instance Learning","authors":["Petar Sušac","Sebastian P. Bayerl","Hrvoje Džapo"],"year":2026,"doi":"10.21437/Interspeech.2026-1091","isca_url":"https://www.isca-archive.org/interspeech_2026/susac26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/susac26_interspeech.pdf","session":"Speech and Language Technologies in Healthcare","topics":["paralinguistics","self-supervised","evaluation"],"category":"health-clinical","labels":["self-supervised"],"institutions":["University of Zagreb","Rosenheim Technical University of Applied Sciences"],"funding":["European Union NextGenerationEU","NPOO VISTAHealth"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"susac26_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["University of Zagreb","Rosenheim Technical University of Applied Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1091","pdf":"https://www.isca-archive.org/interspeech_2026/susac26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/susac26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/susac26_interspeech/markdown.md"},{"id":"suzuki26_interspeech","title":"ELSA: Acoustic Event-Level Semantic Alignment for Fine-Grained Reference-Free Text-to-Audio Evaluation","authors":["Shuntaro Suzuki","Kento Tokura","Daichi Yashima","Kanon Amemiya","Komei Sugiura","Shinnosuke Takamichi"],"year":2026,"doi":"10.21437/Interspeech.2026-914","isca_url":"https://www.isca-archive.org/interspeech_2026/suzuki26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/suzuki26_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["evaluation","self-supervised","speech-llm"],"category":"resources-evaluation","labels":["self-supervised"],"institutions":["Keio University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"suzuki26_interspeech","category":"resources-evaluation","labels":["self-supervised"],"institutions":["Keio University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-914","pdf":"https://www.isca-archive.org/interspeech_2026/suzuki26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/suzuki26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/suzuki26_interspeech/markdown.md"},{"id":"swietojanski26_interspeech","title":"Segmental Attention Decoding With Long Form Acoustic Encodings","authors":["Pawel Swietojanski","Xinwei Li","Mingbin Xu","Takaaki Hori","Dogan Can","Xiaodan Zhuang"],"year":2026,"doi":"10.21437/Interspeech.2026-341","isca_url":"https://www.isca-archive.org/interspeech_2026/swietojanski26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/swietojanski26_interspeech.pdf","session":"Long-form Audio & New Attention Approaches","topics":["asr","self-supervised","on-device"],"category":"asr","labels":["self-supervised"],"institutions":["Apple"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"swietojanski26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Apple"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-341","pdf":"https://www.isca-archive.org/interspeech_2026/swietojanski26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/swietojanski26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/swietojanski26_interspeech/markdown.md"},{"id":"syed26_interspeech","title":"corpusgen: An Open-Source Toolkit for Phoneme-Coverage-Optimized Speech Corpus Design Across Languages","authors":["Muntaser Syed","Marius Silaghi","Sharun Akter Khushbu","Fariha Jaigirdar"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/syed26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/syed26_interspeech.pdf","session":"Speech Analysis, Data Resources and Research Tools","topics":["tts","asr","low-resource"],"category":"resources-evaluation","labels":["low-resource","multilingual"],"institutions":["Florida Institute of Technology","Daffodil International University","Deakin University"],"code":{"url":"https://github.com/jemsbhai/corpusgen","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"syed26_interspeech","category":"resources-evaluation","labels":["low-resource","multilingual"],"institutions":["Florida Institute of Technology","Daffodil International University","Deakin University"],"code":"https://github.com/jemsbhai/corpusgen","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/syed26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/syed26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/syed26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/syed26_interspeech/markdown.md"},{"id":"syllas26_interspeech","title":"Deterministic Prompting for Speaker-Stable Low-Resource Greek TTS","authors":["Georgios Syllas","Efthymios Georgiou","Kosmas Kritsis","Alexandros Potamianos"],"year":2026,"doi":"10.21437/Interspeech.2026-2481","isca_url":"https://www.isca-archive.org/interspeech_2026/syllas26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/syllas26_interspeech.pdf","session":"Text Processing for Speech Synthesis","topics":["tts","low-resource","multilingual"],"category":"tts","labels":["low-resource","multilingual","generative-model"],"institutions":["Athena R.C","University of Bern","National Technical University of Athens"],"funding":["European High-Performance Computing Joint Undertaking","Greek Ministry of Digital Governance and Artificial Intelligence"],"code":{"url":"https://github.com/gsyllas/greek-stable-tts/tree/main/scripts/data","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"syllas26_interspeech","category":"tts","labels":["low-resource","multilingual","generative-model"],"institutions":["Athena R.C","University of Bern","National Technical University of Athens"],"code":"https://github.com/gsyllas/greek-stable-tts/tree/main/scripts/data","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2481","pdf":"https://www.isca-archive.org/interspeech_2026/syllas26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/syllas26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/syllas26_interspeech/markdown.md"},{"id":"ta26_interspeech","title":"Age-related Differences in the Perception of Vowel Length Contrast in Northern Vietnamese: The Case of Hoang Van (Bac Ninh) Variety","authors":["Van Dat Ta","Baoya Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1167","isca_url":"https://www.isca-archive.org/interspeech_2026/ta26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ta26_interspeech.pdf","session":"Gender- and Age-Related Speech Characteristics","topics":["phonetics","evaluation","speech-perception"],"category":"phonetics-linguistics","institutions":["Peking University"],"funding":["National Social Science Fund of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ta26_interspeech","category":"phonetics-linguistics","institutions":["Peking University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1167","pdf":"https://www.isca-archive.org/interspeech_2026/ta26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ta26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ta26_interspeech/markdown.md"},{"id":"ta26b_interspeech","title":"Progressive Weak Supervision for Speech Emotion Recognition","authors":["Bao Thang Ta","Huynh Thi Thanh Binh","Van Hai Do"],"year":2026,"doi":"10.21437/Interspeech.2026-1587","isca_url":"https://www.isca-archive.org/interspeech_2026/ta26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ta26b_interspeech.pdf","session":"Speech Emotion Recognition and Representation 1","topics":["speech-emotion-recognition","self-supervised","low-resource"],"category":"paralinguistics-emotion","institutions":["Viettel AI","Viettel Group","Hanoi University of Science and Technology","Thuyloi University"],"funding":["Vingroup Innovation Foundation"],"code":{"url":"https://github.com/skyemo47/PWS","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ta26b_interspeech","category":"paralinguistics-emotion","institutions":["Viettel AI","Viettel Group","Hanoi University of Science and Technology","Thuyloi University"],"code":"https://github.com/skyemo47/PWS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1587","pdf":"https://www.isca-archive.org/interspeech_2026/ta26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ta26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ta26b_interspeech/markdown.md"},{"id":"tabatabaee26_interspeech","title":"Towards Language-Agnostic Speech Inversion","authors":["Saba Tabatabaee","Mark Tiede","Suzanne Boyce","Liran Oren","Carol Espy-Wilson"],"year":2026,"doi":"10.21437/Interspeech.2026-1633","isca_url":"https://www.isca-archive.org/interspeech_2026/tabatabaee26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tabatabaee26_interspeech.pdf","session":"Methods and Data for Vocal Tract Shape and Articulation Analysis","topics":["phonetics","self-supervised","multilingual"],"category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["University of Maryland","Yale University","University of Cincinnati"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tabatabaee26_interspeech","category":"phonetics-linguistics","labels":["multilingual","self-supervised"],"institutions":["University of Maryland","Yale University","University of Cincinnati"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1633","pdf":"https://www.isca-archive.org/interspeech_2026/tabatabaee26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tabatabaee26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tabatabaee26_interspeech/markdown.md"},{"id":"taghibeyglou26_interspeech","title":"Multi-Phonation Graph Learning with Self-Supervised Speech Embeddings for ALS Detection and Progression Prediction","authors":["Behrad TaghiBeyglou","Fatemeh Bagheri","Ervin Sejdic"],"year":2026,"doi":"10.21437/Interspeech.2026-844","isca_url":"https://www.isca-archive.org/interspeech_2026/taghibeyglou26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/taghibeyglou26_interspeech.pdf","session":"Pathological Speech Assessment 3","topics":["paralinguistics","self-supervised","health"],"category":"health-clinical","labels":["self-supervised"],"institutions":["University of Toronto","North York General Hospital"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"taghibeyglou26_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["University of Toronto","North York General Hospital"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-844","pdf":"https://www.isca-archive.org/interspeech_2026/taghibeyglou26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/taghibeyglou26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/taghibeyglou26_interspeech/markdown.md"},{"id":"taguchi26_interspeech","title":"Pretrained self-supervised speech models can recognize unseen consonants","authors":["Chihiro Taguchi","Éric Le Ferrand","Hirosi Nakagawa","Hitomi Ono","Kanji Kato","Emily Prud'hommeaux","David Chiang"],"year":2026,"doi":"10.21437/Interspeech.2026-2848","isca_url":"https://www.isca-archive.org/interspeech_2026/taguchi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/taguchi26_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["asr","low-resource","self-supervised"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Notre Dame","University at Buffalo","Tokyo University of Foreign Studies","Reitaku University","Boston College"],"funding":["National Science Foundation","JSPS KAKENHI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"taguchi26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Notre Dame","University at Buffalo","Tokyo University of Foreign Studies","Reitaku University","Boston College"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2848","pdf":"https://www.isca-archive.org/interspeech_2026/taguchi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/taguchi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/taguchi26_interspeech/markdown.md"},{"id":"takagi26_interspeech","title":"Investigating Human-Model Discrepancies in Speech Quality Assessment via Acoustic and Prosodic Perturbations","authors":["Masato Takagi","Masaya Kawamura","Reo Shimizu","Yuma Shirahata"],"year":2026,"doi":"10.21437/Interspeech.2026-1478","isca_url":"https://www.isca-archive.org/interspeech_2026/takagi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/takagi26_interspeech.pdf","session":"Speech Production and Perception 1","topics":["tts","evaluation","self-supervised"],"category":"resources-evaluation","institutions":["Nagoya Institute of Technology","LY Corporation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"takagi26_interspeech","category":"resources-evaluation","institutions":["Nagoya Institute of Technology","LY Corporation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1478","pdf":"https://www.isca-archive.org/interspeech_2026/takagi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/takagi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/takagi26_interspeech/markdown.md"},{"id":"takaki26_interspeech","title":"Head-Worn Dipole Microphone Array-based Speech Enhancement System for Single-Sided Deafness","authors":["Ken Takaki","Kouei Yamaoka","Yoshihiro Kawahara"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/takaki26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/takaki26_interspeech.pdf","session":"Speech and Language Processing for Health and Accessibility","topics":["speech-enhancement","evaluation","on-device"],"category":"enhancement-separation","institutions":["University of Tokyo"],"funding":["JSPS KAKENHI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"takaki26_interspeech","category":"enhancement-separation","institutions":["University of Tokyo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/takaki26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/takaki26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/takaki26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/takaki26_interspeech/markdown.md"},{"id":"takizawa26_interspeech","title":"Dissecting Sensitivity to Training Language in Self-Supervised Speech Learning Using Neural Audio Codec Tokens","authors":["Daigo Takizawa","Tomohiko Nakamura","Samuele Cornell","William Chen","Satoru Fukayama","Shinji Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-3002","isca_url":"https://www.isca-archive.org/interspeech_2026/takizawa26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/takizawa26_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 2","topics":["self-supervised","multilingual","asr"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["National Institute of Advanced Industrial Science and Technology","Carnegie Mellon University"],"funding":["Programs for Bridging the gap between R&D and the IDeal society (society 5.0) and Generating Economic and social value (BRIDGE)","R&D on Generative AI Foundation Models for the Physical Domain"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"takizawa26_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["National Institute of Advanced Industrial Science and Technology","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3002","pdf":"https://www.isca-archive.org/interspeech_2026/takizawa26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/takizawa26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/takizawa26_interspeech/markdown.md"},{"id":"tamiru26_interspeech","title":"High-Quality Speech Synthesis for Under-Resourced Ethiopian Languages","authors":["Rahel Mekonen Tamiru","Solomon Teferra Abate","Martha Yifiru Tachbelie","Abel Mulat Alemu","Samuel Rahimeto Kebede","Rosa Tsegaye Aga"],"year":2026,"doi":"10.21437/Interspeech.2026-2658","isca_url":"https://www.isca-archive.org/interspeech_2026/tamiru26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tamiru26_interspeech.pdf","session":"Low-Resource Speech Synthesis","topics":["tts","low-resource","multilingual"],"category":"tts","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release","generative-model"],"institutions":["Ethiopian Artificial Intelligence Institute","Addis Ababa University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tamiru26_interspeech","category":"tts","labels":["low-resource","multilingual","self-supervised","dataset-or-benchmark-release","generative-model"],"institutions":["Ethiopian Artificial Intelligence Institute","Addis Ababa University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2658","pdf":"https://www.isca-archive.org/interspeech_2026/tamiru26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tamiru26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tamiru26_interspeech/markdown.md"},{"id":"tanabu26_interspeech","title":"SSL-GMMVC: Interpretable Voice Conversion via Locally Linear GMM Transforms in Self-Supervised Representation Space","authors":["Tomoya Tanabu","Hiroshi Nishijima","Daisuke Saito","Nobuaki Minematsu"],"year":2026,"doi":"10.21437/Interspeech.2026-1688","isca_url":"https://www.isca-archive.org/interspeech_2026/tanabu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tanabu26_interspeech.pdf","session":"Voice Conversion","topics":["voice-conversion","self-supervised","evaluation"],"category":"tts","labels":["self-supervised"],"institutions":["University of Tokyo"],"funding":["JSPS KAKENHI"],"code":{"url":"https://github.com/tomoya-san/ssl-gmmvc","stars":5,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tanabu26_interspeech","category":"tts","labels":["self-supervised"],"institutions":["University of Tokyo"],"code":"https://github.com/tomoya-san/ssl-gmmvc","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1688","pdf":"https://www.isca-archive.org/interspeech_2026/tanabu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tanabu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tanabu26_interspeech/markdown.md"},{"id":"tang26_interspeech","title":"ADALA: A Wake-up Word Detection Framework Based on Adaptive Semi-supervised learning and Large Language Model","authors":["Nianhang Tang","Chaoyi Sun","Chao Cai"],"year":2026,"doi":"10.21437/Interspeech.2026-493","isca_url":"https://www.isca-archive.org/interspeech_2026/tang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tang26_interspeech.pdf","session":"Audio Language Models: Reasoning, Reliability, and Multimodal Understanding","topics":["keyword-spotting","self-supervised","multilingual"],"category":"asr","institutions":["Huazhong University of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tang26_interspeech","category":"asr","institutions":["Huazhong University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-493","pdf":"https://www.isca-archive.org/interspeech_2026/tang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tang26_interspeech/markdown.md"},{"id":"tang26b_interspeech","title":"DeSRPA: Decoupled Speech Role-Playing Agent via Inference-Time Intervention","authors":["Wenqiu Tang","Zhen Wan","Takahiro Komamizu","Ichiro Ide"],"year":2026,"doi":"10.21437/Interspeech.2026-1627","isca_url":"https://www.isca-archive.org/interspeech_2026/tang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tang26b_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["speech-llm","tts","self-supervised"],"category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["Nagoya University","National Institute of Informatics"],"funding":["JSPS KAKENHI"],"code":{"url":"https://github.com/steeremo971-commits/DeSRPA","stars":22,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tang26b_interspeech","category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["Nagoya University","National Institute of Informatics"],"code":"https://github.com/steeremo971-commits/DeSRPA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1627","pdf":"https://www.isca-archive.org/interspeech_2026/tang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tang26b_interspeech/markdown.md"},{"id":"tanner26_interspeech","title":"wav2VOT: automatic estimation of voice onset time, closure duration, and burst realisation with wav2vec2","authors":["James Tanner","Morgan Sonderegger","Jane Stuart-Smith","Tyler Kendall","Jeff Mielke"],"year":2026,"doi":"10.21437/Interspeech.2026-743","isca_url":"https://www.isca-archive.org/interspeech_2026/tanner26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tanner26_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["phonetics","self-supervised","dataset"],"category":"phonetics-linguistics","labels":["self-supervised"],"institutions":["University of Glasgow","McGill University","University of Oregon","North Carolina State University"],"funding":["Economic and Social Research Council","Natural Sciences and Engineering Research Council of Canada","Social Sciences and Humanities Research Council","National Science Foundation","Canada Research Chairs","British Academy","University of Glasgow"],"code":{"url":"https://github.com/james-tanner/wav2VOT","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tanner26_interspeech","category":"phonetics-linguistics","labels":["self-supervised"],"institutions":["University of Glasgow","McGill University","University of Oregon","North Carolina State University"],"code":"https://github.com/james-tanner/wav2VOT","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-743","pdf":"https://www.isca-archive.org/interspeech_2026/tanner26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tanner26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tanner26_interspeech/markdown.md"},{"id":"tao26_interspeech","title":"ANCHOR: Autoregressive Non-intrusive Chunk-Ordered Refinement for Joint Multi-Resolution Speech Quality Modeling","authors":["Zhuoyan Tao","Jiatong Shi","Hye-jin Shim","Shinji Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-927","isca_url":"https://www.isca-archive.org/interspeech_2026/tao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tao26_interspeech.pdf","session":"Speech and Audio Quality Assessment","topics":["speech-enhancement","self-supervised","evaluation"],"category":"resources-evaluation","labels":["streaming-real-time"],"institutions":["University of Southern California","Carnegie Mellon University"],"funding":["Advanced Cyberinfrastructure Coordination Ecosystem: Services & Support","National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tao26_interspeech","category":"resources-evaluation","labels":["streaming-real-time"],"institutions":["University of Southern California","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-927","pdf":"https://www.isca-archive.org/interspeech_2026/tao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tao26_interspeech/markdown.md"},{"id":"tatsumi26_interspeech","title":"Universality of Speech Emotion Recognition in Humans and Speech Language Models","authors":["Yuka Tatsumi","Nathan Roll","Robert D. Hawkins","Meghan Sumner","Dan Jurafsky"],"year":2026,"doi":"10.21437/Interspeech.2026-3061","isca_url":"https://www.isca-archive.org/interspeech_2026/tatsumi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tatsumi26_interspeech.pdf","session":"Multilingual and Cross-Lingual Paralinguistic Analysis and Processing","topics":["speech-emotion-recognition","self-supervised","evaluation"],"category":"paralinguistics-emotion","labels":["multilingual","self-supervised"],"institutions":["Stanford University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tatsumi26_interspeech","category":"paralinguistics-emotion","labels":["multilingual","self-supervised"],"institutions":["Stanford University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3061","pdf":"https://www.isca-archive.org/interspeech_2026/tatsumi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tatsumi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tatsumi26_interspeech/markdown.md"},{"id":"tawara26_interspeech","title":"Who Spoke What When? Evaluating Spoken Language Models for Conversational ASR with Semantic and Overlap-Aware Metrics","authors":["Naohiro Tawara","Samuele Cornell","Alexander Polok","Marc Delcroix","Lukáš Burget","Shinji Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-2912","isca_url":"https://www.isca-archive.org/interspeech_2026/tawara26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tawara26_interspeech.pdf","session":"Multi-Talker ASR & Speaker Diarization","topics":["asr","speaker-diarization","speech-llm"],"category":"asr","institutions":["NTT","Carnegie Mellon University","Brno University of Technology"],"funding":["Ministry of Education, Youth and Sports of the Czech Republic"],"code":{"url":"https://pcspeech-demo.fit.vut.cz/wsw2","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tawara26_interspeech","category":"asr","institutions":["NTT","Carnegie Mellon University","Brno University of Technology"],"code":"https://pcspeech-demo.fit.vut.cz/wsw2","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2912","pdf":"https://www.isca-archive.org/interspeech_2026/tawara26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tawara26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tawara26_interspeech/markdown.md"},{"id":"teikitohe26_interspeech","title":"Speech Recognition to Accelerate Documentation of Marquesan and Cook Islands Māori","authors":["Marie Teikitohe","Rolando Coto-Solano","Sally Akevai Nicholas"],"year":2026,"doi":"10.21437/Interspeech.2026-3276","isca_url":"https://www.isca-archive.org/interspeech_2026/teikitohe26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/teikitohe26_interspeech.pdf","session":"Pacific Voices: Speech Science and Technology for the Languages of the Pacific Ocean","topics":["asr","low-resource","multilingual"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of French Polynesia","Dartmouth College","University of Auckland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"teikitohe26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of French Polynesia","Dartmouth College","University of Auckland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3276","pdf":"https://www.isca-archive.org/interspeech_2026/teikitohe26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/teikitohe26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/teikitohe26_interspeech/markdown.md"},{"id":"thahmid26_interspeech","title":"Layer-wise Probing of Whisper's Encoder Representations for Bengali Phone-like Units","authors":["Munim Thahmid","Sadia Sharmin"],"year":2026,"doi":"10.21437/Interspeech.2026-2199","isca_url":"https://www.isca-archive.org/interspeech_2026/thahmid26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/thahmid26_interspeech.pdf","session":"Audio signal analysis","topics":["self-supervised","evaluation","phonetics"],"category":"asr","labels":["self-supervised"],"institutions":["Bangladesh University of Engineering and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"thahmid26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Bangladesh University of Engineering and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2199","pdf":"https://www.isca-archive.org/interspeech_2026/thahmid26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/thahmid26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/thahmid26_interspeech/markdown.md"},{"id":"thai26_interspeech","title":"Contrastive Regularization for Accent-Robust ASR","authors":["Van-Phat Thai","Aradhya Dhruv","Duc-Thinh Pham","Sameer Alam"],"year":2026,"doi":"10.21437/Interspeech.2026-949","isca_url":"https://www.isca-archive.org/interspeech_2026/thai26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/thai26_interspeech.pdf","session":"Domain Adaptation & Accented ASR","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["Nanyang Technological University","VinUniversity"],"funding":["National Research Foundation, Singapore","Civil Aviation Authority of Singapore","Aviation Transformation Programme"],"code":{"url":"https://github.com/thaivanphat95/robust-atc-asr","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"thai26_interspeech","category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["Nanyang Technological University","VinUniversity"],"code":"https://github.com/thaivanphat95/robust-atc-asr","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-949","pdf":"https://www.isca-archive.org/interspeech_2026/thai26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/thai26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/thai26_interspeech/markdown.md"},{"id":"thavarasa26_interspeech","title":"KuralHub: Exposing Typological Capability Frontiers in Multilingual Speech Emotion Recognition","authors":["Luxshan Thavarasa","Jubeerathan Thevakumar","Thanikan Sivatheepan","Uthayasanker Thayasivam"],"year":2026,"doi":"10.21437/Interspeech.2026-3502","isca_url":"https://www.isca-archive.org/interspeech_2026/thavarasa26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/thavarasa26_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["speech-emotion-recognition","multilingual","self-supervised"],"category":"paralinguistics-emotion","labels":["multilingual","self-supervised"],"institutions":["University of Moratuwa"],"code":{"url":"https://github.com/aaivu/KuralHub","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"thavarasa26_interspeech","category":"paralinguistics-emotion","labels":["multilingual","self-supervised"],"institutions":["University of Moratuwa"],"code":"https://github.com/aaivu/KuralHub","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3502","pdf":"https://www.isca-archive.org/interspeech_2026/thavarasa26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/thavarasa26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/thavarasa26_interspeech/markdown.md"},{"id":"thebaud26_interspeech","title":"Speaker Verification with Speech-Aware LLMs: Evaluation and Augmentation","authors":["Thomas Thebaud","Yuzhe Wang","Laureano Moro-Velázquez","Jesús Villalba-Lopez","Najim Dehak"],"year":2026,"doi":"10.21437/Interspeech.2026-2670","isca_url":"https://www.isca-archive.org/interspeech_2026/thebaud26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/thebaud26_interspeech.pdf","session":"Speaker Verification: Architectures, Losses, and LLMs","topics":["speaker-verification","speech-llm","self-supervised"],"category":"speaker","institutions":["Johns Hopkins University"],"code":{"url":"https://github.com/thomasthebaud/ASV-with-SpeechLLMs","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"thebaud26_interspeech","category":"speaker","institutions":["Johns Hopkins University"],"code":"https://github.com/thomasthebaud/ASV-with-SpeechLLMs","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2670","pdf":"https://www.isca-archive.org/interspeech_2026/thebaud26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/thebaud26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/thebaud26_interspeech/markdown.md"},{"id":"thienpondt26_interspeech","title":"Multi-Speaker Embeddings With Weakly Supervised Speaker Activity Detection For Granular Speaker Diarization","authors":["Jenthe Thienpondt","Kris Demuynck"],"year":2026,"doi":"10.21437/Interspeech.2026-2471","isca_url":"https://www.isca-archive.org/interspeech_2026/thienpondt26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/thienpondt26_interspeech.pdf","session":"Speaker Diarization 2","topics":["speaker-diarization","speaker-verification","self-supervised"],"category":"speaker","institutions":["Ghent University","imec"],"funding":["Research Foundation Flanders"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"thienpondt26_interspeech","category":"speaker","institutions":["Ghent University","imec"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2471","pdf":"https://www.isca-archive.org/interspeech_2026/thienpondt26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/thienpondt26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/thienpondt26_interspeech/markdown.md"},{"id":"tian26_interspeech","title":"Bagpiper-TTS: Natural Language Guided Universal Speech Synthesis","authors":["Jinchuan Tian","Haoran Wang","Siddhant Arora","Takashi Maekaku","Keita Goto","Jin Sakuma","Yusuke Shinohara","Chao-Han Huck Yang","Shinji Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-873","isca_url":"https://www.isca-archive.org/interspeech_2026/tian26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tian26_interspeech.pdf","session":"LLM Based Speech Synthesis","topics":["tts","speech-llm","multilingual"],"category":"tts","labels":["generative-model"],"institutions":["Carnegie Mellon University","LY Corporation","NVIDIA"],"funding":["ACCESS program","National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tian26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Carnegie Mellon University","LY Corporation","NVIDIA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-873","pdf":"https://www.isca-archive.org/interspeech_2026/tian26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tian26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tian26_interspeech/markdown.md"},{"id":"tian26b_interspeech","title":"DDSN: A Physics-Aware Decoupled Dual-Stream Network for Speech Packet Loss Concealment","authors":["Hao Tian","Yonghui Liu","Jianbing Liu","Kai Niu","Zhiqiang He"],"year":2026,"doi":"10.21437/Interspeech.2026-1295","isca_url":"https://www.isca-archive.org/interspeech_2026/tian26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tian26b_interspeech.pdf","session":"Active Noise and Echo Control, Sound Zones and Packet-Loss Concealment","topics":["speech-enhancement"],"category":"speech-coding","institutions":["Beijing University of Posts and Telecommunications","Fanvil Link Technology Co., Ltd"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tian26b_interspeech","category":"speech-coding","institutions":["Beijing University of Posts and Telecommunications","Fanvil Link Technology Co., Ltd"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1295","pdf":"https://www.isca-archive.org/interspeech_2026/tian26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tian26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tian26b_interspeech/markdown.md"},{"id":"tipaksorn26_interspeech","title":"AV-FlowSep: Audio-Visual Target Speaker Separation via Flow Matching","authors":["Pattara Tipaksorn","Wayupuk Sommuang","Kwanchiva Thangthai"],"year":2026,"doi":"10.21437/Interspeech.2026-1960","isca_url":"https://www.isca-archive.org/interspeech_2026/tipaksorn26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tipaksorn26_interspeech.pdf","session":"Source Separation 2","topics":["speech-enhancement","self-supervised","multilingual"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["NECTEC","Thammasat University"],"funding":["NECTEC","NSTDA"],"code":{"url":"https://github.com/CAI-NECTEC/AV-FlowSep","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tipaksorn26_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["NECTEC","Thammasat University"],"code":"https://github.com/CAI-NECTEC/AV-FlowSep","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1960","pdf":"https://www.isca-archive.org/interspeech_2026/tipaksorn26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tipaksorn26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tipaksorn26_interspeech/markdown.md"},{"id":"tiwari26_interspeech","title":"Say That Again: Visualizing Paralinguistic Cues with Prosody-Aware Diffusion","authors":["Shyamji Tiwari"],"year":2026,"doi":"10.21437/Interspeech.2026-2102","isca_url":"https://www.isca-archive.org/interspeech_2026/tiwari26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tiwari26_interspeech.pdf","session":"Paralinguistics","topics":["paralinguistics","emotion-recognition","dataset"],"category":"paralinguistics-emotion","labels":["generative-model"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tiwari26_interspeech","category":"paralinguistics-emotion","labels":["generative-model"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2102","pdf":"https://www.isca-archive.org/interspeech_2026/tiwari26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tiwari26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tiwari26_interspeech/markdown.md"},{"id":"tokuma26_interspeech","title":"Perception of English /iː/–/ɪ/ by Japanese Listeners under Silent-Centre and Devoiced Vowel Conditions","authors":["Shinichi Tokuma"],"year":2026,"doi":"10.21437/Interspeech.2026-211","isca_url":"https://www.isca-archive.org/interspeech_2026/tokuma26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tokuma26_interspeech.pdf","session":"Cross-Linguistic and L2 Phonetic Studies","topics":["phonetics","evaluation","low-resource"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Chuo University"],"funding":["Chuo University Personal Research Grant"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tokuma26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Chuo University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-211","pdf":"https://www.isca-archive.org/interspeech_2026/tokuma26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tokuma26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tokuma26_interspeech/markdown.md"},{"id":"toussaint26_interspeech","title":"Emergence of Phonetic Representations in EMG-based Silent Speech Interfaces","authors":["Guillaume Toussaint","Deborah Pereg","Kevin Scheck","Tanja Schultz","Jürgen Schmidhuber","Michael Wand"],"year":2026,"doi":"10.21437/Interspeech.2026-2499","isca_url":"https://www.isca-archive.org/interspeech_2026/toussaint26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/toussaint26_interspeech.pdf","session":"Assistive Technologies 2","topics":["speech-synthesis","self-supervised","low-resource"],"category":"tts","labels":["self-supervised"],"institutions":["SUPSI","University of Bremen","King Abdullah University of Science and Technology"],"funding":["Swiss National Science Foundation","German Research Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"toussaint26_interspeech","category":"tts","labels":["self-supervised"],"institutions":["SUPSI","University of Bremen","King Abdullah University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2499","pdf":"https://www.isca-archive.org/interspeech_2026/toussaint26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/toussaint26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/toussaint26_interspeech/markdown.md"},{"id":"toyin26_interspeech","title":"What Counts as an Error? Dual-Reference Benchmarking for Atypical ASR","authors":["Hawau Olamide Toyin","S Umesh","Hanan Aldarmaki"],"year":2026,"doi":"10.21437/Interspeech.2026-750","isca_url":"https://www.isca-archive.org/interspeech_2026/toyin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/toyin26_interspeech.pdf","session":"Assistive Technologies 2","topics":["asr","evaluation","low-resource"],"category":"resources-evaluation","institutions":["Mohamed bin Zayed University of Artificial Intelligence","Indian Institute of Technology Madras"],"code":{"url":"https://github.com/Theehawau/usecase_asr","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"toyin26_interspeech","category":"resources-evaluation","institutions":["Mohamed bin Zayed University of Artificial Intelligence","Indian Institute of Technology Madras"],"code":"https://github.com/Theehawau/usecase_asr","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-750","pdf":"https://www.isca-archive.org/interspeech_2026/toyin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/toyin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/toyin26_interspeech/markdown.md"},{"id":"toyin26b_interspeech","title":"Aligning Stuttered-Speech Research with End-User Needs: Scoping Review, Survey, and Guidelines","authors":["Hawau Olamide Toyin","Mutiah Apampa","Toluwani Aremu","Humaid Alblooshi","Ana Rita Valente","Gonçalo Leal","Zhengjun Yue","Zeerak Talat","Hanan Aldarmaki"],"year":2026,"doi":"10.21437/Interspeech.2026-1704","isca_url":"https://www.isca-archive.org/interspeech_2026/toyin26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/toyin26b_interspeech.pdf","session":"Clinical and Inclusive Speech Technology","topics":["asr","speech-llm","evaluation"],"category":"asr","institutions":["MBZUAI","SpeechCare","SLAI","Chinese University of Hong Kong, Shenzhen","University of Edinburgh","University of Aveiro"],"code":{"url":"https://github.com/Theehawau/stutterresearch_survey","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"toyin26b_interspeech","category":"asr","institutions":["MBZUAI","SpeechCare","SLAI","Chinese University of Hong Kong, Shenzhen","University of Edinburgh","University of Aveiro"],"code":"https://github.com/Theehawau/stutterresearch_survey","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1704","pdf":"https://www.isca-archive.org/interspeech_2026/toyin26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/toyin26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/toyin26b_interspeech/markdown.md"},{"id":"tran26_interspeech","title":"Deepfake Word Detection by Next-token Prediction using Fine-tuned Whisper","authors":["Hoan My Tran","Xin Wang","Wanying Ge","Xuechen Liu","Junichi Yamagishi"],"year":2026,"doi":"10.21437/Interspeech.2026-628","isca_url":"https://www.isca-archive.org/interspeech_2026/tran26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tran26_interspeech.pdf","session":"Speech Deepfake Detection: Robustness, Generalization, Attribution","topics":["speech-anti-spoofing","deepfake-detection","asr"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["Universite de Rennes","National Institute of Informatics"],"funding":["JST","PRESTO","NII International Internship Program"],"code":{"url":"https://github.com/nii-yamagishilab/Whisper-deepfake-word-detection","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tran26_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["Universite de Rennes","National Institute of Informatics"],"code":"https://github.com/nii-yamagishilab/Whisper-deepfake-word-detection","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-628","pdf":"https://www.isca-archive.org/interspeech_2026/tran26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tran26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tran26_interspeech/markdown.md"},{"id":"tran26b_interspeech","title":"Measuring English and Vietnamese language input and output in an Australian preschool – A longitudinal study","authors":["Ha Chi Tran","Minh Anh Tran","Mai Linh Tran","Weicong Li","Paola Escudero"],"year":2026,"doi":"10.21437/Interspeech.2026-3213","isca_url":"https://www.isca-archive.org/interspeech_2026/tran26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tran26b_interspeech.pdf","session":"CHILDSPACE: Child Home Interaction & Language Dynamics: Speech, Psychology, Affect, Computation, and Environments","topics":["paralinguistics","multilingual","dataset"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Western Sydney University","Deakin University"],"funding":["Australian Research Council","Australian Government Research Training Program Scholarship","ViêtSpeak"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tran26b_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Western Sydney University","Deakin University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3213","pdf":"https://www.isca-archive.org/interspeech_2026/tran26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tran26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tran26b_interspeech/markdown.md"},{"id":"tran26c_interspeech","title":"From Single to Multi-Label SER: Dataset and Mamba-Based Fusion Model","authors":["Vu Long Tran","Long Duc Do","Vinh Quang Nguyen","Ngoc Minh Nguyen","Quang Minh Le Pham","Hieu Trung Nguyen","Trang Thu Thi Nguyen"],"year":2026,"doi":"10.21437/Interspeech.2026-3458","isca_url":"https://www.isca-archive.org/interspeech_2026/tran26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tran26c_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["paralinguistics","emotion-recognition","self-supervised"],"category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["Hanoi University of Science and Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tran26c_interspeech","category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["Hanoi University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3458","pdf":"https://www.isca-archive.org/interspeech_2026/tran26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tran26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tran26c_interspeech/markdown.md"},{"id":"treffehn26_interspeech","title":"Screening Matters: A Comparative Study of Conventional and Crowdsourced Listening Tests","authors":["Anika Treffehn","Andrea Eichenseer","Emily Kratsch","Nicola Pia"],"year":2026,"doi":"10.21437/Interspeech.2026-1387","isca_url":"https://www.isca-archive.org/interspeech_2026/treffehn26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/treffehn26_interspeech.pdf","session":"Quality, Intelligibility and Evaluation of Speech and Codecs","topics":["speech-coding","evaluation","dataset"],"category":"resources-evaluation","institutions":["Fraunhofer-Gesellschaft"],"funding":["Free State of Bavaria"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"treffehn26_interspeech","category":"resources-evaluation","institutions":["Fraunhofer-Gesellschaft"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1387","pdf":"https://www.isca-archive.org/interspeech_2026/treffehn26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/treffehn26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/treffehn26_interspeech/markdown.md"},{"id":"truong26_interspeech","title":"QAMO: Quality-aware Multi-centroid One-class Learning For Speech Deepfake Detection","authors":["Duc-Tuan Truong","Tianchi Liu","Ruijie Tao","Junjie Li","Kong Aik Lee","Eng Siong Chng"],"year":2026,"doi":"10.21437/Interspeech.2026-1098","isca_url":"https://www.isca-archive.org/interspeech_2026/truong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/truong26_interspeech.pdf","session":"Speech Deepfake Detection, Attribution and Characterization","topics":["speech-deepfake-detection","self-supervised","evaluation"],"category":"deepfake-security","institutions":["Nanyang Technological University","National University of Singapore","Hong Kong Polytechnic University"],"code":{"url":"https://github.com/ductuantruong/QAMO","stars":11,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"truong26_interspeech","category":"deepfake-security","institutions":["Nanyang Technological University","National University of Singapore","Hong Kong Polytechnic University"],"code":"https://github.com/ductuantruong/QAMO","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1098","pdf":"https://www.isca-archive.org/interspeech_2026/truong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/truong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/truong26_interspeech/markdown.md"},{"id":"tsai26_interspeech","title":"The False Resonance: A Critical Examination of Emotion Embedding Similarity for Speech Generation Evaluation","authors":["Yun-Shao Tsai","Yi-Cheng Lin","Huang-Cheng Chou","Tzu-Wen Hsu","Yun-Man Hsu","Chun Wei Chen","Shrikanth Narayanan","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-39","isca_url":"https://www.isca-archive.org/interspeech_2026/tsai26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tsai26_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["evaluation","speech-generation","self-supervised"],"category":"resources-evaluation","labels":["self-supervised","robustness-noise"],"institutions":["National Taiwan University","University of Southern California"],"funding":["Ministry of Education","NSTC Taiwan","US NSF","ODNI IARPA ARTS"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tsai26_interspeech","category":"resources-evaluation","labels":["self-supervised","robustness-noise"],"institutions":["National Taiwan University","University of Southern California"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-39","pdf":"https://www.isca-archive.org/interspeech_2026/tsai26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tsai26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tsai26_interspeech/markdown.md"},{"id":"tse26_interspeech","title":"AudioNoisePrints: Model-free audio watermarking using spatial correlation in flow matching TTS","authors":["Timothy Tin-Long Tse","Jian Zhu","Aidan Pine","Mengzhe Geng"],"year":2026,"doi":"10.21437/Interspeech.2026-2165","isca_url":"https://www.isca-archive.org/interspeech_2026/tse26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tse26_interspeech.pdf","session":"Audio Watermarking and Source Verification","topics":["audio-deepfake","self-supervised","dataset"],"category":"deepfake-security","labels":["generative-model","robustness-noise"],"institutions":["National Research Council Canada","University of British Columbia"],"code":{"url":"https://github.com/SWivid/F5-TTS","stars":15311,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tse26_interspeech","category":"deepfake-security","labels":["generative-model","robustness-noise"],"institutions":["National Research Council Canada","University of British Columbia"],"code":"https://github.com/SWivid/F5-TTS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2165","pdf":"https://www.isca-archive.org/interspeech_2026/tse26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tse26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tse26_interspeech/markdown.md"},{"id":"tseng26_interspeech","title":"Time-normalized spectrograms reveal segmental differences in English heterographic homophones","authors":["Yu-Hsiang Tseng","Harald Baayen"],"year":2026,"doi":"10.21437/Interspeech.2026-793","isca_url":"https://www.isca-archive.org/interspeech_2026/tseng26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tseng26_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["phonetics","self-supervised","evaluation"],"category":"phonetics-linguistics","institutions":["University of Tubingen"],"funding":["European Research Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tseng26_interspeech","category":"phonetics-linguistics","institutions":["University of Tubingen"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-793","pdf":"https://www.isca-archive.org/interspeech_2026/tseng26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tseng26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tseng26_interspeech/markdown.md"},{"id":"tseng26b_interspeech","title":"TASTE-Streaming: Towards Streamable Text-Aligned Speech Tokenization and Embedding for Spoken Language Modeling","authors":["Liang-Hsuan Tseng","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1686","isca_url":"https://www.isca-archive.org/interspeech_2026/tseng26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tseng26b_interspeech.pdf","session":"Audio Language Models","topics":["speech-llm","asr","self-supervised"],"category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["National Taiwan University"],"code":{"url":"https://andybi7676.github.io/taste_s_demo","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tseng26b_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["National Taiwan University"],"code":"https://andybi7676.github.io/taste_s_demo","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1686","pdf":"https://www.isca-archive.org/interspeech_2026/tseng26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tseng26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tseng26b_interspeech/markdown.md"},{"id":"tseng26c_interspeech","title":"VOSSA: Voiceprint Optimization for Streaming Speech Architectures","authors":["Mu-Ruei Tseng","Waris Quamer","Ghady Nasrallah","Ricardo Gutierrez-Osuna"],"year":2026,"doi":"10.21437/Interspeech.2026-2763","isca_url":"https://www.isca-archive.org/interspeech_2026/tseng26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tseng26c_interspeech.pdf","session":"Streaming Speech Synthesis","topics":["voice-conversion","self-supervised","streaming"],"category":"tts","labels":["efficient-on-device","self-supervised","streaming-real-time","generative-model"],"institutions":["Texas A&M University"],"funding":["Intelligence Advanced Research Projects Activity","Department of Interior","Interior Business Center"],"code":{"url":"https://morris88826.github.io/VOSSA/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tseng26c_interspeech","category":"tts","labels":["efficient-on-device","self-supervised","streaming-real-time","generative-model"],"institutions":["Texas A&M University"],"code":"https://morris88826.github.io/VOSSA/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2763","pdf":"https://www.isca-archive.org/interspeech_2026/tseng26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tseng26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tseng26c_interspeech/markdown.md"},{"id":"tsoi26_interspeech","title":"Next-Turn: Duration-Aware Streaming Endpoint Detection via Time-to-Next-Speech-Onset Prediction","authors":["Tristan Tsoi","Jiajun Deng","Yingke Zhu","Huu Quyen Dang","Tianxiang Cao","Nikita Kuzmin","Tao Zhong","Simon Lui"],"year":2026,"doi":"10.21437/Interspeech.2026-1053","isca_url":"https://www.isca-archive.org/interspeech_2026/tsoi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tsoi26_interspeech.pdf","session":"Audio segmentation","topics":["speech-llm","self-supervised","on-device"],"category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["Huawei","Chinese University of Hong Kong","Nanyang Technological University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tsoi26_interspeech","category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["Huawei","Chinese University of Hong Kong","Nanyang Technological University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1053","pdf":"https://www.isca-archive.org/interspeech_2026/tsoi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tsoi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tsoi26_interspeech/markdown.md"},{"id":"tsukagoshi26_interspeech","title":"Distilling Structured Reasoning into SpeechLLMs for Spoken Language Understanding","authors":["Toshihiro Tsukagoshi","Natsuo Yamashita","Kota Dohi","Hiroaki Kokubo","Masaaki Yamamoto"],"year":2026,"doi":"10.21437/Interspeech.2026-1540","isca_url":"https://www.isca-archive.org/interspeech_2026/tsukagoshi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tsukagoshi26_interspeech.pdf","session":"Spoken Language Understanding","topics":["spoken-language-understanding","speech-llm","self-supervised"],"category":"speech-llm-dialogue","institutions":["Shizuoka University","Hitachi"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tsukagoshi26_interspeech","category":"speech-llm-dialogue","institutions":["Shizuoka University","Hitachi"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1540","pdf":"https://www.isca-archive.org/interspeech_2026/tsukagoshi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tsukagoshi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tsukagoshi26_interspeech/markdown.md"},{"id":"tu26_interspeech","title":"Duration-aware self-attention for speech deepfake detection","authors":["Youzhi Tu","Xin Fang","Liping Chen","Zhen-Hua Ling","Kong Aik Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-2200","isca_url":"https://www.isca-archive.org/interspeech_2026/tu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tu26_interspeech.pdf","session":"Speech Deepfake Detection, Attribution and Characterization","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","institutions":["Hong Kong Polytechnic University","University of Science and Technology of China","iFLYTEK"],"funding":["Innovation and Technology Fund","National Key R&D Program of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tu26_interspeech","category":"deepfake-security","institutions":["Hong Kong Polytechnic University","University of Science and Technology of China","iFLYTEK"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2200","pdf":"https://www.isca-archive.org/interspeech_2026/tu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tu26_interspeech/markdown.md"},{"id":"tu26b_interspeech","title":"VISA: A Visual Information Strengthened Audio-Reasoning System for the Interspeech 2026 ARC Agent Track","authors":["Wenming Tu","Jian Gao","Yanru Huo","Yixuan Wang","Jing Peng","Bohan Li","Ziyang Ma","Tao Liu","Shuai Fan","Kai Yu","Xie Chen","Zilong Zheng"],"year":2026,"doi":"10.21437/Interspeech.2026-2381","isca_url":"https://www.isca-archive.org/interspeech_2026/tu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tu26b_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","evaluation","self-supervised"],"category":"speech-llm-dialogue","institutions":["Shanghai Jiao Tong University","Shanghai Innovation Institute","AISpeech","Beijing Institute of General Artificial Intelligence"],"funding":["Science and Technology Innovation 2030-Major Project","National Natural Science Foundation of China","Shanghai Municipal Science and Technology Major Project","Yangtze River Delta Science and Technology Innovation Community Joint Research Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tu26b_interspeech","category":"speech-llm-dialogue","institutions":["Shanghai Jiao Tong University","Shanghai Innovation Institute","AISpeech","Beijing Institute of General Artificial Intelligence"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2381","pdf":"https://www.isca-archive.org/interspeech_2026/tu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tu26b_interspeech/markdown.md"},{"id":"tu26c_interspeech","title":"AV-SNINet: A multi-channel audio-visual speech-noise interaction network for Target Speaker Extraction with cross-beam attention","authors":["Yanhui Tu","Runxiang Yu","Yi Fang"],"year":2026,"doi":"10.21437/Interspeech.2026-3227","isca_url":"https://www.isca-archive.org/interspeech_2026/tu26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tu26c_interspeech.pdf","session":"Target Speaker Extraction, Speech Separation and Audio Understanding","topics":["speech-enhancement","self-supervised","multilingual"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Anhui University","iFLYTEK","Acousnet Technology Company","Chinese Academy of Sciences"],"funding":["State Key Laboratory of Optoelectronic Information Acquisition and Protection, Anhui University","Anhui University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tu26c_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Anhui University","iFLYTEK","Acousnet Technology Company","Chinese Academy of Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3227","pdf":"https://www.isca-archive.org/interspeech_2026/tu26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tu26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tu26c_interspeech/markdown.md"},{"id":"tuncay26_interspeech","title":"BEST-RQ-2: Contextualize-Then-Predict, a Two-Step Approach for Self-Supervised Audio Representations","authors":["Ludovic Tuncay","Étienne Labbé","Thomas Pellegrini"],"year":2026,"doi":"10.21437/Interspeech.2026-2488","isca_url":"https://www.isca-archive.org/interspeech_2026/tuncay26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tuncay26_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["self-supervised","speech-llm","evaluation"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["IRIT","Universite de Toulouse","CNRS","Toulouse INP"],"funding":["ANR-3IA Artificial and Natural Intelligence Toulouse Institute ANITI"],"code":{"url":"https://github.com/LudovicTuncay/audio-embeddings","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tuncay26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["IRIT","Universite de Toulouse","CNRS","Toulouse INP"],"code":"https://github.com/LudovicTuncay/audio-embeddings","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2488","pdf":"https://www.isca-archive.org/interspeech_2026/tuncay26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tuncay26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tuncay26_interspeech/markdown.md"},{"id":"turavecino26_interspeech","title":"Learnable Classifier-Free Guidance Null Embeddings for Enhanced Controllable Speech Synthesis","authors":["Biel Tura-Vecino","Yoach Lacombe","Julian Weber","Zbigniew Latka","Haitong Zhang","Logan Hart","Eren Golge"],"year":2026,"doi":"10.21437/Interspeech.2026-803","isca_url":"https://www.isca-archive.org/interspeech_2026/turavecino26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/turavecino26_interspeech.pdf","session":"Controllable and Expressive Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["Cantina Labs"],"code":{"url":"https://airtimemedia.github.io/IS2026-LearnableCFG/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"turavecino26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Cantina Labs"],"code":"https://airtimemedia.github.io/IS2026-LearnableCFG/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-803","pdf":"https://www.isca-archive.org/interspeech_2026/turavecino26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/turavecino26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/turavecino26_interspeech/markdown.md"},{"id":"turetzky26_interspeech","title":"Knowing What to Stress: A Discourse-Conditioned Text-to-Speech Benchmark","authors":["Arnon Turetzky","Avihu Dekel","Hagai Aronowitz","Ron Hoory","Yossi Adi"],"year":2026,"doi":"10.21437/Interspeech.2026-2743","isca_url":"https://www.isca-archive.org/interspeech_2026/turetzky26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/turetzky26_interspeech.pdf","session":"Text Processing for Speech Synthesis","topics":["tts","prosody","evaluation"],"category":"tts","labels":["dataset-or-benchmark-release"],"institutions":["Hebrew University of Jerusalem","IBM"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"turetzky26_interspeech","category":"tts","labels":["dataset-or-benchmark-release"],"institutions":["Hebrew University of Jerusalem","IBM"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2743","pdf":"https://www.isca-archive.org/interspeech_2026/turetzky26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/turetzky26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/turetzky26_interspeech/markdown.md"},{"id":"turk26_interspeech","title":"When “yeah” means “not quite”: Multimodal detection of backchannels expressing incomplete understanding","authors":["Olcay Türk","Stefan Lazarov","Yu Wang","Angela Grimminger","Hendrik Buschmeier","Petra Wagner"],"year":2026,"doi":"10.21437/Interspeech.2026-713","isca_url":"https://www.isca-archive.org/interspeech_2026/turk26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/turk26_interspeech.pdf","session":"Entrainment and Dialogue Coordination","topics":["spoken-language-understanding","paralinguistics","multilingual"],"category":"paralinguistics-emotion","institutions":["Bielefeld University","Paderborn University","SFB/Transregio 318 ‘Constructing Explainability’"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"turk26_interspeech","category":"paralinguistics-emotion","institutions":["Bielefeld University","Paderborn University","SFB/Transregio 318 ‘Constructing Explainability’"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-713","pdf":"https://www.isca-archive.org/interspeech_2026/turk26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/turk26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/turk26_interspeech/markdown.md"},{"id":"tushar26_interspeech","title":"Child-Centric Voice Anonymization in Single and Multi-Speaker Speech via Domain-Adapted SSL Models","authors":["Pranav Tushar","Xiaoxiao Miao","Rong Tong"],"year":2026,"doi":"10.21437/Interspeech.2026-2191","isca_url":"https://www.isca-archive.org/interspeech_2026/tushar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/tushar26_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["speech-enhancement","speaker-verification","self-supervised"],"category":"deepfake-security","labels":["self-supervised","generative-model"],"institutions":["Singapore Institute of Technology","Duke Kunshan University"],"funding":["Singapore Ministry of Education Academic Research Fund Tier 1"],"code":{"url":"https://github.com/pranavtushar/SSL-CVA","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"tushar26_interspeech","category":"deepfake-security","labels":["self-supervised","generative-model"],"institutions":["Singapore Institute of Technology","Duke Kunshan University"],"code":"https://github.com/pranavtushar/SSL-CVA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2191","pdf":"https://www.isca-archive.org/interspeech_2026/tushar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/tushar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/tushar26_interspeech/markdown.md"},{"id":"udawatta26_interspeech","title":"Phonetically Grounded Vowel Space Metrics for Evaluating Synthetic Speech During TTS Model Training","authors":["Pasindu Udawatta","Jesin James","Sally Akevai Nicholas","B. T. Balamurali","C. I. Watson"],"year":2026,"doi":"10.21437/Interspeech.2026-1579","isca_url":"https://www.isca-archive.org/interspeech_2026/udawatta26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/udawatta26_interspeech.pdf","session":"Speech Synthesis Evaluation 1","topics":["tts","evaluation","phonetics"],"category":"resources-evaluation","institutions":["University of Auckland"],"funding":["Marsden Fund"],"code":{"url":"https://github.com/pasindu-ud/vowel-space-metrics/tree/interspeech-2026","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"udawatta26_interspeech","category":"resources-evaluation","institutions":["University of Auckland"],"code":"https://github.com/pasindu-ud/vowel-space-metrics/tree/interspeech-2026","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1579","pdf":"https://www.isca-archive.org/interspeech_2026/udawatta26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/udawatta26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/udawatta26_interspeech/markdown.md"},{"id":"udeogu26_interspeech","title":"From Continuous Speech to Subglottal Resonances: Automatic Signal Generation, Estimation, and Tracking Framework","authors":["Chigozie Uzochukwu Udeogu","Carol Espy-Wilson"],"year":2026,"doi":"10.21437/Interspeech.2026-2464","isca_url":"https://www.isca-archive.org/interspeech_2026/udeogu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/udeogu26_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["speech-enhancement","phonetics","self-supervised"],"category":"phonetics-linguistics","institutions":["University of Maryland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"udeogu26_interspeech","category":"phonetics-linguistics","institutions":["University of Maryland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2464","pdf":"https://www.isca-archive.org/interspeech_2026/udeogu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/udeogu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/udeogu26_interspeech/markdown.md"},{"id":"udupa26_interspeech","title":"Endpoint Anticipation for Low-Latency Spoken Dialogue","authors":["Sathvik Udupa","Shinji Watanabe","Petr Schwarz","Honza Černocký"],"year":2026,"doi":"10.21437/Interspeech.2026-2196","isca_url":"https://www.isca-archive.org/interspeech_2026/udupa26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/udupa26_interspeech.pdf","session":"Turn-taking","topics":["speech-llm","spoken-language-understanding","self-supervised"],"category":"speech-llm-dialogue","labels":["efficient-on-device","streaming-real-time"],"institutions":["Brno University of Technology","Carnegie Mellon University"],"funding":["Technology Agency of the Czech Republic","Czech Ministry of Education, Youth and Sports"],"code":{"url":"https://github.com/bloodraven66/EndpointAnticipation","stars":7,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"udupa26_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","streaming-real-time"],"institutions":["Brno University of Technology","Carnegie Mellon University"],"code":"https://github.com/bloodraven66/EndpointAnticipation","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2196","pdf":"https://www.isca-archive.org/interspeech_2026/udupa26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/udupa26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/udupa26_interspeech/markdown.md"},{"id":"ugan26_interspeech","title":"Adding Robust Code-Switching Capabilities to High Performance Multilingual ASR","authors":["Enes Yavuz Ugan","Alexander Waibel"],"year":2026,"doi":"10.21437/Interspeech.2026-1099","isca_url":"https://www.isca-archive.org/interspeech_2026/ugan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ugan26_interspeech.pdf","session":"Code-Switching ASR","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["Karlsruhe Institute of Technology","Carnegie Mellon University"],"funding":["European Union","Federal Ministry of Education and Research","Ministry of Science, Research and the Arts of Baden-Wurttemberg","DFG"],"code":{"url":"https://github.com/enesyugan/robust-code-switching-asr","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ugan26_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["Karlsruhe Institute of Technology","Carnegie Mellon University"],"code":"https://github.com/enesyugan/robust-code-switching-asr","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1099","pdf":"https://www.isca-archive.org/interspeech_2026/ugan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ugan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ugan26_interspeech/markdown.md"},{"id":"ulgen26_interspeech","title":"Rethinking Speaker Embeddings for Speech Generation: Sub-Center Modeling for Capturing Intra-Speaker Diversity","authors":["Ismail Rasim Ulgen","John Hansen","Carlos Busso","Berrak Sisman"],"year":2026,"doi":"10.21437/Interspeech.2026-942","isca_url":"https://www.isca-archive.org/interspeech_2026/ulgen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ulgen26_interspeech.pdf","session":"Speaker Verification: Advances in Speaker Embeddings","topics":["voice-conversion","speaker-verification","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["Johns Hopkins University","University of Texas at Dallas","Carnegie Mellon University"],"funding":["National Science Foundation"],"code":{"url":"https://choughtotem.github.io/subcentervc_demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ulgen26_interspeech","category":"tts","labels":["generative-model"],"institutions":["Johns Hopkins University","University of Texas at Dallas","Carnegie Mellon University"],"code":"https://choughtotem.github.io/subcentervc_demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-942","pdf":"https://www.isca-archive.org/interspeech_2026/ulgen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ulgen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ulgen26_interspeech/markdown.md"},{"id":"ulgen26b_interspeech","title":"DiffAnon: Diffusion-based Prosody Control for Voice Anonymization","authors":["Ismail Rasim Ulgen","Zexin Cai","Nicholas Andrews","Philipp Koehn","Berrak Sisman"],"year":2026,"doi":"10.21437/Interspeech.2026-1331","isca_url":"https://www.isca-archive.org/interspeech_2026/ulgen26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ulgen26b_interspeech.pdf","session":"Speaker Privacy and Anonymization","topics":["voice-conversion","self-supervised","evaluation"],"category":"deepfake-security","labels":["generative-model"],"institutions":["Johns Hopkins University"],"funding":["National Science Foundation","Office of the Director of National Intelligence"],"code":{"url":"https://github.com/rsmlgen/diffanon","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ulgen26b_interspeech","category":"deepfake-security","labels":["generative-model"],"institutions":["Johns Hopkins University"],"code":"https://github.com/rsmlgen/diffanon","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1331","pdf":"https://www.isca-archive.org/interspeech_2026/ulgen26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ulgen26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ulgen26b_interspeech/markdown.md"},{"id":"ullah26_interspeech","title":"Music Artistic Captioning: Towards Translating Music into Expressive Language","authors":["Ubaid Ullah","Hyun-Chul Choi","Zied Bouraoui"],"year":2026,"doi":"10.21437/Interspeech.2026-2618","isca_url":"https://www.isca-archive.org/interspeech_2026/ullah26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ullah26_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["speech-llm","audio-captioning","multilingual"],"category":"audio-understanding","labels":["generative-model"],"institutions":["Yeungnam University","Univ. Artois","CRIL CNRS"],"funding":["National Research Foundation of Korea","Ministry of Science, ICT and Future Planning","Yeungnam University","ANR"],"code":{"url":"https://github.com/uu95/MAC","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ullah26_interspeech","category":"audio-understanding","labels":["generative-model"],"institutions":["Yeungnam University","Univ. Artois","CRIL CNRS"],"code":"https://github.com/uu95/MAC","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2618","pdf":"https://www.isca-archive.org/interspeech_2026/ullah26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ullah26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ullah26_interspeech/markdown.md"},{"id":"vaaras26_interspeech","title":"TSExplorer: An interactive data annotation and exploration tool for time-series data","authors":["Einari Vaaras","Manu Airaksinen","Okko Räsänen"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/vaaras26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/vaaras26_interspeech.pdf","session":"Speech Analysis, Data Resources and Research Tools","topics":["evaluation","dataset","self-supervised"],"category":"resources-evaluation","institutions":["Tampere University","University of Helsinki"],"funding":["Research Council of Finland","Sigrid Juselius Foundation"],"code":{"url":"https://github.com/SPEECHCOG/TSExplorer","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"vaaras26_interspeech","category":"resources-evaluation","institutions":["Tampere University","University of Helsinki"],"code":"https://github.com/SPEECHCOG/TSExplorer","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://www.isca-archive.org/interspeech_2026/vaaras26_interspeech.html","pdf":"https://www.isca-archive.org/interspeech_2026/vaaras26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/vaaras26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/vaaras26_interspeech/markdown.md"},{"id":"valentinibotinhao26_interspeech","title":"Exploring Active Sampling Strategies for Pairwise Comparisons in Speech Synthesis Evaluation","authors":["Cassia Valentini-Botinhao","Andrea Lorena Aldana Blanco","Dan Wells","Aidan Pine","Korin Richmond"],"year":2026,"doi":"10.21437/Interspeech.2026-446","isca_url":"https://www.isca-archive.org/interspeech_2026/valentinibotinhao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/valentinibotinhao26_interspeech.pdf","session":"Speech Synthesis Evaluation 2","topics":["tts","evaluation","low-resource"],"category":"resources-evaluation","institutions":["University of Edinburgh","National Research Council Canada"],"funding":["UK Research and Innovation","National Research Council Canada"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"valentinibotinhao26_interspeech","category":"resources-evaluation","institutions":["University of Edinburgh","National Research Council Canada"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-446","pdf":"https://www.isca-archive.org/interspeech_2026/valentinibotinhao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/valentinibotinhao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/valentinibotinhao26_interspeech/markdown.md"},{"id":"varadhan26_interspeech","title":"IN-F5: Adapting an English TTS Foundation Model for Multilingual and Zero-Resource Indian Speech Synthesis","authors":["Praveen Srinivasa Varadhan","Srija Anand","Siddhartha Soma","Mitesh M Khapra"],"year":2026,"doi":"10.21437/Interspeech.2026-3366","isca_url":"https://www.isca-archive.org/interspeech_2026/varadhan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/varadhan26_interspeech.pdf","session":"Low-Resource Speech Synthesis","topics":["tts","multilingual","low-resource"],"category":"tts","labels":["low-resource","multilingual","self-supervised","generative-model"],"institutions":["Indian Institute of Technology Madras","Saryps Labs"],"funding":["Digital India Bhashini","MeitY, Government of India","EkStep Foundation","Nilekani Philanthropies"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"varadhan26_interspeech","category":"tts","labels":["low-resource","multilingual","self-supervised","generative-model"],"institutions":["Indian Institute of Technology Madras","Saryps Labs"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3366","pdf":"https://www.isca-archive.org/interspeech_2026/varadhan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/varadhan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/varadhan26_interspeech/markdown.md"},{"id":"variani26_interspeech","title":"Representational Instability in Decoupled Audio Encoders","authors":["Ehsan Variani","Tom Bagby","Georg Heigold","Ke Wu","Cyril Allauzen"],"year":2026,"doi":"10.21437/Interspeech.2026-2487","isca_url":"https://www.isca-archive.org/interspeech_2026/variani26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/variani26_interspeech.pdf","session":"Audio Coding and Signal Analysis","topics":["self-supervised","multilingual","speech-llm"],"category":"speech-coding","labels":["multilingual","self-supervised"],"institutions":["Google"],"code":{"url":"https://github.com/google-research/mseb/tree/main/mseb","stars":67,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"variani26_interspeech","category":"speech-coding","labels":["multilingual","self-supervised"],"institutions":["Google"],"code":"https://github.com/google-research/mseb/tree/main/mseb","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2487","pdf":"https://www.isca-archive.org/interspeech_2026/variani26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/variani26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/variani26_interspeech/markdown.md"},{"id":"vendrame26_interspeech","title":"Joint Speech And Text Training For LLM-based End-To-End Spoken Dialogue State Tracking","authors":["Katia Vendrame","Bolaji Yusuf","Santosh Kesiraju","Šimon Sedláček","Oldřich Plchot","Honza Černocký"],"year":2026,"doi":"10.21437/Interspeech.2026-2769","isca_url":"https://www.isca-archive.org/interspeech_2026/vendrame26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/vendrame26_interspeech.pdf","session":"Spoken Language Understanding","topics":["spoken-language-understanding","speech-llm","multilingual"],"category":"speech-llm-dialogue","labels":["low-resource"],"institutions":["Brno University of Technology"],"funding":["PRINS","European Union","Horizon Europe","MoE"],"code":{"url":"https://github.com/kackav/dialogue_state_tracking","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"vendrame26_interspeech","category":"speech-llm-dialogue","labels":["low-resource"],"institutions":["Brno University of Technology"],"code":"https://github.com/kackav/dialogue_state_tracking","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2769","pdf":"https://www.isca-archive.org/interspeech_2026/vendrame26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/vendrame26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/vendrame26_interspeech/markdown.md"},{"id":"viakhirev26_interspeech","title":"From Dispersion to Attraction: Spectral Dynamics of Hallucination Across Whisper Model Scales","authors":["Ivan Viakhirev","Kirill Borodin","Grach Mkrtchian"],"year":2026,"doi":"10.21437/Interspeech.2026-1420","isca_url":"https://www.isca-archive.org/interspeech_2026/viakhirev26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/viakhirev26_interspeech.pdf","session":"Robust ASR: Hallucinations and Biases","topics":["asr","self-supervised","evaluation"],"category":"asr","labels":["self-supervised"],"institutions":["Information Technologies, Mechanics and Optics University","Moscow Technical University of Communications and Informatics","BitmanagerAI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"viakhirev26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Information Technologies, Mechanics and Optics University","Moscow Technical University of Communications and Informatics","BitmanagerAI"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1420","pdf":"https://www.isca-archive.org/interspeech_2026/viakhirev26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/viakhirev26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/viakhirev26_interspeech/markdown.md"},{"id":"villatorotello26_interspeech","title":"Context Projector: Complementary Keyword and Dialogue Context Embeddings for LLM-based ASR","authors":["Esaú Villatoro-Tello","Sergio Burdisso","Shashi Kumar","Hasindri Watawana","Srikanth Madikeri","Manjunath K E","Jeena Prakash","Thibault Bañeras-Roux","Kadri Hacioglu","Petr Motlicek","Andreas Stolcke"],"year":2026,"doi":"10.21437/Interspeech.2026-3326","isca_url":"https://www.isca-archive.org/interspeech_2026/villatorotello26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/villatorotello26_interspeech.pdf","session":"Spoken Dialogue Systems","topics":["speech-llm","asr","spoken-language-understanding"],"category":"asr","institutions":["Idiap Research Institute","EPFL","University of Zurich","Uniphore","Brno University of Technology"],"funding":["Idiap Research Institute","Uniphore","EU Horizon 2020"],"code":{"url":"https://github.com/idiap/llm-asr-context-projector","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"villatorotello26_interspeech","category":"asr","institutions":["Idiap Research Institute","EPFL","University of Zurich","Uniphore","Brno University of Technology"],"code":"https://github.com/idiap/llm-asr-context-projector","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3326","pdf":"https://www.isca-archive.org/interspeech_2026/villatorotello26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/villatorotello26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/villatorotello26_interspeech/markdown.md"},{"id":"visser26_interspeech","title":"ZeroSyl: Simple Zero-Resource Syllable Tokenization for Spoken Language Modeling","authors":["Nicol Visser","Simon Malan","Danel Slabbert","Herman Kamper"],"year":2026,"doi":"10.21437/Interspeech.2026-315","isca_url":"https://www.isca-archive.org/interspeech_2026/visser26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/visser26_interspeech.pdf","session":"Efficient Inference for ASR and Speech LMs","topics":["spoken-language-understanding","self-supervised","evaluation"],"category":"asr","labels":["self-supervised"],"institutions":["Stellenbosch University"],"funding":["Google PhD Fellowship Program"],"code":{"url":"https://github.com/nicolvisser/ZeroSyl","stars":9,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"visser26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Stellenbosch University"],"code":"https://github.com/nicolvisser/ZeroSyl","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-315","pdf":"https://www.isca-archive.org/interspeech_2026/visser26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/visser26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/visser26_interspeech/markdown.md"},{"id":"vo26_interspeech","title":"SETEAB: Multiscale approach with Squeeze-and-Excitation Temporal Enhanced Aware Block for Speech Emotion Recognition","authors":["Duy Vo","Kiet Anh Hoang","Hao Do"],"year":2026,"doi":"10.21437/Interspeech.2026-1208","isca_url":"https://www.isca-archive.org/interspeech_2026/vo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/vo26_interspeech.pdf","session":"Speech Emotion Recognition and Representation 3","topics":["speech-emotion-recognition","self-supervised","on-device"],"category":"paralinguistics-emotion","labels":["efficient-on-device"],"institutions":["University of Science, Ho Chi Minh City","Vietnam National University, Ho Chi Minh City","UNEY"],"funding":["University of Science, VNU-HCM"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"vo26_interspeech","category":"paralinguistics-emotion","labels":["efficient-on-device"],"institutions":["University of Science, Ho Chi Minh City","Vietnam National University, Ho Chi Minh City","UNEY"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1208","pdf":"https://www.isca-archive.org/interspeech_2026/vo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/vo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/vo26_interspeech/markdown.md"},{"id":"vonaspern26_interspeech","title":"Leveraging Mutual Intra-Modal Similarity Supervision for Text and Audio","authors":["Julian Miguel von Aspern","Bruno Defraene","Anastasios Vafeiadis","Oleksiy Kutscher","Ernst Seidel","Tim Fingscheidt"],"year":2026,"doi":"10.21437/Interspeech.2026-3300","isca_url":"https://www.isca-archive.org/interspeech_2026/vonaspern26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/vonaspern26_interspeech.pdf","session":"Audio segmentation","topics":["speech-llm","self-supervised","evaluation"],"category":"audio-understanding","labels":["self-supervised"],"institutions":["Technische Universitat Braunschweig","NXP Semiconductors"],"code":{"url":"https://github.com/OptimusPrimus/salsa","stars":8,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"vonaspern26_interspeech","category":"audio-understanding","labels":["self-supervised"],"institutions":["Technische Universitat Braunschweig","NXP Semiconductors"],"code":"https://github.com/OptimusPrimus/salsa","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3300","pdf":"https://www.isca-archive.org/interspeech_2026/vonaspern26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/vonaspern26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/vonaspern26_interspeech/markdown.md"},{"id":"vu26_interspeech","title":"Adaptive Multimodal Expert Specialization by Meta-Learning for Spoken English Assessment","authors":["Cong-Thanh Vu","Candy Olivia Mawalim","Hung Le","Chee Wee Leong","Guy Sivan","Shogo Okada"],"year":2026,"doi":"10.21437/Interspeech.2026-1650","isca_url":"https://www.isca-archive.org/interspeech_2026/vu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/vu26_interspeech.pdf","session":"Speech Technologies for Language Learning & Assessment","topics":["paralinguistics","self-supervised","low-resource"],"category":"applications-other","labels":["low-resource"],"institutions":["Japan Advanced Institute of Science and Technology","Educational Testing Service","Vericant"],"funding":["JSPS KAKENHI","JST CREST","JST CRONOS","AMED"],"code":{"url":"https://github.com/cngthnh/meta_see","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"vu26_interspeech","category":"applications-other","labels":["low-resource"],"institutions":["Japan Advanced Institute of Science and Technology","Educational Testing Service","Vericant"],"code":"https://github.com/cngthnh/meta_see","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1650","pdf":"https://www.isca-archive.org/interspeech_2026/vu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/vu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/vu26_interspeech/markdown.md"},{"id":"wagner26_interspeech","title":"Transcription Policy as a Latent Variable: Activating Controllable Verbatim ASR with Word-Level Timing","authors":["Laurin Wagner"],"year":2026,"doi":"10.21437/Interspeech.2026-2792","isca_url":"https://www.isca-archive.org/interspeech_2026/wagner26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wagner26_interspeech.pdf","session":"Robust and Efficient ASR","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["self-supervised"],"institutions":["nyra health"],"code":{"url":"https://github.com/nyrahealth/CrisperWhisper","stars":1431,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wagner26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["nyra health"],"code":"https://github.com/nyrahealth/CrisperWhisper","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2792","pdf":"https://www.isca-archive.org/interspeech_2026/wagner26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wagner26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wagner26_interspeech/markdown.md"},{"id":"wagner26b_interspeech","title":"Content is What Remains: Invariant Speech Tokenization from Parallel Utterances","authors":["Laurin Wagner"],"year":2026,"doi":"10.21437/Interspeech.2026-2817","isca_url":"https://www.isca-archive.org/interspeech_2026/wagner26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wagner26b_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["self-supervised","speech-coding","speech-llm"],"category":"speech-coding","labels":["self-supervised"],"institutions":["nyra health"],"code":{"url":"https://github.com/nyrahealth/PINT","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wagner26b_interspeech","category":"speech-coding","labels":["self-supervised"],"institutions":["nyra health"],"code":"https://github.com/nyrahealth/PINT","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2817","pdf":"https://www.isca-archive.org/interspeech_2026/wagner26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wagner26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wagner26b_interspeech/markdown.md"},{"id":"walczak26_interspeech","title":"From onset to coda: spectral variation in normative Polish /s/ produced by children","authors":["Joanna Walczak","Oliwia Skórzewska","Zuzanna Miodońska"],"year":2026,"doi":"10.21437/Interspeech.2026-1936","isca_url":"https://www.isca-archive.org/interspeech_2026/walczak26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/walczak26_interspeech.pdf","session":"Speaker-Specific, Forensic and Segmental Characteristics of Typical and Atypical Speech","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["Silesian University of Technology"],"funding":["National Science Centre, Poland","European Funds for Silesia","Just Transition Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"walczak26_interspeech","category":"phonetics-linguistics","institutions":["Silesian University of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1936","pdf":"https://www.isca-archive.org/interspeech_2026/walczak26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/walczak26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/walczak26_interspeech/markdown.md"},{"id":"wan26_interspeech","title":"Enhancing Visual Paralinguistics: Motion-Guided Spatial Denoising for Non-Verbal Interaction Analysis","authors":["Junjie Wan"],"year":2026,"doi":"10.21437/Interspeech.2026-868","isca_url":"https://www.isca-archive.org/interspeech_2026/wan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wan26_interspeech.pdf","session":"Multimodal Emotion Recognition","topics":["paralinguistics","self-supervised"],"category":"paralinguistics-emotion","institutions":["Harbin Institute of Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wan26_interspeech","category":"paralinguistics-emotion","institutions":["Harbin Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-868","pdf":"https://www.isca-archive.org/interspeech_2026/wan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wan26_interspeech/markdown.md"},{"id":"wan26b_interspeech","title":"Continuous Time-Varying Emotion Control Zero-Shot Text-To-Speech With Emotion Orthogonal LoRA","authors":["Chenchen Wan","Minchuan Chen","Yan Shi","Peng Qi","Shaojun Wang","Jing Xiao"],"year":2026,"doi":"10.21437/Interspeech.2026-1798","isca_url":"https://www.isca-archive.org/interspeech_2026/wan26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wan26b_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["tts","self-supervised","prosody"],"category":"tts","labels":["generative-model"],"institutions":["Shanghai Jiao Tong University","Ping An Technology","Shanghai Jiao Tong University Chongqing Artificial Intelligence Research Institute"],"code":{"url":"https://wancc-p.github.io/EoLoRA/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wan26b_interspeech","category":"tts","labels":["generative-model"],"institutions":["Shanghai Jiao Tong University","Ping An Technology","Shanghai Jiao Tong University Chongqing Artificial Intelligence Research Institute"],"code":"https://wancc-p.github.io/EoLoRA/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1798","pdf":"https://www.isca-archive.org/interspeech_2026/wan26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wan26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wan26b_interspeech/markdown.md"},{"id":"wang26_interspeech","title":"WildElder: A Chinese Elderly Speech Dataset from the Wild with Fine-Grained Manual Annotations","authors":["Hui Wang","Jiaming Zhou","Jiabei He","Haoqin Sun","Yong Qin"],"year":2026,"doi":"10.21437/Interspeech.2026-102","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26_interspeech.pdf","session":"Robust ASR: Hallucinations and Biases","topics":["asr","dataset","multilingual"],"category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Nankai University"],"funding":["NSF China"],"code":{"url":"https://github.com/NKU-HLT/WildElder","stars":9,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26_interspeech","category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Nankai University"],"code":"https://github.com/NKU-HLT/WildElder","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-102","pdf":"https://www.isca-archive.org/interspeech_2026/wang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26_interspeech/markdown.md"},{"id":"wang26aa_interspeech","title":"URGENT-MOS: Unified Multi-Metric and Preference Learning for Robust Speech Quality Assessment","authors":["Wei Wang","Wangyou Zhang","Chenda Li","Jiahe Wang","Samuele Cornell","Marvin Sach","Kohei Saijo","Yihui Fu","Zhaoheng Ni","Mengxiao Bi","Tim Fingscheidt","Shinji Watanabe","Yanmin Qian"],"year":2026,"doi":"10.21437/Interspeech.2026-1671","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26aa_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26aa_interspeech.pdf","session":"Speech and Audio Quality Assessment","topics":["speech-enhancement","tts","evaluation"],"category":"resources-evaluation","labels":["robustness-noise"],"institutions":["Shanghai Jiao Tong University","Carnegie Mellon University","Technische Universitat Braunschweig","Meta","Waseda University","VUI Labs"],"funding":["National Key Research and Development Program of China","China NSFC Project","SJTU Med-X Translational Research Grant"],"code":{"url":"https://github.com/vvwangvv/URGENT-MOS","stars":9,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26aa_interspeech","category":"resources-evaluation","labels":["robustness-noise"],"institutions":["Shanghai Jiao Tong University","Carnegie Mellon University","Technische Universitat Braunschweig","Meta","Waseda University","VUI Labs"],"code":"https://github.com/vvwangvv/URGENT-MOS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1671","pdf":"https://www.isca-archive.org/interspeech_2026/wang26aa_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26aa_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26aa_interspeech/markdown.md"},{"id":"wang26b_interspeech","title":"FoleyGenEx: Unified Video-to-Audio Generation with Multi-Modal Control, Temporal Alignment, and Semantic Precision","authors":["Shiyao Wang","Xijuan Zeng","Hui Wang","Shiwan Zhao","Feng Deng","Chen Zhang","Yong Qin"],"year":2026,"doi":"10.21437/Interspeech.2026-112","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26b_interspeech.pdf","session":"Generative Audio and Music","topics":["speech-enhancement","multilingual","self-supervised"],"category":"audio-understanding","labels":["generative-model"],"institutions":["Nankai University","Kuaishou Technology"],"code":{"url":"https://foleygenex.github.io/FoleyGenEx","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26b_interspeech","category":"audio-understanding","labels":["generative-model"],"institutions":["Nankai University","Kuaishou Technology"],"code":"https://foleygenex.github.io/FoleyGenEx","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-112","pdf":"https://www.isca-archive.org/interspeech_2026/wang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26b_interspeech/markdown.md"},{"id":"wang26ba_interspeech","title":"Optimal Source Placement for TDoA-based Geometry Calibration of Distributed Microphone Arrays","authors":["Xu Wang","Qintuya Si","Qingying Zhao","De Hu"],"year":2026,"doi":"10.21437/Interspeech.2026-1691","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26ba_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26ba_interspeech.pdf","session":"Spatial Audio 4","topics":["source-separation","evaluation","speech-enhancement"],"category":"enhancement-separation","institutions":["Inner Mongolia University"],"funding":["National Natural Science Foundation of China","Natural Science Foundation of Inner Mongolia Autonomous Region"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26ba_interspeech","category":"enhancement-separation","institutions":["Inner Mongolia University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1691","pdf":"https://www.isca-archive.org/interspeech_2026/wang26ba_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26ba_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26ba_interspeech/markdown.md"},{"id":"wang26c_interspeech","title":"Blind Room Impulse Response Identification via Reverberant Speech Spectrum Reconstruction","authors":["Pengyu Wang","Xiaofei Li"],"year":2026,"doi":"10.21437/Interspeech.2026-217","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26c_interspeech.pdf","session":"Spatial Audio 4","topics":["speech-enhancement","self-supervised","dataset"],"category":"enhancement-separation","institutions":["Zhejiang University","Westlake University","Westlake Institute for Advanced Study"],"code":{"url":"https://github.com/Audio-WestlakeU/Rec-RIR","stars":39,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26c_interspeech","category":"enhancement-separation","institutions":["Zhejiang University","Westlake University","Westlake Institute for Advanced Study"],"code":"https://github.com/Audio-WestlakeU/Rec-RIR","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-217","pdf":"https://www.isca-archive.org/interspeech_2026/wang26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26c_interspeech/markdown.md"},{"id":"wang26ca_interspeech","title":"FreqGuard: Leveraging Frequency-Domain Feature Priors for Universal Proactive Voice Defense","authors":["Yankai Wang","Zhipeng Chen","Yuxuan Du","Rong Zheng","Jing Deng"],"year":2026,"doi":"10.21437/Interspeech.2026-2069","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26ca_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26ca_interspeech.pdf","session":"Spoofing and Deepfake Detection 1","topics":["audio-deepfake","speaker-verification","speech-enhancement"],"category":"deepfake-security","institutions":["Beijing Fosafer Information Technology"],"funding":["National Key Research and Development Program of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26ca_interspeech","category":"deepfake-security","institutions":["Beijing Fosafer Information Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2069","pdf":"https://www.isca-archive.org/interspeech_2026/wang26ca_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26ca_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26ca_interspeech/markdown.md"},{"id":"wang26d_interspeech","title":"Back to Ear: Perceptually Driven High Fidelity Music Reconstruction","authors":["Kangdi Wang","Zhiyue Wu","Rui Lin","Junyu Dai","Tao Jiang"],"year":2026,"doi":"10.21437/Interspeech.2026-219","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26d_interspeech.pdf","session":"Singing Voice and Music Generation","topics":["self-supervised","speech-enhancement","evaluation"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Chinese Academy of Sciences","Ear-LAB","Initi-AI Ltd"],"code":{"url":"https://github.com/Eps-Acoustic-Revolution-Lab/EAR_VAE","stars":98,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26d_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Chinese Academy of Sciences","Ear-LAB","Initi-AI Ltd"],"code":"https://github.com/Eps-Acoustic-Revolution-Lab/EAR_VAE","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-219","pdf":"https://www.isca-archive.org/interspeech_2026/wang26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26d_interspeech/markdown.md"},{"id":"wang26da_interspeech","title":"GACA-DiT: Diffusion-based Dance-to-Music Generation with Genre-Adaptive Rhythm and Context-Aware Alignment","authors":["Jinting Wang","Yan Rong","Chenxing Li","Li Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-2348","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26da_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26da_interspeech.pdf","session":"Audio Foundation Models and Generation","topics":["tts","self-supervised","evaluation"],"category":"audio-understanding","labels":["generative-model"],"institutions":["Hong Kong University of Science and Technology","Tencent"],"funding":["National Natural Science Foundation of China","Guangdong Basic and Applied Basic Research Foundation","Tencent AI Lab Rhino-Bird Program"],"code":{"url":"https://anonymous.4open.science/w/GACA-DiT/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26da_interspeech","category":"audio-understanding","labels":["generative-model"],"institutions":["Hong Kong University of Science and Technology","Tencent"],"code":"https://anonymous.4open.science/w/GACA-DiT/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2348","pdf":"https://www.isca-archive.org/interspeech_2026/wang26da_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26da_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26da_interspeech/markdown.md"},{"id":"wang26e_interspeech","title":"Predictive Directional Selective Fixed-Filter Active Noise Control for Moving Sources via a Convolutional Recurrent Neural Network","authors":["Boxiang Wang","Zhengding Luo","Dongyuan Shi","Junwei Ji","Xiruo Su","Woon-Seng Gan"],"year":2026,"doi":"10.21437/Interspeech.2026-271","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26e_interspeech.pdf","session":"Active Noise and Echo Control, Sound Zones and Packet-Loss Concealment","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Nanyang Technological University","Northwestern Polytechnical University"],"funding":["Ministry of Education, Singapore"],"code":{"url":"https://github.com/Wang-Boxiang/PD-SFANC","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26e_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Nanyang Technological University","Northwestern Polytechnical University"],"code":"https://github.com/Wang-Boxiang/PD-SFANC","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-271","pdf":"https://www.isca-archive.org/interspeech_2026/wang26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26e_interspeech/markdown.md"},{"id":"wang26ea_interspeech","title":"Not Flat, But Dissociated: Prosodic and Segmental Divergence in Neural TTS","authors":["Rong Wang","Kun Sun","Harald Baayen"],"year":2026,"doi":"10.21437/Interspeech.2026-2730","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26ea_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26ea_interspeech.pdf","session":"Long-Form Speech Synthesis","topics":["tts","evaluation","phonetics"],"category":"tts","institutions":["University of Tuebingen","Tongji University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26ea_interspeech","category":"tts","institutions":["University of Tuebingen","Tongji University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2730","pdf":"https://www.isca-archive.org/interspeech_2026/wang26ea_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26ea_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26ea_interspeech/markdown.md"},{"id":"wang26f_interspeech","title":"Dual-Space Constrained Face-Based Zero-Shot Text-to-Speech Synthesis","authors":["Jianrong Wang","Shengjie Zhou","Ju Zhang","Dengcheng Hu","Qi Li"],"year":2026,"doi":"10.21437/Interspeech.2026-409","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26f_interspeech.pdf","session":"Scaling and Zero-Shot Speech Synthesis","topics":["tts","self-supervised","multilingual"],"category":"tts","labels":["generative-model"],"institutions":["Tianjin University","Tianjin University of Technology"],"funding":["Key R&D Program of the Nanning Science Research and Technology Development Plan","Tianjin Science and Technology Program, Special Project for High-Quality Development of Manufacturing Industry"],"code":{"url":"https://zsj23.github.io/dsc-tts","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26f_interspeech","category":"tts","labels":["generative-model"],"institutions":["Tianjin University","Tianjin University of Technology"],"code":"https://zsj23.github.io/dsc-tts","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-409","pdf":"https://www.isca-archive.org/interspeech_2026/wang26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26f_interspeech/markdown.md"},{"id":"wang26fa_interspeech","title":"StanceBench: A Benchmark for Audio LLM-Based Interpersonal Stance Evaluation from Speech","authors":["Yuzhe Wang","Thomas Thebaud","Jennifer Hu","Jesús Villalba-Lopez","Venkatesh Ravichandran","Georgi Tinchev","Najim Dehak","Laureano Moro-Velázquez"],"year":2026,"doi":"10.21437/Interspeech.2026-2938","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26fa_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26fa_interspeech.pdf","session":"Benchmarking Foundation Models","topics":["speech-llm","paralinguistics","evaluation"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Johns Hopkins University","Amazon"],"funding":["Amazon"],"code":{"url":"https://github.com/YuzheWangjhu/StanceBench","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26fa_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Johns Hopkins University","Amazon"],"code":"https://github.com/YuzheWangjhu/StanceBench","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2938","pdf":"https://www.isca-archive.org/interspeech_2026/wang26fa_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26fa_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26fa_interspeech/markdown.md"},{"id":"wang26g_interspeech","title":"MS-GNN: Multi-Scale Graph Neural Network for Detecting Local Audio-Visual Forgery Traces","authors":["Jianrong Wang","Hengyang Guo","Jie Liu","Ju Zhang","Qi Li","Ying Guo","Jing Zhao"],"year":2026,"doi":"10.21437/Interspeech.2026-416","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26g_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","institutions":["Tianjin University","Tianjin Renai College","Tianjin Electronic Information College","Tianjin University of Technology","Tianjin Beiyang Rongke Intelligent Technology Co., Ltd"],"funding":["Key R&D Program of the Nanning Science Research and Technology Development Plan","Tianjin Science and Technology Program, Special Project for High-Quality Development of Manufacturing Industry"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26g_interspeech","category":"deepfake-security","institutions":["Tianjin University","Tianjin Renai College","Tianjin Electronic Information College","Tianjin University of Technology","Tianjin Beiyang Rongke Intelligent Technology Co., Ltd"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-416","pdf":"https://www.isca-archive.org/interspeech_2026/wang26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26g_interspeech/markdown.md"},{"id":"wang26ga_interspeech","title":"Clinically-Supervised Hierarchical LoRA-MoE: A Parameter-Efficient Framework for Severity-Aware Dysarthric Speech Assessment","authors":["Jiaqi Wang","Guanghua Xu","Hongjie Zhou","Shuwen Bai","Sicong Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-3043","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26ga_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26ga_interspeech.pdf","session":"Pathological Speech Assessment 3","topics":["paralinguistics","self-supervised","low-resource"],"category":"health-clinical","labels":["efficient-on-device","self-supervised"],"institutions":["Xi'an Jiaotong University"],"funding":["Scientific and Technological Innovation 2030 Major Project","Key Research and Development Program of Shaanxi Province"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26ga_interspeech","category":"health-clinical","labels":["efficient-on-device","self-supervised"],"institutions":["Xi'an Jiaotong University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3043","pdf":"https://www.isca-archive.org/interspeech_2026/wang26ga_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26ga_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26ga_interspeech/markdown.md"},{"id":"wang26h_interspeech","title":"GMOD: Voice-Face Association Learning via Graph Mining and Orthogonal Disentanglement","authors":["Jianrong Wang","Kaibin Bi","Jinghui Li","Ju Zhang","Qi Li","Ying Guo","Jing Zhao"],"year":2026,"doi":"10.21437/Interspeech.2026-432","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26h_interspeech.pdf","session":"Speech and Language Representation","topics":["self-supervised","speaker-verification","multimodal"],"category":"speaker","institutions":["Tianjin University","Tianjin Renai College","Tianjin University of Technology","Tianjin Beiyang Rongke Intelligent Technology Co., Ltd"],"funding":["Key R&D Program of the Nanning Science Research and Technology Development Plan","Tianjin Science and Technology Program, Special Project for High-Quality Development of Manufacturing Industry"],"code":{"url":"https://github.com/BKB00001/GMOD","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26h_interspeech","category":"speaker","institutions":["Tianjin University","Tianjin Renai College","Tianjin University of Technology","Tianjin Beiyang Rongke Intelligent Technology Co., Ltd"],"code":"https://github.com/BKB00001/GMOD","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-432","pdf":"https://www.isca-archive.org/interspeech_2026/wang26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26h_interspeech/markdown.md"},{"id":"wang26i_interspeech","title":"ES-3DF: Editable Speech-Driven 3D Face Reconstruction via Geometry Texture Disentanglement","authors":["Jianrong Wang","Kaibin Bi","Jinghui Li","Ju Zhang","Qi Li","Ying Guo","Jing Zhao"],"year":2026,"doi":"10.21437/Interspeech.2026-433","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26i_interspeech.pdf","session":"Articulatory, EMG, and Visual Speech Generation","topics":["speech-llm","self-supervised","evaluation"],"category":"applications-other","labels":["generative-model"],"institutions":["Tianjin University","Tianjin Renai College","Tianjin University of Technology","Tianjin Beiyang Rongke Intelligent Technology Co., Ltd"],"funding":["Key R&D Program of the Nanning Science Research and Technology Development Plan","Tianjin Science and Technology Program, Special Project for High-Quality Development of Manufacturing Industry"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26i_interspeech","category":"applications-other","labels":["generative-model"],"institutions":["Tianjin University","Tianjin Renai College","Tianjin University of Technology","Tianjin Beiyang Rongke Intelligent Technology Co., Ltd"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-433","pdf":"https://www.isca-archive.org/interspeech_2026/wang26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26i_interspeech/markdown.md"},{"id":"wang26j_interspeech","title":"Layer-wise Multi-factor Adaptive Disentanglement for Cross-corpus Speech Depression Detection","authors":["Minggang Wang","Shohei Kato","Wen Gu","Fenghui Ren","Jun Yan"],"year":2026,"doi":"10.21437/Interspeech.2026-465","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26j_interspeech.pdf","session":"Pathological Speech Assessment 3","topics":["paralinguistics","emotion-recognition","self-supervised"],"category":"health-clinical","labels":["robustness-noise"],"institutions":["Nagoya Institute of Technology","University of Wollongong"],"funding":["Ministry of Education, Culture, Sports, Science and Technology-Japan","National Institute of Information and Communications Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26j_interspeech","category":"health-clinical","labels":["robustness-noise"],"institutions":["Nagoya Institute of Technology","University of Wollongong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-465","pdf":"https://www.isca-archive.org/interspeech_2026/wang26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26j_interspeech/markdown.md"},{"id":"wang26k_interspeech","title":"Does Fine-tuning by Reinforcement Learning Improve Generalization in Binary Speech Deepfake Detection?","authors":["Xin Wang","Wanying Ge","Junichi Yamagishi"],"year":2026,"doi":"10.21437/Interspeech.2026-589","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26k_interspeech.pdf","session":"Speech Deepfake Detection, Attribution and Characterization","topics":["speech-anti-spoofing","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised","robustness-noise"],"institutions":["National Institute of Informatics"],"funding":["Japan Science and Technology Agency","New Energy and Industrial Technology Development Organization"],"code":{"url":"https://github.com/nii-yamagishilab/AntiDeepfake","stars":56,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26k_interspeech","category":"deepfake-security","labels":["self-supervised","robustness-noise"],"institutions":["National Institute of Informatics"],"code":"https://github.com/nii-yamagishilab/AntiDeepfake","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-589","pdf":"https://www.isca-archive.org/interspeech_2026/wang26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26k_interspeech/markdown.md"},{"id":"wang26l_interspeech","title":"An All-neural Distributed Filtering Algorithm for Speech Extraction in Wireless Acoustic Sensor Networks","authors":["Jiawei Wang","Xiaoqing Hu","Feiran Yang","Jianfei Tong","Guohua Sun"],"year":2026,"doi":"10.21437/Interspeech.2026-648","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26l_interspeech.pdf","session":"Multi-Channel, Beamforming and Spatial Speech Enhancement","topics":["speech-enhancement","source-separation","multilingual"],"category":"enhancement-separation","institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26l_interspeech","category":"enhancement-separation","institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-648","pdf":"https://www.isca-archive.org/interspeech_2026/wang26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26l_interspeech/markdown.md"},{"id":"wang26m_interspeech","title":"Position-Aware Target Speaker Extraction for Long-Form Multi-Party Conversations: A Diarization-Free Framework for ASR","authors":["Yichi Wang","Junzhe Chen","Wangjin Zhou","Tatsuya Kawahara"],"year":2026,"doi":"10.21437/Interspeech.2026-787","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26m_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["target-speaker-extraction","speech-enhancement","asr"],"category":"enhancement-separation","institutions":["Kyoto University"],"funding":["JST BOOST","JST Moonshot R&D"],"code":{"url":"https://huggingface.co/datasets/real-recordings/LibriReplay-DOA","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26m_interspeech","category":"enhancement-separation","institutions":["Kyoto University"],"code":"https://huggingface.co/datasets/real-recordings/LibriReplay-DOA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-787","pdf":"https://www.isca-archive.org/interspeech_2026/wang26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26m_interspeech/markdown.md"},{"id":"wang26n_interspeech","title":"Towards Interpretable Framework for Neural Audio Codecs via Sparse Autoencoders: A Case Study on Accent Information","authors":["Shih-Heng Wang","Tiantian Feng","Aditya Kommineni","Thanathai Lertpetchpun","Bowen Yi","Xuan Shi","Shrikanth Narayanan"],"year":2026,"doi":"10.21437/Interspeech.2026-811","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26n_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["self-supervised","paralinguistic","evaluation"],"category":"speech-coding","institutions":["University of Southern California"],"funding":["Office of the Director of National Intelligence","Intelligence Advanced Research Projects Activity","ARTS Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26n_interspeech","category":"speech-coding","institutions":["University of Southern California"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-811","pdf":"https://www.isca-archive.org/interspeech_2026/wang26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26n_interspeech/markdown.md"},{"id":"wang26o_interspeech","title":"Gumbel-BEARD: Automatic Layer Selection for Self-Supervised Adaptation of Whisper in Low-Resource Domains","authors":["Zilai Wang","Natarajan Balaji Shankar","Mohan Shi","Kaiyuan Zhang","Abeer Alwan"],"year":2026,"doi":"10.21437/Interspeech.2026-825","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26o_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26o_interspeech.pdf","session":"Self-supervised Speech Representation Learning","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["low-resource","self-supervised"],"institutions":["University of California, Los Angeles"],"funding":["National Science Foundation","Institute of Education Sciences","U.S. Department of Education"],"code":{"url":"https://github.com/Zilai-WANG/Gumbel_Beard","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26o_interspeech","category":"asr","labels":["low-resource","self-supervised"],"institutions":["University of California, Los Angeles"],"code":"https://github.com/Zilai-WANG/Gumbel_Beard","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-825","pdf":"https://www.isca-archive.org/interspeech_2026/wang26o_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26o_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26o_interspeech/markdown.md"},{"id":"wang26p_interspeech","title":"SAGE: Switch-Aware EEG-Guided Soft Gating for Target Speaker Extraction with In-Trial Switching","authors":["Xuefei Wang","Ximin Chen","Yuting Ding","Chunlin Li","Fei Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-864","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26p_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26p_interspeech.pdf","session":"Brain Studies and Speech","topics":["speech-enhancement","self-supervised","spoken-language-understanding"],"category":"enhancement-separation","labels":["streaming-real-time"],"institutions":["Southern University of Science and Technology","Capital Medical University"],"funding":["National Key Research and Development Program of China","National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26p_interspeech","category":"enhancement-separation","labels":["streaming-real-time"],"institutions":["Southern University of Science and Technology","Capital Medical University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-864","pdf":"https://www.isca-archive.org/interspeech_2026/wang26p_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26p_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26p_interspeech/markdown.md"},{"id":"wang26q_interspeech","title":"Empathy Omni: Enabling Empathetic Speech Response Generation Through Large Language Models","authors":["Haoyu Wang","Guangyan Zhang","Jiale Chen","Jingyu Li","Yuehai Wang","Yiwen Guo"],"year":2026,"doi":"10.21437/Interspeech.2026-984","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26q_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26q_interspeech.pdf","session":"Empathetic Dialogue and Interaction Dynamics","topics":["speech-llm","paralinguistics","emotion-recognition"],"category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["Zhejiang University","LIGHTSPEED"],"code":{"url":"https://anonymous.4open.science/w/omni_demo-4876/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26q_interspeech","category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["Zhejiang University","LIGHTSPEED"],"code":"https://anonymous.4open.science/w/omni_demo-4876/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-984","pdf":"https://www.isca-archive.org/interspeech_2026/wang26q_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26q_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26q_interspeech/markdown.md"},{"id":"wang26r_interspeech","title":"Is Speaker Identity a Unitary Construct? Neural Evidence for Distinct Trait Processing","authors":["Yike Wang","Kaile Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1018","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26r_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26r_interspeech.pdf","session":"Brain Studies and Speech","topics":["paralinguistics","speaker-verification","evaluation"],"category":"paralinguistics-emotion","institutions":["Hong Kong Polytechnic University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26r_interspeech","category":"paralinguistics-emotion","institutions":["Hong Kong Polytechnic University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1018","pdf":"https://www.isca-archive.org/interspeech_2026/wang26r_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26r_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26r_interspeech/markdown.md"},{"id":"wang26s_interspeech","title":"FlowTTS-GRPO: Online Reinforcement Learning with Multi-Objective Reward Optimization for Flow-Matching Based Text-to-Speech","authors":["Haoxu Wang","Biao Tian","Weiqing Li","Xiang Lv","Han Zhao","Xiangang Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1102","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26s_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26s_interspeech.pdf","session":"Text-to-Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["Alibaba Group"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26s_interspeech","category":"tts","labels":["generative-model"],"institutions":["Alibaba Group"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1102","pdf":"https://www.isca-archive.org/interspeech_2026/wang26s_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26s_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26s_interspeech/markdown.md"},{"id":"wang26t_interspeech","title":"MATA: A Training-Free Approach to Mitigate Cross-Modal Attention Imbalance in Large Audio Language Models","authors":["Junyu Wang","Jian Zong","Tianrui Wang","Zhengding Luo","Meng Ge","Xiaobao Wang","Longbiao Wang","Jianwu Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-1212","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26t_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26t_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","evaluation","self-supervised"],"category":"speech-llm-dialogue","institutions":["Tianjin University","Nanyang Technological University","Huiyan Technology","Shenzhen Institute of Advanced Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26t_interspeech","category":"speech-llm-dialogue","institutions":["Tianjin University","Nanyang Technological University","Huiyan Technology","Shenzhen Institute of Advanced Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1212","pdf":"https://www.isca-archive.org/interspeech_2026/wang26t_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26t_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26t_interspeech/markdown.md"},{"id":"wang26u_interspeech","title":"Multi-Loss Learning for Speech Emotion Recognition with Energy-Adaptive Mixup and Frame-Level Attention","authors":["Cong Wang","Yizhong Geng","Yuhua Wen","Qifei Li","Yingming Gao","Ruimin Wang","Chunfeng Wang","Hao Li","Wei Chen","Ya Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1219","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26u_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26u_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["speech-emotion-recognition","self-supervised","paralinguistics"],"category":"paralinguistics-emotion","institutions":["Beijing University of Posts and Telecommunications","Li Auto"],"funding":["National Key R&D Program of China","National Natural Science Foundation of China","National Language Commission","National Social Science Fund of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26u_interspeech","category":"paralinguistics-emotion","institutions":["Beijing University of Posts and Telecommunications","Li Auto"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1219","pdf":"https://www.isca-archive.org/interspeech_2026/wang26u_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26u_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26u_interspeech/markdown.md"},{"id":"wang26v_interspeech","title":"TC-DBI: A Plug-and-Play Trajectory Confidence-Guided Dynamic Block Inference Strategy for Speech Synthesis with Continuous Block Flow Matching","authors":["Ren Wang","Zhiyu Cui","Shun Lei","Jiawei Jin","Dongfu Song","Zhiyong Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-1242","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26v_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26v_interspeech.pdf","session":"Flow Matching for Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["Tsinghua University","Ant Group"],"funding":["National Natural Science Foundation of China","Ant Group"],"code":{"url":"https://thuhcsi.github.io/interspeech2026-TC-DBI/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26v_interspeech","category":"tts","labels":["generative-model"],"institutions":["Tsinghua University","Ant Group"],"code":"https://thuhcsi.github.io/interspeech2026-TC-DBI/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1242","pdf":"https://www.isca-archive.org/interspeech_2026/wang26v_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26v_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26v_interspeech/markdown.md"},{"id":"wang26w_interspeech","title":"An Efficient vLLM-Based Inference Pipeline for Unified Audio Understanding and Generation","authors":["Haoran Wang","Jinchuan Tian","Siddhant Arora","Shinji Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-1244","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26w_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26w_interspeech.pdf","session":"Efficient Inference for ASR and Speech LMs","topics":["speech-llm","tts","asr"],"category":"speech-llm-dialogue","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Carnegie Mellon University","Shanghai Jiao Tong University"],"funding":["ACCESS program","National Science Foundation"],"code":{"url":"https://github.com/whr-a/vLLM","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26w_interspeech","category":"speech-llm-dialogue","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Carnegie Mellon University","Shanghai Jiao Tong University"],"code":"https://github.com/whr-a/vLLM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1244","pdf":"https://www.isca-archive.org/interspeech_2026/wang26w_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26w_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26w_interspeech/markdown.md"},{"id":"wang26x_interspeech","title":"Eliminating Stability Hallucinations in LLM-based TTS models via Attention Guidance","authors":["Shiming Wang","Zhihao Du","Yang Xiang","Tianyu Zhao","Han Zhao","Qian Chen","Xiangang Li","Hanjie Guo","Zhen-Hua Ling"],"year":2026,"doi":"10.21437/Interspeech.2026-1345","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26x_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26x_interspeech.pdf","session":"LLM Based Speech Synthesis","topics":["tts","self-supervised","speech-llm"],"category":"tts","labels":["generative-model"],"institutions":["University of Science and Technology of China"],"code":{"url":"https://wsmzzz.github.io/llm_attn/index.html","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26x_interspeech","category":"tts","labels":["generative-model"],"institutions":["University of Science and Technology of China"],"code":"https://wsmzzz.github.io/llm_attn/index.html","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1345","pdf":"https://www.isca-archive.org/interspeech_2026/wang26x_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26x_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26x_interspeech/markdown.md"},{"id":"wang26y_interspeech","title":"AugCodec: A Low-Bitrate Disentangled Neural Speech Codec via Data Augmentation","authors":["Dongmei Wang","Xiaohang Sun","Yang Liu","Fanjie Kong","Abhishek Yanamandra","Abhinav Jain","Daniel Tompkins","Woohyun Kang","Najmeh Sadoughi","Sunil Hadap","Xiang Hao","Zhu Liu","Caren Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1490","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26y_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26y_interspeech.pdf","session":"Neural Speech Codecs: Low-Bitrate and Disentangled Coding","topics":["self-supervised","speech-coding","voice-conversion"],"category":"speech-coding","labels":["generative-model"],"institutions":["Amazon"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26y_interspeech","category":"speech-coding","labels":["generative-model"],"institutions":["Amazon"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1490","pdf":"https://www.isca-archive.org/interspeech_2026/wang26y_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26y_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26y_interspeech/markdown.md"},{"id":"wang26z_interspeech","title":"Categorical Perception of Mandarin Tones in Jingpo Native Speakers","authors":["Binghao Wang","Xinyuan Li","Yao Lu"],"year":2026,"doi":"10.21437/Interspeech.2026-1602","isca_url":"https://www.isca-archive.org/interspeech_2026/wang26z_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wang26z_interspeech.pdf","session":"Tones","topics":["phonetics","prosody","evaluation"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Peking University"],"funding":["National Social Science Fund of China"],"code":{"url":"https://github.com/wbh-XC/Interspeech2026","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wang26z_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Peking University"],"code":"https://github.com/wbh-XC/Interspeech2026","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1602","pdf":"https://www.isca-archive.org/interspeech_2026/wang26z_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26z_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wang26z_interspeech/markdown.md"},{"id":"ward26_interspeech","title":"The Interspeech 2026 Challenge on Transfer of Pragmatic Intent in Speech-to-Speech Translation","authors":["Nigel G. Ward","Marcel de Korte","Javier Vazquez","Montserrat G. Molina","Vanessa Bolado","Carol Figueroa","Eliya Nachmani","John E. Ortega","Satoshi Nakamura"],"year":2026,"doi":"10.21437/Interspeech.2026-390","isca_url":"https://www.isca-archive.org/interspeech_2026/ward26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ward26_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["speech-translation","prosody","evaluation"],"category":"translation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["University of Texas at El Paso","Ben Gurion University","Northeastern University","Chinese University of Hong Kong"],"funding":["National Science Foundation"],"code":{"url":"https://www.cs.utep.edu/topi/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ward26_interspeech","category":"translation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["University of Texas at El Paso","Ben Gurion University","Northeastern University","Chinese University of Hong Kong"],"code":"https://www.cs.utep.edu/topi/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-390","pdf":"https://www.isca-archive.org/interspeech_2026/ward26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ward26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ward26_interspeech/markdown.md"},{"id":"warlewski26_interspeech","title":"Sub-Model Short-Term Memory Convolutions for Keyword Spotting Systems on Device","authors":["Paweł Warlewski","Artur Czeczko","Artur Szumaczuk","Grzegorz Stefański","Szymon Klimaszewski"],"year":2026,"doi":"10.21437/Interspeech.2026-1343","isca_url":"https://www.isca-archive.org/interspeech_2026/warlewski26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/warlewski26_interspeech.pdf","session":"Efficient Inference for ASR and Speech LMs","topics":["keyword-spotting","speech-enhancement","on-device"],"category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Samsung R&D Institute Poland","Samsung AI Center Warsaw"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"warlewski26_interspeech","category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Samsung R&D Institute Poland","Samsung AI Center Warsaw"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1343","pdf":"https://www.isca-archive.org/interspeech_2026/warlewski26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/warlewski26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/warlewski26_interspeech/markdown.md"},{"id":"wazed26_interspeech","title":"CLEAR: Clinical LLM Embedding and Attention-based Reconstruction","authors":["Eashita Wazed","Hieyong Jeong","Choonsung Shin","Shima Okada"],"year":2026,"doi":"10.21437/Interspeech.2026-883","isca_url":"https://www.isca-archive.org/interspeech_2026/wazed26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wazed26_interspeech.pdf","session":"Acoustic Event Detection 3","topics":["speech-enhancement","source-separation","self-supervised"],"category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Chonnam National University","Ritsumeikan University"],"funding":["National Research Foundation of Korea","Institute of Information & Communications Technology Planning & Evaluation","Korea Creative Content Agency"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wazed26_interspeech","category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Chonnam National University","Ritsumeikan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-883","pdf":"https://www.isca-archive.org/interspeech_2026/wazed26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wazed26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wazed26_interspeech/markdown.md"},{"id":"weber26_interspeech","title":"Multilingual Word-Level Forced Alignment with Self-Supervised Representations and Learned Dynamic Programming","authors":["Roy Weber","Meidan Zehavi","Rotem Rousso","Joseph Keshet"],"year":2026,"doi":"10.21437/Interspeech.2026-296","isca_url":"https://www.isca-archive.org/interspeech_2026/weber26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/weber26_interspeech.pdf","session":"Speech signal analysis","topics":["speech-alignment","self-supervised","multilingual"],"category":"asr","labels":["multilingual","self-supervised"],"institutions":["Technion – Israel Institute of Technology"],"funding":["NSF","BSF"],"code":{"url":"https://github.com/MLSpeech/Multilingual-Word-Aligner","stars":12,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"weber26_interspeech","category":"asr","labels":["multilingual","self-supervised"],"institutions":["Technion – Israel Institute of Technology"],"code":"https://github.com/MLSpeech/Multilingual-Word-Aligner","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-296","pdf":"https://www.isca-archive.org/interspeech_2026/weber26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/weber26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/weber26_interspeech/markdown.md"},{"id":"wei26_interspeech","title":"Vowel Nasalization in Upper Airway Diseases: An Analysis Using the CUCO Database","authors":["Yilan Wei","Qi Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-46","isca_url":"https://www.isca-archive.org/interspeech_2026/wei26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wei26_interspeech.pdf","session":"Pathological Speech Assessment 3","topics":["paralinguistics","evaluation","health"],"category":"health-clinical","institutions":["Hunan University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wei26_interspeech","category":"health-clinical","institutions":["Hunan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-46","pdf":"https://www.isca-archive.org/interspeech_2026/wei26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wei26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wei26_interspeech/markdown.md"},{"id":"wei26b_interspeech","title":"Speaker Identity in Non-Verbal Vocalizations: Conditional Distillation and Mixture of Experts Approach","authors":["Tzu-Chieh Wei","Yi-Cheng Lin","Huang-Cheng Chou","Kuan-Yu Chen","Hsin-Yen Sung","Shrikanth Narayanan","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-77","isca_url":"https://www.isca-archive.org/interspeech_2026/wei26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wei26b_interspeech.pdf","session":"Speech and Language Representation","topics":["speaker-verification","self-supervised","tts"],"category":"speaker","institutions":["University of Michigan","National Taiwan University","University of Southern California"],"funding":["Ministry of Education","Taiwan Centers of Excellence in Artificial Intelligence","NTU Artificial Intelligence Center of Research Excellence","NSTC","US National Science Foundation","ODNI IARPA ARTS Program"],"code":{"url":"https://github.com/wiizzz/nonverbal-sv","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wei26b_interspeech","category":"speaker","institutions":["University of Michigan","National Taiwan University","University of Southern California"],"code":"https://github.com/wiizzz/nonverbal-sv","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-77","pdf":"https://www.isca-archive.org/interspeech_2026/wei26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wei26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wei26b_interspeech/markdown.md"},{"id":"wei26c_interspeech","title":"Perceptually Weighted Minimum Mean Square Error Precoding for Acoustic Multi-User MIMO in Vehicular Personal Sound Zones","authors":["Huihui Wei","Dengke Deng","Pengcheng Luo","Weiyu You","Xiaojuan Zhang","Genke Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-1191","isca_url":"https://www.isca-archive.org/interspeech_2026/wei26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wei26c_interspeech.pdf","session":"Spatial Audio 3","topics":["source-separation","speech-enhancement","evaluation"],"category":"enhancement-separation","institutions":["Shanghai Jiao Tong University","Ningbo Artificial Intelligence Institute","Institute of Advanced Intelligence and Computing"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wei26c_interspeech","category":"enhancement-separation","institutions":["Shanghai Jiao Tong University","Ningbo Artificial Intelligence Institute","Institute of Advanced Intelligence and Computing"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1191","pdf":"https://www.isca-archive.org/interspeech_2026/wei26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wei26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wei26c_interspeech/markdown.md"},{"id":"wei26d_interspeech","title":"USV-DETR: High-Resolution and Densely Supervised Detection of Ultrasonic Vocalizations","authors":["Yilan Wei","Kumiko Long","Arielle Granston","Adrian Rodriguez-Contreras"],"year":2026,"doi":"10.21437/Interspeech.2026-1492","isca_url":"https://www.isca-archive.org/interspeech_2026/wei26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wei26d_interspeech.pdf","session":"Acoustic Event Detection 2","topics":["paralinguistics","dataset","evaluation"],"category":"audio-understanding","institutions":["Northwestern University"],"code":{"url":"https://github.com/weiyilan9/USV-DETR","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wei26d_interspeech","category":"audio-understanding","institutions":["Northwestern University"],"code":"https://github.com/weiyilan9/USV-DETR","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1492","pdf":"https://www.isca-archive.org/interspeech_2026/wei26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wei26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wei26d_interspeech/markdown.md"},{"id":"wei26e_interspeech","title":"Do Speech Emphasis Models Generalize across Languages and Emotions?","authors":["Megan Wei","Deepali Aneja","Jiaqi Su","Yunyun Wang","Haonan Chen","Zeyu Jin"],"year":2026,"doi":"10.21437/Interspeech.2026-2783","isca_url":"https://www.isca-archive.org/interspeech_2026/wei26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wei26e_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["prosody","multilingual","self-supervised"],"category":"phonetics-linguistics","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Adobe Research","Brown University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wei26e_interspeech","category":"phonetics-linguistics","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Adobe Research","Brown University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2783","pdf":"https://www.isca-archive.org/interspeech_2026/wei26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wei26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wei26e_interspeech/markdown.md"},{"id":"weller26_interspeech","title":"HaessigDB: A Database of Irritable Speech with Intensity Grading","authors":["Niklas Weller","Marc Grau","Ivo Blohm"],"year":2026,"doi":"10.21437/Interspeech.2026-3304","isca_url":"https://www.isca-archive.org/interspeech_2026/weller26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/weller26_interspeech.pdf","session":"Datasets","topics":["paralinguistics","emotion-recognition","dataset"],"category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["University of St. Gallen"],"funding":["Innosuisse"],"code":{"url":"https://huggingface.co/datasets/nwllr/haessigDB","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"weller26_interspeech","category":"resources-evaluation","labels":["self-supervised","dataset-or-benchmark-release"],"institutions":["University of St. Gallen"],"code":"https://huggingface.co/datasets/nwllr/haessigDB","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3304","pdf":"https://www.isca-archive.org/interspeech_2026/weller26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/weller26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/weller26_interspeech/markdown.md"},{"id":"wellington26_interspeech","title":"Shared Phone-Level Neural Representations of Auditory Perception and ‘Inner Voice’ Production: One-to-One Mapping using a Single-Subject EEG Corpus of Heard and Imagined Natural Speech","authors":["Scott Wellington","Oliver Watts","Damien Coyle","Benjamin Metcalfe"],"year":2026,"doi":"10.21437/Interspeech.2026-2683","isca_url":"https://www.isca-archive.org/interspeech_2026/wellington26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wellington26_interspeech.pdf","session":"Assistive Technologies 2","topics":["speech-llm","self-supervised","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["University of Bath","SpeakUnique"],"funding":["United Kingdom Research Institute"],"code":{"url":"https://www.speakunique.co.uk/research/CHINS","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wellington26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["University of Bath","SpeakUnique"],"code":"https://www.speakunique.co.uk/research/CHINS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2683","pdf":"https://www.isca-archive.org/interspeech_2026/wellington26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wellington26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wellington26_interspeech/markdown.md"},{"id":"wen26_interspeech","title":"Addressing random spatial translations in measured microphone directional responses by maximizing finite order energy","authors":["Xue Wen"],"year":2026,"doi":"10.21437/Interspeech.2026-149","isca_url":"https://www.isca-archive.org/interspeech_2026/wen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wen26_interspeech.pdf","session":"Spatial Audio 4","topics":["spatial-audio","evaluation","self-supervised"],"category":"applications-other","institutions":["Samsung"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wen26_interspeech","category":"applications-other","institutions":["Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-149","pdf":"https://www.isca-archive.org/interspeech_2026/wen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wen26_interspeech/markdown.md"},{"id":"wen26b_interspeech","title":"YUE-PUB-Speech: A Speech-based Pragmatic Understanding Benchmark for Cantonese","authors":["Yajie Wen","Ziwei Gong","Chengyan Wu","Xiyun Gong","Yun Xue","Julia Hirschberg","Bolei Ma"],"year":2026,"doi":"10.21437/Interspeech.2026-289","isca_url":"https://www.isca-archive.org/interspeech_2026/wen26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wen26b_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["spoken-language-understanding","low-resource","self-supervised"],"category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["South China Normal University","Columbia University","Sun Yat-sen University","Guangzhou College of Commerce","LMU Munich","Munich Center for Machine Learning"],"funding":["National Science Foundation","Guangdong Basic and Applied Basic Research Foundation","National Natural Science Foundation of China","Characteristic Innovation Projects of Guangdong Colleges and Universities","Guangdong Provincial Key Laboratory"],"code":{"url":"https://huggingface.co/datasets/Multilingual-NLP/YUE-PUB-Speech","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wen26b_interspeech","category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["South China Normal University","Columbia University","Sun Yat-sen University","Guangzhou College of Commerce","LMU Munich","Munich Center for Machine Learning"],"code":"https://huggingface.co/datasets/Multilingual-NLP/YUE-PUB-Speech","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-289","pdf":"https://www.isca-archive.org/interspeech_2026/wen26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wen26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wen26b_interspeech/markdown.md"},{"id":"wen26c_interspeech","title":"End-Fire Degradation-Robust DOA Estimation for Compact Linear Microphone Arrays","authors":["Zheng Wen","Huayang Wang","Zhongxin Bai","Xin Guo","Gongping Huang","Qiqi Tong","Yaling Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-2156","isca_url":"https://www.isca-archive.org/interspeech_2026/wen26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wen26c_interspeech.pdf","session":"Spatial Audio 1","topics":["speaker-diarization","speech-enhancement","evaluation"],"category":"enhancement-separation","institutions":["Wuhan University","Harbin Engineering University","Guangdong Murora Intelligent Lighting Co., Ltd"],"code":{"url":"https://github.com/WwHhYy666/ULA_End_Fire_Robust_ASL","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wen26c_interspeech","category":"enhancement-separation","institutions":["Wuhan University","Harbin Engineering University","Guangdong Murora Intelligent Lighting Co., Ltd"],"code":"https://github.com/WwHhYy666/ULA_End_Fire_Robust_ASL","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2156","pdf":"https://www.isca-archive.org/interspeech_2026/wen26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wen26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wen26c_interspeech/markdown.md"},{"id":"wen26d_interspeech","title":"Towards Robust Ultrasound-based Silent Speech Recognition Learning Physics-Aware and Context-Rich Representations","authors":["Hongyu Wen","Yi Su","Yudong Yang","Siwen Guo","Qisheng Xu","Kele Xu"],"year":2026,"doi":"10.21437/Interspeech.2026-2646","isca_url":"https://www.isca-archive.org/interspeech_2026/wen26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wen26d_interspeech.pdf","session":"Speech Production and Perception 2","topics":["silent-speech-recognition","self-supervised","dataset"],"category":"asr","labels":["self-supervised"],"institutions":["National University of Defense Technology","Chinese Academy of Sciences"],"funding":["National Science and Technology Major Project","National University of Defense Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wen26d_interspeech","category":"asr","labels":["self-supervised"],"institutions":["National University of Defense Technology","Chinese Academy of Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2646","pdf":"https://www.isca-archive.org/interspeech_2026/wen26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wen26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wen26d_interspeech/markdown.md"},{"id":"weng26_interspeech","title":"Tone-space Distribution Modulates Transfer from Non-linguistic Pitch Training to Cantonese Tone-in-Noise Perception in Native Speakers","authors":["Yi Weng","Junjie Zhang","Yanyuan Ye","Gang Peng"],"year":2026,"doi":"10.21437/Interspeech.2026-1274","isca_url":"https://www.isca-archive.org/interspeech_2026/weng26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/weng26_interspeech.pdf","session":"Tones","topics":["phonetics","evaluation","low-resource"],"category":"phonetics-linguistics","labels":["robustness-noise"],"institutions":["Hong Kong Polytechnic University"],"funding":["Research Grants Council of the Hong Kong SAR, China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"weng26_interspeech","category":"phonetics-linguistics","labels":["robustness-noise"],"institutions":["Hong Kong Polytechnic University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1274","pdf":"https://www.isca-archive.org/interspeech_2026/weng26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/weng26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/weng26_interspeech/markdown.md"},{"id":"wiechmann26_interspeech","title":"Can deep learning based voice editing enhance voice quality perception skills in speech therapy students?","authors":["Jana Wiechmann","Frederik Rautenberg","Reinhold Haeb-Umbach","Petra Wagner"],"year":2026,"doi":"10.21437/Interspeech.2026-2475","isca_url":"https://www.isca-archive.org/interspeech_2026/wiechmann26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wiechmann26_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 1","topics":["speech-enhancement","evaluation","paralinguistics"],"category":"health-clinical","labels":["generative-model"],"institutions":["Bielefeld University","Paderborn University"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wiechmann26_interspeech","category":"health-clinical","labels":["generative-model"],"institutions":["Bielefeld University","Paderborn University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2475","pdf":"https://www.isca-archive.org/interspeech_2026/wiechmann26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wiechmann26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wiechmann26_interspeech/markdown.md"},{"id":"wiesner26_interspeech","title":"Modeling Overlapped Speech with Shuffles","authors":["Matthew Wiesner","Samuele Cornell","Alexander Polok","Lucas Ondel-Yang","Lukáš Burget","Sanjeev Khudanpur"],"year":2026,"doi":"10.21437/Interspeech.2026-2462","isca_url":"https://www.isca-archive.org/interspeech_2026/wiesner26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wiesner26_interspeech.pdf","session":"Speaker Diarization and Recognition","topics":["multi-talker-asr","self-supervised","alignment"],"category":"asr","labels":["self-supervised"],"institutions":["Johns Hopkins University","CNRS","Carnegie Mellon University","Brno University of Technology"],"funding":["Jelinek Memorial Summer Workshop on Speech and Language Technologies","Advanced Cyberinfrastructure Coordination Ecosystem: Services Support","National Science Foundation","National Science and Technology Council","Ministry of Education, Youth and Sports of the Czech Republic"],"code":{"url":"https://github.com/geolocation-from-speech/jsalt2025.git","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wiesner26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Johns Hopkins University","CNRS","Carnegie Mellon University","Brno University of Technology"],"code":"https://github.com/geolocation-from-speech/jsalt2025.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2462","pdf":"https://www.isca-archive.org/interspeech_2026/wiesner26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wiesner26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wiesner26_interspeech/markdown.md"},{"id":"wilkinghoff26_interspeech","title":"Mind the Gap: Detecting Cluster Exits for Robust Local Density-Based Score Normalization in Anomalous Sound Detection","authors":["Kevin Wilkinghoff","Gordon Wichern","Jonathan Le Roux","Zheng-Hua Tan"],"year":2026,"doi":"10.21437/Interspeech.2026-177","isca_url":"https://www.isca-archive.org/interspeech_2026/wilkinghoff26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wilkinghoff26_interspeech.pdf","session":"Acoustic Event Detection 1","topics":["paralinguistics","dataset","evaluation"],"category":"audio-understanding","institutions":["Aalborg University","Pioneer Centre for Artificial Intelligence","Mitsubishi Electric Research Laboratories"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wilkinghoff26_interspeech","category":"audio-understanding","institutions":["Aalborg University","Pioneer Centre for Artificial Intelligence","Mitsubishi Electric Research Laboratories"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-177","pdf":"https://www.isca-archive.org/interspeech_2026/wilkinghoff26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wilkinghoff26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wilkinghoff26_interspeech/markdown.md"},{"id":"williams26_interspeech","title":"AI Regulation and the Technical Language of Speech Synthesis","authors":["Jennifer Williams"],"year":2026,"doi":"10.21437/Interspeech.2026-210","isca_url":"https://www.isca-archive.org/interspeech_2026/williams26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/williams26_interspeech.pdf","session":"Safeguards for Synthetic Speech: Ethical, Technical, and Legal Perspectives","topics":["speech-llm","speaker-verification","voice-conversion"],"category":"deepfake-security","institutions":["University of Southampton"],"funding":["EPSRC National EdgeAI Hub","EPSRC Responsible AI UK","Research England Higher Education Innovation Fund WSI Knowledge Exchange Fund"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"williams26_interspeech","category":"deepfake-security","institutions":["University of Southampton"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-210","pdf":"https://www.isca-archive.org/interspeech_2026/williams26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/williams26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/williams26_interspeech/markdown.md"},{"id":"withanage26_interspeech","title":"Articulatory Entrainment and Coordination Complexity in Spontaneous Autistic and Non-autistic Dialogue","authors":["Thanushi Withanage","Carol Espy-Wilson","Elizabeth Redcay","Desi Jones","Noah Sasson"],"year":2026,"doi":"10.21437/Interspeech.2026-2855","isca_url":"https://www.isca-archive.org/interspeech_2026/withanage26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/withanage26_interspeech.pdf","session":"Entrainment and Dialogue Coordination","topics":["paralinguistics","evaluation","phonetics"],"category":"phonetics-linguistics","institutions":["University of Maryland","University of Texas at Dallas"],"funding":["Grand Challenge grant from University of Maryland"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"withanage26_interspeech","category":"phonetics-linguistics","institutions":["University of Maryland","University of Texas at Dallas"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2855","pdf":"https://www.isca-archive.org/interspeech_2026/withanage26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/withanage26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/withanage26_interspeech/markdown.md"},{"id":"wong26_interspeech","title":"TMASC: Transmasculine Attitude and Speech Corpus","authors":["Sidney Wong"],"year":2026,"doi":"10.21437/Interspeech.2026-2637","isca_url":"https://www.isca-archive.org/interspeech_2026/wong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wong26_interspeech.pdf","session":"Queer and Trans Speech Science and Technology","topics":["paralinguistics","dataset","evaluation"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["University of Otago","Te Punaha Matatini"],"funding":["UC College of Arts","School of Language, Social and Political Science","Te Punaha Matatini"],"code":{"url":"https://osf.io/tg8bc/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wong26_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["University of Otago","Te Punaha Matatini"],"code":"https://osf.io/tg8bc/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2637","pdf":"https://www.isca-archive.org/interspeech_2026/wong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wong26_interspeech/markdown.md"},{"id":"woszczyk26_interspeech","title":"Is Natural Always Appropriate? Investigating Naturalness and Appropriateness Across Different Domains for TTS Evaluation","authors":["Dominika Woszczyk","Andreas Triantafyllopoulos","Jura Miniota","Éva Székely","Bjoern Schuller"],"year":2026,"doi":"10.21437/Interspeech.2026-3392","isca_url":"https://www.isca-archive.org/interspeech_2026/woszczyk26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/woszczyk26_interspeech.pdf","session":"Speech Synthesis Evaluation 1","topics":["tts","evaluation","prosody"],"category":"resources-evaluation","institutions":["Iconic","Technische Universitat Munchen","KTH Royal Institute of Technology","Imperial College London"],"code":{"url":"https://github.com/domiwk/domain-aware-tts-eval","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"woszczyk26_interspeech","category":"resources-evaluation","institutions":["Iconic","Technische Universitat Munchen","KTH Royal Institute of Technology","Imperial College London"],"code":"https://github.com/domiwk/domain-aware-tts-eval","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3392","pdf":"https://www.isca-archive.org/interspeech_2026/woszczyk26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/woszczyk26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/woszczyk26_interspeech/markdown.md"},{"id":"wright26_interspeech","title":"Not all language switching is equal: Language brokering and code-switching are associated with working memory and inhibitory control in young adults","authors":["Sarah M. Wright","Mark Antoniou","Michael Tyler"],"year":2026,"doi":"10.21437/Interspeech.2026-3404","isca_url":"https://www.isca-archive.org/interspeech_2026/wright26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wright26_interspeech.pdf","session":"Brain Studies and Speech","topics":["spoken-language-understanding","multilingual","evaluation"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Western Sydney University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wright26_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Western Sydney University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3404","pdf":"https://www.isca-archive.org/interspeech_2026/wright26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wright26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wright26_interspeech/markdown.md"},{"id":"wu26_interspeech","title":"LISE : Listenable Interpretable Speaker Embeddings","authors":["Xiaoliang Wu","Chong-xin Gan","Ke Liu","Peter Bell","Jennifer Williams"],"year":2026,"doi":"10.21437/Interspeech.2026-537","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["speaker-verification","self-supervised","evaluation"],"category":"speaker","labels":["self-supervised"],"institutions":["University of Southampton","Hong Kong Polytechnic University","University of Edinburgh"],"funding":["Engineering and Physical Sciences Research Council","National Edge AI Hub for Real Data: Edge Intelligence for Cyberdisturbances and Data Quality","Responsible AI UK"],"code":{"url":"https://sites.google.com/view/components-samples/home","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26_interspeech","category":"speaker","labels":["self-supervised"],"institutions":["University of Southampton","Hong Kong Polytechnic University","University of Edinburgh"],"code":"https://sites.google.com/view/components-samples/home","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-537","pdf":"https://www.isca-archive.org/interspeech_2026/wu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26_interspeech/markdown.md"},{"id":"wu26b_interspeech","title":"Towards Dys-XAI: Influence-Based Explanations for Dysarthria Severity Assessment","authors":["Xiaoliang Wu","Qiyang Sun","Yupei Li","Erfan Loweimi","Jennifer Williams","Zhengjun Yue"],"year":2026,"doi":"10.21437/Interspeech.2026-538","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26b_interspeech.pdf","session":"Explainability for Compliance and Trust in Speech AI","topics":["paralinguistics","evaluation","health"],"category":"health-clinical","institutions":["University of Southampton","Imperial College London","University of Edinburgh","King's College London"],"funding":["Engineering and Physical Sciences Research Council","National Edge AI Hub for Real Data: Edge Intelligence for Cyberdisturbances and Data Quality","Responsible AI UK"],"code":{"url":"https://sites.google.com/view/infx-dys-samples","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26b_interspeech","category":"health-clinical","institutions":["University of Southampton","Imperial College London","University of Edinburgh","King's College London"],"code":"https://sites.google.com/view/infx-dys-samples","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-538","pdf":"https://www.isca-archive.org/interspeech_2026/wu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26b_interspeech/markdown.md"},{"id":"wu26c_interspeech","title":"SEA-MDD: Self-adapting Mispronunciation Detection and Diagnosis Models via Test-Time Training","authors":["Minglin Wu","Helen Meng"],"year":2026,"doi":"10.21437/Interspeech.2026-856","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26c_interspeech.pdf","session":"Domain Adaptation & Accented ASR","topics":["mdd","self-supervised","low-resource"],"category":"asr","labels":["low-resource"],"institutions":["Chinese University of Hong Kong"],"funding":["Centre for Perceptual and Interactive Intelligence","Innovation and Technology Commission of the Hong Kong Special Administrative Region Government"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26c_interspeech","category":"asr","labels":["low-resource"],"institutions":["Chinese University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-856","pdf":"https://www.isca-archive.org/interspeech_2026/wu26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26c_interspeech/markdown.md"},{"id":"wu26d_interspeech","title":"Articulatory Analysis of the Mandarin Alveolar–Retroflex Contrast Using Real-Time MRI","authors":["Qi Wu","Tatsuya Kitamura"],"year":2026,"doi":"10.21437/Interspeech.2026-1593","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26d_interspeech.pdf","session":"Methods and Data for Vocal Tract Shape and Articulation Analysis","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["University of Tsukuba","Konan University"],"funding":["JSPS KAKENHI","Kawai Foundation for Sound & Music"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26d_interspeech","category":"phonetics-linguistics","institutions":["University of Tsukuba","Konan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1593","pdf":"https://www.isca-archive.org/interspeech_2026/wu26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26d_interspeech/markdown.md"},{"id":"wu26e_interspeech","title":"CrossPhon-Tonal: Streamlining Cross-language Modeling for Forced Alignment in Low-resource Tonal Languages","authors":["Hongchen Wu","Yixin Gu"],"year":2026,"doi":"10.21437/Interspeech.2026-1770","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26e_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["speech-enhancement","low-resource","multilingual"],"category":"asr","labels":["low-resource","multilingual"],"institutions":["Georgia Institute of Technology","University of Illinois Urbana-Champaign"],"funding":["Georgia Institute of Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26e_interspeech","category":"asr","labels":["low-resource","multilingual"],"institutions":["Georgia Institute of Technology","University of Illinois Urbana-Champaign"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1770","pdf":"https://www.isca-archive.org/interspeech_2026/wu26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26e_interspeech/markdown.md"},{"id":"wu26f_interspeech","title":"EmoInstruct-TTS: Dual-Path Instruction-Guided Emotional Speech Synthesis","authors":["Minghui Wu","Ganjun Liu","Zikun Fang","Ting Meng","Hongchuan Wu","Bingao Xu","Yonglong Cai","Jiasheng Chen","Jun Du"],"year":2026,"doi":"10.21437/Interspeech.2026-1834","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26f_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["tts","speech-llm","paralinguistics"],"category":"tts","labels":["generative-model"],"institutions":["University of Science and Technology of China","iFLYTEK","Huawei Technologies"],"code":{"url":"https://huanyulab.github.io/EMOINSTRUCT-TTS","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26f_interspeech","category":"tts","labels":["generative-model"],"institutions":["University of Science and Technology of China","iFLYTEK","Huawei Technologies"],"code":"https://huanyulab.github.io/EMOINSTRUCT-TTS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1834","pdf":"https://www.isca-archive.org/interspeech_2026/wu26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26f_interspeech/markdown.md"},{"id":"wu26g_interspeech","title":"Accelerating End-to-End ASR via Semi-Autoregressive Speculative Decoding","authors":["Long Wu","Lingchao Zhao","Yuanzhong Zheng","Haojun Fei","Qing Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-1953","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26g_interspeech.pdf","session":"Search Methods and Inference Algorithms","topics":["asr","self-supervised","on-device"],"category":"asr","labels":["efficient-on-device"],"institutions":["Qifu Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26g_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["Qifu Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1953","pdf":"https://www.isca-archive.org/interspeech_2026/wu26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26g_interspeech/markdown.md"},{"id":"wu26h_interspeech","title":"SongBench: A Fine-Grained Multi-Aspect Benchmark for Song Quality Assessment","authors":["Dapeng Wu","Shun Lei","Wei Tan","Guangzheng Li","Yunzhe Wang","Huaicheng Zhang","Lishi Zuo","Zhiyong Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-1985","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26h_interspeech.pdf","session":"Speech Synthesis Evaluation and Benchmarking","topics":["evaluation","dataset","tts"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Tsinghua University","Tencent"],"funding":["National Natural Science Foundation of China","Shenzhen Science and Technology Program"],"code":{"url":"https://github.com/Tencent/SongBench","stars":60,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26h_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Tsinghua University","Tencent"],"code":"https://github.com/Tencent/SongBench","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1985","pdf":"https://www.isca-archive.org/interspeech_2026/wu26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26h_interspeech/markdown.md"},{"id":"wu26i_interspeech","title":"AFG-Bias: Acoustic-Fusion-Gated Biasing for Plug-and-Play Hotword Customization in LLM-Based ASR","authors":["Long Wu","Lingchao Zhao","Yuanzhong Zheng","Haojun Fei","Qing Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-2029","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26i_interspeech.pdf","session":"Multi-Speaker Processing, Personalization, and Adaptation","topics":["asr","speech-llm","low-resource"],"category":"asr","institutions":["Qifu Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26i_interspeech","category":"asr","institutions":["Qifu Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2029","pdf":"https://www.isca-archive.org/interspeech_2026/wu26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26i_interspeech/markdown.md"},{"id":"wu26j_interspeech","title":"AuscuTSLM: Patient-Level Multimodal Question Answering from Multi-Site Auscultation Recordings","authors":["Fan Wu","Tsai-Ning Wang","Nicolas Zumarraga","Ning Wang","Markus Kreft","Kevin O'Sullivan","Paula Manso Zorrilla","Elgar Fleisch","Oliver Aalami","Paul Schmiedmayer","Robert Jakob","Patrick Langer"],"year":2026,"doi":"10.21437/Interspeech.2026-2037","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26j_interspeech.pdf","session":"Medical Dialogue and Conversational Understanding","topics":["speech-llm","paralinguistics","health"],"category":"health-clinical","institutions":["ETH Zurich","Eindhoven University of Technology","University of St. Gallen","Stanford University"],"code":{"url":"https://github.com/Fan-loewe/AuscuTSLM","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26j_interspeech","category":"health-clinical","institutions":["ETH Zurich","Eindhoven University of Technology","University of St. Gallen","Stanford University"],"code":"https://github.com/Fan-loewe/AuscuTSLM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2037","pdf":"https://www.isca-archive.org/interspeech_2026/wu26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26j_interspeech/markdown.md"},{"id":"wu26k_interspeech","title":"Leveraging Temporal Redundancy via Layer-wise Key-Value Pooling Attention for Efficient ASR","authors":["Yi Wu","Guibin Zheng","Chenhao Jing","Jiqing Han","Jiarui Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-2124","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26k_interspeech.pdf","session":"Resource Constrained Speech Recognition","topics":["asr","self-supervised","on-device"],"category":"asr","labels":["efficient-on-device"],"institutions":["Harbin Institute of Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26k_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["Harbin Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2124","pdf":"https://www.isca-archive.org/interspeech_2026/wu26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26k_interspeech/markdown.md"},{"id":"wu26l_interspeech","title":"One-to-Many Electrolaryngeal Voice Conversion with Synthetic Data","authors":["Bowen Wu","Haruto Ueno","Carlos Toshinori Ishi","Chaoran Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-2150","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26l_interspeech.pdf","session":"Beyond Speech Technologies in Healthcare","topics":["speech-enhancement","voice-conversion","low-resource"],"category":"tts","labels":["low-resource","generative-model"],"institutions":["RIKEN","University of Osaka","Advanced Telecommunications Research Institute International","National Institute of Informatics"],"funding":["RIKEN Special Postdoctoral Researcher Program","JST Moonshot R&D"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26l_interspeech","category":"tts","labels":["low-resource","generative-model"],"institutions":["RIKEN","University of Osaka","Advanced Telecommunications Research Institute International","National Institute of Informatics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2150","pdf":"https://www.isca-archive.org/interspeech_2026/wu26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26l_interspeech/markdown.md"},{"id":"wu26m_interspeech","title":"SEA-Spoof: Bridging the Gap in Multilingual Audio Deepfake Detection for South-East Asia","authors":["Jinyang Wu","Nana Hou","Zihan Pan","Qiquan Zhang","Sailor Hardik","Soumik Mondal"],"year":2026,"doi":"10.21437/Interspeech.2026-3019","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26m_interspeech.pdf","session":"Spoofing and Deepfake Detection 1","topics":["speech-deepfake-detection","multilingual","self-supervised"],"category":"deepfake-security","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Agency for Science, Technology and Research","Nanyang Technological University","University of New South Wales"],"funding":["National Research Foundation","Ministry of Digital Development and Information"],"code":{"url":"https://huggingface.co/datasets/Jack-ppkdczgx/SEA-Spoof/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26m_interspeech","category":"deepfake-security","labels":["low-resource","multilingual","dataset-or-benchmark-release"],"institutions":["Agency for Science, Technology and Research","Nanyang Technological University","University of New South Wales"],"code":"https://huggingface.co/datasets/Jack-ppkdczgx/SEA-Spoof/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3019","pdf":"https://www.isca-archive.org/interspeech_2026/wu26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26m_interspeech/markdown.md"},{"id":"wu26n_interspeech","title":"Quantizer-Aware Hierarchical Neural Codec Modeling for Speech Deepfake Detection","authors":["Jinyang Wu","Zihan Pan","Qiquan Zhang","Sailor Hardik","Soumik Mondal"],"year":2026,"doi":"10.21437/Interspeech.2026-3212","isca_url":"https://www.isca-archive.org/interspeech_2026/wu26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wu26n_interspeech.pdf","session":"Spoofing and Deepfake Detection 3","topics":["speech-deepfake-detection","self-supervised","audio-codec"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["Agency for Science, Technology and Research","University of New South Wales"],"funding":["National Research Foundation, Prime Minister’s Office, Singapore","Ministry of Digital Development and Information","Online Trust and Safety Research Programme"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wu26n_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["Agency for Science, Technology and Research","University of New South Wales"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3212","pdf":"https://www.isca-archive.org/interspeech_2026/wu26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wu26n_interspeech/markdown.md"},{"id":"wust26_interspeech","title":"Applying the Simplified Vocal Profile Analysis to Swiss German Dialects","authors":["Alessia Wüst","Adrian Leemann"],"year":2026,"doi":"10.21437/Interspeech.2026-329","isca_url":"https://www.isca-archive.org/interspeech_2026/wust26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/wust26_interspeech.pdf","session":"Cross-Linguistic and L2 Phonetic Studies","topics":["sociophonetics","paralinguistics","phonetics"],"category":"phonetics-linguistics","institutions":["University of Berne"],"funding":["Swiss National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"wust26_interspeech","category":"phonetics-linguistics","institutions":["University of Berne"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-329","pdf":"https://www.isca-archive.org/interspeech_2026/wust26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/wust26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/wust26_interspeech/markdown.md"},{"id":"xia26_interspeech","title":"Eye and Mouth Cues in Audiovisual Perception of Mandarin Irony: Evidence from Eye-Tracking","authors":["Shifeng Xia","Shanpeng Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1544","isca_url":"https://www.isca-archive.org/interspeech_2026/xia26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xia26_interspeech.pdf","session":"Behavioral, Cross-lingual, and Multimodal Speech Analysis","topics":["paralinguistics","evaluation","spoken-language-understanding"],"category":"paralinguistics-emotion","institutions":["Nanjing University of Science and Technology"],"funding":["Ministry of Education Humanities and Social Sciences Research Youth Fund Project","Jiangsu Social Science Fund Youth Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xia26_interspeech","category":"paralinguistics-emotion","institutions":["Nanjing University of Science and Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1544","pdf":"https://www.isca-archive.org/interspeech_2026/xia26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xia26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xia26_interspeech/markdown.md"},{"id":"xiang26_interspeech","title":"Quantifying the Uncertainty of Blindly Estimated Room Embeddings Using a Dispersion-Calibrated Score","authors":["Yang Xiang","Philipp Götz","Emanuël A. P. Habets","Andreas Walther","Wenwu Wang","Philip J.B. Jackson"],"year":2026,"doi":"10.21437/Interspeech.2026-1357","isca_url":"https://www.isca-archive.org/interspeech_2026/xiang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xiang26_interspeech.pdf","session":"Spatial Audio 4","topics":["self-supervised","speech-enhancement","evaluation"],"category":"enhancement-separation","labels":["self-supervised","robustness-noise"],"institutions":["University of Surrey","International Audio Laboratories Erlangen","Fraunhofer Institute for Integrated Circuits IIS"],"funding":["Fraunhofer IIS"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xiang26_interspeech","category":"enhancement-separation","labels":["self-supervised","robustness-noise"],"institutions":["University of Surrey","International Audio Laboratories Erlangen","Fraunhofer Institute for Integrated Circuits IIS"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1357","pdf":"https://www.isca-archive.org/interspeech_2026/xiang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xiang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xiang26_interspeech/markdown.md"},{"id":"xiang26b_interspeech","title":"Revisiting Delay Compensation via Feature-Level Temporal Accumulation in Continuous Emotion Recognition","authors":["Jian Xiang","Jingyao Wu","Ting Dang","Vidhyasaharan Sethu","Eliathamby Ambikairajah"],"year":2026,"doi":"10.21437/Interspeech.2026-3030","isca_url":"https://www.isca-archive.org/interspeech_2026/xiang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xiang26b_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["paralinguistics","emotion-recognition"],"category":"paralinguistics-emotion","institutions":["University of New South Wales","Massachusetts Institute of Technology","University of Melbourne"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xiang26b_interspeech","category":"paralinguistics-emotion","institutions":["University of New South Wales","Massachusetts Institute of Technology","University of Melbourne"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3030","pdf":"https://www.isca-archive.org/interspeech_2026/xiang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xiang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xiang26b_interspeech/markdown.md"},{"id":"xiao26_interspeech","title":"Continual Adaptation for Pacific Indigenous Speech Recognition","authors":["Yang Xiao","Aso Mahmudi","Nick Thieberger","Eliathamby Ambikairajah","Eun-Jung Holden","Ting Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-2215","isca_url":"https://www.isca-archive.org/interspeech_2026/xiao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xiao26_interspeech.pdf","session":"Pacific Voices: Speech Science and Technology for the Languages of the Pacific Ocean","topics":["asr","low-resource","self-supervised"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Melbourne","UNSW Sydney"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xiao26_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["University of Melbourne","UNSW Sydney"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2215","pdf":"https://www.isca-archive.org/interspeech_2026/xiao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xiao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xiao26_interspeech/markdown.md"},{"id":"xiao26b_interspeech","title":"WSG: Clinically-Informed Weighted Speech Graphs for Dementia Detection","authors":["Yao Xiao","Fritz Peters","Madhurananda Pahar","Dorota A Braun","Caitlin H Illingworth","Stefan Goetze","Daniel Blackburn","Heidi Christensen"],"year":2026,"doi":"10.21437/Interspeech.2026-2266","isca_url":"https://www.isca-archive.org/interspeech_2026/xiao26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xiao26b_interspeech.pdf","session":"Speech and Language Technologies for Health Applications 2","topics":["speech-llm","paralinguistics","health"],"category":"health-clinical","institutions":["University of Sheffield"],"code":{"url":"https://github.com/yaoxiao1999/weighted-speech-graphs","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xiao26b_interspeech","category":"health-clinical","institutions":["University of Sheffield"],"code":"https://github.com/yaoxiao1999/weighted-speech-graphs","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2266","pdf":"https://www.isca-archive.org/interspeech_2026/xiao26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xiao26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xiao26b_interspeech/markdown.md"},{"id":"xiao26c_interspeech","title":"Evidence Subspace Projection: Measuring How Much Evidence Explains Deepfake Detection in Self-Supervised Speech Models","authors":["Yixuan Xiao","Cheng-Wei Lin","Xin Wang","Yassine El Kheir","Arnab Das","Tim Polzehl","Sebastian Möller","Ngoc Thang Vu"],"year":2026,"doi":"10.21437/Interspeech.2026-3210","isca_url":"https://www.isca-archive.org/interspeech_2026/xiao26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xiao26c_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["self-supervised"],"institutions":["University of Stuttgart","National Institute of Informatics","German Research Center for Artificial Intelligence","Technical University of Berlin"],"code":{"url":"https://github.com/XIAOYixuan/ESP","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xiao26c_interspeech","category":"deepfake-security","labels":["self-supervised"],"institutions":["University of Stuttgart","National Institute of Informatics","German Research Center for Artificial Intelligence","Technical University of Berlin"],"code":"https://github.com/XIAOYixuan/ESP","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3210","pdf":"https://www.isca-archive.org/interspeech_2026/xiao26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xiao26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xiao26c_interspeech/markdown.md"},{"id":"xie26_interspeech","title":"FakeSound2: A Benchmark for Explainable, Traceable, and Generalizable Deepfake Sound Detection","authors":["Zeyu Xie","Yaoyun Zhang","Xuenan Xu","Yongkang Yin","Chenxing Li","Mengyue Wu","Yuexian Zou"],"year":2026,"doi":"10.21437/Interspeech.2026-1157","isca_url":"https://www.isca-archive.org/interspeech_2026/xie26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xie26_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["Guangdong Provincial Key Laboratory of Ultra High Definition Immersive Media Technology","Peking University","Tencent AI Lab","Shanghai Jiao Tong University"],"funding":["National Natural Science Foundation of China","Tencent AI Lab Rhino-Bird Program"],"code":{"url":"https://zeyuxie29.github.io/FakeSound2/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xie26_interspeech","category":"deepfake-security","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["Guangdong Provincial Key Laboratory of Ultra High Definition Immersive Media Technology","Peking University","Tencent AI Lab","Shanghai Jiao Tong University"],"code":"https://zeyuxie29.github.io/FakeSound2/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1157","pdf":"https://www.isca-archive.org/interspeech_2026/xie26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xie26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xie26_interspeech/markdown.md"},{"id":"xie26b_interspeech","title":"FlashTTS: Fast Streaming TTS with MTP Acceleration and X-pred Mean Flow Distillation","authors":["Hanke Xie","Xiaming Ren","Dake Guo","Ruonan You","Wenhao Li","Jingbin Hu","Guobin Ma","Huakang Chen","Kejie Xu","Rui Huang","Weiguo Tan","Xianrong Wang","Lei Xie"],"year":2026,"doi":"10.21437/Interspeech.2026-1692","isca_url":"https://www.isca-archive.org/interspeech_2026/xie26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xie26b_interspeech.pdf","session":"Streaming Speech Synthesis","topics":["tts","speech-llm","self-supervised"],"category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Northwestern Polytechnical University","Huawei Technologies"],"code":{"url":"https://github.com/ASLP-lab/FlashTTS","stars":76,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xie26b_interspeech","category":"tts","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["Northwestern Polytechnical University","Huawei Technologies"],"code":"https://github.com/ASLP-lab/FlashTTS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1692","pdf":"https://www.isca-archive.org/interspeech_2026/xie26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xie26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xie26b_interspeech/markdown.md"},{"id":"xie26c_interspeech","title":"VoiceTTA: Enhancing Zero-Shot Text-to-Speech via Reinforcement Learning-Based Test-Time Adaptation","authors":["Tianxin Xie","Chenxing Li","Dong Yu","Li Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-1757","isca_url":"https://www.isca-archive.org/interspeech_2026/xie26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xie26c_interspeech.pdf","session":"Scaling and Zero-Shot Speech Synthesis","topics":["tts","self-supervised","on-device"],"category":"tts","labels":["generative-model"],"institutions":["Hong Kong University of Science and Technology","Tencent"],"funding":["National Natural Science Foundation of China","Guangdong Basic and Applied Basic Research Foundation","Tencent AI Lab Rhino-Bird Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xie26c_interspeech","category":"tts","labels":["generative-model"],"institutions":["Hong Kong University of Science and Technology","Tencent"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1757","pdf":"https://www.isca-archive.org/interspeech_2026/xie26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xie26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xie26c_interspeech/markdown.md"},{"id":"xie26d_interspeech","title":"Acoustic Differences Between Citation and Sandhi Tones Across Three Generations in Xiamen Southern Min","authors":["Huangyang Xie","Xiuwei Zeng","Weijun Zhang","Peggy Pik Ki Mok"],"year":2026,"doi":"10.21437/Interspeech.2026-2138","isca_url":"https://www.isca-archive.org/interspeech_2026/xie26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xie26d_interspeech.pdf","session":"Tones","topics":["phonetics","prosody","paralinguistics"],"category":"phonetics-linguistics","institutions":["Chinese University of Hong Kong","Hong Kong Metropolitan University"],"funding":["Research Grants Council of Hong Kong"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xie26d_interspeech","category":"phonetics-linguistics","institutions":["Chinese University of Hong Kong","Hong Kong Metropolitan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2138","pdf":"https://www.isca-archive.org/interspeech_2026/xie26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xie26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xie26d_interspeech/markdown.md"},{"id":"xinyuan26_interspeech","title":"Universal Speech Content Factorization","authors":["Henry Li Xinyuan","Zexin Cai","Lin Zhang","Leibny Paola Garcia-Perera","Berrak Sisman","Sanjeev Khudanpur","Nicholas Andrews","Matthew Wiesner"],"year":2026,"doi":"10.21437/Interspeech.2026-198","isca_url":"https://www.isca-archive.org/interspeech_2026/xinyuan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xinyuan26_interspeech.pdf","session":"Voice Conversion","topics":["voice-conversion","tts","self-supervised"],"category":"tts","labels":["low-resource"],"institutions":["Johns Hopkins University"],"funding":["Office of the Director of National Intelligence","Intelligence Advanced Research Projects Activity","ARTS Program"],"code":{"url":"https://github.com/HSTEHSTEHSTE/uscf","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xinyuan26_interspeech","category":"tts","labels":["low-resource"],"institutions":["Johns Hopkins University"],"code":"https://github.com/HSTEHSTEHSTE/uscf","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-198","pdf":"https://www.isca-archive.org/interspeech_2026/xinyuan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xinyuan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xinyuan26_interspeech/markdown.md"},{"id":"xu26_interspeech","title":"A Sparsity-Aware Robust Nonlinear Active Noise Control for Impulsive Noise Environments","authors":["Hengwei Xu","Hongqing Liu","Liming Shi","Lu Gan"],"year":2026,"doi":"10.21437/Interspeech.2026-122","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26_interspeech.pdf","session":"Active Noise and Echo Control, Sound Zones and Packet-Loss Concealment","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","labels":["streaming-real-time","robustness-noise"],"institutions":["Chongqing University of Posts and Telecommunications","Brunel University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26_interspeech","category":"enhancement-separation","labels":["streaming-real-time","robustness-noise"],"institutions":["Chongqing University of Posts and Telecommunications","Brunel University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-122","pdf":"https://www.isca-archive.org/interspeech_2026/xu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26_interspeech/markdown.md"},{"id":"xu26b_interspeech","title":"Enhancing BEST-RQ Pseudo-Label Quality Through Online Refinement for Automatic Speech Recognition","authors":["Jingjing Xu","Zijian Yang","Mohammad Zeineldeen","Eugen Beck","Ralf Schlüter","Hermann Ney"],"year":2026,"doi":"10.21437/Interspeech.2026-650","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26b_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["asr","self-supervised"],"category":"asr","labels":["self-supervised"],"institutions":["RWTH Aachen University","AppTek"],"funding":["Federal Ministry for the Environment, Nature Conservation, Nuclear Safety and Consumer Protection","RESCALE"],"code":{"url":"https://github.com/rwth-i6/returnn-experiments/tree/master/2026enhance-bestrq","stars":163,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26b_interspeech","category":"asr","labels":["self-supervised"],"institutions":["RWTH Aachen University","AppTek"],"code":"https://github.com/rwth-i6/returnn-experiments/tree/master/2026enhance-bestrq","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-650","pdf":"https://www.isca-archive.org/interspeech_2026/xu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26b_interspeech/markdown.md"},{"id":"xu26c_interspeech","title":"HRIR-Former: Grid-Free Time-Domain Reconstruction of Head-Related Impulse Responses with a Spatially Encoded Transformer","authors":["Shaoheng Xu","Chunyi Sun","Jihui Zhang","Amy Bastine","Prasanga N. Samarasinghe","Thushara D. Abhayapala","Hongdong Li"],"year":2026,"doi":"10.21437/Interspeech.2026-702","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26c_interspeech.pdf","session":"Spatial Audio 1","topics":["spatial-audio","binaural-rendering","self-supervised"],"category":"applications-other","institutions":["Australian National University","University of Queensland"],"funding":["ANU PhD Scholarship","ANU HDR Fee Merit Scholarship"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26c_interspeech","category":"applications-other","institutions":["Australian National University","University of Queensland"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-702","pdf":"https://www.isca-archive.org/interspeech_2026/xu26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26c_interspeech/markdown.md"},{"id":"xu26d_interspeech","title":"Speech Enhancement Based on Drifting Models","authors":["Liang Xu","Diego Caviedes-Nozal","W. Bastiaan Kleijn","Longfei Felix Yan","Rasmus Kongsgaard Olsson"],"year":2026,"doi":"10.21437/Interspeech.2026-833","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26d_interspeech.pdf","session":"Neural Speech Enhancement: Survey, Diffusion and Flow Matching","topics":["speech-enhancement","self-supervised","generative-model"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Victoria University of Wellington","Lincoln University","GN Advanced Science"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26d_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Victoria University of Wellington","Lincoln University","GN Advanced Science"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-833","pdf":"https://www.isca-archive.org/interspeech_2026/xu26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26d_interspeech/markdown.md"},{"id":"xu26e_interspeech","title":"SCNet: Enhancing GAN-based Speech Generation with Subband Condition Network and Magnitude-aware Phase Loss","authors":["Nan Xu","Mingxue Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-843","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26e_interspeech.pdf","session":"Speech Enhancement and Restoration","topics":["tts","voice-conversion","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["Tencent","University of Electronic Science and Technology of China"],"code":{"url":"https://github.com/vspeech/SCNet","stars":15,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26e_interspeech","category":"tts","labels":["generative-model"],"institutions":["Tencent","University of Electronic Science and Technology of China"],"code":"https://github.com/vspeech/SCNet","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-843","pdf":"https://www.isca-archive.org/interspeech_2026/xu26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26e_interspeech/markdown.md"},{"id":"xu26f_interspeech","title":"Human-like cross-language generalisation in deep neural speaker embeddings and its acoustic foundations","authors":["Tianze Xu","Xiyang Li","Xiaoming Jiang","Volker Dellwo"],"year":2026,"doi":"10.21437/Interspeech.2026-858","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26f_interspeech.pdf","session":"Multilingual and Cross-Lingual Paralinguistic Analysis and Processing","topics":["speaker-verification","multilingual","self-supervised"],"category":"speaker","labels":["multilingual"],"institutions":["University of Zurich","Shanghai International Studies University"],"funding":["Marie Skłodowska-Curie Actions Doctoral Networks","European Union","Swiss State Secretariat for Education, Research and Innovation"],"code":{"url":"https://github.com/xiyangg12/wespeaker","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26f_interspeech","category":"speaker","labels":["multilingual"],"institutions":["University of Zurich","Shanghai International Studies University"],"code":"https://github.com/xiyangg12/wespeaker","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-858","pdf":"https://www.isca-archive.org/interspeech_2026/xu26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26f_interspeech/markdown.md"},{"id":"xu26g_interspeech","title":"Whisper-Aware LLM: Self-Supervised Uncertainty Learning for Robust Whispered Speech Recognition","authors":["Gaopeng Xu","Zhenyu Wang","Zheng Xue","Yinfeng Xia","Haitao Yao"],"year":2026,"doi":"10.21437/Interspeech.2026-879","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26g_interspeech.pdf","session":"Robust ASR: Uncertainty and Confidence","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["Alibaba"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26g_interspeech","category":"asr","labels":["self-supervised","robustness-noise"],"institutions":["Alibaba"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-879","pdf":"https://www.isca-archive.org/interspeech_2026/xu26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26g_interspeech/markdown.md"},{"id":"xu26h_interspeech","title":"Towards Data-free and Training-free Compression for Speech Foundation Models Using Parameter Clustering","authors":["Haoning Xu","Zhaoqing Li","Huimeng Wang","Youjun Chen","Chengxi Deng","Mengzhe Geng","Xunying Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-1010","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26h_interspeech.pdf","session":"Efficient Inference for ASR and Speech LMs","topics":["asr","self-supervised","on-device"],"category":"asr","labels":["efficient-on-device","self-supervised"],"institutions":["Chinese University of Hong Kong","National Research Council Canada"],"funding":["Hong Kong RGC GRF"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26h_interspeech","category":"asr","labels":["efficient-on-device","self-supervised"],"institutions":["Chinese University of Hong Kong","National Research Council Canada"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1010","pdf":"https://www.isca-archive.org/interspeech_2026/xu26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26h_interspeech/markdown.md"},{"id":"xu26i_interspeech","title":"A barrier or a booster? Familiarity effects on Mandarin emotion prosody recognition using AI-powered voice cloning","authors":["Feng Xu","Gaoyuan Zhang","Shanshan Xue","Yixiang Chen","Hanrui Zhou","Xurong Xie","Hui Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1038","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26i_interspeech.pdf","session":"Phonetic Aspects of TTS and ASR Systems","topics":["speech-emotion-recognition","voice-conversion","paralinguistics"],"category":"paralinguistics-emotion","labels":["generative-model"],"institutions":["Chinese Academy of Sciences","Macquarie University"],"funding":["National Key R&D Program of China","NSFC","China Disabled Persons Federation","Youth Innovation Promotion Association CAS Grant","China Postdoctoral Science Foundation"],"code":{"url":"https://github.com/Plachtaa/seed-vc","stars":3893,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26i_interspeech","category":"paralinguistics-emotion","labels":["generative-model"],"institutions":["Chinese Academy of Sciences","Macquarie University"],"code":"https://github.com/Plachtaa/seed-vc","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1038","pdf":"https://www.isca-archive.org/interspeech_2026/xu26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26i_interspeech/markdown.md"},{"id":"xu26j_interspeech","title":"GLAD-CSpeech: A Dialectologically Comprehensive Benchmark for Genuine Chinese Dialect Speech","authors":["Ke Xu","Lihan Xu","Jiayi Lin","Bin Zhang","Yunfei Chu","Shuting Yuan","Ruiye Lv","Guangxuan Zheng","Qi Han","Jin Xu","Bing Zhao","Hu Wei","Yang Bai","Ziyi Cheng","Qibin Ran"],"year":2026,"doi":"10.21437/Interspeech.2026-1128","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26j_interspeech.pdf","session":"Datasets","topics":["asr","tts","speaker-identification"],"category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Alibaba Group","Nankai University","Fudan University"],"funding":["Alibaba Innovative Research Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26j_interspeech","category":"resources-evaluation","labels":["low-resource","dataset-or-benchmark-release"],"institutions":["Alibaba Group","Nankai University","Fudan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1128","pdf":"https://www.isca-archive.org/interspeech_2026/xu26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26j_interspeech/markdown.md"},{"id":"xu26k_interspeech","title":"From Reactive to Proactive: Assessing the Proactivity of Voice Agents via ProVoice-Bench","authors":["Ke Xu","Yuhao Wang","Yu Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1160","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26k_interspeech.pdf","session":"Spoken Dialogue Systems","topics":["speech-llm","evaluation","dataset"],"category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release"],"institutions":["Shanghai Jiao Tong University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26k_interspeech","category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release"],"institutions":["Shanghai Jiao Tong University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1160","pdf":"https://www.isca-archive.org/interspeech_2026/xu26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26k_interspeech/markdown.md"},{"id":"xu26l_interspeech","title":"Audio-Language Prompt Learning for Few-Shot Audio Classification","authors":["Qisheng Xu","Xiaoyi Tan","Wuyang Chen","Yutao Dou","Kele Xu"],"year":2026,"doi":"10.21437/Interspeech.2026-1173","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26l_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["self-supervised","few-shot","speech-llm"],"category":"audio-understanding","labels":["low-resource","self-supervised"],"institutions":["National University of Defense Technology","Hunan Normal University","Hunan University"],"funding":["National Science and Technology Major Project","National University of Defense Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26l_interspeech","category":"audio-understanding","labels":["low-resource","self-supervised"],"institutions":["National University of Defense Technology","Hunan Normal University","Hunan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1173","pdf":"https://www.isca-archive.org/interspeech_2026/xu26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26l_interspeech/markdown.md"},{"id":"xu26m_interspeech","title":"Learning speaker identities in dialogue: Conversational familiarisation modulates response bias and confidence in voice recognition","authors":["Tianze Xu","Volker Dellwo"],"year":2026,"doi":"10.21437/Interspeech.2026-1174","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26m_interspeech.pdf","session":"Speaker Identity, States, and Traits in Paralinguistics","topics":["speaker-verification","evaluation","paralinguistics"],"category":"speaker","institutions":["University of Zurich"],"funding":["Marie Skłodowska-Curie Actions Doctoral Networks","European Union","Swiss State Secretariat for Education, Research and Innovation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26m_interspeech","category":"speaker","institutions":["University of Zurich"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1174","pdf":"https://www.isca-archive.org/interspeech_2026/xu26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26m_interspeech/markdown.md"},{"id":"xu26n_interspeech","title":"Automatic Graphical Representations of Language for Dementia Detection","authors":["Lingfeng Xu","Si-Ioi Ng","Pranav S. Ambadi","Fan Lei","Kimberly D. Mueller","Julie Liss","Visar Berisha"],"year":2026,"doi":"10.21437/Interspeech.2026-1233","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26n_interspeech.pdf","session":"Clinically Useful Speech Representations 1","topics":["paralinguistics","speech-llm","health"],"category":"health-clinical","labels":["self-supervised"],"institutions":["Arizona State University","University of Wisconsin-Madison","University of Waterloo"],"funding":["NIH-NIA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26n_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["Arizona State University","University of Wisconsin-Madison","University of Waterloo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1233","pdf":"https://www.isca-archive.org/interspeech_2026/xu26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26n_interspeech/markdown.md"},{"id":"xu26o_interspeech","title":"BACON: Boundary-Aware Convolution for Streaming Conformer Models","authors":["Hainan Xu","Kunal Dhawan","Dongji Gao","Jagadeesh Balam"],"year":2026,"doi":"10.21437/Interspeech.2026-1455","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26o_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26o_interspeech.pdf","session":"New Training Methods for ASR","topics":["asr","speech-translation","self-supervised"],"category":"asr","labels":["streaming-real-time"],"institutions":["NVIDIA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26o_interspeech","category":"asr","labels":["streaming-real-time"],"institutions":["NVIDIA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1455","pdf":"https://www.isca-archive.org/interspeech_2026/xu26o_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26o_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26o_interspeech/markdown.md"},{"id":"xu26p_interspeech","title":"Room Impulse Response Completion Using Signal-Prediction Diffusion Models Conditioned on Simulated Early Reflections","authors":["Zeyu Xu","Andreas Brendel","Albert G. Prinn","Emanuël A. P. Habets"],"year":2026,"doi":"10.21437/Interspeech.2026-2025","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26p_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26p_interspeech.pdf","session":"Spatial Audio 2","topics":["self-supervised","dataset","speech-enhancement"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["International Audio Laboratories Erlangen","Fraunhofer Institute for Integrated Circuits","Friedrich-Alexander-Universitat Erlangen-Nurnberg"],"funding":["German Research Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26p_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["International Audio Laboratories Erlangen","Fraunhofer Institute for Integrated Circuits","Friedrich-Alexander-Universitat Erlangen-Nurnberg"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2025","pdf":"https://www.isca-archive.org/interspeech_2026/xu26p_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26p_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26p_interspeech/markdown.md"},{"id":"xu26q_interspeech","title":"Continual Generalized Category Discovery for Acoustic Signals via Instance-Adaptive Regularization and Dynamic Teacher Guidance","authors":["Qisheng Xu","Shanhao Han","Hui Geng","Yulu Fang","Yunsheng Xiong","Yutao Dou","Kele Xu"],"year":2026,"doi":"10.21437/Interspeech.2026-2614","isca_url":"https://www.isca-archive.org/interspeech_2026/xu26q_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xu26q_interspeech.pdf","session":"Acoustic Event Detection 3","topics":["self-supervised","speech-llm","paralinguistics"],"category":"audio-understanding","institutions":["National University of Defense Technology","Hunan University"],"funding":["National Science and Technology Major Project","National University of Defense Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xu26q_interspeech","category":"audio-understanding","institutions":["National University of Defense Technology","Hunan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2614","pdf":"https://www.isca-archive.org/interspeech_2026/xu26q_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26q_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xu26q_interspeech/markdown.md"},{"id":"xuan26_interspeech","title":"Disentangling Speaker Traits for Deepfake Source Verification via Chebyshev Polynomial and Riemannian Metric Learning","authors":["Xi Xuan","Wenxin Zhang","Zhiyu Li","Jennifer Williams","Ville Hautamäki","Tomi H. Kinnunen"],"year":2026,"doi":"10.21437/Interspeech.2026-36","isca_url":"https://www.isca-archive.org/interspeech_2026/xuan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xuan26_interspeech.pdf","session":"Spoofing and Deepfake Detection 3","topics":["audio-deepfake","speaker-verification","self-supervised"],"category":"deepfake-security","institutions":["University of Eastern Finland","University of Illinois Urbana-Champaign","University of Southampton","University of Chinese Academy of Sciences","University of Science and Technology of China"],"funding":["Finnish AI-DOC project","Research Council of Finland"],"code":{"url":"https://github.com/xxuan-acoustics/RiemannSD-Net","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xuan26_interspeech","category":"deepfake-security","institutions":["University of Eastern Finland","University of Illinois Urbana-Champaign","University of Southampton","University of Chinese Academy of Sciences","University of Science and Technology of China"],"code":"https://github.com/xxuan-acoustics/RiemannSD-Net","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-36","pdf":"https://www.isca-archive.org/interspeech_2026/xuan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xuan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xuan26_interspeech/markdown.md"},{"id":"xue26_interspeech","title":"Imperceptible Voiceprint Protection via Human-Machine Perception Discrepancy Feature Disentanglement","authors":["Chenlong Xue","Meng Sun","Qiang Zhang","Xiongwei Zhang","Kunyuan Li","Yuan Liao","Xiaoyi Ge","Kui Yao"],"year":2026,"doi":"10.21437/Interspeech.2026-105","isca_url":"https://www.isca-archive.org/interspeech_2026/xue26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xue26_interspeech.pdf","session":"Speaker Privacy Preservation and Anonymization","topics":["speaker-verification","voice-conversion","self-supervised"],"category":"deepfake-security","institutions":["Army Engineering University of PLA","Chinese University of Hong Kong","Information Support Force Engineering University"],"funding":["National Natural Science Foundation of China","Natural Science Foundation of Jiangsu Province","China Postdoctoral Science Foundation"],"code":{"url":"https://cero529.github.io/voiceprint-protection-demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xue26_interspeech","category":"deepfake-security","institutions":["Army Engineering University of PLA","Chinese University of Hong Kong","Information Support Force Engineering University"],"code":"https://cero529.github.io/voiceprint-protection-demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-105","pdf":"https://www.isca-archive.org/interspeech_2026/xue26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xue26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xue26_interspeech/markdown.md"},{"id":"xue26b_interspeech","title":"Edge–Cloud Collaborative Speech Emotion Captioning via Token-Level Speculative Decoding in Audio-Language Models","authors":["Xiangyuan Xue","Jiajun Lu","Yan Gao","Gongping Huang","Ting Dang","Hong Jia"],"year":2026,"doi":"10.21437/Interspeech.2026-901","isca_url":"https://www.isca-archive.org/interspeech_2026/xue26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xue26b_interspeech.pdf","session":"Speech Emotion Recognition and Representation 3","topics":["speech-llm","paralinguistics","emotion-recognition"],"category":"paralinguistics-emotion","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["University of Auckland","University of Melbourne","University of Cambridge","Wuhan University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xue26b_interspeech","category":"paralinguistics-emotion","labels":["efficient-on-device","streaming-real-time","generative-model"],"institutions":["University of Auckland","University of Melbourne","University of Cambridge","Wuhan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-901","pdf":"https://www.isca-archive.org/interspeech_2026/xue26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xue26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xue26b_interspeech/markdown.md"},{"id":"xue26c_interspeech","title":"NVV-SuperBench: Beyond Words, Beyond Quality—Benchmarking Nonverbal Vocalizations in Speech Generation","authors":["Liumeng Xue","Weizhen Bian","Jiahao Pan","Wenxuan Wu","Yilin Ren","Boyi Kang","Jingbin Hu","Ziyang Ma","Shuai Wang","Xinyuan Qian","Hung-yi Lee","Yike Guo"],"year":2026,"doi":"10.21437/Interspeech.2026-2513","isca_url":"https://www.isca-archive.org/interspeech_2026/xue26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xue26c_interspeech.pdf","session":"Benchmarking Foundation Models","topics":["tts","paralinguistics","evaluation"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Nanjing University","Hong Kong University of Science and Technology","Chinese University of Hong Kong","University of Science and Technology Beijing","Northwestern Polytechnical University","Shanghai Jiao Tong University","National Taiwan University"],"code":{"url":"https://lmxue.github.io/NVV-SuperBench/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xue26c_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Nanjing University","Hong Kong University of Science and Technology","Chinese University of Hong Kong","University of Science and Technology Beijing","Northwestern Polytechnical University","Shanghai Jiao Tong University","National Taiwan University"],"code":"https://lmxue.github.io/NVV-SuperBench/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2513","pdf":"https://www.isca-archive.org/interspeech_2026/xue26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xue26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xue26c_interspeech/markdown.md"},{"id":"xue26d_interspeech","title":"Preserving Acoustic Cues for Video Reasoning: An Efficient Uniqueness-Driven Token Compression Framework","authors":["Haiwei Xue","Zichao Nie","Zhiyong Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-3512","isca_url":"https://www.isca-archive.org/interspeech_2026/xue26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/xue26d_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["speech-llm","multilingual","self-supervised"],"category":"audio-understanding","labels":["efficient-on-device"],"institutions":["Tsinghua University","Hong Kong University of Science and Technology","Chinese University of Hong Kong"],"funding":["National Natural Science Foundation of China","Shenzhen Science and Technology Program"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"xue26d_interspeech","category":"audio-understanding","labels":["efficient-on-device"],"institutions":["Tsinghua University","Hong Kong University of Science and Technology","Chinese University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3512","pdf":"https://www.isca-archive.org/interspeech_2026/xue26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/xue26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/xue26d_interspeech/markdown.md"},{"id":"yadav26_interspeech","title":"ARTIST: Universal Articulatory Space Modeling for Multilingual Indic-to-English Speech-to-Speech Translation","authors":["Khushal Yadav","Vinayak Abrol"],"year":2026,"doi":"10.21437/Interspeech.2026-2384","isca_url":"https://www.isca-archive.org/interspeech_2026/yadav26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yadav26_interspeech.pdf","session":"Translation","topics":["speech-translation","low-resource","multilingual"],"category":"translation","labels":["low-resource","multilingual","efficient-on-device","generative-model"],"institutions":["Indraprastha Institute of Information Technology Delhi"],"funding":["Nebius Research Grant","Infosys Foundation"],"code":{"url":"https://sites.google.com/view/artist-demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yadav26_interspeech","category":"translation","labels":["low-resource","multilingual","efficient-on-device","generative-model"],"institutions":["Indraprastha Institute of Information Technology Delhi"],"code":"https://sites.google.com/view/artist-demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2384","pdf":"https://www.isca-archive.org/interspeech_2026/yadav26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yadav26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yadav26_interspeech/markdown.md"},{"id":"yadav26b_interspeech","title":"I'll Keep an Ear Out: Teaching AudioLLMs Proactive Audio Assistance","authors":["Amit Kumar Singh Yadav","Ritvik Shrivastava","Xuan Zhang","Seungwhan Moon","Shashank Jain","Pinar Donmez","Babak Damavandi"],"year":2026,"doi":"10.21437/Interspeech.2026-2807","isca_url":"https://www.isca-archive.org/interspeech_2026/yadav26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yadav26b_interspeech.pdf","session":"Acoustic Event Detection 3","topics":["audio-llm","speech-llm","streaming"],"category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["Meta"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yadav26b_interspeech","category":"speech-llm-dialogue","labels":["streaming-real-time"],"institutions":["Meta"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2807","pdf":"https://www.isca-archive.org/interspeech_2026/yadav26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yadav26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yadav26b_interspeech/markdown.md"},{"id":"yadla26_interspeech","title":"Extreme Few-Shot Phoneme Discovery for Indigenous Australian and Pacific Languages via Typological Transfer Learning","authors":["Prasanth Yadla"],"year":2026,"doi":"10.21437/Interspeech.2026-284","isca_url":"https://www.isca-archive.org/interspeech_2026/yadla26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yadla26_interspeech.pdf","session":"Speech Production and Perception 2","topics":["low-resource","multilingual","self-supervised"],"category":"phonetics-linguistics","labels":["low-resource","multilingual","self-supervised","generative-model"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yadla26_interspeech","category":"phonetics-linguistics","labels":["low-resource","multilingual","self-supervised","generative-model"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-284","pdf":"https://www.isca-archive.org/interspeech_2026/yadla26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yadla26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yadla26_interspeech/markdown.md"},{"id":"yaish26_interspeech","title":"Active Constructive Interference for Speech","authors":["Ofir Yaish","Yehuda Mishaly","Eliya Nachmani"],"year":2026,"doi":"10.21437/Interspeech.2026-222","isca_url":"https://www.isca-archive.org/interspeech_2026/yaish26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yaish26_interspeech.pdf","session":"Active Noise and Echo Control, Sound Zones and Packet-Loss Concealment","topics":["speech-enhancement","self-supervised"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Ben-Gurion University of the Negev","Tel Aviv University"],"code":{"url":"https://github.com/ofiryaish/ASE-TM","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yaish26_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Ben-Gurion University of the Negev","Tel Aviv University"],"code":"https://github.com/ofiryaish/ASE-TM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-222","pdf":"https://www.isca-archive.org/interspeech_2026/yaish26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yaish26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yaish26_interspeech/markdown.md"},{"id":"yakovlev26_interspeech","title":"ReDimNet2: Scaling Speaker Verification via Time-Pooled Dimension Reshaping","authors":["Ivan Yakovlev","Anton Okhotnikov"],"year":2026,"doi":"10.21437/Interspeech.2026-1447","isca_url":"https://www.isca-archive.org/interspeech_2026/yakovlev26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yakovlev26_interspeech.pdf","session":"Speaker Verification: Architectures, Losses, and LLMs","topics":["speaker-verification","self-supervised","audio-deepfake"],"category":"speaker","labels":["efficient-on-device"],"institutions":["Palabra AI"],"code":{"url":"https://github.com/PalabraAI/redimnet2","stars":95,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yakovlev26_interspeech","category":"speaker","labels":["efficient-on-device"],"institutions":["Palabra AI"],"code":"https://github.com/PalabraAI/redimnet2","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1447","pdf":"https://www.isca-archive.org/interspeech_2026/yakovlev26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yakovlev26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yakovlev26_interspeech/markdown.md"},{"id":"yalegama26_interspeech","title":"Deep Learning Based Relative Transfer Matrix Estimation for Multiple Sources and Multiple Microphones","authors":["Oshan A. B. Yalegama","Wageesha N. Manamperi"],"year":2026,"doi":"10.21437/Interspeech.2026-2524","isca_url":"https://www.isca-archive.org/interspeech_2026/yalegama26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yalegama26_interspeech.pdf","session":"Multi-Channel Processing and Specialized Acquisition (UAV, Radar, Hearables)","topics":["speech-enhancement","source-separation","self-supervised"],"category":"enhancement-separation","institutions":["University of Moratuwa","Australian National University"],"funding":["Accelerating Higher Education Expansion and Development","World Bank"],"code":{"url":"https://github.com/oshanyalegama/Denoised_ReTM_DL","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yalegama26_interspeech","category":"enhancement-separation","institutions":["University of Moratuwa","Australian National University"],"code":"https://github.com/oshanyalegama/Denoised_ReTM_DL","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2524","pdf":"https://www.isca-archive.org/interspeech_2026/yalegama26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yalegama26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yalegama26_interspeech/markdown.md"},{"id":"yamagishi26_interspeech","title":"Countermeasures Against Misuse of Speech Generative AI","authors":["Junichi Yamagishi"],"year":2026,"isca_url":"https://www.isca-archive.org/interspeech_2026/yamagishi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yamagishi26_interspeech.pdf","session":"Keynote4 - Junichi Yamagishi: Countermeasures Against Misuse of Speech Generative AI","topics":["audio-deepfake"],"category":"deepfake-security","institutions":["National Institute of Informatics"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yamagishi26_interspeech","category":"deepfake-security","institutions":["National Institute of Informatics"],"updated":"2026-09-28","confidence":"abstract-only","source":"https://www.isca-archive.org/interspeech_2026/yamagishi26_interspeech.html"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yamagishi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yamagishi26_interspeech/markdown.md"},{"id":"yamauchi26_interspeech","title":"QC-GAN: A Parameter-Efficient Quaternion Conformer GAN for High-Fidelity Speech Enhancement","authors":["Shogo Yamauchi","Hideaki Tamori","Makoto Sakai","Yosuke Yamano","Tohru Nitta"],"year":2026,"doi":"10.21437/Interspeech.2026-889","isca_url":"https://www.isca-archive.org/interspeech_2026/yamauchi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yamauchi26_interspeech.pdf","session":"Speech Enhancement and Restoration","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","generative-model","robustness-noise"],"institutions":["Asahi Shimbun Company","Tokyo Woman's Christian University"],"code":{"url":"https://github.com/asahi-research/QC-GAN","stars":3,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yamauchi26_interspeech","category":"enhancement-separation","labels":["efficient-on-device","generative-model","robustness-noise"],"institutions":["Asahi Shimbun Company","Tokyo Woman's Christian University"],"code":"https://github.com/asahi-research/QC-GAN","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-889","pdf":"https://www.isca-archive.org/interspeech_2026/yamauchi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yamauchi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yamauchi26_interspeech/markdown.md"},{"id":"yan26_interspeech","title":"UniSE: A Unified Framework for Decoder-Only Autoregressive LM-Based Speech Enhancement","authors":["Haoyin Yan","Chengwei Liu","Shaofei Xue","Xiaotao Liang","Yinghao Liu","Yuxiang Kong","Zheng Xue"],"year":2026,"doi":"10.21437/Interspeech.2026-192","isca_url":"https://www.isca-archive.org/interspeech_2026/yan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yan26_interspeech.pdf","session":"Language-Model and Codec-Token Speech Enhancement","topics":["speech-enhancement","speech-separation","speaker-verification"],"category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Alibaba"],"code":{"url":"https://github.com/alibaba/unified-audio","stars":543,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yan26_interspeech","category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Alibaba"],"code":"https://github.com/alibaba/unified-audio","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-192","pdf":"https://www.isca-archive.org/interspeech_2026/yan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yan26_interspeech/markdown.md"},{"id":"yan26b_interspeech","title":"Consistent and Coherent Audio-Visual Understanding with Cross-Frame Patch Differential Attention and Cross-Modal Temporal Alignment","authors":["Lecheng Yan","Chenyang Lyu","Haoqin Sun","Wenxi Li","Mohamed Fazli Imam","Jiahui Geng","Qing Li","Shaochen Jiang"],"year":2026,"doi":"10.21437/Interspeech.2026-619","isca_url":"https://www.isca-archive.org/interspeech_2026/yan26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yan26b_interspeech.pdf","session":"Audio-Visual Grounding, Synchronization & Video Understanding","topics":["speech-llm","multilingual","self-supervised"],"category":"speech-llm-dialogue","institutions":["University of Science and Technology of China","Alibaba Group","East China Normal University","Mohamed bin Zayed University of Artificial Intelligence","Linkoping University","University of Groningen","Xinjiang University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yan26b_interspeech","category":"speech-llm-dialogue","institutions":["University of Science and Technology of China","Alibaba Group","East China Normal University","Mohamed bin Zayed University of Artificial Intelligence","Linkoping University","University of Groningen","Xinjiang University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-619","pdf":"https://www.isca-archive.org/interspeech_2026/yan26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yan26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yan26b_interspeech/markdown.md"},{"id":"yan26c_interspeech","title":"Probing and Mitigating Hallucinations in Speech-augmented Language Models for Automatic Speech Recognition via Small Language Models","authors":["Bi-Cheng Yan","Jhih-Rong Guo","Fu-An Chao","Berlin Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-1278","isca_url":"https://www.isca-archive.org/interspeech_2026/yan26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yan26c_interspeech.pdf","session":"Robust ASR: Uncertainty and Confidence","topics":["asr","speech-llm","self-supervised"],"category":"asr","institutions":["National Taiwan Normal University"],"funding":["Realtek Semiconductor Corporation"],"code":{"url":"https://github.com/bicheng1225/AudioSLM","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yan26c_interspeech","category":"asr","institutions":["National Taiwan Normal University"],"code":"https://github.com/bicheng1225/AudioSLM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1278","pdf":"https://www.isca-archive.org/interspeech_2026/yan26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yan26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yan26c_interspeech/markdown.md"},{"id":"yang26_interspeech","title":"Multi-Channel Differential ASR for Robust Wearer Speech Recognition on Smart Glasses","authors":["Yufeng Yang","Yiteng Huang","Yong Xu","Li Wan","Suwon Shon","Yang Liu","Yifeng Fan","Zhaojun Yang","Olivier Siohan","Yue Liu","Ming Sun","Florian Metze"],"year":2026,"doi":"10.21437/Interspeech.2026-127","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26_interspeech.pdf","session":"Robust and Real-World ASR Systems","topics":["asr","speech-enhancement","self-supervised"],"category":"asr","labels":["robustness-noise"],"institutions":["Ohio State University","Meta"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26_interspeech","category":"asr","labels":["robustness-noise"],"institutions":["Ohio State University","Meta"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-127","pdf":"https://www.isca-archive.org/interspeech_2026/yang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26_interspeech/markdown.md"},{"id":"yang26b_interspeech","title":"Enroll-on-Wakeup: A First Comparative Study of Target Speech Extraction for Seamless Interaction in Real Noisy Human-Machine Dialogue Scenarios","authors":["Yiming Yang","Guangyong Wang","Haixin Guan","Yanhua Long"],"year":2026,"doi":"10.21437/Interspeech.2026-259","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26b_interspeech.pdf","session":"Target Speaker Extraction, Speech Separation and Audio Understanding","topics":["speech-enhancement","speaker-verification","asr"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Shanghai Normal University","Unisound AI Technology Co., Ltd"],"code":{"url":"https://github.com/Yym-line/EoW-TSE","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26b_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Shanghai Normal University","Unisound AI Technology Co., Ltd"],"code":"https://github.com/Yym-line/EoW-TSE","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-259","pdf":"https://www.isca-archive.org/interspeech_2026/yang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26b_interspeech/markdown.md"},{"id":"yang26c_interspeech","title":"MUGEN: Evaluating and Improving Multi-audio Understanding of Large Audio-Language Models","authors":["Chih-Kai Yang","Yun-Shao Tsai","Yu-Kai Guo","Ping-Le Tsai","Yen-Ting Piao","Hung-Wei Chen","Ting-Lin Hsiao","Yun-Man Hsu","Ke-Han Lu","Hung-yi Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-530","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26c_interspeech.pdf","session":"Audio Language Models: Reasoning, Reliability, and Multimodal Understanding","topics":["speech-llm","evaluation","self-supervised"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["National Taiwan University"],"funding":["Ministry of Education","NTU Artificial Intelligence Center of Research Excellence","Taiwan Centers of Excellence in Artificial Intelligence"],"code":{"url":"https://github.com/danielqwer/MUGEN","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26c_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["National Taiwan University"],"code":"https://github.com/danielqwer/MUGEN","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-530","pdf":"https://www.isca-archive.org/interspeech_2026/yang26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26c_interspeech/markdown.md"},{"id":"yang26d_interspeech","title":"K-DIALECT : Korean Dialect-Aware Face-Based Speech Synthesis","authors":["Seongyeon Yang","Juyeob Lee","Eunil Park"],"year":2026,"doi":"10.21437/Interspeech.2026-616","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26d_interspeech.pdf","session":"Articulatory, EMG, and Visual Speech Generation","topics":["tts","multilingual","self-supervised"],"category":"tts","labels":["low-resource","generative-model"],"institutions":["Sungkyunkwan University","Jaume I University"],"funding":["MSIT, Korea","Global Research Support Program","ICAN","ITRC","IITP"],"code":{"url":"https://dxlabskku.github.io/K-Dialect/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26d_interspeech","category":"tts","labels":["low-resource","generative-model"],"institutions":["Sungkyunkwan University","Jaume I University"],"code":"https://dxlabskku.github.io/K-Dialect/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-616","pdf":"https://www.isca-archive.org/interspeech_2026/yang26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26d_interspeech/markdown.md"},{"id":"yang26e_interspeech","title":"Schrödinger Bridge Mamba for One-Step Speech Enhancement","authors":["Jing Yang","Sirui Wang","Chao Wu","Lei Guo","Fan Fan"],"year":2026,"doi":"10.21437/Interspeech.2026-682","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26e_interspeech.pdf","session":"Neural Speech Enhancement: Survey, Diffusion and Flow Matching","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","generative-model","robustness-noise"],"institutions":["Huawei"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26e_interspeech","category":"enhancement-separation","labels":["efficient-on-device","generative-model","robustness-noise"],"institutions":["Huawei"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-682","pdf":"https://www.isca-archive.org/interspeech_2026/yang26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26e_interspeech/markdown.md"},{"id":"yang26f_interspeech","title":"Geometry-Informed Distributed Acoustic Scene Understanding","authors":["Yiyuan Yang","Shitong Xu","Niki Trigoni","Andrew Markham"],"year":2026,"doi":"10.21437/Interspeech.2026-821","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26f_interspeech.pdf","session":"Spatial Audio 3","topics":["spoken-language-understanding","speech-llm","dataset"],"category":"audio-understanding","institutions":["University of Oxford"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26f_interspeech","category":"audio-understanding","institutions":["University of Oxford"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-821","pdf":"https://www.isca-archive.org/interspeech_2026/yang26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26f_interspeech/markdown.md"},{"id":"yang26g_interspeech","title":"Pitch-Injected Residual Adapter for Tonal Language in Neural Audio Codec","authors":["Jie-Shiang Yang","Ya-Tse Wu","Chi-Chun Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-1224","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26g_interspeech.pdf","session":"Neural Speech Codecs: Low-Bitrate and Disentangled Coding","topics":["tts","speech-llm","low-resource"],"category":"speech-coding","labels":["multilingual"],"institutions":["National Tsing Hua University"],"code":{"url":"https://github.com/Jie-shiang/Pitch-Injected-Residual-Adapter","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26g_interspeech","category":"speech-coding","labels":["multilingual"],"institutions":["National Tsing Hua University"],"code":"https://github.com/Jie-shiang/Pitch-Injected-Residual-Adapter","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1224","pdf":"https://www.isca-archive.org/interspeech_2026/yang26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26g_interspeech/markdown.md"},{"id":"yang26h_interspeech","title":"Robust Streaming ASR with Decoupled Separation and Recognition","authors":["Yufeng Yang","Cheng Yu","Vahid A. Kalkhorani","DeLiang Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1503","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26h_interspeech.pdf","session":"ASR Under Real-World Constraints: Streaming, Adaptation, and Efficiency","topics":["asr","speech-enhancement","self-supervised"],"category":"asr","labels":["streaming-real-time","robustness-noise"],"institutions":["Ohio State University","Chinese University of Hong Kong"],"funding":["National Science Foundation","Ohio Supercomputer Center"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26h_interspeech","category":"asr","labels":["streaming-real-time","robustness-noise"],"institutions":["Ohio State University","Chinese University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1503","pdf":"https://www.isca-archive.org/interspeech_2026/yang26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26h_interspeech/markdown.md"},{"id":"yang26i_interspeech","title":"A Dynamic Knowledge Distillation Framework for Mitigating Spatial Ambiguity in Lightweight Dual-Channel Speech Enhancement","authors":["Yifei Yang","Zheng Wang","Yu Sun","Wentao Hua","Xiaobin Rong","Kai Chen","Jing Lu"],"year":2026,"doi":"10.21437/Interspeech.2026-1524","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26i_interspeech.pdf","session":"Multi-Channel, Beamforming and Spatial Speech Enhancement","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["Nanjing University","Horizon Robotics","Samsung Electronics"],"funding":["National Natural Science Foundation of China","AI & AI for Science Project of Nanjing University"],"code":{"url":"https://muefy.github.io/D-SKD-Audio-Demo","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26i_interspeech","category":"enhancement-separation","labels":["efficient-on-device"],"institutions":["Nanjing University","Horizon Robotics","Samsung Electronics"],"code":"https://muefy.github.io/D-SKD-Audio-Demo","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1524","pdf":"https://www.isca-archive.org/interspeech_2026/yang26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26i_interspeech/markdown.md"},{"id":"yang26j_interspeech","title":"A Fusion-Aware Two-Stage Framework for Mispronunciation Detection and Diagnosis in Low-Resource Modern Standard Arabic","authors":["Jing Yang","Shuqing Zhang","Yongyi Deng","Pan Li","Ting Dang","Gongping Huang","Jingdong Chen","Jacob Benesty"],"year":2026,"doi":"10.21437/Interspeech.2026-1553","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26j_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["asr","low-resource","self-supervised"],"category":"asr","labels":["low-resource","self-supervised"],"institutions":["Wuhan University","University of Melbourne","Northwestern Polytechnical University","University of Quebec"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://hf.co/spaces/IqraEval","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26j_interspeech","category":"asr","labels":["low-resource","self-supervised"],"institutions":["Wuhan University","University of Melbourne","Northwestern Polytechnical University","University of Quebec"],"code":"https://hf.co/spaces/IqraEval","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1553","pdf":"https://www.isca-archive.org/interspeech_2026/yang26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26j_interspeech/markdown.md"},{"id":"yang26k_interspeech","title":"HARP: Harmonic-Aware Residual Partitioning for Neural Audio Codecs","authors":["Qiaoyu Yang","Lixing He","Binyue Deng","Weifeng Zhao"],"year":2026,"doi":"10.21437/Interspeech.2026-1759","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26k_interspeech.pdf","session":"Audio Coding and Signal Analysis","topics":["speech-coding","self-supervised","tts"],"category":"speech-coding","institutions":["Georgia Institute of Technology","Chinese University of Hong Kong","Tencent Music Entertainment"],"code":{"url":"https://github.com/QiaoyuYang/HARP","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26k_interspeech","category":"speech-coding","institutions":["Georgia Institute of Technology","Chinese University of Hong Kong","Tencent Music Entertainment"],"code":"https://github.com/QiaoyuYang/HARP","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1759","pdf":"https://www.isca-archive.org/interspeech_2026/yang26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26k_interspeech/markdown.md"},{"id":"yang26l_interspeech","title":"CraftTTS: Fine-Grained Prosody Control for Text-to-Speech","authors":["Wenbing Yang","Qihang Lu","Bingsong Bai","Zihan Sun","Yueran Hou","Peilei Jia","Yingming Gao","Ya Li","Jun Gao"],"year":2026,"doi":"10.21437/Interspeech.2026-2018","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26l_interspeech.pdf","session":"Long-Form Speech Synthesis","topics":["tts","speech-llm","prosody"],"category":"tts","labels":["generative-model"],"institutions":["Beijing University of Posts and Telecommunications","Hello Group Inc"],"funding":["National Key R&D Program of China","National Natural Science Foundation of China","National Language Commission","National Social Science Fund of China"],"code":{"url":"https://ywbn.github.io/CraftTTS/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26l_interspeech","category":"tts","labels":["generative-model"],"institutions":["Beijing University of Posts and Telecommunications","Hello Group Inc"],"code":"https://ywbn.github.io/CraftTTS/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2018","pdf":"https://www.isca-archive.org/interspeech_2026/yang26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26l_interspeech/markdown.md"},{"id":"yang26m_interspeech","title":"Multi-View Based Audio Visual Target Speaker Extraction","authors":["Peijun Yang","Zhan Jin","Juan Liu","Ming Li"],"year":2026,"doi":"10.21437/Interspeech.2026-2035","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26m_interspeech.pdf","session":"Target Speaker Extraction, Speech Separation and Audio Understanding","topics":["source-separation","self-supervised","multilingual"],"category":"enhancement-separation","institutions":["Wuhan University"],"funding":["National Key Research and Development Program of China"],"code":{"url":"https://b23ca07a.github.io/MVTF-Gridnet/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26m_interspeech","category":"enhancement-separation","institutions":["Wuhan University"],"code":"https://b23ca07a.github.io/MVTF-Gridnet/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2035","pdf":"https://www.isca-archive.org/interspeech_2026/yang26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26m_interspeech/markdown.md"},{"id":"yang26n_interspeech","title":"U-Codec: Neural Speech Codec under Extreme Temporal Compression for Fast High-Fidelity Speech Generation","authors":["Xusheng Yang","Long Zhou","Wenfu Wang","Kai Hu","Zixiang Wan","Yushen Chen","Shulin Feng","Chenxing Li","Meng Yu","Dong Yu","Yuexian Zou"],"year":2026,"doi":"10.21437/Interspeech.2026-2398","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26n_interspeech.pdf","session":"Neural Audio Codec Architectures","topics":["tts","speech-llm","self-supervised"],"category":"speech-coding","labels":["efficient-on-device","generative-model"],"institutions":["Peking University","Tencent","Shanghai Jiao Tong University"],"funding":["Guangdong Provincial Key Laboratory of Ultra High Definition Immersive Media Technology"],"code":{"url":"https://anonymous666-speech.github.io/CodecFormer_5Hz/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26n_interspeech","category":"speech-coding","labels":["efficient-on-device","generative-model"],"institutions":["Peking University","Tencent","Shanghai Jiao Tong University"],"code":"https://anonymous666-speech.github.io/CodecFormer_5Hz/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2398","pdf":"https://www.isca-archive.org/interspeech_2026/yang26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26n_interspeech/markdown.md"},{"id":"yang26o_interspeech","title":"Speech Recognition on TV Series with Video-Guided Post-ASR Correction","authors":["Haoyuan Yang","Yue Zhang","Liqiang Jing","John Hansen"],"year":2026,"doi":"10.21437/Interspeech.2026-2970","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26o_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26o_interspeech.pdf","session":"Multimodal Speech Processing and Speech LLM Systems","topics":["asr","speech-llm","evaluation"],"category":"asr","institutions":["University of Texas at Dallas"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26o_interspeech","category":"asr","institutions":["University of Texas at Dallas"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2970","pdf":"https://www.isca-archive.org/interspeech_2026/yang26o_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26o_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26o_interspeech/markdown.md"},{"id":"yang26p_interspeech","title":"RobustSpeechFlow: Learning Robust Text-to-Speech Trajectories via Augmentation-based Contrastive Flow Matching","authors":["Jinhyeok Yang","Hyeongju Kim","Yechan Yu","Joon Byun","Frederik Bous","Juheon Lee"],"year":2026,"doi":"10.21437/Interspeech.2026-3086","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26p_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26p_interspeech.pdf","session":"Flow Matching for Speech Synthesis","topics":["tts","self-supervised","low-resource"],"category":"tts","labels":["generative-model"],"institutions":["Supertone","Out of Set"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26p_interspeech","category":"tts","labels":["generative-model"],"institutions":["Supertone","Out of Set"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3086","pdf":"https://www.isca-archive.org/interspeech_2026/yang26p_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26p_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26p_interspeech/markdown.md"},{"id":"yang26q_interspeech","title":"The Perception of Korean Stop Coda Consonants by Mandarin Speakers in Taiwan: A Comparative Study of Proficiency Levels","authors":["Shu-Wei Yang","Jung-yueh Tu"],"year":2026,"doi":"10.21437/Interspeech.2026-3292","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26q_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26q_interspeech.pdf","session":"Perception and Appraisal of Prosody","topics":["phonetics","speech-perception","low-resource"],"category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Seoul National University","National Chengchi University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26q_interspeech","category":"phonetics-linguistics","labels":["multilingual"],"institutions":["Seoul National University","National Chengchi University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3292","pdf":"https://www.isca-archive.org/interspeech_2026/yang26q_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26q_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26q_interspeech/markdown.md"},{"id":"yang26r_interspeech","title":"A Compact Fully-Open Cache-Aware Streaming Model for Japanese ASR","authors":["Yinchang Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-3380","isca_url":"https://www.isca-archive.org/interspeech_2026/yang26r_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yang26r_interspeech.pdf","session":"ASR Under Real-World Constraints: Streaming, Adaptation, and Efficiency","topics":["asr","streaming","low-resource"],"category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Kyoto College of Graduate Studies for Informatics"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yang26r_interspeech","category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Kyoto College of Graduate Studies for Informatics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3380","pdf":"https://www.isca-archive.org/interspeech_2026/yang26r_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26r_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yang26r_interspeech/markdown.md"},{"id":"yano26_interspeech","title":"Adapting Text LLMs to Speech via Multimodal Depth Up-Scaling","authors":["Kazuki Yano","Jun Suzuki","Shinji Watanabe"],"year":2026,"doi":"10.21437/Interspeech.2026-2099","isca_url":"https://www.isca-archive.org/interspeech_2026/yano26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yano26_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["self-supervised"],"institutions":["Tohoku University","Carnegie Mellon University"],"funding":["JST Moonshot R&D"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yano26_interspeech","category":"asr","labels":["self-supervised"],"institutions":["Tohoku University","Carnegie Mellon University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2099","pdf":"https://www.isca-archive.org/interspeech_2026/yano26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yano26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yano26_interspeech/markdown.md"},{"id":"ye26_interspeech","title":"Which Speech Representation Better Matches Text-Native Reasoning? A Study of Speech-Text Alignment on Frame Rate and Representation","authors":["Zhen Ye","Xu Tan","Yiming Li","Guangyan Zhang","Chimin Chan","Haohe Liu","Zhengxi Liu","Hongzhan Lin","Zheqi Dai","Xinshen Zhang","Peiwen Sun","Qiuqiang Kong","Wei Xue"],"year":2026,"doi":"10.21437/Interspeech.2026-21","isca_url":"https://www.isca-archive.org/interspeech_2026/ye26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ye26_interspeech.pdf","session":"Speech Representations and Alignment","topics":["speech-llm","self-supervised","spoken-language-understanding"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Hong Kong University of Science and Technology","Tencent","University of Surrey","Chinese University of Hong Kong","Hong Kong Baptist University","Hong Kong Polytechnic University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ye26_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Hong Kong University of Science and Technology","Tencent","University of Surrey","Chinese University of Hong Kong","Hong Kong Baptist University","Hong Kong Polytechnic University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-21","pdf":"https://www.isca-archive.org/interspeech_2026/ye26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ye26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ye26_interspeech/markdown.md"},{"id":"ye26b_interspeech","title":"Refining Emphasis Control in Flow-Matching TTS via Preference Alignment and Reinforcement Learning","authors":["Jiangnan Ye","Jiawei Jin","Pengfei Tan","Chao Yan","Xuerui Yang","Zhiyong Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-2284","isca_url":"https://www.isca-archive.org/interspeech_2026/ye26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ye26b_interspeech.pdf","session":"Long-Form Speech Synthesis","topics":["tts","self-supervised","prosody"],"category":"tts","labels":["generative-model"],"institutions":["Tsinghua University","StepFun"],"funding":["National Natural Science Foundation of China","National Social Science Foundation of China"],"code":{"url":"https://thuhcsi.github.io/interspeech2026-F5Emphasis","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ye26b_interspeech","category":"tts","labels":["generative-model"],"institutions":["Tsinghua University","StepFun"],"code":"https://thuhcsi.github.io/interspeech2026-F5Emphasis","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2284","pdf":"https://www.isca-archive.org/interspeech_2026/ye26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ye26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ye26b_interspeech/markdown.md"},{"id":"ye26c_interspeech","title":"Reinforcement Learning for Data-Efficient Code-Switched ASR","authors":["Ziwei Ye","Peter Vickers"],"year":2026,"doi":"10.21437/Interspeech.2026-2667","isca_url":"https://www.isca-archive.org/interspeech_2026/ye26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ye26c_interspeech.pdf","session":"Code-Switching ASR","topics":["asr","self-supervised","multilingual"],"category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Spotify"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ye26c_interspeech","category":"asr","labels":["low-resource","multilingual","self-supervised"],"institutions":["Spotify"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2667","pdf":"https://www.isca-archive.org/interspeech_2026/ye26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ye26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ye26c_interspeech/markdown.md"},{"id":"ye26d_interspeech","title":"ZipCodec: Simple and Pretrained-Model-Free Speech Tokenizer via Flow-Matching","authors":["Lingxuan Ye","Han Zhu","Liyong Guo","Zengwei Yao","Wei Kang","Fangjun Kuang","Zhifeng Han","Long Lin","Daniel Povey"],"year":2026,"doi":"10.21437/Interspeech.2026-2941","isca_url":"https://www.isca-archive.org/interspeech_2026/ye26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ye26d_interspeech.pdf","session":"From Self-Supervised Pre-training to Phonetic Analysis of Speech Models","topics":["tts","self-supervised","speech-llm"],"category":"speech-coding","labels":["efficient-on-device","generative-model"],"institutions":["Xiaomi"],"code":{"url":"https://github.com/winlaic/ZipCodec","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ye26d_interspeech","category":"speech-coding","labels":["efficient-on-device","generative-model"],"institutions":["Xiaomi"],"code":"https://github.com/winlaic/ZipCodec","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2941","pdf":"https://www.isca-archive.org/interspeech_2026/ye26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ye26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ye26d_interspeech/markdown.md"},{"id":"ye26e_interspeech","title":"OPERA-Net: Octave-aware Phase-sensitive Enhanced Recognition Architecture for Singing Voice Deepfake Detection","authors":["Fengwei Ye","Kun Zeng"],"year":2026,"doi":"10.21437/Interspeech.2026-3341","isca_url":"https://www.isca-archive.org/interspeech_2026/ye26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/ye26e_interspeech.pdf","session":"Spoofing and Deepfake Detection 2","topics":["audio-deepfake","self-supervised","paralinguistics"],"category":"deepfake-security","institutions":["Sun Yat-sen University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"ye26e_interspeech","category":"deepfake-security","institutions":["Sun Yat-sen University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3341","pdf":"https://www.isca-archive.org/interspeech_2026/ye26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/ye26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/ye26e_interspeech/markdown.md"},{"id":"yeh26_interspeech","title":"Who is Speaking or Who is Depressed? A Controlled Study of Speaker Leakage in Speech-Based Depression Detection","authors":["Hsiang-Chen Yeh","Luqi Sun","Aurosweta Mahapatra","Shreeram Suresh Chandra","Emily Mower Provost","Berrak Sisman"],"year":2026,"doi":"10.21437/Interspeech.2026-1394","isca_url":"https://www.isca-archive.org/interspeech_2026/yeh26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yeh26_interspeech.pdf","session":"Pathological Speech Assessment 1","topics":["paralinguistics","self-supervised","evaluation"],"category":"health-clinical","institutions":["Johns Hopkins University","University of Michigan"],"funding":["Johns Hopkins University Data Science and AI Institute"],"code":{"url":"https://github.com/jen900704/Speech-Depression-Speaker-Leakage","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yeh26_interspeech","category":"health-clinical","institutions":["Johns Hopkins University","University of Michigan"],"code":"https://github.com/jen900704/Speech-Depression-Speaker-Leakage","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1394","pdf":"https://www.isca-archive.org/interspeech_2026/yeh26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yeh26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yeh26_interspeech/markdown.md"},{"id":"yen26_interspeech","title":"MDM-ASR: Bridging Accuracy and Efficiency in ASR with Diffusion-Based Non-Autoregressive Decoding","authors":["Hao Yen","Pin-Jui Ku","Ante Jukić","Sabato Marco Siniscalchi"],"year":2026,"doi":"10.21437/Interspeech.2026-488","isca_url":"https://www.isca-archive.org/interspeech_2026/yen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yen26_interspeech.pdf","session":"Robust and Efficient ASR","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["efficient-on-device","generative-model"],"institutions":["Georgia Institute of Technology","Universita degli Studi di Palermo","NVIDIA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yen26_interspeech","category":"asr","labels":["efficient-on-device","generative-model"],"institutions":["Georgia Institute of Technology","Universita degli Studi di Palermo","NVIDIA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-488","pdf":"https://www.isca-archive.org/interspeech_2026/yen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yen26_interspeech/markdown.md"},{"id":"yeo26_interspeech","title":"Pushing the Limits of Compression: Sub-1-Bit Conformer via Variable-Rank Binary Decomposition","authors":["Jinsu Yeo","Banseok Lee","Youngmin Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-2063","isca_url":"https://www.isca-archive.org/interspeech_2026/yeo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yeo26_interspeech.pdf","session":"Resource Constrained Speech Recognition","topics":["asr","self-supervised","low-resource"],"category":"asr","labels":["efficient-on-device"],"institutions":["Samsung Electronics"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yeo26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["Samsung Electronics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2063","pdf":"https://www.isca-archive.org/interspeech_2026/yeo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yeo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yeo26_interspeech/markdown.md"},{"id":"yerpude26_interspeech","title":"Attention-Based Multiple Instance Learning with Tabular Stacking for Ambulatory Detection of PVH and NPVH","authors":["Kiran Yerpude","Seung Gyu Jeong","Seong-Eun Kim"],"year":2026,"doi":"10.21437/Interspeech.2026-2355","isca_url":"https://www.isca-archive.org/interspeech_2026/yerpude26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yerpude26_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["paralinguistics","self-supervised","dataset"],"category":"health-clinical","institutions":["Seoul National University of Science and Technology","Medisensing"],"funding":["National Research Foundation of Korea","AI Seoul Tech Research Support Program","Seoul Future Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yerpude26_interspeech","category":"health-clinical","institutions":["Seoul National University of Science and Technology","Medisensing"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2355","pdf":"https://www.isca-archive.org/interspeech_2026/yerpude26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yerpude26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yerpude26_interspeech/markdown.md"},{"id":"yi26_interspeech","title":"Exploring the Scale and Diversity of Speech Anti-spoofing Datasets: Experiments and Analysis","authors":["Zhuolin Yi","Jun Xue","Yanzhen Ren","Yihuan Huang","Yi Chai","Daixian Li","Guanxiang Feng","Jiajun Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-157","isca_url":"https://www.isca-archive.org/interspeech_2026/yi26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yi26_interspeech.pdf","session":"Spoofing and Deepfake Detection 1","topics":["speech-enhancement","self-supervised","evaluation"],"category":"deepfake-security","labels":["robustness-noise"],"institutions":["Wuhan University"],"funding":["Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yi26_interspeech","category":"deepfake-security","labels":["robustness-noise"],"institutions":["Wuhan University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-157","pdf":"https://www.isca-archive.org/interspeech_2026/yi26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yi26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yi26_interspeech/markdown.md"},{"id":"yin26_interspeech","title":"The First Environmental Sound Deepfake Detection Challenge: Benchmarking Robustness, Evaluation, and Insights","authors":["Han Yin","Yang Xiao","Rohan Kumar Das","Jisheng Bai","Ting Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-1599","isca_url":"https://www.isca-archive.org/interspeech_2026/yin26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yin26_interspeech.pdf","session":"Evaluation, Benchmarking, and Reliability of Audio Systems","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","labels":["low-resource","self-supervised","dataset-or-benchmark-release","robustness-noise"],"institutions":["KAIST","University of Melbourne","Fortemedia Singapore","Xi'an University of Posts & Telecommunications","Xi'an Lianfeng Acoustic Technologies Co., Ltd"],"code":{"url":"https://github.com/apple-yinhan/ESDD-Review","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yin26_interspeech","category":"deepfake-security","labels":["low-resource","self-supervised","dataset-or-benchmark-release","robustness-noise"],"institutions":["KAIST","University of Melbourne","Fortemedia Singapore","Xi'an University of Posts & Telecommunications","Xi'an Lianfeng Acoustic Technologies Co., Ltd"],"code":"https://github.com/apple-yinhan/ESDD-Review","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1599","pdf":"https://www.isca-archive.org/interspeech_2026/yin26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yin26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yin26_interspeech/markdown.md"},{"id":"yin26b_interspeech","title":"STArK: Towards Synthesizing Articulatory Kinematics from Text","authors":["Xavier Yin","Carlos Busso"],"year":2026,"doi":"10.21437/Interspeech.2026-2842","isca_url":"https://www.isca-archive.org/interspeech_2026/yin26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yin26b_interspeech.pdf","session":"Modeling Articulation","topics":["tts","speech-synthesis","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["Carnegie Mellon University"],"code":{"url":"https://github.com/Lab-MSP/STArK/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yin26b_interspeech","category":"tts","labels":["generative-model"],"institutions":["Carnegie Mellon University"],"code":"https://github.com/Lab-MSP/STArK/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2842","pdf":"https://www.isca-archive.org/interspeech_2026/yin26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yin26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yin26b_interspeech/markdown.md"},{"id":"yokota26_interspeech","title":"Physics-Informed Neural Operator for Speech Production Analysis","authors":["Kazuya Yokota","Xinmeng Luan","Debasish Ray Mohapatra","Gary Scavone","Sidney Fels"],"year":2026,"doi":"10.21437/Interspeech.2026-2023","isca_url":"https://www.isca-archive.org/interspeech_2026/yokota26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yokota26_interspeech.pdf","session":"Modeling Articulation","topics":["speech-production","self-supervised","tts"],"category":"phonetics-linguistics","institutions":["Nagaoka University of Technology","McGill University","University of British Columbia"],"funding":["JSPS KAKENHI","JSPS Program for Forming Japan's Peak Research Universities","Ono Charitable Trust for Acoustics"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yokota26_interspeech","category":"phonetics-linguistics","institutions":["Nagaoka University of Technology","McGill University","University of British Columbia"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2023","pdf":"https://www.isca-archive.org/interspeech_2026/yokota26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yokota26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yokota26_interspeech/markdown.md"},{"id":"yoon26_interspeech","title":"Robust Multi-Source-Free Domain Adaptation via Posterior Adjustment and Label Agreement","authors":["Hoyoung Yoon","U Kang"],"year":2026,"doi":"10.21437/Interspeech.2026-370","isca_url":"https://www.isca-archive.org/interspeech_2026/yoon26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yoon26_interspeech.pdf","session":"Spatial Audio 1","topics":["acoustic-scene-classification","self-supervised","multilingual"],"category":"audio-understanding","labels":["robustness-noise"],"institutions":["Seoul National University"],"funding":["Institute of Information & Communications Technology Planning & Evaluation","Korea government (MSIT)","XVoice: Multi-Modal Voice Meta Learning","AI Star Fellowship Support Program","Global AI Frontier Lab","Artificial Intelligence Graduate School Program"],"code":{"url":"https://github.com/snudatalab/FASOLA","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yoon26_interspeech","category":"audio-understanding","labels":["robustness-noise"],"institutions":["Seoul National University"],"code":"https://github.com/snudatalab/FASOLA","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-370","pdf":"https://www.isca-archive.org/interspeech_2026/yoon26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yoon26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yoon26_interspeech/markdown.md"},{"id":"yoshinaga26_interspeech","title":"Noise Scaling Factor for the One-Dimensional Voice Production Model","authors":["Tsukasa Yoshinaga","Takeshi Ikuma","Brad H. Story"],"year":2026,"doi":"10.21437/Interspeech.2026-559","isca_url":"https://www.isca-archive.org/interspeech_2026/yoshinaga26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yoshinaga26_interspeech.pdf","session":"Speech Production and Perception 1","topics":["tts","phonetics","prosody"],"category":"phonetics-linguistics","institutions":["Osaka University","Louisiana State University Health Sciences Center","University of Arizona"],"funding":["JSPS KAKENHI"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yoshinaga26_interspeech","category":"phonetics-linguistics","institutions":["Osaka University","Louisiana State University Health Sciences Center","University of Arizona"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-559","pdf":"https://www.isca-archive.org/interspeech_2026/yoshinaga26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yoshinaga26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yoshinaga26_interspeech/markdown.md"},{"id":"you26_interspeech","title":"Uncovering the Impact of G2P Precision on Korean TTS: A Large-Scale Statistical Validation via a Novel Morphological Engine","authors":["Heejo You","Sungwoo Moon"],"year":2026,"doi":"10.21437/Interspeech.2026-887","isca_url":"https://www.isca-archive.org/interspeech_2026/you26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/you26_interspeech.pdf","session":"Text Processing for Speech Synthesis","topics":["tts","evaluation","low-resource"],"category":"tts","labels":["efficient-on-device"],"institutions":["Hyundai Motor Group"],"funding":["Robotics Lab at Hyundai Motor Group"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"you26_interspeech","category":"tts","labels":["efficient-on-device"],"institutions":["Hyundai Motor Group"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-887","pdf":"https://www.isca-archive.org/interspeech_2026/you26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/you26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/you26_interspeech/markdown.md"},{"id":"you26b_interspeech","title":"Bidirectional Retention Network-based Segmentation Model for Speaker Diarization","authors":["Jian You","Xiangfeng Li","Tengfei Zhou","Erwan Zerhouni"],"year":2026,"doi":"10.21437/Interspeech.2026-1032","isca_url":"https://www.isca-archive.org/interspeech_2026/you26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/you26b_interspeech.pdf","session":"Speaker Diarization 1","topics":["speaker-diarization","self-supervised","multilingual"],"category":"speaker","labels":["self-supervised"],"institutions":["Cisco Systems"],"code":{"url":"https://github.com/frankyoujian/BiRetNetDiarization","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"you26b_interspeech","category":"speaker","labels":["self-supervised"],"institutions":["Cisco Systems"],"code":"https://github.com/frankyoujian/BiRetNetDiarization","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1032","pdf":"https://www.isca-archive.org/interspeech_2026/you26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/you26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/you26b_interspeech/markdown.md"},{"id":"yousef26_interspeech","title":"Modeling Lombard Effects in Voice Disorders Using Daily-Life Monitoring of Ambient Noise and Voice Acoustics","authors":["Ahmed Yousef","Trishul Chowdhury","Gregory Ciccarelli","Thomas F. Quatieri","Robert Hillman","Daryush Mehta"],"year":2026,"doi":"10.21437/Interspeech.2026-2777","isca_url":"https://www.isca-archive.org/interspeech_2026/yousef26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yousef26_interspeech.pdf","session":"Speech, Voice and Language Disorders","topics":["paralinguistics","health","speech-enhancement"],"category":"health-clinical","institutions":["Massachusetts General Hospital","Harvard Medical School","Northeastern University","MIT Lincoln Laboratory"],"funding":["National Institutes of Health","National Institute on Deafness and Other Communication Disorders"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yousef26_interspeech","category":"health-clinical","institutions":["Massachusetts General Hospital","Harvard Medical School","Northeastern University","MIT Lincoln Laboratory"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2777","pdf":"https://www.isca-archive.org/interspeech_2026/yousef26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yousef26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yousef26_interspeech/markdown.md"},{"id":"yousef26b_interspeech","title":"The Interspeech 2026 NeckVibe Challenge: Voice Disorder Detection via Real-World Monitoring of Neck-Surface Vibration","authors":["Ahmed Yousef","Robert Hillman","Jarrad Van Stan","Matías Zañartu","Hamzeh Ghasemzadeh","Ben Kevelson","Daryush Mehta"],"year":2026,"doi":"10.21437/Interspeech.2026-3049","isca_url":"https://www.isca-archive.org/interspeech_2026/yousef26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yousef26b_interspeech.pdf","session":"Grand Special Challenges Poster Showcase","topics":["paralinguistics","dataset","self-supervised"],"category":"health-clinical","labels":["dataset-or-benchmark-release"],"institutions":["Massachusetts General Hospital","Harvard Medical School","Universidad Tecnica Federico Santa Maria","University of Central Florida"],"funding":["National Institutes of Health","National Institute on Deafness and Other Communication Disorders"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yousef26b_interspeech","category":"health-clinical","labels":["dataset-or-benchmark-release"],"institutions":["Massachusetts General Hospital","Harvard Medical School","Universidad Tecnica Federico Santa Maria","University of Central Florida"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3049","pdf":"https://www.isca-archive.org/interspeech_2026/yousef26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yousef26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yousef26b_interspeech/markdown.md"},{"id":"yousef26c_interspeech","title":"Measuring Vocal Efficiency in Daily Life in Patients with Voice Disorders Using Wireless Accelerometer and Microphone Sensors","authors":["Ahmed Yousef","Emma Willis","Vahni Tagirisa","Robert Hillman","Daryush Mehta"],"year":2026,"doi":"10.21437/Interspeech.2026-3457","isca_url":"https://www.isca-archive.org/interspeech_2026/yousef26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yousef26c_interspeech.pdf","session":"Speech, Voice and Language Disorders","topics":["paralinguistics","health","evaluation"],"category":"health-clinical","institutions":["Massachusetts General Hospital","Harvard Medical School"],"funding":["National Institutes of Health","National Institute on Deafness and Other Communication Disorders"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yousef26c_interspeech","category":"health-clinical","institutions":["Massachusetts General Hospital","Harvard Medical School"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3457","pdf":"https://www.isca-archive.org/interspeech_2026/yousef26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yousef26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yousef26c_interspeech/markdown.md"},{"id":"yu26_interspeech","title":"Investigating LLMs Behavior in Depression Severity Prediction","authors":["Jiawei Yu","Yun Hao","Heysem Kaya"],"year":2026,"doi":"10.21437/Interspeech.2026-456","isca_url":"https://www.isca-archive.org/interspeech_2026/yu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yu26_interspeech.pdf","session":"Medical Dialogue and Conversational Understanding","topics":["speech-llm","spoken-language-understanding","health"],"category":"health-clinical","institutions":["Utrecht University","University of Groningen"],"funding":["China Scholarship Council"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yu26_interspeech","category":"health-clinical","institutions":["Utrecht University","University of Groningen"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-456","pdf":"https://www.isca-archive.org/interspeech_2026/yu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26_interspeech/markdown.md"},{"id":"yu26b_interspeech","title":"Enhancing Flow Matching with A Unified Guidance Framework for Efficient and Robust Speech Synthesis","authors":["Zuda Yu","Qianhui Xu","Ting Chen","Junhui Zhang","Tao Fu","Hongjiang Yu","Qiangqiang Wang","Yang Song"],"year":2026,"doi":"10.21437/Interspeech.2026-1015","isca_url":"https://www.isca-archive.org/interspeech_2026/yu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yu26b_interspeech.pdf","session":"Flow Matching for Speech Synthesis","topics":["tts","voice-conversion","self-supervised"],"category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Zuoyebang"],"code":{"url":"https://yuzuda283.github.io/unified-guidanc%20e-flow-matching/Interspeech2026_demo_samples/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yu26b_interspeech","category":"tts","labels":["efficient-on-device","generative-model"],"institutions":["Zuoyebang"],"code":"https://yuzuda283.github.io/unified-guidanc%20e-flow-matching/Interspeech2026_demo_samples/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1015","pdf":"https://www.isca-archive.org/interspeech_2026/yu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26b_interspeech/markdown.md"},{"id":"yu26c_interspeech","title":"Learning to Attend to Depression-Related Patterns: An Adaptive Cross-Modal Gating Network for Depression Detection","authors":["Hangbin Yu","Yudong Yang","Rongfeng Su","Nan Yan","Lan Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1075","isca_url":"https://www.isca-archive.org/interspeech_2026/yu26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yu26c_interspeech.pdf","session":"Pathological Speech Assessment 3","topics":["paralinguistics","speech-llm","emotion-recognition"],"category":"health-clinical","institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences"],"funding":["National Key R&D Program of China","National Natural Science Foundation of China","Shenzhen Peacock Team Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yu26c_interspeech","category":"health-clinical","institutions":["Chinese Academy of Sciences","University of Chinese Academy of Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1075","pdf":"https://www.isca-archive.org/interspeech_2026/yu26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26c_interspeech/markdown.md"},{"id":"yu26d_interspeech","title":"Sweep-RSE: Streaming Region-of-Interest Speech Extraction in Multi-Talker Scenarios via Explicit Spatial Sweeping","authors":["Hogeon Yu","SeongHun Noh","Hyunsik Choi","Sihyun Joo"],"year":2026,"doi":"10.21437/Interspeech.2026-1631","isca_url":"https://www.isca-archive.org/interspeech_2026/yu26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yu26d_interspeech.pdf","session":"Source Separation 1","topics":["speech-enhancement","spatial-audio","self-supervised"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Hyundai Motor Company"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yu26d_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Hyundai Motor Company"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1631","pdf":"https://www.isca-archive.org/interspeech_2026/yu26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26d_interspeech/markdown.md"},{"id":"yu26e_interspeech","title":"Online Audiovisual Speaker Separation Using Efficient Visual Knowledge Distillation","authors":["Cheng Yu","Vahid A. Kalkhorani","Ashutosh Pandey","Daniel Wong","Jacob Donley","Buye Xu","DeLiang Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1999","isca_url":"https://www.isca-archive.org/interspeech_2026/yu26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yu26e_interspeech.pdf","session":"Source Separation 1","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Ohio State University","Meta","Chinese University of Hong Kong, Shenzhen"],"funding":["Meta","National Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yu26e_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Ohio State University","Meta","Chinese University of Hong Kong, Shenzhen"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1999","pdf":"https://www.isca-archive.org/interspeech_2026/yu26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26e_interspeech/markdown.md"},{"id":"yu26f_interspeech","title":"Disentangling Reasoning in Large Audio-Language Models for Ambiguous Emotion Prediction","authors":["Xiaofeng Yu","Jiaheng Dong","Jean Honorio","Abhirup Ghosh","Hong Jia","Ting Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-2031","isca_url":"https://www.isca-archive.org/interspeech_2026/yu26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yu26f_interspeech.pdf","session":"Speech Emotion Recognition and Representation 3","topics":["emotion-recognition","speech-llm","self-supervised"],"category":"paralinguistics-emotion","institutions":["University of Auckland","University of Melbourne","University of Birmingham","ARC OPTIMA"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yu26f_interspeech","category":"paralinguistics-emotion","institutions":["University of Auckland","University of Melbourne","University of Birmingham","ARC OPTIMA"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2031","pdf":"https://www.isca-archive.org/interspeech_2026/yu26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26f_interspeech/markdown.md"},{"id":"yu26g_interspeech","title":"SDR-LLM: Speech-LLM Based End-to-End Speaker Diarization and Recognition with Sentence-Level Temporal Modeling","authors":["Renjie Yu","Yixuan Zhou","Shun Lei","Xiang Li","Runchuan Ye","Yikai Huang","Guoyang Zeng","Zhiyong Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-2854","isca_url":"https://www.isca-archive.org/interspeech_2026/yu26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yu26g_interspeech.pdf","session":"Speaker Diarization and Recognition","topics":["speech-llm","speaker-diarization","asr"],"category":"speaker","institutions":["Tsinghua University","ModelBest"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yu26g_interspeech","category":"speaker","institutions":["Tsinghua University","ModelBest"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2854","pdf":"https://www.isca-archive.org/interspeech_2026/yu26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yu26g_interspeech/markdown.md"},{"id":"yuan26_interspeech","title":"STSR: High-Fidelity Speech Super-Resolution via Spectral-Transient Context Modeling","authors":["Jiajun Yuan","Xiaochen Wang","Yulin Wu","Chenhao Hu","Xueyang Lv"],"year":2026,"doi":"10.21437/Interspeech.2026-27","isca_url":"https://www.isca-archive.org/interspeech_2026/yuan26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yuan26_interspeech.pdf","session":"Dereverberation, Bandwidth Extension and Restoration","topics":["speech-enhancement","self-supervised","generative-ai"],"category":"speech-coding","labels":["generative-model"],"institutions":["Wuhan University","Jianghan University","Xiaomi Corporation"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yuan26_interspeech","category":"speech-coding","labels":["generative-model"],"institutions":["Wuhan University","Jianghan University","Xiaomi Corporation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-27","pdf":"https://www.isca-archive.org/interspeech_2026/yuan26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yuan26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yuan26_interspeech/markdown.md"},{"id":"yuan26b_interspeech","title":"DelayGSE: A Generative Speech Enhancement Framework with Delayed Text-Aware Conditioning","authors":["Xin Yuan","Junling Lv","Zezhou Xu","Xingjun Tan","Liangliang Li","Yanqiang Lei"],"year":2026,"doi":"10.21437/Interspeech.2026-232","isca_url":"https://www.isca-archive.org/interspeech_2026/yuan26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yuan26b_interspeech.pdf","session":"Language-Model and Codec-Token Speech Enhancement","topics":["speech-enhancement","self-supervised","speech-llm"],"category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Guangzhou Shiyuan Electronic Technology Company Limited","Shanghai University of Finance and Economics"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yuan26b_interspeech","category":"enhancement-separation","labels":["self-supervised","generative-model","robustness-noise"],"institutions":["Guangzhou Shiyuan Electronic Technology Company Limited","Shanghai University of Finance and Economics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-232","pdf":"https://www.isca-archive.org/interspeech_2026/yuan26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yuan26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yuan26b_interspeech/markdown.md"},{"id":"yue26_interspeech","title":"G2C-NET: A Grid-to-Continuous Neural Network for Sound Source Localization in Distributed Microphone Arrays","authors":["Zhiyuan Yue","De Hu"],"year":2026,"doi":"10.21437/Interspeech.2026-1733","isca_url":"https://www.isca-archive.org/interspeech_2026/yue26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yue26_interspeech.pdf","session":"Spatial Audio 4","topics":["speech-enhancement","self-supervised","evaluation"],"category":"enhancement-separation","institutions":["Inner Mongolia University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://github.com/Zhiyuan-Yue/G2C-NET.git","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yue26_interspeech","category":"enhancement-separation","institutions":["Inner Mongolia University"],"code":"https://github.com/Zhiyuan-Yue/G2C-NET.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1733","pdf":"https://www.isca-archive.org/interspeech_2026/yue26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yue26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yue26_interspeech/markdown.md"},{"id":"yuen26_interspeech","title":"How do word frequency and syllable surprisal affect response time and acoustic duration in sentence formulation?","authors":["Ivan Yuen","Bernd Möbius","Bistra Andreeva","Mitko Sabev"],"year":2026,"doi":"10.21437/Interspeech.2026-1080","isca_url":"https://www.isca-archive.org/interspeech_2026/yuen26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/yuen26_interspeech.pdf","session":"Speech Production and Perception 2","topics":["phonetics","prosody","evaluation"],"category":"phonetics-linguistics","institutions":["Saarland University"],"funding":["Deutsche Forschungsgemeinschaft"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"yuen26_interspeech","category":"phonetics-linguistics","institutions":["Saarland University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1080","pdf":"https://www.isca-archive.org/interspeech_2026/yuen26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/yuen26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/yuen26_interspeech/markdown.md"},{"id":"zafar26_interspeech","title":"Rethinking Acoustic Variability Of ADReSS and ADReSSo Datasets For Dementia Detection","authors":["Muhammad Abdullah Zafar","Mostafa Shahin","Beena Ahmed"],"year":2026,"doi":"10.21437/Interspeech.2026-2862","isca_url":"https://www.isca-archive.org/interspeech_2026/zafar26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zafar26_interspeech.pdf","session":"Clinically Useful Speech Representations 1","topics":["paralinguistics","evaluation","health"],"category":"health-clinical","labels":["robustness-noise"],"institutions":["University of New South Wales"],"funding":["National Institutes of Health"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zafar26_interspeech","category":"health-clinical","labels":["robustness-noise"],"institutions":["University of New South Wales"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2862","pdf":"https://www.isca-archive.org/interspeech_2026/zafar26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zafar26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zafar26_interspeech/markdown.md"},{"id":"zhang26_interspeech","title":"AURA: Audio-Geometry Conditioned U-Net Refinement with Flow Matching for High-Fidelity Monaural-to-Binaural Synthesis","authors":["Wenjie Zhang","Changjun He","Yinghan Cao","Shiyun Xu","Mingjiang Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-87","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26_interspeech.pdf","session":"Spatial Audio 3","topics":["source-separation","self-supervised","evaluation"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Harbin Institute of Technology"],"funding":["National Natural Science Foundation of China","Key Research and Development Program of Xinjiang Uygur Autonomous Region","Shenzhen Higher Education Institutions Stability Support Program","Guangdong Basic and Applied Basic Research Foundation"],"code":{"url":"https://ethuil.github.io/AURA/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Harbin Institute of Technology"],"code":"https://ethuil.github.io/AURA/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-87","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26_interspeech/markdown.md"},{"id":"zhang26aa_interspeech","title":"PART: Progressive Alignment Representation Training for Multilingual Speech-To-Text with LLMs","authors":["Pei Zhang","Andong Chen","Xi Chen","Baosong Yang","Derek F. Wong","Fei Huang"],"year":2026,"doi":"10.21437/Interspeech.2026-1734","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26aa_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26aa_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["asr","speech-translation","speech-llm"],"category":"asr","labels":["multilingual"],"institutions":["Alibaba Group","University of Macau","Chinese University of Hong Kong"],"funding":["Science and Technology Development Fund of Macau SAR","UM and UMDF"],"code":{"url":"https://github.com/ChenX17/PART","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26aa_interspeech","category":"asr","labels":["multilingual"],"institutions":["Alibaba Group","University of Macau","Chinese University of Hong Kong"],"code":"https://github.com/ChenX17/PART","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1734","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26aa_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26aa_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26aa_interspeech/markdown.md"},{"id":"zhang26b_interspeech","title":"Step-Audio-R1: Why Audio LLMs Fail at Reasoning — The Trap of Textual Surrogates","authors":["Yuxin Zhang","Daijiao Liu","Haoyang Zhang","Xiangyu Zhang","Yuxin Li","Fei Tian","Yayue Deng","Donghang Wu","Jun Chen","Liang Zhao","Chengyuan Yao","Gaolei Li","Quanhai Zhang","Qiquan Zhang","Hexin Liu","Eng Siong Chng","Xuerui Yang","Xiangyu Zhang","Daxin Jiang","Gang Yu"],"year":2026,"doi":"10.21437/Interspeech.2026-256","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26b_interspeech.pdf","session":"Post-Training of Speech Foundation Models","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","institutions":["Shanghai Jiao Tong University","University of New South Wales","Nanyang Technological University","StepFun"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26b_interspeech","category":"speech-llm-dialogue","institutions":["Shanghai Jiao Tong University","University of New South Wales","Nanyang Technological University","StepFun"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-256","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26b_interspeech/markdown.md"},{"id":"zhang26ba_interspeech","title":"Neural Oscillatory Mechanisms of Speaker Normalization Under Cognitive Load: Evidence from Cantonese Tone Perception","authors":["Kaile Zhang","Gang Peng"],"year":2026,"doi":"10.21437/Interspeech.2026-1940","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26ba_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26ba_interspeech.pdf","session":"Brain Studies and Speech","topics":["speech-perception","paralinguistics","phonetics"],"category":"phonetics-linguistics","institutions":["Hong Kong Polytechnic University"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26ba_interspeech","category":"phonetics-linguistics","institutions":["Hong Kong Polytechnic University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1940","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26ba_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ba_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ba_interspeech/markdown.md"},{"id":"zhang26c_interspeech","title":"AQA-TTRL: Self-Adaptation in Audio Question Answering with Test-Time Reinforcement Learning","authors":["Haoyu Zhang","Jiaxian Guo","Dong Yang","Yusuke Iwasawa","Yutaka Matsuo"],"year":2026,"doi":"10.21437/Interspeech.2026-288","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26c_interspeech.pdf","session":"Multi-Speaker Processing, Personalization, and Adaptation","topics":["speech-llm","self-supervised","evaluation"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["University of Tokyo"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26c_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["University of Tokyo"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-288","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26c_interspeech/markdown.md"},{"id":"zhang26ca_interspeech","title":"ACR-Net: Mitigating Semantic Dominance via Contrastive Acoustic-Semantic Decoupling","authors":["Mengke Zhang","Yanda Shao","Tianhe Wu","Kai Feng"],"year":2026,"doi":"10.21437/Interspeech.2026-2134","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26ca_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26ca_interspeech.pdf","session":"Speech Emotion Recognition and Representation 1","topics":["speech-emotion-recognition","self-supervised","evaluation"],"category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["Beijing University of Posts and Telecommunications"],"code":{"url":"https://github.com/zmkshakespar/ACR-Net","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26ca_interspeech","category":"paralinguistics-emotion","labels":["dataset-or-benchmark-release"],"institutions":["Beijing University of Posts and Telecommunications"],"code":"https://github.com/zmkshakespar/ACR-Net","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2134","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26ca_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ca_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ca_interspeech/markdown.md"},{"id":"zhang26d_interspeech","title":"Dual-Encoder Fusion with Explicit and Implicit Injection for the Interspeech 2026 Audio Encoder Capability Challenge","authors":["Yucong Zhang","Zhang Chen","Juan Liu","Wei Ju","Hongbin Suo","Ming Li"],"year":2026,"doi":"10.21437/Interspeech.2026-463","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26d_interspeech.pdf","session":"SE Architectures, Adaptation and Audio Front-Ends","topics":["self-supervised","speech-llm","multilingual"],"category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Wuhan University","Chinese University of Hong Kong, Shenzhen","OPPO","Duke Kunshan University"],"funding":["National Natural Science Foundation of China","Yangtze River Delta Science and Technology Innovation Community Joint Research Project","OPPO"],"code":{"url":"https://huggingface.co/yucongzh/implicit_fusion","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26d_interspeech","category":"speech-llm-dialogue","labels":["self-supervised"],"institutions":["Wuhan University","Chinese University of Hong Kong, Shenzhen","OPPO","Duke Kunshan University"],"code":"https://huggingface.co/yucongzh/implicit_fusion","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-463","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26d_interspeech/markdown.md"},{"id":"zhang26da_interspeech","title":"Grammar-Guided Hierarchical Parsing for Long-form Audio Activity Recognition","authors":["Peng Zhang","Qingyu Luo","Philip J.B. Jackson","Wenwu Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-2157","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26da_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26da_interspeech.pdf","session":"Audio Understanding and Representation Learning","topics":["spoken-language-understanding","speech-llm","dataset"],"category":"audio-understanding","institutions":["University of Surrey"],"funding":["Bang & Olufsen A/S","AURIC Project"],"code":{"url":"https://github.com/PennyZhang9/MultiAct","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26da_interspeech","category":"audio-understanding","institutions":["University of Surrey"],"code":"https://github.com/PennyZhang9/MultiAct","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2157","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26da_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26da_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26da_interspeech/markdown.md"},{"id":"zhang26e_interspeech","title":"Towards Unified Song Generation and Singing Voice Conversion with Accompaniment Co-Generation","authors":["Ziyu Zhang","Chunyu Qiang","Xiaopeng Wang","Yuxin Guo","Kang Yin","Wenjie Tian","Jingbin Hu","Tianlun Zuo","Zhao Guo","Teng Ma","Yuzhe Liang","Chen Zhang","Lei Xie"],"year":2026,"doi":"10.21437/Interspeech.2026-481","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26e_interspeech.pdf","session":"Singing Voice and Music Generation","topics":["tts","voice-conversion","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["Northwestern Polytechnical University","Kuaishou Technology","Beijing Institute of Technology","Chinese Academy of Sciences"],"code":{"url":"https://ziyu6.github.io/UniSinger/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26e_interspeech","category":"tts","labels":["generative-model"],"institutions":["Northwestern Polytechnical University","Kuaishou Technology","Beijing Institute of Technology","Chinese Academy of Sciences"],"code":"https://ziyu6.github.io/UniSinger/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-481","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26e_interspeech/markdown.md"},{"id":"zhang26ea_interspeech","title":"AcoustEmo: An Utterance-Aware Acoustic Q-Former for Open-Vocabulary Emotion Reasoning","authors":["Liyun Zhang","Xuanmeng Sha","Shuqiong Wu","Fengkai Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-2364","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26ea_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26ea_interspeech.pdf","session":"Speech Emotion Recognition and Representation 3","topics":["speech-llm","paralinguistics","emotion-recognition"],"category":"paralinguistics-emotion","institutions":["University of Tokyo","University of Osaka"],"funding":["Japan Society for the Promotion of Science"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26ea_interspeech","category":"paralinguistics-emotion","institutions":["University of Tokyo","University of Osaka"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2364","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26ea_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ea_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ea_interspeech/markdown.md"},{"id":"zhang26f_interspeech","title":"CoRE: Contrastive Evidence-Aware Rescoring for Multiple-Choice Audio Question Answering","authors":["Peihong Zhang","Zhixin Li","Yuxuan Liu","Yiqiang Cai","Yizhou Tan","Shengchen Li"],"year":2026,"doi":"10.21437/Interspeech.2026-656","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26f_interspeech.pdf","session":"Reasoning with Speech/Audio Language Models","topics":["speech-llm","evaluation","self-supervised"],"category":"audio-understanding","institutions":["Xi'an Jiaotong-Liverpool University"],"funding":["Jiangsu Provincial Major Science and Technology Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26f_interspeech","category":"audio-understanding","institutions":["Xi'an Jiaotong-Liverpool University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-656","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26f_interspeech/markdown.md"},{"id":"zhang26fa_interspeech","title":"MPA-KWS: Multi-Modal Phoneme-Level Alignment for Streaming Open-Vocabulary Keyword Spotting","authors":["Jue Zhang","Guibin Zheng","Jiarui Zhang","Jiqing Han","Chenhao Jing"],"year":2026,"doi":"10.21437/Interspeech.2026-2485","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26fa_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26fa_interspeech.pdf","session":"Information Extraction and Retrieval","topics":["keyword-spotting","self-supervised","on-device"],"category":"asr","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Harbin Institute of Technology"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26fa_interspeech","category":"asr","labels":["efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Harbin Institute of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2485","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26fa_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26fa_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26fa_interspeech/markdown.md"},{"id":"zhang26g_interspeech","title":"Prosodic Realization of Focus in Yi-Mandarin Bilingual Speakers: On-Focus Expansion without Post-Focus Compression","authors":["Ziyu Zhang","Chenyu Li"],"year":2026,"doi":"10.21437/Interspeech.2026-660","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26g_interspeech.pdf","session":"Prominence, Stress and Focus","topics":["paralinguistics","prosody","low-resource"],"category":"phonetics-linguistics","labels":["low-resource","multilingual"],"institutions":["Flinders University","Johns Hopkins University"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26g_interspeech","category":"phonetics-linguistics","labels":["low-resource","multilingual"],"institutions":["Flinders University","Johns Hopkins University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-660","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26g_interspeech/markdown.md"},{"id":"zhang26ga_interspeech","title":"A Dual-Stream Discrete Neural Codec with Fixed-Length Global Speaker Tokens and Dynamic Frame Rates for Low-Bitrate Speech Tokenization","authors":["Boyang Zhang","Yechang Huang","Xuerui Yang","Ziyue Jiang"],"year":2026,"doi":"10.21437/Interspeech.2026-3314","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26ga_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26ga_interspeech.pdf","session":"Neural Audio Codec Architectures","topics":["speech-coding","self-supervised","voice-conversion"],"category":"speech-coding","labels":["generative-model"],"institutions":["Zhejiang University","StepFun"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26ga_interspeech","category":"speech-coding","labels":["generative-model"],"institutions":["Zhejiang University","StepFun"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3314","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26ga_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ga_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ga_interspeech/markdown.md"},{"id":"zhang26h_interspeech","title":"Gender differences in the phonetic realization of the checked tone in Kaihui Xiang","authors":["Yi Zhang","Aini Li"],"year":2026,"doi":"10.21437/Interspeech.2026-691","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26h_interspeech.pdf","session":"Gender- and Age-Related Speech Characteristics","topics":["phonetics","prosody","multilingual"],"category":"phonetics-linguistics","institutions":["City University of Hong Kong"],"funding":["City University of Hong Kong"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26h_interspeech","category":"phonetics-linguistics","institutions":["City University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-691","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26h_interspeech/markdown.md"},{"id":"zhang26ha_interspeech","title":"Margin-Aware Contrastive Regularization for Robust Streaming Keyword Spotting under Strict False-Alarm Constraints","authors":["Hanwen Zhang","Guosong Zhu","Zhen Qin"],"year":2026,"doi":"10.21437/Interspeech.2026-3545","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26ha_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26ha_interspeech.pdf","session":"ASR Under Real-World Constraints: Streaming, Adaptation, and Efficiency","topics":["keyword-spotting","self-supervised","on-device"],"category":"asr","labels":["streaming-real-time"],"institutions":["Network and Data Security Key Laboratory of Sichuan Province","University of Electronic Science and Technology of China"],"funding":["National Natural Science Foundation of China","Sichuan Science and Technology Support Plan"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26ha_interspeech","category":"asr","labels":["streaming-real-time"],"institutions":["Network and Data Security Key Laboratory of Sichuan Province","University of Electronic Science and Technology of China"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3545","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26ha_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ha_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ha_interspeech/markdown.md"},{"id":"zhang26i_interspeech","title":"Tense Voice, Not Falsetto: An F0-specific Physiological Byproduct of Extreme High-Pitch Tone in Kaihui Xiang","authors":["Yi Zhang","Aini Li"],"year":2026,"doi":"10.21437/Interspeech.2026-693","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26i_interspeech.pdf","session":"Voice Quality Aspects of Speech","topics":["phonetics","prosody","paralinguistics"],"category":"phonetics-linguistics","institutions":["City University of Hong Kong"],"funding":["CityU StartUp Grant"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26i_interspeech","category":"phonetics-linguistics","institutions":["City University of Hong Kong"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-693","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26i_interspeech/markdown.md"},{"id":"zhang26ia_interspeech","title":"Mitigating Causality Mismatch with Causal Temporal Relation Distillation for Streaming Keyword Spotting","authors":["Hanwen Zhang","Guosong Zhu","Zhen Qin"],"year":2026,"doi":"10.21437/Interspeech.2026-3546","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26ia_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26ia_interspeech.pdf","session":"ASR Under Real-World Constraints: Streaming, Adaptation, and Efficiency","topics":["keyword-spotting","self-supervised","on-device"],"category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Network and Data Security Key Laboratory of Sichuan Province","University of Electronic Science and Technology of China"],"funding":["National Natural Science Foundation of China","Sichuan Science and Technology Support Plan"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26ia_interspeech","category":"asr","labels":["efficient-on-device","streaming-real-time"],"institutions":["Network and Data Security Key Laboratory of Sichuan Province","University of Electronic Science and Technology of China"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3546","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26ia_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ia_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26ia_interspeech/markdown.md"},{"id":"zhang26j_interspeech","title":"Improved modeling of vocal fold contacting and de-contacting in a geometric vocal fold model","authors":["Tianyi Zhang","Zihao Huang","Peter Birkholz"],"year":2026,"doi":"10.21437/Interspeech.2026-753","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26j_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26j_interspeech.pdf","session":"Speech Production and Perception 1","topics":["tts","phonetics","prosody"],"category":"tts","institutions":["TU Dresden"],"funding":["German Research Foundation","German Federal Ministry for Economic Affairs and Energy","Zentrales Innovationsprogramm Mittelstand"],"code":{"url":"https://www.vocaltractlab.de/index.php?page=birkholz-supplements","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26j_interspeech","category":"tts","institutions":["TU Dresden"],"code":"https://www.vocaltractlab.de/index.php?page=birkholz-supplements","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-753","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26j_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26j_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26j_interspeech/markdown.md"},{"id":"zhang26k_interspeech","title":"WeSep: A Modular and Cue-Composable Framework for Target Speaker Extraction","authors":["Ke Zhang","Xiaoyang Yu","Haoyu Li","Shuai Wang","Shuhan Zhang","Haizhou Li"],"year":2026,"doi":"10.21437/Interspeech.2026-784","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26k_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26k_interspeech.pdf","session":"Source Separation 1","topics":["speech-enhancement","self-supervised","multilingual"],"category":"enhancement-separation","institutions":["Chinese University of Hong Kong, Shenzhen","Nanjing University","Shenzhen Loop Area Institute"],"funding":["National Natural Science Foundation of China","Yangtze River Delta Science and Technology Innovation Community Joint Research Project","Shenzhen Science and Technology Program","Program for Guangdong Introducing Innovative and Enterpreneurial Teams","Shenzhen Stability Science Program"],"code":{"url":"https://github.com/wenet-e2e/WeSep","stars":317,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26k_interspeech","category":"enhancement-separation","institutions":["Chinese University of Hong Kong, Shenzhen","Nanjing University","Shenzhen Loop Area Institute"],"code":"https://github.com/wenet-e2e/WeSep","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-784","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26k_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26k_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26k_interspeech/markdown.md"},{"id":"zhang26l_interspeech","title":"CTMusic: A Traditional Chinese Instrumental Music Dataset Towards Text-to-Music Generation","authors":["Zixing Zhang","Yimin Cao","Haotian Guo","Bin Wang","Jing Han"],"year":2026,"doi":"10.21437/Interspeech.2026-882","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26l_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26l_interspeech.pdf","session":"Singing Voice and Music Generation","topics":["tts","self-supervised","dataset"],"category":"audio-understanding","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Hunan University","Yuelushan Center for Industrial Innovation"],"funding":["National Natural Science Foundation of China","National Science and Technology Major Project of China","Science and Technology Innovation Program of Hunan Province","Guangdong Basic and Applied Basic Research Foundation","Shenzhen Natural Science Foundation","Project of Yuelushan Center for Industrial Innovation"],"code":{"url":"https://frei-2.github.io/CTMusic","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26l_interspeech","category":"audio-understanding","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Hunan University","Yuelushan Center for Industrial Innovation"],"code":"https://frei-2.github.io/CTMusic","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-882","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26l_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26l_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26l_interspeech/markdown.md"},{"id":"zhang26m_interspeech","title":"Poly-InstructTTS: Learning In-the-Wild Expressive Speech Synthesis from Open-Ended Instructions","authors":["Junhui Zhang","Qianhui Xu","Qingxiang Guo","Dawei Yang","Ling Miao","Qiangqiang Wang","Yang Song"],"year":2026,"doi":"10.21437/Interspeech.2026-930","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26m_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26m_interspeech.pdf","session":"Instruction-following and Controllable Speech Synthesis","topics":["tts","speech-llm","self-supervised"],"category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["ZuoYeBang Technology"],"code":{"url":"https://zhangjh915.github.io/PolyInstructTTS-demo/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26m_interspeech","category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["ZuoYeBang Technology"],"code":"https://zhangjh915.github.io/PolyInstructTTS-demo/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-930","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26m_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26m_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26m_interspeech/markdown.md"},{"id":"zhang26n_interspeech","title":"SE-AGCNet: An End-to-End Framework for Joint Speech Enhancement and Loudness Control in Meeting Scenarios","authors":["Jinming Zhang","Wei Rao","Xionghu Zhong","Eng Siong Chng"],"year":2026,"doi":"10.21437/Interspeech.2026-1023","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26n_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26n_interspeech.pdf","session":"SE Architectures, Adaptation and Audio Front-Ends","topics":["speech-enhancement","asr","dataset"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Zhejiang University","Nanyang Technological University","Hunan University"],"code":{"url":"https://jinming00.github.io/SE-AGCNet/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26n_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Zhejiang University","Nanyang Technological University","Hunan University"],"code":"https://jinming00.github.io/SE-AGCNet/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1023","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26n_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26n_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26n_interspeech/markdown.md"},{"id":"zhang26o_interspeech","title":"Rubric-Aligned Disentangled Evaluation of Human Simultaneous Interpreting","authors":["Ziyu Zhang","Satoshi Nakamura"],"year":2026,"doi":"10.21437/Interspeech.2026-1105","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26o_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26o_interspeech.pdf","session":"Translation","topics":["speech-translation","evaluation","self-supervised"],"category":"resources-evaluation","labels":["multilingual"],"institutions":["Chinese University of Hong Kong","Shenzhen Loop Area Institute"],"funding":["National Natural Science Foundation of China","Program for Guangdong Introducing Innovative and Entrepreneurial Teams"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26o_interspeech","category":"resources-evaluation","labels":["multilingual"],"institutions":["Chinese University of Hong Kong","Shenzhen Loop Area Institute"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1105","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26o_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26o_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26o_interspeech/markdown.md"},{"id":"zhang26p_interspeech","title":"MultiAPI Spoof: A Multi-API Dataset and Local-Attention Network for Speech Anti-spoofing Detection","authors":["Xueping Zhang","Zhenshan Zhang","Yechen Wang","Linxi Li","Liwei Jin","Ming Li"],"year":2026,"doi":"10.21437/Interspeech.2026-1187","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26p_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26p_interspeech.pdf","session":"Spoofing and Deepfake Detection 1","topics":["speech-anti-spoofing","self-supervised","dataset"],"category":"deepfake-security","labels":["dataset-or-benchmark-release"],"institutions":["Duke Kunshan University","Chinese University of Hong Kong, Shenzhen","OfSpectrum"],"code":{"url":"https://github.com/XuepingZhang/MultiAPI-Spoof","stars":1,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26p_interspeech","category":"deepfake-security","labels":["dataset-or-benchmark-release"],"institutions":["Duke Kunshan University","Chinese University of Hong Kong, Shenzhen","OfSpectrum"],"code":"https://github.com/XuepingZhang/MultiAPI-Spoof","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1187","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26p_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26p_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26p_interspeech/markdown.md"},{"id":"zhang26q_interspeech","title":"Age-related Differences in Acoustic Realization of Aspirated Fricatives in Shaxi Bai","authors":["Xiaofang Zhang","Infat Lo","Yao Lu"],"year":2026,"doi":"10.21437/Interspeech.2026-1200","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26q_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26q_interspeech.pdf","session":"Gender- and Age-Related Speech Characteristics","topics":["phonetics","prosody","dataset"],"category":"phonetics-linguistics","institutions":["University of Macau","Peking University"],"funding":["National Social Science Fund of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26q_interspeech","category":"phonetics-linguistics","institutions":["University of Macau","Peking University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1200","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26q_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26q_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26q_interspeech/markdown.md"},{"id":"zhang26r_interspeech","title":"PRISM: Prosody-Integrated Multi-Agent Reasoning Framework for Empathetic Spoken Dialogue","authors":["Wen Zhang","Xiaocui Yang","Zhuoyue Gao","Shi Feng","Daling Wang","Yifei Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1214","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26r_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26r_interspeech.pdf","session":"Empathetic Dialogue and Interaction Dynamics","topics":["speech-llm","tts","asr"],"category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["Northeastern University"],"funding":["National Natural Science Foundation of China","Fundamental Research Funds for the Central Universities"],"code":{"url":"https://github.com/Bxzfrm/PRISM","stars":2,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26r_interspeech","category":"speech-llm-dialogue","labels":["generative-model"],"institutions":["Northeastern University"],"code":"https://github.com/Bxzfrm/PRISM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1214","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26r_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26r_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26r_interspeech/markdown.md"},{"id":"zhang26s_interspeech","title":"DASR-CPO: Reference-Free Contrastive Preference Optimization for Correcting Mandarin Semantic Drift in Low-Resource Chinese Dialect ASR","authors":["Tao Zhang","haiyang Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1228","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26s_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26s_interspeech.pdf","session":"Language and Dialect Recognition","topics":["asr","low-resource","self-supervised"],"category":"asr","labels":["low-resource"],"institutions":["Beijing University of Posts and Telecommunications"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26s_interspeech","category":"asr","labels":["low-resource"],"institutions":["Beijing University of Posts and Telecommunications"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1228","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26s_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26s_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26s_interspeech/markdown.md"},{"id":"zhang26t_interspeech","title":"EChO-Agent: Evidence Chain Orchestration Agent for Audio Reasoning","authors":["Siyuan Zhang","Jian Zong","Junyu Wang","Peiyuan Jiang","Jiahao Yan","Jingyu Zhang","Tianrui Wang","Xiaobao Wang","Longbiao Wang","Jianwu Dang"],"year":2026,"doi":"10.21437/Interspeech.2026-1313","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26t_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26t_interspeech.pdf","session":"Challenge - Audio Reasoning Challenge","topics":["speech-llm","spoken-language-understanding","evaluation"],"category":"speech-llm-dialogue","institutions":["Tianjin University","Chinese Academy of Sciences"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26t_interspeech","category":"speech-llm-dialogue","institutions":["Tianjin University","Chinese Academy of Sciences"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1313","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26t_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26t_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26t_interspeech/markdown.md"},{"id":"zhang26u_interspeech","title":"Learning to Wait: Real Streaming Speech-to-Text Translation with an LLM","authors":["Shucong Zhang","Titouan Parcollet","Rogier van Dalen"],"year":2026,"doi":"10.21437/Interspeech.2026-1323","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26u_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26u_interspeech.pdf","session":"Multilingual Speech 2","topics":["speech-translation","self-supervised","speech-llm"],"category":"translation","labels":["multilingual","efficient-on-device","streaming-real-time"],"institutions":["Samsung"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26u_interspeech","category":"translation","labels":["multilingual","efficient-on-device","streaming-real-time"],"institutions":["Samsung"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1323","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26u_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26u_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26u_interspeech/markdown.md"},{"id":"zhang26v_interspeech","title":"Larynx segmentation in mid-sagittal speech production real-time MRI","authors":["Yubin Zhang","Xuan Shi","Kevin Huang","Prakash Kumar","Kevin Lee","Louis Goldstein","Krishna Nayak","Shrikanth Narayanan"],"year":2026,"doi":"10.21437/Interspeech.2026-1402","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26v_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26v_interspeech.pdf","session":"Tools and Techniques for Phonetic Analysis","topics":["speech-production","phonetics","self-supervised"],"category":"phonetics-linguistics","institutions":["University of Southern California"],"funding":["National Science Foundation"],"code":{"url":"https://github.com/pkuzyb/larynx_segmentation","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26v_interspeech","category":"phonetics-linguistics","institutions":["University of Southern California"],"code":"https://github.com/pkuzyb/larynx_segmentation","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1402","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26v_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26v_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26v_interspeech/markdown.md"},{"id":"zhang26w_interspeech","title":"BACH: Benchmarking Audio Codecs for Bio-Acoustic Health","authors":["Zixing Zhang","Xiaojun Mo","Zhongren Dong","Bin Wang","Jing Han"],"year":2026,"doi":"10.21437/Interspeech.2026-1588","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26w_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26w_interspeech.pdf","session":"Multimodal and Non-Speech Healthcare Applications","topics":["speech-enhancement","evaluation","health"],"category":"speech-coding","labels":["dataset-or-benchmark-release"],"institutions":["Hunan University","Xiaomi","Yuelushan Center for Industrial Innovation"],"funding":["Beijing Xiaomi Mobile Software Co., Ltd","National Natural Science Foundation of China","National Science and Technology Major Project of China","Science and Technology Innovation Program of Hunan Province"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26w_interspeech","category":"speech-coding","labels":["dataset-or-benchmark-release"],"institutions":["Hunan University","Xiaomi","Yuelushan Center for Industrial Innovation"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1588","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26w_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26w_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26w_interspeech/markdown.md"},{"id":"zhang26x_interspeech","title":"VoxEffects: A Speech-Oriented Audio Effects Dataset and Benchmark","authors":["Zhe Zhang","Yigitcan Özer","Junichi Yamagishi"],"year":2026,"doi":"10.21437/Interspeech.2026-1621","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26x_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26x_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["speech-enhancement","evaluation","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["National Institute of Informatics"],"funding":["New Energy and Industrial Technology Development Organization"],"code":{"url":"https://github.com/nii-yamagishilab/VoxEffects","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26x_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release","robustness-noise"],"institutions":["National Institute of Informatics"],"code":"https://github.com/nii-yamagishilab/VoxEffects","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1621","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26x_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26x_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26x_interspeech/markdown.md"},{"id":"zhang26y_interspeech","title":"SoniSpeech: A Large-Scale Open-Vocabulary Tri-Modal Dataset for Wearable Silent Speech Interfaces","authors":["Ruidong Zhang","Jiacheng Liu","François Guimbretière","Cheng Zhang"],"year":2026,"doi":"10.21437/Interspeech.2026-1625","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26y_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26y_interspeech.pdf","session":"Assistive Technologies 1","topics":["asr","self-supervised","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Cornell University"],"funding":["National Science Foundation","Qualcomm Innovation Fellowship"],"code":{"url":"https://doi.org/10.7298/xjjr-9m85","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26y_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Cornell University"],"code":"https://doi.org/10.7298/xjjr-9m85","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1625","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26y_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26y_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26y_interspeech/markdown.md"},{"id":"zhang26z_interspeech","title":"Time-Unconditional Generative Speech Enhancement via Autonomous Rectified Flow","authors":["Wen Zhang","Wenbin Jiang","Yang Zhang","Xiaofei Zhou"],"year":2026,"doi":"10.21437/Interspeech.2026-1679","isca_url":"https://www.isca-archive.org/interspeech_2026/zhang26z_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhang26z_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","generative-model"],"institutions":["Hangzhou Dianzi University"],"funding":["Yangtze River Delta Science and Technology Innovation Community Joint Research","Zhejiang Provincial Natural Science Foundation of China"],"code":{"url":"https://github.com/zhangwen0821/ARFSE.git","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhang26z_interspeech","category":"enhancement-separation","labels":["efficient-on-device","generative-model"],"institutions":["Hangzhou Dianzi University"],"code":"https://github.com/zhangwen0821/ARFSE.git","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1679","pdf":"https://www.isca-archive.org/interspeech_2026/zhang26z_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26z_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhang26z_interspeech/markdown.md"},{"id":"zhao26_interspeech","title":"TF-MossFormer: Integrating Convolution Gated Local-Global Attentions for Enhanced Time-Frequency Domain Monaural Speech Separation","authors":["Shengkui Zhao","Zexu Pan","Haoxu Wang","Biao Tian","Bin Ma","Xiangang Li"],"year":2026,"doi":"10.21437/Interspeech.2026-218","isca_url":"https://www.isca-archive.org/interspeech_2026/zhao26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhao26_interspeech.pdf","session":"Target Speaker Extraction, Speech Separation and Audio Understanding","topics":["source-separation","speech-enhancement","self-supervised"],"category":"enhancement-separation","institutions":["Alibaba Group"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhao26_interspeech","category":"enhancement-separation","institutions":["Alibaba Group"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-218","pdf":"https://www.isca-archive.org/interspeech_2026/zhao26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26_interspeech/markdown.md"},{"id":"zhao26b_interspeech","title":"F0 realization of prosodic focus across adulthood in Jianghuai Mandarin","authors":["Xinxian Zhao","Xiaohu Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-255","isca_url":"https://www.isca-archive.org/interspeech_2026/zhao26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhao26b_interspeech.pdf","session":"Prominence, Stress and Focus","topics":["prosody","phonetics","evaluation"],"category":"phonetics-linguistics","institutions":["Shanghai Jiao Tong University","National Research Center for Language and Well-being","Tongji University"],"funding":["China Postdoctoral Science Foundation"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhao26b_interspeech","category":"phonetics-linguistics","institutions":["Shanghai Jiao Tong University","National Research Center for Language and Well-being","Tongji University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-255","pdf":"https://www.isca-archive.org/interspeech_2026/zhao26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26b_interspeech/markdown.md"},{"id":"zhao26c_interspeech","title":"HALO: Half-Frame-Rate Adaptive Learnable Operator for Lightweight STFT-Based Speech Enhancement","authors":["Jiadong Zhao","Dahan Wang","Yu Sun","Leyan Yang","Xiaobin Rong","Shiruo Sun","Yuxiang Hu","Jing Lu"],"year":2026,"doi":"10.21437/Interspeech.2026-601","isca_url":"https://www.isca-archive.org/interspeech_2026/zhao26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhao26c_interspeech.pdf","session":"Real-Time, Low-Latency and Edge Speech Enhancement","topics":["speech-enhancement","self-supervised","on-device"],"category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Nanjing University","Horizon Robotics","Samsung Electronics"],"funding":["National Natural Science Foundation of China","AI & AI for Science Project of Nanjing University"],"code":{"url":"https://github.com/dddaniel-z/HALO","stars":21,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhao26c_interspeech","category":"enhancement-separation","labels":["efficient-on-device","streaming-real-time"],"institutions":["Nanjing University","Horizon Robotics","Samsung Electronics"],"code":"https://github.com/dddaniel-z/HALO","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-601","pdf":"https://www.isca-archive.org/interspeech_2026/zhao26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26c_interspeech/markdown.md"},{"id":"zhao26d_interspeech","title":"Speech-Worthy Alignment for Japanese SpeechLLMs via Direct Preference Optimization","authors":["Mengjie Zhao","Lianbo Liu","Yusuke Fujita","Hao Shi","Yuan Gao","Roman Koshkin","Yui Sudo"],"year":2026,"doi":"10.21437/Interspeech.2026-976","isca_url":"https://www.isca-archive.org/interspeech_2026/zhao26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhao26d_interspeech.pdf","session":"Text Processing for Speech Synthesis","topics":["speech-llm","spoken-language-understanding","multilingual"],"category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["SB Intuitions"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhao26d_interspeech","category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["SB Intuitions"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-976","pdf":"https://www.isca-archive.org/interspeech_2026/zhao26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26d_interspeech/markdown.md"},{"id":"zhao26e_interspeech","title":"Decoding Order Matters in Autoregressive Speech Synthesis","authors":["Minghui Zhao","Anton Ragni"],"year":2026,"doi":"10.21437/Interspeech.2026-1339","isca_url":"https://www.isca-archive.org/interspeech_2026/zhao26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhao26e_interspeech.pdf","session":"LLM Based Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["generative-model"],"institutions":["University of Sheffield"],"funding":["UK Research and Innovation","UKRI AI Centre for Doctoral Training in Speech and Language Technologies (SLT) and their Applications"],"code":{"url":"https://minghuizhao39.github.io/sample-page-order/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhao26e_interspeech","category":"tts","labels":["generative-model"],"institutions":["University of Sheffield"],"code":"https://minghuizhao39.github.io/sample-page-order/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1339","pdf":"https://www.isca-archive.org/interspeech_2026/zhao26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26e_interspeech/markdown.md"},{"id":"zhao26f_interspeech","title":"SA-HRTF: A Sound-Assisted Approach to Personalized HRTF Modeling","authors":["Qingying Zhao","Siyuan Chen","De Hu"],"year":2026,"doi":"10.21437/Interspeech.2026-1995","isca_url":"https://www.isca-archive.org/interspeech_2026/zhao26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhao26f_interspeech.pdf","session":"Spatial Audio 3","topics":["spatial-audio","self-supervised","evaluation"],"category":"applications-other","labels":["low-resource"],"institutions":["Inner Mongolia University"],"funding":["Natural Science Foundation of Inner Mongolia Autonomous Region","National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhao26f_interspeech","category":"applications-other","labels":["low-resource"],"institutions":["Inner Mongolia University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1995","pdf":"https://www.isca-archive.org/interspeech_2026/zhao26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26f_interspeech/markdown.md"},{"id":"zhao26g_interspeech","title":"MSpoofTTS: Multi-Resolution Spoof-Guided Inference for Discrete Speech Synthesis","authors":["Junchuan Zhao","Minh Duc Vu","Ye Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-2159","isca_url":"https://www.isca-archive.org/interspeech_2026/zhao26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhao26g_interspeech.pdf","session":"Spoofing and Deepfake Detection 3","topics":["tts","speech-llm","self-supervised"],"category":"tts","labels":["generative-model"],"institutions":["National University of Singapore"],"code":{"url":"https://github.com/neuphonic/neutts","stars":6296,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhao26g_interspeech","category":"tts","labels":["generative-model"],"institutions":["National University of Singapore"],"code":"https://github.com/neuphonic/neutts","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2159","pdf":"https://www.isca-archive.org/interspeech_2026/zhao26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26g_interspeech/markdown.md"},{"id":"zhao26h_interspeech","title":"Beyond Semantic Dominance: Cognitive Affective Reasoning and Empathetic Response Alignment in Audio Language Models","authors":["Zhixian Zhao","Shuiyuan Wang","Wenjie Tian","Jingbin Hu","Ziyu Zhang","Lei Xie"],"year":2026,"doi":"10.21437/Interspeech.2026-2400","isca_url":"https://www.isca-archive.org/interspeech_2026/zhao26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhao26h_interspeech.pdf","session":"Reasoning with Speech/Audio Language Models","topics":["speech-llm","paralinguistics","emotion-recognition"],"category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release"],"institutions":["Northwestern Polytechnical University"],"code":{"url":"https://github.com/zxzhao0/CogAudio-LLM","stars":5,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhao26h_interspeech","category":"speech-llm-dialogue","labels":["dataset-or-benchmark-release"],"institutions":["Northwestern Polytechnical University"],"code":"https://github.com/zxzhao0/CogAudio-LLM","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2400","pdf":"https://www.isca-archive.org/interspeech_2026/zhao26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26h_interspeech/markdown.md"},{"id":"zhao26i_interspeech","title":"Towards Robust Generative Speech Enhancement Using Vector Quantisation-Based Neural Audio Codec","authors":["Haixin Zhao","Nilesh Madhu"],"year":2026,"doi":"10.21437/Interspeech.2026-2564","isca_url":"https://www.isca-archive.org/interspeech_2026/zhao26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhao26i_interspeech.pdf","session":"Language-Model and Codec-Token Speech Enhancement","topics":["speech-enhancement","self-supervised","generative-model"],"category":"enhancement-separation","labels":["generative-model"],"institutions":["Ghent University","imec"],"code":{"url":"https://aspire.ugent.be/demos/INTERSPEECH2026HZ/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhao26i_interspeech","category":"enhancement-separation","labels":["generative-model"],"institutions":["Ghent University","imec"],"code":"https://aspire.ugent.be/demos/INTERSPEECH2026HZ/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2564","pdf":"https://www.isca-archive.org/interspeech_2026/zhao26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhao26i_interspeech/markdown.md"},{"id":"zheng26_interspeech","title":"CycleCodec: Distillation-Free Factorized Neural Speech Codec via Cycle-Consistent Speaker Swapping","authors":["Rui-Chen Zheng","Nicholas Sanders","Jinzuomu Zhong","Yang Ai","Zhen-Hua Ling","Korin Richmond"],"year":2026,"doi":"10.21437/Interspeech.2026-806","isca_url":"https://www.isca-archive.org/interspeech_2026/zheng26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zheng26_interspeech.pdf","session":"Speech Synthesis: Speech Features, Codec and Representations","topics":["speech-coding","voice-conversion","self-supervised"],"category":"speech-coding","labels":["multilingual","generative-model"],"institutions":["University of Science and Technology of China","University of Edinburgh"],"funding":["National Natural Science Foundation of China","Speech Generation for Indigenous Language Education project"],"code":{"url":"https://zhengrachel.github.io/CycleCodec/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zheng26_interspeech","category":"speech-coding","labels":["multilingual","generative-model"],"institutions":["University of Science and Technology of China","University of Edinburgh"],"code":"https://zhengrachel.github.io/CycleCodec/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-806","pdf":"https://www.isca-archive.org/interspeech_2026/zheng26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zheng26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zheng26_interspeech/markdown.md"},{"id":"zheng26b_interspeech","title":"Balancing ASR and diarization in end-to-end LLMs for multi-talker speech recognition","authors":["Naijun Zheng","Yuke Lin","Sanli Tian","Mengtian Li","Zhiwei Lin","Longshuai Xiao","Dandan Tu"],"year":2026,"doi":"10.21437/Interspeech.2026-1124","isca_url":"https://www.isca-archive.org/interspeech_2026/zheng26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zheng26b_interspeech.pdf","session":"Robust ASR: Hallucinations and Biases","topics":["asr","speaker-diarization","speech-llm"],"category":"asr","institutions":["Huawei Technologies"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zheng26b_interspeech","category":"asr","institutions":["Huawei Technologies"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1124","pdf":"https://www.isca-archive.org/interspeech_2026/zheng26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zheng26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zheng26b_interspeech/markdown.md"},{"id":"zheng26c_interspeech","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","authors":["Zhisheng Zheng","Xiaohang Sun","Zhu Liu","Caren Chen","Rohith Kumar","Manoj Aggarwal","Gerard Medioni","David Harwath"],"year":2026,"doi":"10.21437/Interspeech.2026-1760","isca_url":"https://www.isca-archive.org/interspeech_2026/zheng26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zheng26c_interspeech.pdf","session":"Controllable and Expressive Speech Synthesis","topics":["tts","self-supervised","prosody"],"category":"tts","labels":["generative-model"],"institutions":["University of Texas at Austin","Amazon"],"funding":["Amazon.com"],"code":{"url":"https://www.modelscope.cn/models/iic/CosyVoice-300M/file/view/master/campplus.onnx","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zheng26c_interspeech","category":"tts","labels":["generative-model"],"institutions":["University of Texas at Austin","Amazon"],"code":"https://www.modelscope.cn/models/iic/CosyVoice-300M/file/view/master/campplus.onnx","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1760","pdf":"https://www.isca-archive.org/interspeech_2026/zheng26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zheng26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zheng26c_interspeech/markdown.md"},{"id":"zhong26_interspeech","title":"Uncovering Dimension-Specific Layer Preferences in Wav2Vec2 for Fine-Grained Perceptual Assessment of Dysarthric Speech","authors":["Zihan Zhong","Qianli Wang","Satwinder Singh","Clarion Mendes","Mark Hasegawa-Johnson","Waleed Abdulla","Seyed Reza Shahamiri"],"year":2026,"doi":"10.21437/Interspeech.2026-692","isca_url":"https://www.isca-archive.org/interspeech_2026/zhong26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhong26_interspeech.pdf","session":"Pathological Speech Assessment 4","topics":["paralinguistics","self-supervised","evaluation"],"category":"health-clinical","labels":["self-supervised"],"institutions":["DeepNet Discovery Network","University of Auckland","University of Illinois Urbana-Champaign"],"code":{"url":"https://github.com/Kanelmis/UDS","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhong26_interspeech","category":"health-clinical","labels":["self-supervised"],"institutions":["DeepNet Discovery Network","University of Auckland","University of Illinois Urbana-Champaign"],"code":"https://github.com/Kanelmis/UDS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-692","pdf":"https://www.isca-archive.org/interspeech_2026/zhong26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhong26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhong26_interspeech/markdown.md"},{"id":"zhong26b_interspeech","title":"A Benchmark for Early-stage Parkinson's Disease Detection from Speech","authors":["Terry Yi Zhong","Cristian Tejedor-Garcia","Khiet Truong","Janna Maas","Louis ten Bosch","Bastiaan R. Bloem"],"year":2026,"doi":"10.21437/Interspeech.2026-1057","isca_url":"https://www.isca-archive.org/interspeech_2026/zhong26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhong26b_interspeech.pdf","session":"Pathological Speech Assessment 4","topics":["paralinguistics","dataset","evaluation"],"category":"health-clinical","labels":["dataset-or-benchmark-release"],"institutions":["Radboud University","Radboud University Medical Center"],"funding":["Dutch Research Council","SURF"],"code":{"url":"https://github.com/terryyizhongru/B-EarlyPD-Speech","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhong26b_interspeech","category":"health-clinical","labels":["dataset-or-benchmark-release"],"institutions":["Radboud University","Radboud University Medical Center"],"code":"https://github.com/terryyizhongru/B-EarlyPD-Speech","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1057","pdf":"https://www.isca-archive.org/interspeech_2026/zhong26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhong26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhong26b_interspeech/markdown.md"},{"id":"zhong26c_interspeech","title":"Phoneme Error and Uncertainty Features for Interpretable Dysarthric Speech Assessment","authors":["Zihan Zhong","Qianli Wang","Satwinder Singh","Clarion Mendes","Mark Hasegawa-Johnson","Waleed Abdulla","Seyed Reza Shahamiri"],"year":2026,"doi":"10.21437/Interspeech.2026-1138","isca_url":"https://www.isca-archive.org/interspeech_2026/zhong26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhong26c_interspeech.pdf","session":"Speech and Language Technologies in Healthcare","topics":["paralinguistics","speech-enhancement","self-supervised"],"category":"health-clinical","institutions":["DeepNet Discovery Network","University of Auckland","University of Illinois Urbana-Champaign"],"code":{"url":"https://github.com/Kanelmis/PED","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhong26c_interspeech","category":"health-clinical","institutions":["DeepNet Discovery Network","University of Auckland","University of Illinois Urbana-Champaign"],"code":"https://github.com/Kanelmis/PED","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1138","pdf":"https://www.isca-archive.org/interspeech_2026/zhong26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhong26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhong26c_interspeech/markdown.md"},{"id":"zhong26d_interspeech","title":"Towards Personalized Federated Learning for Dysarthric Speech Recognition","authors":["Tao Zhong","Mengzhe Geng","Jiajun Deng","Shujie Hu","Xunying Liu"],"year":2026,"doi":"10.21437/Interspeech.2026-1559","isca_url":"https://www.isca-archive.org/interspeech_2026/zhong26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhong26d_interspeech.pdf","session":"Assistive Technologies 2","topics":["asr","speech-llm","low-resource"],"category":"asr","labels":["low-resource"],"institutions":["Chinese University of Hong Kong","National Research Council Canada"],"funding":["Hong Kong RGC GRF"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhong26d_interspeech","category":"asr","labels":["low-resource"],"institutions":["Chinese University of Hong Kong","National Research Council Canada"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1559","pdf":"https://www.isca-archive.org/interspeech_2026/zhong26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhong26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhong26d_interspeech/markdown.md"},{"id":"zhou26_interspeech","title":"Ouroboros: Self-Referential Backdoor Attacks on Speech Enhancement via Clean Audio Triggers","authors":["Yunjie Zhou","Yuheng Huang","Diqun Yan"],"year":2026,"doi":"10.21437/Interspeech.2026-455","isca_url":"https://www.isca-archive.org/interspeech_2026/zhou26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhou26_interspeech.pdf","session":"Spoofing and Deepfake Detection 1","topics":["speech-enhancement","evaluation","self-supervised"],"category":"deepfake-security","institutions":["Ningbo University","Ningbo University of Finance and Economics"],"funding":["National Natural Science Foundation of China","Zhejiang Provincial Collaborative Innovation Center for Digital Supply Chain and Artificial Intelligence of Bulk Commodities"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhou26_interspeech","category":"deepfake-security","institutions":["Ningbo University","Ningbo University of Finance and Economics"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-455","pdf":"https://www.isca-archive.org/interspeech_2026/zhou26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26_interspeech/markdown.md"},{"id":"zhou26b_interspeech","title":"Intonation Perception in Real and Synthetic Speech across Varying Familiarity Levels: A Pilot Study of Equivalence Assessment","authors":["Hanrui Zhou","Gaoyuan Zhang","Yixiang Chen","Yujie Xing","Feng Xu","Xurong Xie","Hui Chen"],"year":2026,"doi":"10.21437/Interspeech.2026-996","isca_url":"https://www.isca-archive.org/interspeech_2026/zhou26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhou26b_interspeech.pdf","session":"Phonetic Aspects of TTS and ASR Systems","topics":["evaluation","voice-conversion","prosody"],"category":"tts","labels":["generative-model"],"institutions":["Chinese Academy of Sciences","Capital Normal University"],"funding":["National Key R&D Program of China","NSFC","China Disabled Persons Federation","Youth Innovation Promotion Association CAS Grant","China Postdoctoral Science Foundation"],"code":{"url":"https://github.com/Plachtaa/seed-vc","stars":3893,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhou26b_interspeech","category":"tts","labels":["generative-model"],"institutions":["Chinese Academy of Sciences","Capital Normal University"],"code":"https://github.com/Plachtaa/seed-vc","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-996","pdf":"https://www.isca-archive.org/interspeech_2026/zhou26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26b_interspeech/markdown.md"},{"id":"zhou26c_interspeech","title":"UG-Bench: A Comprehensive Benchmark for Evaluating Large Audio-Language Models","authors":["Jiaming Zhou","Haoqin Sun","Hui Wang","Jinghua Zhao","Yuhang Jia","Shiyao Wang","Enzhi Wang","Shiwan Zhao","Yong Qin"],"year":2026,"doi":"10.21437/Interspeech.2026-1517","isca_url":"https://www.isca-archive.org/interspeech_2026/zhou26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhou26c_interspeech.pdf","session":"Evaluation of Speech and Audio Analysis","topics":["speech-llm","evaluation","dataset"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Nankai University"],"funding":["National Key R&D Program of China","NSF China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhou26c_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Nankai University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1517","pdf":"https://www.isca-archive.org/interspeech_2026/zhou26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26c_interspeech/markdown.md"},{"id":"zhou26d_interspeech","title":"BiSASV: Bidirectional Feature Modulation with Dual-Granularity Fusion for Spoofing-Robust ASV","authors":["Yiqun Zhou","Kong Aik Lee","Ji Liu","Longbiao Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-1532","isca_url":"https://www.isca-archive.org/interspeech_2026/zhou26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhou26d_interspeech.pdf","session":"Speaker Verification and Anti-Spoofing","topics":["speaker-verification","self-supervised"],"category":"speaker","institutions":["Tianjin University","Hong Kong Polytechnic University","Huiyan Technology (Tianjin) Co., Ltd"],"funding":["National Natural Science Foundation of China"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhou26d_interspeech","category":"speaker","institutions":["Tianjin University","Hong Kong Polytechnic University","Huiyan Technology (Tianjin) Co., Ltd"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1532","pdf":"https://www.isca-archive.org/interspeech_2026/zhou26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26d_interspeech/markdown.md"},{"id":"zhou26e_interspeech","title":"Beyond One-Size-Fits-All: Personalized and Culturally Adaptive Emotional TTS via Interactive Optimization of Individual Emotion Perception Spaces","authors":["Wangzixi Zhou","Bagus Tris Atmaja","Sakriani Sakti"],"year":2026,"doi":"10.21437/Interspeech.2026-1696","isca_url":"https://www.isca-archive.org/interspeech_2026/zhou26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhou26e_interspeech.pdf","session":"Emotional Speech Synthesis","topics":["tts","self-supervised","evaluation"],"category":"tts","labels":["multilingual","generative-model"],"institutions":["Nara Institute of Science and Technology"],"funding":["JSPS KAKENHI","JST NEXUS"],"code":{"url":"https://37integer.github.io/Beyond-One-Size-Fits-All/","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhou26e_interspeech","category":"tts","labels":["multilingual","generative-model"],"institutions":["Nara Institute of Science and Technology"],"code":"https://37integer.github.io/Beyond-One-Size-Fits-All/","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1696","pdf":"https://www.isca-archive.org/interspeech_2026/zhou26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26e_interspeech/markdown.md"},{"id":"zhou26f_interspeech","title":"Conflict-Aware Pseudo-Labeling via Acoustic Signals for Multi-Task Speech Emotion Recognition","authors":["Haojie Zhou","Shunfei Liang","Chengze Li","Zhizhong Bai","Yicheng Feng","Ning Wang"],"year":2026,"doi":"10.21437/Interspeech.2026-2118","isca_url":"https://www.isca-archive.org/interspeech_2026/zhou26f_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhou26f_interspeech.pdf","session":"Speech Emotion Recognition and Representation 2","topics":["speech-emotion-recognition","self-supervised","multitask-learning"],"category":"paralinguistics-emotion","institutions":["Jiangnan University"],"code":{"url":"https://github.com/sfxii/CAPL-SER","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhou26f_interspeech","category":"paralinguistics-emotion","institutions":["Jiangnan University"],"code":"https://github.com/sfxii/CAPL-SER","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2118","pdf":"https://www.isca-archive.org/interspeech_2026/zhou26f_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26f_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26f_interspeech/markdown.md"},{"id":"zhou26g_interspeech","title":"AV-SyncBench: Decoupled Benchmarking of Temporal and Semantic Audio-Visual Synchronization","authors":["Tianhong Zhou","Mingyang Han","Boyu Li","Yuxuan Jiang","Jiaxin Ye","Dongxiao Wang","Haoxiang Shi","Kunpeng Wang","Jun Song","Cheng Yu","Bo Zheng"],"year":2026,"doi":"10.21437/Interspeech.2026-2177","isca_url":"https://www.isca-archive.org/interspeech_2026/zhou26g_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhou26g_interspeech.pdf","session":"Audio-Visual Grounding, Synchronization & Video Understanding","topics":["self-supervised","evaluation","multimodal"],"category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Alibaba Group","Tsinghua University","Fudan University"],"code":{"url":"https://fgt7t6g.github.io/AV-SyncBench","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhou26g_interspeech","category":"resources-evaluation","labels":["dataset-or-benchmark-release"],"institutions":["Alibaba Group","Tsinghua University","Fudan University"],"code":"https://fgt7t6g.github.io/AV-SyncBench","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2177","pdf":"https://www.isca-archive.org/interspeech_2026/zhou26g_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26g_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26g_interspeech/markdown.md"},{"id":"zhou26h_interspeech","title":"FineCombo-TTS: Collaborative and Precise Controllable Speech Synthesis Using Text Descriptions and Reference Speech","authors":["Shuoyi Zhou","Yixuan Zhou","Peiji Yang","Yifan Hu","Yicheng Zhong","Zhisheng Wang","Zhiyong Wu"],"year":2026,"doi":"10.21437/Interspeech.2026-2280","isca_url":"https://www.isca-archive.org/interspeech_2026/zhou26h_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhou26h_interspeech.pdf","session":"Voice Editing","topics":["tts","self-supervised","speech-llm"],"category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Tsinghua University","Inner Mongolia University","Tencent"],"funding":["National Natural Science Foundation of China","National Social Science Foundation of China"],"code":{"url":"https://thuhcsi.github.io/interspeech2026-FineCombo-TTS","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhou26h_interspeech","category":"tts","labels":["dataset-or-benchmark-release","generative-model"],"institutions":["Tsinghua University","Inner Mongolia University","Tencent"],"code":"https://thuhcsi.github.io/interspeech2026-FineCombo-TTS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2280","pdf":"https://www.isca-archive.org/interspeech_2026/zhou26h_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26h_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26h_interspeech/markdown.md"},{"id":"zhou26i_interspeech","title":"Rethinking Speech Foundation Model Fine-tuning: Better SFT or Better Match?","authors":["Wangjin Zhou","Yizhou Zhang","Yichi Wang","Tatsuya Kawahara"],"year":2026,"doi":"10.21437/Interspeech.2026-2436","isca_url":"https://www.isca-archive.org/interspeech_2026/zhou26i_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhou26i_interspeech.pdf","session":"Audio Foundation Models and Generation","topics":["self-supervised","evaluation","speech-classification"],"category":"resources-evaluation","labels":["self-supervised"],"institutions":["Kyoto University"],"funding":["JST BOOST"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhou26i_interspeech","category":"resources-evaluation","labels":["self-supervised"],"institutions":["Kyoto University"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2436","pdf":"https://www.isca-archive.org/interspeech_2026/zhou26i_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26i_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhou26i_interspeech/markdown.md"},{"id":"zhu26_interspeech","title":"Content-Aware Dynamic Compression for Efffcient Speech Recognition based on Large Language Model","authors":["Huifeng Zhu","Bingqian Wang","Shaoxun Xiu","Xingqun Jiang","Bin Lv"],"year":2026,"doi":"10.21437/Interspeech.2026-230","isca_url":"https://www.isca-archive.org/interspeech_2026/zhu26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhu26_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 1","topics":["asr","speech-llm","self-supervised"],"category":"asr","labels":["efficient-on-device"],"institutions":["BOE Technology Group"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhu26_interspeech","category":"asr","labels":["efficient-on-device"],"institutions":["BOE Technology Group"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-230","pdf":"https://www.isca-archive.org/interspeech_2026/zhu26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhu26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhu26_interspeech/markdown.md"},{"id":"zhu26b_interspeech","title":"G-MaP-SE: Guided Speech Enhancement via GMM-Based Prior Matching","authors":["Yike Zhu","Ziqian Wang","Zikai Liu","Xingchen Li","Zhuangqi Chen","Xianjun Xia","Chuanzeng Huang","Lei Xie"],"year":2026,"doi":"10.21437/Interspeech.2026-2148","isca_url":"https://www.isca-archive.org/interspeech_2026/zhu26b_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhu26b_interspeech.pdf","session":"Generative and Self-Supervised Speech Enhancement","topics":["speech-enhancement","speaker-verification","self-supervised"],"category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Northwestern Polytechnical University"],"code":{"url":"https://github.com/Hello3orld/G-MaP-SE","stars":10,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhu26b_interspeech","category":"enhancement-separation","labels":["robustness-noise"],"institutions":["Northwestern Polytechnical University"],"code":"https://github.com/Hello3orld/G-MaP-SE","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2148","pdf":"https://www.isca-archive.org/interspeech_2026/zhu26b_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhu26b_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhu26b_interspeech/markdown.md"},{"id":"zhu26c_interspeech","title":"Token-Independent Language Representations for Low-Latency Configurable Multilingual Speech Recognition","authors":["Hongxu Zhu","Lahiru Samarakoon","Ivan Fung"],"year":2026,"doi":"10.21437/Interspeech.2026-2455","isca_url":"https://www.isca-archive.org/interspeech_2026/zhu26c_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhu26c_interspeech.pdf","session":"Cross-Lingual and Multilingual Speech Recognition 1","topics":["asr","multilingual","self-supervised"],"category":"asr","labels":["multilingual","efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Fano"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhu26c_interspeech","category":"asr","labels":["multilingual","efficient-on-device","self-supervised","streaming-real-time"],"institutions":["Fano"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-2455","pdf":"https://www.isca-archive.org/interspeech_2026/zhu26c_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhu26c_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhu26c_interspeech/markdown.md"},{"id":"zhu26d_interspeech","title":"DASM: Detecting AI-Synthetic Music via Authentic Manifold Deviation Modeling","authors":["Xinya Zhu","Mengyu Qiao","Wenqiang Li","Zhihui Yang"],"year":2026,"doi":"10.21437/Interspeech.2026-3245","isca_url":"https://www.isca-archive.org/interspeech_2026/zhu26d_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhu26d_interspeech.pdf","session":"Evaluation, Benchmarking, and Reliability of Audio Systems","topics":["audio-deepfake","self-supervised","evaluation"],"category":"deepfake-security","institutions":["Beijing Key Laboratory of Key Technologies for AI+ Domain Applications","North China University of Technology"],"funding":["NCUT Research Startup Fund","NCUT Education Teaching Reform Research Project Fund","NCUT Graduate Educational Reform Research Project"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhu26d_interspeech","category":"deepfake-security","institutions":["Beijing Key Laboratory of Key Technologies for AI+ Domain Applications","North China University of Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3245","pdf":"https://www.isca-archive.org/interspeech_2026/zhu26d_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhu26d_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhu26d_interspeech/markdown.md"},{"id":"zhu26e_interspeech","title":"OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models","authors":["Han Zhu","Lingxuan Ye","Wei Kang","Zengwei Yao","Liyong Guo","Fangjun Kuang","Zhifeng Han","Weiji Zhuang","Long Lin","Daniel Povey"],"year":2026,"doi":"10.21437/Interspeech.2026-3256","isca_url":"https://www.isca-archive.org/interspeech_2026/zhu26e_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zhu26e_interspeech.pdf","session":"Scaling and Zero-Shot Speech Synthesis","topics":["tts","multilingual","self-supervised"],"category":"tts","labels":["low-resource","multilingual","generative-model"],"institutions":["Xiaomi"],"code":{"url":"https://github.com/k2-fsa/OmniVoice","stars":14021,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zhu26e_interspeech","category":"tts","labels":["low-resource","multilingual","generative-model"],"institutions":["Xiaomi"],"code":"https://github.com/k2-fsa/OmniVoice","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3256","pdf":"https://www.isca-archive.org/interspeech_2026/zhu26e_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zhu26e_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zhu26e_interspeech/markdown.md"},{"id":"znotins26_interspeech","title":"Low-Resource Medical ASR for Rich Transcription in Latvian","authors":["Arturs Znotins","Normunds Gruzitis","Andris Rozentals","Maris Golubovskis","Mikelis Gulbis","Roberts Dargis"],"year":2026,"doi":"10.21437/Interspeech.2026-3469","isca_url":"https://www.isca-archive.org/interspeech_2026/znotins26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/znotins26_interspeech.pdf","session":"Multilingual, Cross-lingual & Low-Resource ASR","topics":["asr","speech-llm","low-resource"],"category":"asr","labels":["low-resource","self-supervised"],"institutions":["University of Latvia","Assistentis","DATI Group","Viroling Technology"],"funding":["European Union's Recovery and Resilience Facility"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"znotins26_interspeech","category":"asr","labels":["low-resource","self-supervised"],"institutions":["University of Latvia","Assistentis","DATI Group","Viroling Technology"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3469","pdf":"https://www.isca-archive.org/interspeech_2026/znotins26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/znotins26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/znotins26_interspeech/markdown.md"},{"id":"zorila26_interspeech","title":"From Noisy Speech to Accurate APIs: LLM-driven Embedding Steering for Resilient Tool Retrieval","authors":["Catalin Zorilă","Qingxiuxiong Dong","Rama Doddipatla"],"year":2026,"doi":"10.21437/Interspeech.2026-3291","isca_url":"https://www.isca-archive.org/interspeech_2026/zorila26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zorila26_interspeech.pdf","session":"Audio Language Models: Reasoning, Reliability, and Multimodal Understanding","topics":["speech-llm","self-supervised","dataset"],"category":"speech-llm-dialogue","labels":["generative-model","robustness-noise"],"institutions":["Toshiba"],"code":{"url":"","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zorila26_interspeech","category":"speech-llm-dialogue","labels":["generative-model","robustness-noise"],"institutions":["Toshiba"],"updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-3291","pdf":"https://www.isca-archive.org/interspeech_2026/zorila26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zorila26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zorila26_interspeech/markdown.md"},{"id":"zou26_interspeech","title":"Less is More: Boosting Bimodal Music Emotion Recognition with Adaptive Audio Sequence Compression","authors":["Dinghao Zou"],"year":2026,"doi":"10.21437/Interspeech.2026-1552","isca_url":"https://www.isca-archive.org/interspeech_2026/zou26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zou26_interspeech.pdf","session":"Audio signal analysis","topics":["paralinguistics","emotion-recognition","self-supervised"],"category":"paralinguistics-emotion","institutions":["Wuhan University of Science and Technology"],"funding":["National Natural Science Foundation of China"],"code":{"url":"https://anonymous.4open.science/r/poolingvq","license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zou26_interspeech","category":"paralinguistics-emotion","institutions":["Wuhan University of Science and Technology"],"code":"https://anonymous.4open.science/r/poolingvq","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-1552","pdf":"https://www.isca-archive.org/interspeech_2026/zou26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zou26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zou26_interspeech/markdown.md"},{"id":"zuazo26_interspeech","title":"MEG-to-MEG Transfer Learning and Cross-Task Speech/Silence Detection with Limited Data","authors":["Xabier de Zuazo","Vincenzo Verbeni","Eva Navas","Ibon Saratxaga","Mathieu Bourguignon","Nicola Molinaro"],"year":2026,"doi":"10.21437/Interspeech.2026-439","isca_url":"https://www.isca-archive.org/interspeech_2026/zuazo26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zuazo26_interspeech.pdf","session":"Brain Studies and Speech","topics":["self-supervised","speech-llm","evaluation"],"category":"asr","labels":["low-resource"],"institutions":["University of the Basque Country","Basque Center on Cognition, Brain and Language","Ikerbasque","Universite libre de Bruxelles","WEL Research Institute"],"funding":["Basque Government","Spanish Government","MICIU","AEI"],"code":{"url":"https://github.com/hitz-zentroa/meg-phone-decoding","stars":4,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zuazo26_interspeech","category":"asr","labels":["low-resource"],"institutions":["University of the Basque Country","Basque Center on Cognition, Brain and Language","Ikerbasque","Universite libre de Bruxelles","WEL Research Institute"],"code":"https://github.com/hitz-zentroa/meg-phone-decoding","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-439","pdf":"https://www.isca-archive.org/interspeech_2026/zuazo26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zuazo26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zuazo26_interspeech/markdown.md"},{"id":"zufle26_interspeech","title":"Do What I Say: A Spoken Prompt Dataset for Instruction-Following","authors":["Maike Züfle","Sara Papi","Fabian Retkowski","Szymon Mazurek","Marek Kasztelnik","Alexander Waibel","Luisa Bentivogli","Jan Niehues"],"year":2026,"doi":"10.21437/Interspeech.2026-685","isca_url":"https://www.isca-archive.org/interspeech_2026/zufle26_interspeech.html","pdf_url":"https://www.isca-archive.org/interspeech_2026/zufle26_interspeech.pdf","session":"Audio Language Models: Reasoning, Reliability, and Multimodal Understanding","topics":["speech-llm","evaluation","multilingual"],"category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Karlsruhe Institute of Technology","Fondazione Bruno Kessler","ACC Cyfronet AGH","AGH University of Krakow","Carnegie Mellon University"],"funding":["European Union","Volkswagen Foundation"],"code":{"url":"https://github.com/MaikeZuefle/DOWIS","stars":13,"license":""},"open_to_collaboration":false,"wiki_frontmatter":{"id":"zufle26_interspeech","category":"resources-evaluation","labels":["multilingual","dataset-or-benchmark-release"],"institutions":["Karlsruhe Institute of Technology","Fondazione Bruno Kessler","ACC Cyfronet AGH","AGH University of Krakow","Carnegie Mellon University"],"code":"https://github.com/MaikeZuefle/DOWIS","updated":"2026-09-29","confidence":"full-paper","digest":"v2","source":"https://doi.org/10.21437/Interspeech.2026-685","pdf":"https://www.isca-archive.org/interspeech_2026/zufle26_interspeech.pdf"},"wiki_url":"https://interspeech-2026-wiki.vercel.app/papers/zufle26_interspeech/","markdown_url":"https://interspeech-2026-wiki.vercel.app/papers/zufle26_interspeech/markdown.md"}]}