Publications
Full list, newest first. † denotes equal contribution.
-
Improving Text-to-Audio Instruction Following via Fine-Grained Feedback from Audio-Aware Large Language Models
TL;DRUses fine-grained feedback from audio-aware LLMs to make text-to-audio models follow instructions more faithfully.
@inproceedings{kuan2026improving, title = {Improving Text-to-Audio Instruction Following via Fine-Grained Feedback from Audio-Aware Large Language Models}, author = {Kuan, Chun-Yi and Kim, Siwon and Kim, Byeonggeun and Kim, Suyoun and Lu, Bo-Ru and Tang, Qingming and Gandhe, Ankur and Lee, Hung-yi and Kao, Chieh-Chi and Wang, Chao}, booktitle = {Interspeech 2026}, year = {2026}, url = {https://arxiv.org/abs/2607.13408}, } -
AQUA-Bench: Beyond Finding Answers to Knowing When There Are None in Audio Question Answering
TL;DRA benchmark that tests whether audio QA models know when a question has no answer, not just when they can find one.
@inproceedings{kuan2026aqua, title = {AQUA-Bench: Beyond Finding Answers to Knowing When There Are None in Audio Question Answering}, author = {Kuan, Chun-Yi and Lee, Hung-yi}, booktitle = {ICASSP 2026 -- 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, pages = {16262--16266}, year = {2026}, organization = {IEEE}, doi = {10.1109/ICASSP55912.2026.11460647}, url = {https://arxiv.org/abs/2601.12248}, } -
Game-Time: Evaluating Temporal Dynamics in Spoken Language Models
TL;DRBenchmarks whether spoken language models can handle timing, tempo, and synchronized speech in real-time conversation.
@inproceedings{chang2025game, title = {Game-Time: Evaluating Temporal Dynamics in Spoken Language Models}, author = {Chang, Kai-Wei and Hu, En-Pei and Kuan, Chun-Yi and Ren, Wenze and Chen, Wei-Chih and Lin, Guan-Ting and Tsao, Yu and Sun, Shao-Hua and Lee, Hung-yi and Glass, James}, equalcontrib = {Chang, Kai-Wei and Hu, En-Pei}, booktitle = {ICASSP 2026 -- 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, pages = {16302--16306}, year = {2026}, organization = {IEEE}, doi = {10.1109/ICASSP55912.2026.11464000}, url = {https://arxiv.org/abs/2509.26388}, } -
AQAScore: Evaluating Semantic Alignment in Text-to-Audio Generation via Audio Question Answering
TL;DRScores text-to-audio alignment from an audio-aware LLM's confidence in answering 'Yes' to targeted questions, catching fine-grained mismatches that similarity metrics like CLAPScore miss.
@article{kuan2026aqascore, title = {AQAScore: Evaluating Semantic Alignment in Text-to-Audio Generation via Audio Question Answering}, author = {Kuan, Chun-Yi and Chang, Kai-Wei and Lee, Hung-yi}, journal = {arXiv preprint arXiv:2601.14728}, year = {2026}, url = {https://arxiv.org/abs/2601.14728}, } -
Walking Through Uncertainty: An Empirical Study of Uncertainty Estimation for Audio-Aware Large Language Models
TL;DRThe systematic study of uncertainty estimation for audio-aware LLMs, finding that semantic and verification-based methods win on general reasoning but their advantage breaks down on hallucination and unanswerable-question benchmarks.
@article{kuan2026walking, title = {Walking Through Uncertainty: An Empirical Study of Uncertainty Estimation for Audio-Aware Large Language Models}, author = {Kuan, Chun-Yi and Huang, Wei-Ping and Lee, Hung-yi}, journal = {arXiv preprint arXiv:2604.25591}, year = {2026}, url = {https://arxiv.org/abs/2604.25591}, }
-
From Alignment to Advancement: Bootstrapping Audio-Language Alignment with Synthetic Data
TL;DRBootstraps audio–language alignment with synthetic data to push audio-aware LLMs from basic alignment toward stronger reasoning.
@article{kuan2025alignment, title = {From Alignment to Advancement: Bootstrapping Audio-Language Alignment with Synthetic Data}, author = {Kuan, Chun-Yi and Lee, Hung-yi}, journal = {IEEE Transactions on Audio, Speech and Language Processing}, year = {2025}, volume = {33}, pages = {4604--4619}, doi = {10.1109/TASLPRO.2025.3626233}, url = {https://arxiv.org/abs/2505.20166}, } -
Teaching Audio-Aware Large Language Models What Does Not Hear: Mitigating Hallucinations through Synthesized Negative Samples
TL;DRCurbs hallucinations in audio-aware LLMs by teaching them what is NOT in the audio using synthesized negative samples.
@inproceedings{kuan2025teaching, title = {Teaching Audio-Aware Large Language Models What Does Not Hear: Mitigating Hallucinations through Synthesized Negative Samples}, author = {Kuan, Chun-Yi and Lee, Hung-yi}, booktitle = {Interspeech 2025}, pages = {2073--2077}, year = {2025}, organization = {ISCA}, doi = {10.21437/Interspeech.2025-324}, url = {https://arxiv.org/abs/2505.14518}, } -
Gender Bias in Instruction-Guided Speech Synthesis Models
TL;DRAudits and quantifies gender bias in instruction-guided speech synthesis models.
@inproceedings{kuan2025gender, title = {Gender Bias in Instruction-Guided Speech Synthesis Models}, author = {Kuan, Chun-Yi and Lee, Hung-yi}, booktitle = {Findings of the Association for Computational Linguistics: NAACL 2025}, pages = {5402--5428}, year = {2025}, publisher = {Association for Computational Linguistics}, doi = {10.18653/v1/2025.findings-naacl.298}, url = {https://aclanthology.org/2025.findings-naacl.298/}, } -
Can Large Audio-Language Models Truly Hear? Tackling Hallucinations with Multi-Task Assessment and Stepwise Audio Reasoning
TL;DRProbes whether audio-LLMs truly 'hear' via multi-task assessment and stepwise audio reasoning to reduce hallucinations.
@inproceedings{kuan2025can, title = {Can Large Audio-Language Models Truly Hear? Tackling Hallucinations with Multi-Task Assessment and Stepwise Audio Reasoning}, author = {Kuan, Chun-Yi and Lee, Hung-yi}, booktitle = {ICASSP 2025 -- 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, pages = {1--5}, year = {2025}, organization = {IEEE}, doi = {10.1109/ICASSP49660.2025.10888384}, url = {https://arxiv.org/abs/2410.16130}, } -
Dynamic-SUPERB Phase-2: A Collaboratively Expanding Benchmark for Measuring the Capabilities of Spoken Language Models with 180 Tasks
TL;DRThe largest collaboratively-built benchmark for instruction-following speech models—180 community-contributed tasks across speech, music, and audio, expanding beyond classification to generation and regression.
@inproceedings{huang2024dynamic2, title = {Dynamic-SUPERB Phase-2: A Collaboratively Expanding Benchmark for Measuring the Capabilities of Spoken Language Models with 180 Tasks}, author = {Huang, Chien-yu and Chen, Wei-Chih and Yang, Shu-wen and Liu, Andy T. and Li, Chen-An and Lin, Yu-Xiang and Tseng, Wei-Cheng and Diwan, Anuj and Shih, Yi-Jen and Shi, Jiatong and Kuan, Chun-Yi and others}, booktitle = {The Thirteenth International Conference on Learning Representations (ICLR 2025)}, year = {2025}, url = {https://openreview.net/forum?id=s7lzZpAW7T}, } -
Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models
TL;DREvaluates whether speech-aware language models can follow textual instructions without forgetting core language abilities.
@inproceedings{lu2025speech, title = {Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models}, author = {Lu, Ke-Han and Kuan, Chun-Yi and Lee, Hung-yi}, booktitle = {Interspeech 2025}, pages = {2078--2082}, year = {2025}, organization = {ISCA}, doi = {10.21437/Interspeech.2025-619}, }
-
Speech-Copilot: Leveraging Large Language Models for Speech Processing via Task Decomposition, Modularization, and Program Generation
TL;DRSpeech-Copilot lets an LLM solve speech tasks by decomposing them into modular, callable programs.
@inproceedings{kuan2024speech, title = {Speech-Copilot: Leveraging Large Language Models for Speech Processing via Task Decomposition, Modularization, and Program Generation}, author = {Kuan, Chun-Yi and Yang, Chih-Kai and Huang, Wei-Ping and Lu, Ke-Han and Lee, Hung-yi}, equalcontrib = {Kuan, Chun-Yi and Yang, Chih-Kai}, booktitle = {2024 IEEE Spoken Language Technology Workshop (SLT)}, pages = {1060--1067}, year = {2024}, organization = {IEEE}, doi = {10.1109/SLT61566.2024.10832184}, url = {https://arxiv.org/abs/2407.09886}, } -
Understanding Sounds, Missing the Questions: The Challenge of Object Hallucination in Large Audio-Language Models
TL;DRShows large audio-language models often hallucinate objects/sounds, and frames the object-hallucination problem.
@inproceedings{kuan2024understanding, title = {Understanding Sounds, Missing the Questions: The Challenge of Object Hallucination in Large Audio-Language Models}, author = {Kuan, Chun-Yi and Huang, Wei-Ping and Lee, Hung-yi}, booktitle = {Interspeech 2024}, pages = {4144--4148}, year = {2024}, organization = {ISCA}, doi = {10.21437/Interspeech.2024-1076}, url = {https://arxiv.org/abs/2406.08402}, } -
Dynamic-SUPERB: Towards a Dynamic, Collaborative, and Comprehensive Instruction-Tuning Benchmark for Speech
TL;DRIntroduces a collaborative benchmark for testing whether speech models can follow instructions across diverse speech tasks.
@inproceedings{huang2024dynamic, title = {Dynamic-SUPERB: Towards a Dynamic, Collaborative, and Comprehensive Instruction-Tuning Benchmark for Speech}, author = {Huang, Chien-yu and Lu, Ke-Han and Wang, Shih-Heng and Hsiao, Chi-Yuan and Kuan, Chun-Yi and Wu, Haibin and Arora, Siddhant and Chang, Kai-Wei and Shi, Jiatong and Peng, Yifan and Sharma, Roshan and Watanabe, Shinji and Ramakrishnan, Bhiksha and Shehata, Shady and Lee, Hung-yi}, equalcontrib = {Hsiao, Chi-Yuan and Kuan, Chun-Yi and Wu, Haibin}, booktitle = {ICASSP 2024 -- 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, pages = {12136--12140}, year = {2024}, organization = {IEEE}, doi = {10.1109/ICASSP48485.2024.10448257}, url = {https://arxiv.org/abs/2309.09510}, } -
Large Language Model as an Assignment Evaluator: Insights, Feedback, and Challenges in a 1000+ Student Course
TL;DRStudies how GPT-4 works as an assignment evaluator in a 1000+ student classroom, revealing both usefulness and risks.
@inproceedings{chiang2024large, title = {Large Language Model as an Assignment Evaluator: Insights, Feedback, and Challenges in a 1000+ Student Course}, author = {Chiang, Cheng-Han and Chen, Wei-Chih and Kuan, Chun-Yi and Yang, Chienchou and Lee, Hung-yi}, equalcontrib = {Chen, Wei-Chih and Kuan, Chun-Yi}, booktitle = {Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing (EMNLP)}, pages = {2489--2513}, year = {2024}, publisher = {Association for Computational Linguistics}, doi = {10.18653/v1/2024.emnlp-main.146}, url = {https://aclanthology.org/2024.emnlp-main.146}, } -
Listen and Speak Fairly: A Study on Semantic Gender Bias in Speech Integrated Large Language Models
TL;DRIntroduces spoken bias evaluation tasks to measure semantic gender bias in speech-integrated large language models.
@inproceedings{lin2024listen, title = {Listen and Speak Fairly: A Study on Semantic Gender Bias in Speech Integrated Large Language Models}, author = {Lin, Yi-Cheng and Lin, Tzu-Quan and Yang, Chih-Kai and Lu, Ke-Han and Chen, Wei-Chih and Kuan, Chun-Yi and Lee, Hung-yi}, booktitle = {2024 IEEE Spoken Language Technology Workshop (SLT)}, pages = {439--446}, year = {2024}, organization = {IEEE}, doi = {10.1109/SLT61566.2024.10832317}, url = {https://arxiv.org/abs/2407.06957}, } -
Investigating Zero-Shot Generalizability on Mandarin-English Code-Switched ASR and Speech-to-Text Translation of Recent Foundation Models with Self-Supervision and Weak Supervision
TL;DRTests how well recent foundation models generalize zero-shot to Mandarin-English code-switched ASR and speech translation.
@inproceedings{yang2024investigating, title = {Investigating Zero-Shot Generalizability on Mandarin-English Code-Switched ASR and Speech-to-Text Translation of Recent Foundation Models with Self-Supervision and Weak Supervision}, author = {Yang, Chih-Kai and Huang, Kuan-Po and Lu, Ke-Han and Kuan, Chun-Yi and Hsiao, Chi-Yuan and Lee, Hung-yi}, booktitle = {2024 IEEE International Conference on Acoustics, Speech, and Signal Processing Workshops (ICASSPW)}, pages = {540--544}, year = {2024}, organization = {IEEE}, doi = {10.1109/ICASSPW62465.2024.10626762}, url = {https://arxiv.org/abs/2401.00273}, } -
Building a Taiwanese Mandarin Spoken Language Model: A First Attempt
TL;DRPresents an initial attempt to build a real-time Taiwanese Mandarin spoken language model for multi-turn speech interaction.
@article{yang2024building, title = {Building a Taiwanese Mandarin Spoken Language Model: A First Attempt}, author = {Yang, Chih-Kai and Fu, Yu-Kuan and Li, Chen-An and Lin, Yi-Cheng and Lin, Yu-Xiang and Chen, Wei-Chih and Chung, Ho-Lam and Kuan, Chun-Yi and Huang, Wei-Ping and Lu, Ke-Han and others}, journal = {arXiv preprint arXiv:2411.07111}, year = {2024}, url = {https://arxiv.org/abs/2411.07111}, }
-
Towards General-Purpose Text-Instruction-Guided Voice Conversion
TL;DRA first step toward general-purpose voice conversion controlled by free-form text instructions.
@inproceedings{kuan2023towards, title = {Towards General-Purpose Text-Instruction-Guided Voice Conversion}, author = {Kuan, Chun-Yi and Li, Chen-An and Hsu, Tsu-Yuan and Lin, Tse-Yang and Chung, Ho-Lam and Chang, Kai-Wei and Chang, Shuo-Yiin and Lee, Hung-yi}, booktitle = {2023 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)}, pages = {1--8}, year = {2023}, organization = {IEEE}, doi = {10.1109/ASRU57964.2023.10389672}, url = {https://arxiv.org/abs/2309.14324}, }