@inproceedings{257a001c690d44029b76aa9fb1cf9d6e,
title = "Leveraging Synthetic Speech for CIF-Based Customized Keyword Spotting",
abstract = "Customized keyword spotting aims to detect user-defined keywords from continuous speech, providing flexibility and personalization. Previous research mainly relied on similarity calculations between keyword text and acoustic features. However, due to the gap between the two modalities, it is challenging to obtain alignment information and model their correlation. In our paper, we propose a novel method to address these issues. Firstly, we introduce a text-to-speech (TTS) module to generate the audio of keywords, effectively addressing the cross-modal challenge of text-based customized keyword spotting. Furthermore, we employ the Continuous Integrate-and-Fire (CIF) mechanism for boundary prediction to get token-level acoustic representations of keywords thus solving the keyword and speech alignment problem. Our experimental results on the Aishell-1 dataset demonstrate the effectiveness of our proposed method. It significantly outperforms both the baseline method and the Dynamic Sequence Partitioning (DSP) method in terms of keyword spotting accuracy. Compared with the DSP method, our model can achieve a significant improvement in the relative wake-up rate of 72.7\% when the false accept rate is fixed at 0.02. And our model represents a 64\% improvement over the baseline model.",
keywords = "Continuous Integrate-and-Fire, Keyword spotting, Speech synthesis",
author = "Shuiyun Liu and Ao Zhang and Kaixun Huang and Lei Xie",
note = "Publisher Copyright: {\textcopyright} The Author(s), under exclusive license to Springer Nature Singapore Pte Ltd. 2024.; 18th National Conference on Man-Machine Speech Communication, NCMMSC 2023 ; Conference date: 08-12-2023 Through 11-12-2023",
year = "2024",
doi = "10.1007/978-981-97-0601-3\_31",
language = "英语",
isbn = "9789819706006",
series = "Communications in Computer and Information Science",
publisher = "Springer Science and Business Media Deutschland GmbH",
pages = "354--365",
editor = "Jia Jia and Zhenhua Ling and Xie Chen and Ya Li and Zixing Zhang",
booktitle = "Man-Machine Speech Communication - 18th National Conference, NCMMSC 2023, Proceedings",
}