@inproceedings{ef5c552aabb641398e0f9def71e4869c,
title = "A comparative study of audio features for audio-to-visual conversion in MPEG-4 compliant facial animation",
abstract = "Audio-to-visual conversion is the basic problem of speech-driven facial animation. Since the conversion problem is to predict facial control parameters from the acoustic speech, the informative representation of audio, i.e., the audio feature, is important to get a good prediction. This paper presents a performance comparison on prosodic features, articulatory features, and perceptual features for the audio-to-visual conversion problem on a common test bed. Experimental results show that the Mel frequency cepstral coefficients (MFCCs) produce the best performance, followed by the perceptual linear prediction coefficients (PLPC), the linear predictive cepstral coefficients (LPCCs), and the prosodie feature set (F0) and energy). The combination of three kinds of features can further improve the prediction performance on facial parameters. It unveils that different audio features carry complementary information relevant to facial animation.",
keywords = "Audio features, Audio-to-visual conversion, Facial animation, MPEG-4, Talking face",
author = "Lei Xie and Liu, \{Zhi Qiang\}",
year = "2006",
doi = "10.1109/ICMLC.2006.259085",
language = "英语",
isbn = "1424400619",
series = "Proceedings of the 2006 International Conference on Machine Learning and Cybernetics",
publisher = "IEEE Computer Society",
pages = "4359--4364",
booktitle = "Proceedings of the 2006 International Conference on Machine Learning and Cybernetics",
note = "5th International Conference on Machine Learning and Cybernetics, ICMLC 2006 ; Conference date: 13-08-2006 Through 16-08-2006",
}