@online{Habibie_2102.06837,
TITLE = {Learning Speech-driven {3D} Conversational Gestures from Video},
AUTHOR = {Habibie, Ikhsanul and Xu, Weipeng and Mehta, Dushyant and Liu, Lingjie and Seidel, Hans-Peter and Pons-Moll, Gerard and Elgharib, Mohamed and Theobalt, Christian},
LANGUAGE = {eng},
URL = {https://arxiv.org/abs/2102.06837},
EPRINT = {2102.06837},
EPRINTTYPE = {arXiv},
YEAR = {2021},
ABSTRACT = {We propose the first approach to automatically and jointly synthesize both<br>the synchronous 3D conversational body and hand gestures, as well as 3D face<br>and head animations, of a virtual character from speech input. Our algorithm<br>uses a CNN architecture that leverages the inherent correlation between facial<br>expression and hand gestures. Synthesis of conversational body gestures is a<br>multi-modal problem since many similar gestures can plausibly accompany the<br>same input speech. To synthesize plausible body gestures in this setting, we<br>train a Generative Adversarial Network (GAN) based model that measures the<br>plausibility of the generated sequences of 3D body motion when paired with the<br>input audio features. We also contribute a new way to create a large corpus of<br>more than 33 hours of annotated body, hand, and face data from in-the-wild<br>videos of talking people. To this end, we apply state-of-the-art monocular<br>approaches for 3D body and hand pose estimation as well as dense 3D face<br>performance capture to the video corpus. In this way, we can train on orders of<br>magnitude more data than previous algorithms that resort to complex in-studio<br>motion capture solutions, and thereby train more expressive synthesis<br>algorithms. Our experiments and user study show the state-of-the-art quality of<br>our speech-synthesized full 3D character animations.<br>},
}
