@techreport{BargmannBlanzSeidel2007,
TITLE = {A nonlinear viseme model for triphone-based speech synthesis},
AUTHOR = {Bargmann, Robert and Blanz, Volker and Seidel, Hans-Peter},
LANGUAGE = {eng},
URL = {http://domino.mpi-inf.mpg.de/internet/reports.nsf/NumberView/2007-4-003},
NUMBER = {MPI-I-2007-4-003},
INSTITUTION = {Max-Planck-Institut f{\"u}r Informatik},
ADDRESS = {Saarbr{\"u}cken},
YEAR = {2007},
DATE = {2007},
ABSTRACT = {This paper presents a representation of visemes that defines a measure<br>of similarity between different visemes, and a system of viseme<br>categories. The representation is derived from a statistical data<br>analysis of feature points on 3D scans, using Locally Linear<br>Embedding (LLE). The similarity measure determines which available<br>viseme and triphones to use to synthesize 3D face animation for a<br>novel audio file. From a corpus of dynamic recorded 3D mouth<br>articulation data, our system is able to find the best suited sequence<br>of triphones over which to interpolate while reusing the<br>coarticulation information to obtain correct mouth movements over<br>time. Due to the similarity measure, the system can deal with<br>relatively small triphone databases and find the most appropriate<br>candidates. With the selected sequence of database triphones, we can<br>finally morph along the successive triphones to produce the final<br>articulation animation.<br>In an entirely data-driven approach, our automated procedure for<br>defining viseme categories reproduces the groups of related visemes<br>that are defined in the phonetics literature.},
TYPE = {Research Report / Max-Planck-Institut für Informatik},
}
