Bibtex export
@incollection{ Weninger2013, title = {Speaker trait characterization in web videos: Uniting speech, language, and facial features}, author = {Weninger, Felix and Wagner, Claudia and Wöllmer, Martin and Schuller, Björn and Morency, Louis-Philipp}, year = {2013}, booktitle = {Proceedings of the 38th International Conference on Acoustics, Speech and Signal Processing (ICASSP 2013)}, pages = {3647-3651}, publisher = {IEEE}, issn = {2379-190X}, isbn = {978-1-4799-0356-6}, doi = {https://doi.org/10.1109/ICASSP.2013.6638338}, urn = {https://nbn-resolving.org/urn:nbn:de:0168-ssoar-66084-2}, abstract = {We present a multi-modal approach to speaker characterization using acoustic, visual and linguistic features. Full realism is provided by evaluation on a database of real-life web videos and automatic feature extraction including face and eye detection, and automatic speech recognition. Different segmentations are evaluated for the audio and video streams, and the statistical relevance of Linguistic Inquiry and Word Count (LIWC) features is confirmed. In the result, late multimodal fusion delivers 73, 92 and 73% average recall in binary age, gender and race classification on unseen test subjects, outperforming the best single modalities for age and race.}, keywords = {Video; video; Video-Clip; video clip; Aufzeichnung; recording; Computerlinguistik; computational linguistics; Internet; Internet; Evaluation; evaluation; Soziale Medien; social media; Experiment; experiment; audiovisuelle Medien; audiovisual media}}