@inproceedings {pub2526pub2615,
	title = {Integrating sequence information in the audio-visual detection of word prominence in a human-machine interaction scenario},
	author = {Andrea Schnall AND Martin Ernst Heckmann},
	year = {2014},
	abstract = {Discrimination of prominent and non-prominent words is an important human ability.
By strongly increasing the prominence of a word, for example, a corrections can be indicated. Using this information in human-machine interaction systems is still a challenging task. It is known that modifying the articulatory parameters to raise the prominence of a segment of an utterance, i.e. hyperarticulating, usually is accompanied by a hypoarticualtion, i.e. reduction of these parameters, for the neighboring segments. 
In this paper we compare different approaches for the automatic labeling of the prominence of words. In particular we investigate how the information in the sequence, i.e. the neighboring words, can be used.
During the recording of the underlying audio-visual database, the subjects were interacting with a computer via speech in a Wizard-of-Oz experiment. Subjects where asked to make corrections for a misunderstanding of a single word of the system by using prosodic cues only. During the interaction we made audio and video recordings.
We extracted an extensive range of features from the audio and visual channel, but, compared to others, we are classifying without knowledge of textual features.
For the classification of word prominence we compare two algorithms. On the one hand SVMs, a local classifier, which we either trained on individual words or on word triples. On the other hand a classifier based on a sequential model, linear chain Conditional Random Fields (CRFs). We again train them either with features derived from individual words or from word triples. For the CRF the whole sentence is used as a word sequence for training and testing.   

},
	publisher = {ISCA},
	booktitle = {INTERSPEECH}
}
