@inproceedings {PUBA150,
	title = {Audiovisual Integration in Dialog Scenarios},
	author = {Rujiao Yan AND Tobias Rodemann AND Britta Wrede},
	year = {2012},
	month = {May},
	abstract = { We present a system of audiovisual integration in dialog scenarios. When a robot is interacting with multiple people, it should understand the dialog situation, e.g. the number and position of interacting partners, as well as who is currently speaking. Since auditory and visual information, derived from the same event, can be integrated across sensory modalities to enhance the representation of the external world, we use audiovisual integration to better fulfill this task. State of the art methods are constrained in several ways, e.g. speakers need to be known in advance for training, or many cameras and microphones are required in a smart-room environment. Our approach can overcome these limits. We employ just one camera an d a pair of microphones mounted on a robot head. The system combines many simple auditory and visual features to achieve a high performance of audiovisual integration, and runs in an unsupervised, real-time, online and incremental manner.
We collect various auditory and visual features in a compressed form as so-called proto objects [1,2], which combine an arbitrary number of features. For example, audio proto objects contain binaural position cues, mean sound energy and sound length, while visual proto objects consist of a face{\textquoteright}s position in the camera picture and the corresponding world coordinate, or color and texture features in clothes, collar and hair areas. We use a face detection algorithm [3] to locate faces, and the base position of clothes, collar and hair is based on the corresponding face position.Auditory and visual features can be linked by means of their position information. The probability that a visual proto object belongs to the current audio proto object can be calculated using their position similarity. Then the uncertainty of the association between the current audio proto object and the visual proto object with the maximal probability is calculated by the entropy of the normalized probabilities of all candidate associations. The use of uncertainty information is application dependent. For example this information could be sent to the dialog manager in a dialog system. If an audiovisual association is unreliable, a dialog manager may ask for confirmation. 
We test audiovisual integration in various dialog scenarios where multiple speakers dynamically enter and leave the room. We can show that our system can successfully assign words to corresponding speakers in real world dialog scenarios. With simple color and texture features in visual proto objects, a person can be recognized again, even when disappearing briefly and then reappearing or moving quickly.},
	publisher = {ICCNS},
	booktitle = {Int. Conf. on Cognitive and Neural Systems}
}
