@inproceedings {PUBARCH186,
	title = {A Multimodal Scheduler for Synchronized Humanoid Robot Gesture and Speech},
	author = {Maha Salem AND Stefan Kopp AND Frank Joublin},
	year = {2011},
	month = {May},
	abstract = {In order to engage in natural and fluent human-robot interaction, humanoid robot companions must be able to produce speech-accompanying non-verbal behavior including hand and arm gestures. In human communication, gestures are considered an integral part of the human thinking process. Accordingly, they are found to be finely synchronized with the accompanying linguistic affiliate [McNeill 2002]. Many researchers have emphasized the importance of this temporal synchrony in terms of coexpressiveness [McNeill et al. 2005; Harris 2002; Blumenthal 1970]. However, for a humanoid robot required to generate speech and gesture, an appropriate synchronization of the two modalities still poses a major challenge. In many existing approaches used for virtual conversational agents or robotic platforms, synchronization of different modalities is either achieved only approximately or by solely adapting one modality to the other, for example by frequently adjusting gesture speed to the timing of running speech. Given the limitations of robotic platforms, e.g. motor velocity limits, these approaches turn out to be obsolete when a fine synchronization of speech and gesture is a fundamental necessity for fluent human-robot interaction.

We present a multimodal scheduler that is capable of synchronizing expressive hand and arm gestures with speech for the Honda humanoid robot. The scheduler is based on a forward model which predicts an estimate of the preparation time required for a gestureprior to the actual gesture taking place (stroke). For this, an internal simulation of the designated arm movement is performed using the robot's whole body motion controller software [Gienger et al. 2005]. This is also used to subsequently generate the actual movement. Scheduling and generation of gesture and speech are flexibly conducted at run-time. Despite the fairly accurate prediction-based timing estimation, the actual execution and timing of multimodal utterances might deviate from the prediction. For this reason, an on-line adjustment of the synchronization process is required once a certain threshold value of deviation is exceeded. This is achieved via a utilization of afferent feedback by mutually adapting the two modalities reactively to one another. Figure 1 shows an outline of the proposed scheduler},
	publisher = {GW2011},
	booktitle = {In: Book of Extended Abstracts of the 9th International Gesture Workshop},
	pages = {64-67}
}
