@phdthesis {PUBA124,
	title = {Focus of Attention on Relevant Multimodal Events},
	author = {Miranda Grahl},
	year = {2011},
	abstract = {Current artificial systems for active vision do not tackle an autonomous learning of a gaze control that focuses on multimodal aspects of the environment. They are either based on a reactive gaze control or presume defined models for visual and auditory input in order to attend to objects. However, it is well known in infancy research that infants do not rely on such predefined models (Spelke, 1981). They rather flexibly learn audiovisual associations: For example associations that comprise temporal synchrony between faces and speech (Flom and Bahrick, 2007), or between moving toys and rhythmic objects{\textquoteright} sound characteristics (Flom and Bahrick, 2010). Evidence in favor of this view is provided by the work of Richardson and Kirkham (2004). The results show a principle trend of infants{\textquoteright} gazing behavior that is desirable for an autonomous gazing of an artificial system. In a learning phase, infants were introduced to look on a screen. On this screen, a bouncing toy was presented with a rhythmic sound. The toy appears in a rectangle either on the left or right side of the screen. The sound was always produced in the center of the screen. Subsequently, in a testing phase the gaze behavior of the learned toy location was analyzed in the presence of an associated sound. The toy location was presented with an empty rectangle whose position was either not modified or rotated. In both conditions, infants showed a longer gaze fixation to the location that was learned with a sound. This gazing behavior was interpreted by Richardson as a kind of visual prediction mechanism to object locations in the presence of an object specific sound. The finding of this experiment can serve as an inspiration for designing a gaze control strategy for robots. As a key aspect, they suggest a modulation of a visual filtering process by auditory modalities. This entails the question on how to equip a system with a minimal innate perceptual knowledge and a mechanism that learns to structure the perceptual information by itself. This also means to overcome the problem of audiovisual correlation in the location cue and hence implies the need for a direct measurement of causality between visual and auditory concepts. Based on these findings, the research goal of this thesis aims to develop a control strategy that autonomously learns to focus audiovisual events in its environment. The approach is inspired by findings from infancy research and aims at modeling the development from initially bottom-up driven reactive eye movement towards learned multimodal top-down attention. For this approach, different factors that contribute to the learning are investigated. The research contribution lies in the development of a computational model that acquires visual concepts in an unsupervised way and further configures those concepts in the presence of acoustic information.},
	publisher = {Uni. Bielefeld},
	booktitle = {University Bielefeld},
	institution = {Uni. Bielefeld}
}
