@techreport {pub6162pub6299,
	title = {Context-Aware Human-Robot Collaboration Through Integrated Vision-Language Models and Action Understanding},
	author = {J{\"o}rg Deigm{\"o}ller AND Stephan Hasler AND Nakul Agarwal AND Chao Wang AND Daniel Tanneberg AND Reza Ghoddoosian AND Fan Zhang AND Felix Ocker AND Anna Belardinelli AND Behzad Dariush AND Michael Gienger},
	year = {1970},
	month = {January},
	abstract = {This work presents a human-robot interaction framework that enables robots to naturally and adaptively
support people in collaborative tasks by deeply understanding situational context. The system integrates Vision-Language Models (VLMs) with object and action detection to extract detailed information about who is involved (person), what they are doing (action), and what they are interacting with (object), besides the general environment context. This structured understanding allows the robot to make context-aware decisions and
provide timely, meaningful assistance within group activities. While existing approaches can adapt to dynamic environments, they often lack the fine-grained understanding of interactions at the level of individuals, their actions and object use{\textemdash}an essential component for effective teamwork. By combining visual perception and language understanding, this framework allows the robot to flexibly respond to evolving situations without relying on predefined task rules. The system has been evaluated in collaborative scenarios, such as food preparation or drink mixing, where the robot{\textquoteright}s ability to accurately interpret group dynamics and intervene appropriately will be assessed. This work advances the development of socially intelligent robots capable of seamless and context-sensitive interaction in real-world environments.},
	publisher = {arXiv},
	booktitle = {arXiv}
}
