@phdthesis {pub6419,
	title = {Exploration in Few-Shot Meta-Reinforcement Learning},
	author = {Radu Stoican},
	year = {2025},
	month = {November},
	abstract = {Reinforcement learning trains policies specialized for a single task. Meta-reinforcement
learning (meta-RL) improves upon this by leveraging prior experience to train policies for
few-shot adaptation to new tasks. However, existing meta-RL approaches often struggle to
explore and learn tasks effectively. We introduce a novel meta-RL algorithm for learning
to learn task-specific, sample-efficient exploration policies. We achieve this through task
reconstruction, an original method for learning to identify and collect small but informative
datasets from tasks. To leverage these datasets, we propose a meta-learned hyper-reward
that encourages policies to learn to adapt. Empirical evaluations demonstrate that our
algorithm adapts to a larger variety of tasks and achieves higher returns than existing meta-
RL methods. Additionally, we show that even with full task information, adaptation is more
challenging than previously assumed. However, policies trained with our hyper-reward adapt
to new tasks successfully. We will also discuss how this new framework can enable applications in supporting human-robot interactions and enhance human-human interaction trust through communicative support actions},
	publisher = {University of Manchester},
	booktitle = {University of Manchester},
	city = {Manchester, UK},
	institution = {School of Engineering, Department of Computer Science, University of Manchester}
}
