@inproceedings {pub3692,
	title = {Domain Randomization for Simulation-Based Policy
Optimization with Transferability Assessment},
	author = {Fabio Muratore AND Felix Treede AND Michael Gienger AND Jan Peters},
	year = {2018},
	month = {October},
	abstract = {Learning new policies on real robot systems is generically time-intensive and may
lead to catastrophic robot failures. Therefore, simulation-based policy search ap-
pears to be an appealing alternative. Unfortunately, running policy search on a
slightly faulty simulator can easily lead to the maximization of the so-called {\textquoteright}op-
timality gap{\textquoteright}, where the policy exploits modeling errors of the simulator such
that the resulting behavior can potentially damage the robot. For this reason,
much work in robot reinforcement learning has focused on model-free methods
that learn on real-world systems.
This lack of using simulators results in an extensive need for robot time during
training and imposes severe constraints on the applicable methods. Current ap-
proaches in robot reinforcement learning are frequently engineered to work with
task-appropriate policy representations and methods that can only locally optimize
behavior. These approaches are unlikely to fundamentally speed-up robot learning
or make the resulting approaches generally deployable.
In this paper, we explore how physics simulations can be utilized for a more robust
policy optimization by perturbing the simulator{\textquoteright}s parameters and training from en-
sembles of models. We propose a new algorithm called Simulation-based Policy
Optimization with Transferability Assessment that uses an unbiased estimator of
the optimality gap to formulate a stopping criterion for training.},
	publisher = {PMLR},
	booktitle = {Conference on Robot Learning (CoRL) - 2018 Edition}
}
