@inproceedings {pub3902,
	title = {Assessing Transferability in Reinforcement Learning from Randomized Simulations},
	author = {Fabio Muratore AND Michael Gienger AND Jan Peters},
	year = {2019},
	month = {July},
	abstract = {Exploration-based reinforcement learning of control policies on physical systems is generally time-intensive and can leadto catastrophic failures. Therefore, simulation-based policy search appears to be an appealing alternative. Unfortunately,running policy search on a slightly faulty simulator can easily lead to the maximization of the {\textquoteleft}Simulation OptimizationBias{\textquoteright} (SOB), where the policy exploits modeling errors of the simulator such that the resulting behavior can potentiallydamage the device. For this reason, much work in reinforcement learning has focused on model-free methods. The resulting lack of safe simulation-based policy learning techniques imposes severe limitations on the application of rein-forcement learning to real-world systems.
In this paper, we explore how physics simulations can be utilized for a robust policy optimization by randomizing thesimulator{\textquoteright}s parameters and training from model ensembles.  We propose an algorithm called Simulation-based PolicyOptimization with Transferability Assessment (SPOTA) that uses an estimator of the SOB to formulate a stopping crite-rion for training. We show that the simulation-based policy search algorithm is able to learn a control policy exclusivelyfrom a randomized simulator that can be applied directly to a different system without using any data from the latter.},
	publisher = {unknown},
	booktitle = {Multidisciplinary Conference on Reinforcement Learning and Decision Making}
}
