In this paper, we propose the Quantile Option Architecture (QUOTA) for exploration based on recent advances in distributional reinforcement learning (RL). In QUOTA, decision making is based on quantiles of a value distribution, not only the mean. QUOTA provides a new dimension for exploration via making use of both optimism and pessimism of a value distribution. We demonstrate the performance advantage of QUOTA in both challenging video games and physical robot simulators.
@article{arxiv.1811.02073,
title = {QUOTA: The Quantile Option Architecture for Reinforcement Learning},
author = {Shangtong Zhang and Borislav Mavrin and Linglong Kong and Bo Liu and Hengshuai Yao},
journal= {arXiv preprint arXiv:1811.02073},
year = {2018}
}