We present an algorithm that achieves almost optimal pseudo-regret bounds against adversarial and stochastic bandits. Against adversarial bandits the pseudo-regret is O(Knlogn) and against stochastic bandits the pseudo-regret is O(∑i(logn)/Δi). We also show that no algorithm with O(logn) pseudo-regret against stochastic bandits can achieve O~(n) expected regret against adaptive adversarial bandits. This complements previous results of Bubeck and Slivkins (2012) that show O~(n) expected adversarial regret with O((logn)2) stochastic pseudo-regret.
@article{arxiv.1605.08722,
title = {An algorithm with nearly optimal pseudo-regret for both stochastic and adversarial bandits},
author = {Peter Auer and Chao-Kai Chiang},
journal= {arXiv preprint arXiv:1605.08722},
year = {2016}
}