Multi-agent reinforcement learning (MARL) has witnessed a remarkable surge in interest, fueled by the empirical success achieved in applications of single-agent reinforcement learning (RL). In this study, we consider a distributed Q-learning scenario, wherein a number of agents cooperatively solve a sequential decision making problem without access to the central reward function which is an average of the local rewards. In particular, we study finite-time analysis of a distributed Q-learning algorithm, and provide a new sample complexity result of O~(min{ϵ21(1−γ)6dmin4tmix,ϵ1(1−σ2(W))(1−γ)4dmin3∣\gS∣∣\gA∣}) under tabular lookup
@article{arxiv.2405.14078,
title = {A finite time analysis of distributed Q-learning},
author = {Han-Dong Lim and Donghwan Lee},
journal= {arXiv preprint arXiv:2405.14078},
year = {2025}
}