In this work, we study the asymptotic behavior of Mixture of Experts (MoE) trained via gradient flow on supervised learning problems. Our main result establishes the propagation of chaos for a MoE as the number of experts diverges. We demonstrate that the corresponding empirical measure of their parameters is close to a probability measure that solves a nonlinear continuity equation, and we provide an explicit convergence rate that depends solely on the number of experts. We apply our results to a MoE generated by a quantum neural network.
@article{arxiv.2501.14660,
title = {Mean-field limit from general mixtures of experts to quantum neural networks},
author = {Anderson Melchor Hernandez and Davide Pastorello and Giacomo De Palma},
journal= {arXiv preprint arXiv:2501.14660},
year = {2026}
}