We study the task of noiseless linear regression under Gaussian covariates in the presence of additive oblivious contamination. Specifically, we are given i.i.d.\ samples from a distribution (x,y) on Rd×R with x∼N(0,Id) and y=x⊤β+z, where z is drawn independently of x from an unknown distribution E. Moreover, z satisfies PE[z=0]=α>0. The goal is to accurately recover the regressor β to small ℓ2-error. Ignoring computational considerations, this problem is known to be solvable using O(d/α) samples. On the other hand, the best known polynomial-time algorithms require Ω(d/α2) samples. Here we provide formal evidence that the quadratic dependence in 1/α is inherent for efficient algorithms. Specifically, we show that any efficient Statistical Query algorithm for this task requires VSTAT complexity at least Ω~(d1/2/α2).
@article{arxiv.2510.10665,
title = {Information-Computation Tradeoffs for Noiseless Linear Regression with Oblivious Contamination},
author = {Ilias Diakonikolas and Chao Gao and Daniel M. Kane and John Lafferty and Ankit Pensia},
journal= {arXiv preprint arXiv:2510.10665},
year = {2025}
}