Recent advances in recommendation scaling laws have led to foundation models of unprecedented complexity. While these models offer superior performance, their computational demands make real-time serving impractical, often forcing practitioners to rely on knowledge distillation-compromising serving quality for efficiency. To address this challenge, we present SOLARIS (Speculative Offloading of Latent-bAsed Representation for Inference Scaling), a novel framework inspired by speculative decoding. SOLARIS proactively precomputes user-item interaction embeddings by predicting which user-item pairs are likely to appear in future requests, and asynchronously generating their foundation model representations ahead of time. This approach decouples the costly foundation model inference from the latency-critical serving path, enabling real-time knowledge transfer from models previously considered too expensive for online use. Deployed across Meta's advertising system serving billions of daily requests, SOLARIS achieves 0.67% revenue-driving top-line metrics gain, demonstrating its effectiveness at scale.
@article{arxiv.2604.12110,
title = {SOLARIS: Speculative Offloading of Latent-bAsed Representation for Inference Scaling},
author = {Zikun Liu and Liang Luo and Qianru Li and Zhengyu Zhang and Wei Ling and Jingyi Shen and Zeliang Chen and Yaning Huang and Jingxian Huang and Abdallah Aboelela and Chonglin Sun and Feifan Gu and Fenggang Wu and Hang Qu and Huayu Li and Jill Pan and Kaidi Pei and Laming Chen and Longhao Jin and Qin Huang and Tongyi Tang and Varna Puvvada and Wenlin Chen and Xiaohan Wei and Xu Cao and Yantao Yao and Yuan Jin and Yunchen Pu and Yuxin Chen and Zijian Shen and Zhengkai Zhang and Dong Liang and Ellie Wen},
journal= {arXiv preprint arXiv:2604.12110},
year = {2026}
}