Large Language Model (LLM) agents, acting on their users' behalf to manipulate and analyze data, are likely to become the dominant workload for data systems in the future. When working with data, agents employ a high-throughput process of exploration and solution formulation for the given task, one we call agentic speculation. The sheer volume and inefficiencies of agentic speculation can pose challenges for present-day data systems. We argue that data systems need to adapt to more natively support agentic workloads. We take advantage of the characteristics of agentic speculation that we identify, i.e., scale, heterogeneity, redundancy, and steerability - to outline a number of new research opportunities for a new agent-first data systems architecture, ranging from new query interfaces, to new query processing techniques, to new agentic memory stores.
@article{arxiv.2509.00997,
title = {Supporting Our AI Overlords: Redesigning Data Systems to be Agent-First},
author = {Shu Liu and Soujanya Ponnapalli and Shreya Shankar and Sepanta Zeighami and Alan Zhu and Shubham Agarwal and Ruiqi Chen and Samion Suwito and Shuo Yuan and Ion Stoica and Matei Zaharia and Alvin Cheung and Natacha Crooks and Joseph E. Gonzalez and Aditya G. Parameswaran},
journal= {arXiv preprint arXiv:2509.00997},
year = {2025}
}