In this paper, we investigate the intersection of large generative AI models and cloud-native computing architectures. Recent large models such as ChatGPT, while revolutionary in their capabilities, face challenges like escalating costs and demand for high-end GPUs. Drawing analogies between large-model-as-a-service (LMaaS) and cloud database-as-a-service (DBaaS), we describe an AI-native computing paradigm that harnesses the power of both cloud-native technologies (e.g., multi-tenancy and serverless computing) and advanced machine learning runtime (e.g., batched LoRA inference). These joint efforts aim to optimize costs-of-goods-sold (COGS) and improve resource accessibility. The journey of merging these two domains is just at the beginning and we hope to stimulate future research and development in this area.
@article{arxiv.2401.12230,
title = {Computing in the Era of Large Generative Models: From Cloud-Native to AI-Native},
author = {Yao Lu and Song Bian and Lequn Chen and Yongjun He and Yulong Hui and Matthew Lentz and Beibin Li and Fei Liu and Jialin Li and Qi Liu and Rui Liu and Xiaoxuan Liu and Lin Ma and Kexin Rong and Jianguo Wang and Yingjun Wu and Yongji Wu and Huanchen Zhang and Minjia Zhang and Qizhen Zhang and Tianyi Zhou and Danyang Zhuo},
journal= {arXiv preprint arXiv:2401.12230},
year = {2024}
}