We introduce Baichuan-M3, a medical-enhanced large language model engineered to shift the paradigm from passive question-answering to active, clinical-grade decision support. Addressing the limitations of existing systems in open-ended consultations, Baichuan-M3 utilizes a specialized training pipeline to model the systematic workflow of a physician. Key capabilities include: (i) proactive information acquisition to resolve ambiguity; (ii) long-horizon reasoning that unifies scattered evidence into coherent diagnoses; and (iii) adaptive hallucination suppression to ensure factual reliability. Empirical evaluations demonstrate that Baichuan-M3 achieves state-of-the-art results on HealthBench, the newly introduced HealthBench-Hallu and ScanBench, significantly outperforming GPT-5.2 in clinical inquiry, advisory and safety. The models are publicly available at https://huggingface.co/collections/baichuan-inc/baichuan-m3.
@article{arxiv.2602.06570,
title = {Baichuan-M3: Modeling Clinical Inquiry for Reliable Medical Decision-Making},
author = {M3 Team and Chengfeng Dou and Fan Yang and Fei Li and Jiyuan Jia and Qiang Ju and Shuai Wang and Tianpeng Li and Xiangrong Zeng and Yijie Zhou and Hongda Zhang and Jinyang Tai and Linzhuang Sun and Peidong Guo and Yichuan Mo and Xiaochuan Wang and Hengfu Cui and Zhishou Zhang},
journal= {arXiv preprint arXiv:2602.06570},
year = {2026}
}