This paper introduces Radiology-Llama2, a large language model specialized for radiology through a process known as instruction tuning. Radiology-Llama2 is based on the Llama2 architecture and further trained on a large dataset of radiology reports to generate coherent and clinically useful impressions from radiological findings. Quantitative evaluations using ROUGE metrics on the MIMIC-CXR and OpenI datasets demonstrate that Radiology-Llama2 achieves state-of-the-art performance compared to other generative language models, with a Rouge-1 score of 0.4834 on MIMIC-CXR and 0.4185 on OpenI. Additional assessments by radiology experts highlight the model's strengths in understandability, coherence, relevance, conciseness, and clinical utility. The work illustrates the potential of localized language models designed and tuned for specialized domains like radiology. When properly evaluated and deployed, such models can transform fields like radiology by automating rote tasks and enhancing human expertise.
@article{arxiv.2309.06419,
title = {Radiology-Llama2: Best-in-Class Large Language Model for Radiology},
author = {Zhengliang Liu and Yiwei Li and Peng Shu and Aoxiao Zhong and Longtao Yang and Chao Ju and Zihao Wu and Chong Ma and Jie Luo and Cheng Chen and Sekeun Kim and Jiang Hu and Haixing Dai and Lin Zhao and Dajiang Zhu and Jun Liu and Wei Liu and Dinggang Shen and Tianming Liu and Quanzheng Li and Xiang Li},
journal= {arXiv preprint arXiv:2309.06419},
year = {2023}
}