This paper presents a comparative study of large language models (LLMs) in interpreting grid-structured geospatial data. We evaluate the performance of a base model through structured prompting and contrast it with a fine-tuned variant trained on a dataset of user-assistant interactions. Our results highlight the strengths and limitations of zero-shot prompting and demonstrate the benefits of fine-tuning for structured geospatial and temporal reasoning.
@article{arxiv.2505.17116,
title = {Comparative Evaluation of Prompting and Fine-Tuning for Applying Large Language Models to Grid-Structured Geospatial Data},
author = {Akash Dhruv and Yangxinyu Xie and Jordan Branham and Tanwi Mallick},
journal= {arXiv preprint arXiv:2505.17116},
year = {2025}
}