Multimodal Empathetic Response Generation (MERG) is crucial for building emotionally intelligent human-computer interactions. Although large language models (LLMs) have improved text-based ERG, challenges remain in handling multimodal emotional content and maintaining identity consistency. Thus, we propose E3RG, an Explicit Emotion-driven Empathetic Response Generation System based on multimodal LLMs which decomposes MERG task into three parts: multimodal empathy understanding, empathy memory retrieval, and multimodal response generation. By integrating advanced expressive speech and video generative models, E3RG delivers natural, emotionally rich, and identity-consistent responses without extra training. Experiments validate the superiority of our system on both zero-shot and few-shot settings, securing Top-1 position in the Avatar-based Multimodal Empathy Challenge on ACM MM 25. Our code is available at https://github.com/RH-Lin/E3RG.
@article{arxiv.2508.12854,
title = {E3RG: Building Explicit Emotion-driven Empathetic Response Generation System with Multimodal Large Language Model},
author = {Ronghao Lin and Shuai Shen and Weipeng Hu and Qiaolin He and Aolin Xiong and Li Huang and Haifeng Hu and Yap-peng Tan},
journal= {arXiv preprint arXiv:2508.12854},
year = {2025}
}