Traditional video-induced physiological datasets usually rely on whole-trial labels, which introduce temporal label noise in dynamic emotion recognition. We present FIRMED, a peak-centered multimodal dataset based on an immediate-recall annotation paradigm, with synchronized EEG, ECG, GSR, PPG, and facial recordings from 35 participants. FIRMED provides event-centered timestamps, emotion labels, and intensity annotations, and its annotation quality is supported by subjective and physiological validation. Benchmark experiments show that FIRMED consistently outperforms whole-trial labeling, yielding an average gain of 3.8 percentage points across eight EEG-based classifiers, with further improvements under multimodal fusion. FIRMED provides a practical benchmark for temporally localized supervision in multimodal affective computing.
@article{arxiv.2507.02350,
title = {FIRMED: A Peak-Centered Multimodal Dataset with Fine-Grained Annotation for Emotion Recognition},
author = {Hao Tang and Songyun Xie and Xinzhou Xie and Can Liao and Bohan Li and Zhongyu Tian and Dalu Zheng},
journal= {arXiv preprint arXiv:2507.02350},
year = {2026}
}