This work focuses on anticipating long-term human actions, particularly using short video segments, which can speed up editing workflows through improved suggestions while fostering creativity by suggesting narratives. To this end, we imbue a transformer network with a symbolic knowledge graph for action anticipation in video segments by boosting certain aspects of the transformer's attention mechanism at run-time. Demonstrated on two benchmark datasets, Breakfast and 50Salads, our approach outperforms current state-of-the-art methods for long-term action anticipation using short video context by up to 9%.
@article{arxiv.2309.05943,
title = {Knowledge-Guided Short-Context Action Anticipation in Human-Centric Videos},
author = {Sarthak Bhagat and Simon Stepputtis and Joseph Campbell and Katia Sycara},
journal= {arXiv preprint arXiv:2309.05943},
year = {2023}
}
Comments
ICCV 2023 Workshop on AI for Creative Video Editing and Understanding