This work explores the use of the AMD Xilinx Versal Adaptable Intelligent Engine (AIE) to accelerate Gated Recurrent Unit (GRU) inference for latency constrained applications. We present a custom workload distribution framework across the AIE's vector processors and propose a hybrid AIE - Programmable Logic (PL) design to optimize computational efficiency. Our approach explores the parallelization over the rows of the matrices by utilizing as many of the AIE vectorized processors effectively computing all the elements of the resulting vector at the same time, an alternative to cascade stream pipelining.
@article{arxiv.2511.15626,
title = {A Latency-Constrained, Gated Recurrent Unit (GRU) Implementation in the Versal AI Engine},
author = {M. Sapkas and A. Triossi and M. Zanetti},
journal= {arXiv preprint arXiv:2511.15626},
year = {2026}
}