@inproceedings{62f9d060-d53f-4b1b-b89b-ccca283124ea,
  abstract     = {{<p>Systolic arrays (SAs) for matrix multiplication are commonly used in machine learning (ML), wireless communication, and signal processing. Inherently offering high throughput with good data reuse, they are well-positioned for both low-power edge devices and accelerator applications in high-performance computing. Current realizations suffer from startup latency, defined as the time required to fully utilize all processing elements (PEs). In this work, this issue is addressed by introducing bidirectional systolic arrays with connected edges that form toroidal dataflows. The proposed systolic arrays significantly reduce computational and readout latency from 4 n-2 to 2.5 n-1 clock cycles for an n × n matrix multiplication, while simultaneously reducing energy per operation by up to 43% compared to conventional SAs. Moreover, a variety of differently shaped SAs are synthesized in a 22 nm CMOS technology, and it is shown that the toroidal designs offer a 5%-12% lower silicon area cost.</p>}},
  author       = {{Svensson, Linus and Gustafsson, Oscar and Rodrigues, Joachim}},
  booktitle    = {{2025 IEEE Workshop on Signal Processing Systems (SiPS). 1-4 Nov. 2025}},
  isbn         = {{979-8-3315-9831-0}},
  keywords     = {{NPU; Systolic arrays; Toroidal dataflow; TPU}},
  language     = {{eng}},
  publisher    = {{IEEE - Institute of Electrical and Electronics Engineers Inc.}},
  title        = {{Scalable low-latency systolic arrays using toroidal and bi-directional dataflows}},
  url          = {{http://dx.doi.org/10.1109/SiPS66314.2025.11261266}},
  doi          = {{10.1109/SiPS66314.2025.11261266}},
  year         = {{2025}},
}

