{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T22:42:29Z","timestamp":1781908949508,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-sa\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100006374","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2213701,2217003,2324864,2328972"],"award-info":[{"award-number":["2213701,2217003,2324864,2328972"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,4]]},"DOI":"10.1145\/3626202.3637569","type":"proceedings-article","created":{"date-parts":[[2024,4,2]],"date-time":"2024-04-02T18:04:51Z","timestamp":1712081091000},"page":"55-66","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":32,"title":["SSR: Spatial Sequential Hybrid Architecture for Latency Throughput Tradeoff in Transformer Acceleration"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3659-339X","authenticated-orcid":false,"given":"Jinming","family":"Zhuang","sequence":"first","affiliation":[{"name":"University of Pittsburgh, Pittsburgh, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7655-4080","authenticated-orcid":false,"given":"Zhuoping","family":"Yang","sequence":"additional","affiliation":[{"name":"University of Pittsburgh, Pittsburgh, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3429-4692","authenticated-orcid":false,"given":"Shixin","family":"Ji","sequence":"additional","affiliation":[{"name":"University of Pittsburgh, Pittsburgh, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3483-8333","authenticated-orcid":false,"given":"Heng","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Maryland, College Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7498-0206","authenticated-orcid":false,"given":"Alex K.","family":"Jones","sequence":"additional","affiliation":[{"name":"University of Pittsburgh, Pittsburgh, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4029-4034","authenticated-orcid":false,"given":"Jingtong","family":"Hu","sequence":"additional","affiliation":[{"name":"University of Pittsburgh, Pittsburgh, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6788-9823","authenticated-orcid":false,"given":"Yiyu","family":"Shi","sequence":"additional","affiliation":[{"name":"University of Notre Dame, Notre Dame, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0493-1844","authenticated-orcid":false,"given":"Peipei","family":"Zhou","sequence":"additional","affiliation":[{"name":"University of Pittsburgh, Pittsburgh, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,4,2]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Manouchehr Rafie. Autonomous vehicles drive ai advances for edge computing. https:\/\/www.3dincites.com\/2021\/07\/autonomous-vehicles-drive-aiadvances- for-edge-computing\/."},{"key":"e_1_3_2_1_2_1","volume-title":"CERN's machine learning could help selfdriving cars","author":"CERN.","year":"2023","unstructured":"CERN. Colliding particles not cars: CERN's machine learning could help selfdriving cars, 2023. Last accessed JANUARY 25, 2023."},{"key":"e_1_3_2_1_3_1","first-page":"5","volume-title":"2019 USENIX Conference on Operational Machine Learning (OpML 19)","author":"Zhang Minjia","year":"2019","unstructured":"Minjia Zhang, Samyam Rajbandari,WenhanWang, Elton Zheng, Olatunji Ruwase, Jeff Rasley, Jason Li, JunhuaWang, and Yuxiong He. Accelerating large scale deep learning inference through {DeepCPU} at microsoft. In 2019 USENIX Conference on Operational Machine Learning (OpML 19), pages 5--7, 2019."},{"key":"e_1_3_2_1_4_1","unstructured":"AMD\/Xilinx. Versal: The First Adaptive Compute Acceleration Platform (ACAP)(WP505)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/2678373.2665678"},{"key":"e_1_3_2_1_6_1","first-page":"1","volume-title":"2016 49th Annual IEEE\/ACM international symposium on microarchitecture (MICRO)","author":"Chung Eric S","year":"2016","unstructured":"AdrianMCaulfield, Eric S Chung, Andrew Putnam, Hari Angepat, Jeremy Fowers, Michael Haselman, Stephen Heil, Matt Humphrey, Puneet Kaur, Joo-Young Kim, et al. A cloud-scale acceleration architecture. In 2016 49th Annual IEEE\/ACM international symposium on microarchitecture (MICRO), pages 1--13. IEEE, 2016."},{"key":"e_1_3_2_1_7_1","first-page":"51","volume-title":"15th USENIX Symposium on Networked Systems Design and Implementation (NSDI","author":"Firestone Daniel","year":"2018","unstructured":"Daniel Firestone, Andrew Putnam, Sambhrama Mundkur, Derek Chiou, Alireza Dabagh, Mike Andrewartha, Hari Angepat, Vivek Bhanu, Adrian Caulfield, Eric Chung, et al. Azure accelerated networking:{SmartNICs} in the public cloud. In 15th USENIX Symposium on Networked Systems Design and Implementation (NSDI , pages 51--66, 2018."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00012"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080246"},{"key":"e_1_3_2_1_10_1","unstructured":"Amazon. Aws inferentia: High performance at the lowest cost in amazon ec2 for deep learning inference."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jiixd.2022.10.001"},{"key":"e_1_3_2_1_12_1","volume-title":"Alex K Jones. REFRESH FPGAs: Sustainable FPGA Chiplet Architectures. In 2023 14th International Green and Sustainable Computing Conference (IGSC)","author":"Zhou Peipei","year":"2023","unstructured":"Peipei Zhou, Jinming Zhuang, Stephen Cahoon, Yue Tang, Zhuoping Yang, Xingzhen Chen, Yiyu Shi, Jingtong Hu, and Alex K Jones. REFRESH FPGAs: Sustainable FPGA Chiplet Architectures. In 2023 14th International Green and Sustainable Computing Conference (IGSC), 2023."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240765.3240801"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Yue Tang Xinyi Zhang Peipei Zhou and Jingtong Hu. Ef-train: Enable efficient on-device cnn training on fpga through data reshaping for online adaptation or personalization. ACM Transactions on Design Automation of Electronic Systems (TODAES) 27(5):1--36 2022.","DOI":"10.1145\/3505633"},{"key":"e_1_3_2_1_15_1","volume-title":"Algorithm- Hardware Co-Design of Attention Mechanism on FPGA Devices. ACM Transactions on Embedded Computing Systems (TECS), 20(5s), sep","author":"Zhang Xinyi","year":"2021","unstructured":"Xinyi Zhang, YawenWu, Peipei Zhou, Xulong Tang, and Jingtong Hu. Algorithm- Hardware Co-Design of Attention Mechanism on FPGA Devices. ACM Transactions on Embedded Computing Systems (TECS), 20(5s), sep 2021."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/FPL53798.2021.00014"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589048"},{"key":"e_1_3_2_1_18_1","volume-title":"Architecture and System Support for Transformer Models (ASSYST@ ISCA","author":"Kim Sehoon","year":"2023","unstructured":"Sehoon Kim, Coleman Hooper, Thanakul Wattanawong, Minwoo Kang, Ruohan Yan, Hasan Genc, Grace Dinh, Qijing Huang, Kurt Keutzer, Michael W Mahoney, et al. Full stack optimization of transformer inference. In Architecture and System Support for Transformer Models (ASSYST@ ISCA 2023), 2023."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543622.3573210"},{"key":"e_1_3_2_1_20_1","first-page":"1","volume-title":"AIM: Accelerating Arbitrary-precision Integer Multiplication on Heterogeneous Reconfigurable Computing Platform Versal ACAP. In 2023 IEEE\/ACM International Conference on Computer Aided Design (ICCAD)","author":"Yang Zhuoping","year":"2023","unstructured":"Zhuoping Yang, Jinming Zhuang, Jiaqi Yin, Cunxi Yu, Alex K Jones, and Peipei Zhou. AIM: Accelerating Arbitrary-precision Integer Multiplication on Heterogeneous Reconfigurable Computing Platform Versal ACAP. In 2023 IEEE\/ACM International Conference on Computer Aided Design (ICCAD), pages 1--9. IEEE, 2023."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASP-DAC58780.2024.10473961"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3508352.3549360"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3489517.3530459"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCAD57390.2023.10323830"},{"key":"e_1_3_2_1_25_1","first-page":"10347","volume-title":"International conference on machine learning","author":"Touvron Hugo","year":"2021","unstructured":"Hugo Touvron, Matthieu Cord, Matthijs Douze, Francisco Massa, Alexandre Sablayrolles, and Herv\u00e9 J\u00e9gou. Training data-efficient image transformers & distillation through attention. In International conference on machine learning, pages 10347--10357. PMLR, 2021."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2017.2785257"},{"key":"e_1_3_2_1_27_1","unstructured":"AMD. Versal AI Core Series."},{"key":"e_1_3_2_1_28_1","volume-title":"GPU Technology Conference","volume":"1","author":"Vanholder Han","year":"2016","unstructured":"Han Vanholder. Efficient inference with tensorrt. In GPU Technology Conference, volume 1, 2016."},{"key":"e_1_3_2_1_29_1","unstructured":"Nvidia. Nvidia aws a10g gpu data sheet."},{"key":"e_1_3_2_1_30_1","unstructured":"AMD\/Xilinx. Versal Adaptive Compute Acceleration Platform."},{"key":"e_1_3_2_1_31_1","volume-title":"Diviml: A module-based heuristic for mapping neural networks onto heterogeneous platforms. arXiv preprint arXiv:2308.00127","author":"Ghannane Yassine","year":"2023","unstructured":"Yassine Ghannane and Mohamed S Abdelfattah. Diviml: A module-based heuristic for mapping neural networks onto heterogeneous platforms. arXiv preprint arXiv:2308.00127, 2023."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00065"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00016"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071027"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071047"},{"key":"e_1_3_2_1_36_1","first-page":"109","volume-title":"2022 32nd International Conference on Field-Programmable Logic and Applications (FPL)","author":"Lit Zhengang","year":"2022","unstructured":"Zhengang Lit, Mengshu Sun, Alec Lu, Haoyu Ma, Geng Yuan, Yanyue Xie, Hao Tang, Yanyu Li, Miriam Leeser, Zhangyang Wang, et al. Auto-vit-acc: An fpgaaware automatic acceleration framework for vision transformer with mixedscheme quantization. In 2022 32nd International Conference on Field-Programmable Logic and Applications (FPL), pages 109--116. IEEE, 2022."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICFPT51103.2020.00011"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3400302.3415609"},{"key":"e_1_3_2_1_39_1","first-page":"1","volume-title":"Proceedings of the 21st ACM SIGPLAN International Conference on Functional Programming, ICFP 2016","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi. TensorFlow: Learning Functions at Scale. In Proceedings of the 21st ACM SIGPLAN International Conference on Functional Programming, ICFP 2016, page 1, New York, NY, USA, 2016. Association for Computing Machinery."},{"key":"e_1_3_2_1_40_1","volume-title":"Advances in Neural Information Processing Systems","volume":"32","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, Alban Desmaison, Andreas Kopf, Edward Yang, Zachary DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. PyTorch: An Imperative Style, High-Performance Deep Learning Library. In H. Wallach, H. Larochelle, A. Beygelzimer, F. d'Alch\u00e9-Buc, E. Fox, and R. Garnett, editors, Advances in Neural Information Processing Systems, volume 32. Curran Associates, Inc., 2019."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18074.2021.9586216"},{"key":"e_1_3_2_1_42_1","first-page":"1","volume-title":"Design Challenges and DSE Perspectives. In 2023 60th ACM\/IEEE Design Automation Conference (DAC)","author":"Zhuang Jinming","year":"2023","unstructured":"Jinming Zhuang, Zhuoping Yang, and Peipei Zhou. High Performance, Low Power Matrix Multiply Design on ACAP: from Architecture, Design Challenges and DSE Perspectives. In 2023 60th ACM\/IEEE Design Automation Conference (DAC), pages 1--6, 2023."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1038\/scientificamerican0792-66"},{"key":"e_1_3_2_1_44_1","volume-title":"Vitis unified software platform","year":"2022","unstructured":"Xilinx. Vitis unified software platform, 2022. Last accessed April 21, 2022."},{"key":"e_1_3_2_1_45_1","unstructured":"Nvidia. System Management Interface SMI | NVIDIA Developer."},{"key":"e_1_3_2_1_46_1","unstructured":"AMD. Zynq UltraScale MPSoC ZCU102 Evaluation Kit ."},{"key":"e_1_3_2_1_47_1","unstructured":"AMD. Alveo U250 Data Center Accelerator Card ."},{"key":"e_1_3_2_1_48_1","unstructured":"AMD\/Xilinx. Board evaluation and management Tool."},{"key":"e_1_3_2_1_49_1","unstructured":"Intel. Stratix10 NX FPGA."}],"event":{"name":"FPGA '24: The 2024 ACM\/SIGDA International Symposium on Field Programmable Gate Arrays","location":"Monterey CA USA","acronym":"FPGA '24","sponsor":["SIGDA ACM Special Interest Group on Design Automation"]},"container-title":["Proceedings of the 2024 ACM\/SIGDA International Symposium on Field Programmable Gate Arrays"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3626202.3637569","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3626202.3637569","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T22:04:02Z","timestamp":1755900242000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3626202.3637569"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4]]},"references-count":49,"alternative-id":["10.1145\/3626202.3637569","10.1145\/3626202"],"URL":"https:\/\/doi.org\/10.1145\/3626202.3637569","relation":{},"subject":[],"published":{"date-parts":[[2024,4]]},"assertion":[{"value":"2024-04-02","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}