{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,3]],"date-time":"2026-01-03T03:29:12Z","timestamp":1767410952287,"version":"3.48.0"},"reference-count":35,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"1","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2026,1,1]]},"DOI":"10.1587\/transinf.2025pap0007","type":"journal-article","created":{"date-parts":[[2025,6,23]],"date-time":"2025-06-23T18:06:40Z","timestamp":1750702000000},"page":"23-31","source":"Crossref","is-referenced-by-count":0,"title":["Loop Unrolling and DFG Partitioning for CGRAs: A Case Study of The Lattice Boltzmann Method"],"prefix":"10.1587","volume":"E109.D","author":[{"given":"Toshiyuki","family":"ICHIBA","sequence":"first","affiliation":[{"name":"Fujitsu Ltd."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yasuhiro","family":"WATANABE","sequence":"additional","affiliation":[{"name":"Fujitsu Ltd."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takahide","family":"YOSHIKAWA","sequence":"additional","affiliation":[{"name":"Fujitsu Ltd."}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"doi-asserted-by":"crossref","unstructured":"[1] B. Adhi, C. Cortes, Y. Tan, T. Kojima, A. Podobas, and K. Sano, \u201cExploration framework for synthesizable cgras targeting hpc: Initial design and evaluation,\u201d IPDPSW, pp.639-646, 2022. 10.1109\/ipdpsw55747.2022.00113","key":"1","DOI":"10.1109\/IPDPSW55747.2022.00113"},{"doi-asserted-by":"crossref","unstructured":"[2] E. Del Sozzo, X. Wang, B. Adhi, C. Cortes, J. Anderson, and K. Sano, \u201cExploration of trade-offs between general-purpose and specialized processing elements in hpc-oriented cgra,\u201d IPDPS, pp.668-680, 2024. 10.1109\/ipdps57955.2024.00065","key":"2","DOI":"10.1109\/IPDPS57955.2024.00065"},{"doi-asserted-by":"publisher","unstructured":"[3] L. Liu, J. Zhu, Z. Li, Y. Lu, Y. Deng, J. Han, S. Yin, and S. Wei, \u201cA survey of coarse-grained reconfigurable architecture and design: Taxonomy, challenges, and applications,\u201d ACM Comput. Surv., vol.52, no.6, pp.1-39, Oct. 2019. 10.1145\/3357375","key":"3","DOI":"10.1145\/3357375"},{"doi-asserted-by":"publisher","unstructured":"[4] A. Podobas, K. Sano, and S. Matsuoka, \u201cA survey on coarse-grained reconfigurable architectures from a performance perspective,\u201d IEEE Access, vol.8, pp.146719-146743, 2020. 10.1109\/access.2020.3012084","key":"4","DOI":"10.1109\/ACCESS.2020.3012084"},{"doi-asserted-by":"crossref","unstructured":"[5] D. Rossinelli, Y.-H. Tang, K. Lykov, D. Alexeev, M. Bernaschi, P. Hadjidoukas, M. Bisson, W. Joubert, C. Conti, G. Karniadakis, M. Fatica, I. Pivkin, and P. Koumoutsakos, \u201cThe in-silico lab-on-a-chip: petascale and high-throughput simulations of microfluidics at cell resolution,\u201d Proceedings of SC \u201915, pp.1-12, 2015. 10.1145\/2807591.2807677","key":"5","DOI":"10.1145\/2807591.2807677"},{"doi-asserted-by":"publisher","unstructured":"[6] K. Ando, R. Bale, C. Li, S. Matsuoka, K. Onishi, and M. Tsubokura, \u201cDigital transformation of droplet\/aerosol infection risk assessment realized on \u201cfugaku\u201d for the fight against covid-19,\u201d Int. J. High Perform. Comput. Appl., vol.36, no.5-6, pp.568-586, Nov. 2022. 10.1177\/10943420221116056","key":"6","DOI":"10.1177\/10943420221116056"},{"doi-asserted-by":"crossref","unstructured":"[7] D. Rossinelli, B. Hejazialhosseini, P. Hadjidoukas, C. Bekas, A. Curioni, A. Bertsch, S. Futral, S.J. Schmidt, N.A. Adams, and P. Koumoutsakos, \u201c11 pflop\/s simulations of cloud cavitation collapse,\u201d Proceedings of SC \u201913, pp.1-13, 2013. 10.1145\/2503210.2504565","key":"7","DOI":"10.1145\/2503210.2504565"},{"doi-asserted-by":"publisher","unstructured":"[8] E. Calore, A. Gabbana, J. Kraus, E. Pellegrini, S.F. Schifano, and R. Tripiccione, \u201cMassively parallel lattice-boltzmann codes on large gpu clusters,\u201d Parallel Computing, vol.58, pp.1-24, Oct. 2016. 10.1016\/j.parco.2016.08.005","key":"8","DOI":"10.1016\/j.parco.2016.08.005"},{"doi-asserted-by":"crossref","unstructured":"[9] J. Li, A. Bobyr, S. Boehm, W. Brantley, H. Brunst, A. Cavelan, S. Chandrasekaran, J. Cheng, F.M. Ciorba, M. Colgrove, T. Curtis, C. Daley, M. Ferrato, M.G. de Souza, N. Hagerty, R. Henschel, G. Juckeland, J. Kelling, K. Li, R. Lieberman, K. McMahon, E. Melnichenko, M.A. Neggaz, H. Ono, C. Ponder, D. Raddatz, S. Schueller, R. Searles, F. Vasilev, V.M. Vergara, B. Wang, B. Wesarg, S. Wienke, and M. Zavala, \u201cSpechpc 2021 benchmark suites for modern hpc systems,\u201d ICPE \u201922, pp.15-16, 2022. 10.1145\/3491204.3527498","key":"9","DOI":"10.1145\/3491204.3527498"},{"doi-asserted-by":"crossref","unstructured":"[10] O. Ragheb, T. Yu, R. Beidas, and J. Anderson, \u201cElastic multi-context cgras,\u201d IPDPSW, pp.655-662, 2022. 10.1109\/ipdpsw55747.2022.00115","key":"10","DOI":"10.1109\/IPDPSW55747.2022.00115"},{"doi-asserted-by":"publisher","unstructured":"[11] T. Kojima, N.A.V. Doan, and H. Amano, \u201cGenmap: A genetic algorithmic approach for optimizing spatial mapping of coarse-grained reconfigurable architectures,\u201d IEEE Trans. VLSI Systems, vol.28, no.11, pp.2383-2396, 2020. 10.1109\/tvlsi.2020.3009225","key":"11","DOI":"10.1109\/TVLSI.2020.3009225"},{"doi-asserted-by":"crossref","unstructured":"[12] V. Chacko and J. Anderson, \u201cPower, performance and area consequences of multi-context support in cgras,\u201d ASAP, pp.49-52, 2021. 10.1109\/asap52443.2021.00014","key":"12","DOI":"10.1109\/ASAP52443.2021.00014"},{"doi-asserted-by":"crossref","unstructured":"[13] O. Ragheb, T. Yu, D. Ma, and J. Anderson, \u201cModeling and exploration of elastic cgras,\u201d FPL, pp.404-410, 2022. 10.1109\/fpl57034.2022.00067","key":"13","DOI":"10.1109\/FPL57034.2022.00067"},{"doi-asserted-by":"crossref","unstructured":"[14] N. Jouppi, G. Kurian, S. Li, P. Ma, R. Nagarajan, L. Nai, N. Patil, S. Subramanian, A. Swing, B. Towles, C. Young, X. Zhou, Z. Zhou, and D.A. Patterson, \u201cTpu v4: An optically reconfigurable supercomputer for machine learning with hardware support for embeddings,\u201d ISCA, pp.1-14, 2023. 10.1145\/3579371.3589350","key":"14","DOI":"10.1145\/3579371.3589350"},{"doi-asserted-by":"crossref","unstructured":"[15] K.J.M. Martin, \u201cTwenty years of automated methods for mapping applications on cgra,\u201d IPDPSW, pp.679-686, 2022. 10.1109\/ipdpsw55747.2022.00118","key":"15","DOI":"10.1109\/IPDPSW55747.2022.00118"},{"doi-asserted-by":"crossref","unstructured":"[16] S.A. Chin, N. Sakamoto, A. Rui, J. Zhao, J.H. Kim, Y. Hara-Azumi, and J. Anderson, \u201cCgra-me: A unified framework for cgra modelling and exploration,\u201d ASAP, pp.184-189, 2017. 10.1109\/asap.2017.7995277","key":"16","DOI":"10.1109\/ASAP.2017.7995277"},{"doi-asserted-by":"crossref","unstructured":"[17] D. Wijerathne, Z. Li, T.K. Bandara, and T. Mitra, \u201cPanorama: divide-and-conquer approach for mapping complex loop kernels on cgra,\u201d Proceedings of DAC \u201922, pp.127-132, 2022. 10.1145\/3489517.3530429","key":"17","DOI":"10.1145\/3489517.3530429"},{"doi-asserted-by":"publisher","unstructured":"[18] S. Chen, C. Cai, S. Zheng, J. Li, G. Zhu, J. Li, Y. Yan, Y. Dai, W. Yin, and L. Wang, \u201cHiercgra: A novel framework for large-scale cgra with hierarchical modeling and automated design space exploration,\u201d ACM Trans. Reconfigurable Technol. Syst., vol.17, no.2, pp.1-31, 2024. 10.1145\/3656176","key":"18","DOI":"10.1145\/3656176"},{"doi-asserted-by":"crossref","unstructured":"[19] O. Ragheb and J.H. Anderson, \u201cClumap: Clustered mapper for cgras with predication,\u201d Proceedings of DAC \u201924, pp.1-6, 2024. 10.1145\/3649329.3658269","key":"19","DOI":"10.1145\/3649329.3658269"},{"doi-asserted-by":"crossref","unstructured":"[20] C. Tirelli, R. Otoni, and L. Pozzi, \u201cMonomorphism-based cgra mapping via space and time decoupling,\u201d Proceedings of DATE 2025, 2025. 10.23919\/date64628.2025.10992940","key":"20","DOI":"10.23919\/DATE64628.2025.10992940"},{"doi-asserted-by":"crossref","unstructured":"[21] F.T. Ramos, P.E.F. Realino, W.A. Junior, A.B. Vieira, R.S. Ferreira, and J.A.M. Nacif, \u201cSmartmap: Architecture-agnostic cgra mapping using graph traversal and reinforcement learning,\u201d Proceedings of DATE 2025, 2025. 10.23919\/date64628.2025.10992752","key":"21","DOI":"10.23919\/DATE64628.2025.10992752"},{"doi-asserted-by":"publisher","unstructured":"[22] L. Biferale, F. Mantovani, M. Pivanti, F. Pozzati, M. Sbragaglia, A. Scagliarini, S.F. Schifano, F. Toschi, and R. Tripiccione, \u201cAn optimized d2q37 lattice boltzmann code on gp-gpus,\u201d Computers &amp; Fluids, vol.80, pp.55-62, 2013. 10.1016\/j.compfluid.2012.06.003","key":"22","DOI":"10.1016\/j.compfluid.2012.06.003"},{"doi-asserted-by":"crossref","unstructured":"[23] T. Kojima, B. Adhi, C. Cortes, Y. Tan, and K. Sano, \u201cAn architecture- independent cgra compiler enabling openmp applications,\u201d IPDPSW, pp.631-638, 2022. 10.1109\/ipdpsw55747.2022.00112","key":"23","DOI":"10.1109\/IPDPSW55747.2022.00112"},{"doi-asserted-by":"crossref","unstructured":"[24] N. Sambhus and T.S. Abdelrahman, \u201cReuse-aware partitioning of dataflow graphs on a tightly-coupled cgra,\u201d ISPA\/BDCloud\/SocialCom\/SustainCom, pp.458-467, 2022. 10.1109\/ispa-bdcloud-socialcom-sustaincom57177.2022.00065","key":"24","DOI":"10.1109\/ISPA-BDCloud-SocialCom-SustainCom57177.2022.00065"},{"doi-asserted-by":"crossref","unstructured":"[25] T. Achterberg, T. Berthold, T. Koch, and K. Wolter, \u201cConstraint integer programming: A new approach to integrate cp and mip,\u201d CPAIOR 2008, pp.6-20, 2008. 10.1007\/978-3-540-68155-7_4","key":"25","DOI":"10.1007\/978-3-540-68155-7_4"},{"unstructured":"[26] G. Karypis and V. Kumar, \u201cMetis: A software package for partitioning unstructured graphs, partitioning meshes, and computing fill-reducing orderings of sparse matrices,\u201d tech. rep., Department of Computer Science and Engineering, University of Minnesota, 1997.","key":"26"},{"doi-asserted-by":"crossref","unstructured":"[27] G. Karypis, R. Aggarwal, V. Kumar, and S. Shekhar, \u201cMultilevel hypergraph partitioning: Application in vlsi domain,\u201d Proceedings of DAC \u201997, pp.526-529, 1997. 10.1145\/266021.266273","key":"27","DOI":"10.1145\/266021.266273"},{"doi-asserted-by":"crossref","unstructured":"[28] H. Lee, D. Nguyen, and J. Lee, \u201cOptimizing stream program performance on cgra-based systems,\u201d Proceedings of DAC \u201915, pp.1-6, 2015. 10.1145\/2744769.2744884","key":"28","DOI":"10.1145\/2744769.2744884"},{"doi-asserted-by":"crossref","unstructured":"[29] J. Weng, S. Liu, V. Dadu, Z. Wang, P. Shah, and T. Nowatzki, \u201cDsagen: Synthesizing programmable spatial accelerators,\u201d ISCA, pp.268-281, 2020. 10.1109\/isca45697.2020.00032","key":"29","DOI":"10.1109\/ISCA45697.2020.00032"},{"doi-asserted-by":"crossref","unstructured":"[30] T. Kojima, A. Ohwada, and H. Amano, \u201cMapping-aware kernel partitioning method for cgras assisted by deep learning,\u201d IEEE Trans. Parallel and Distributed Systems, vol.33, no.5, pp.1213-1230, 2022. 10.1109\/tpds.2021.3107746","key":"30","DOI":"10.1109\/TPDS.2021.3107746"},{"doi-asserted-by":"crossref","unstructured":"[31] D. LaSalle and G. Karypis, \u201cMulti-threaded graph partitioning,\u201d IPDPS 2013, pp.225-236, IEEE, 2013. 10.1109\/ipdps.2013.50","key":"31","DOI":"10.1109\/IPDPS.2013.50"},{"doi-asserted-by":"crossref","unstructured":"[32] W.L. Lee, D.-L. Lin, T.-W. Huang, S. Jiang, T.-Y. Ho, Y. Lin, and B. Yu, \u201cG-kway: Multilevel gpu-accelerated k-way graph partitioner,\u201d Proceedings of DAC \u201924, pp.1-6, 2024. 10.1145\/3649329.3656238","key":"32","DOI":"10.1145\/3649329.3656238"},{"doi-asserted-by":"publisher","unstructured":"[33] B. Wu, Z. Xiao, P. Lin, Z. Tang, and K. Li, \u201cCritical path awareness techniques for large-scale graph partitioning,\u201d IEEE Trans. on Sustainable Computing, vol.8, no.3, pp.412-422, 2023. 10.1109\/tsusc.2023.3263172","key":"33","DOI":"10.1109\/TSUSC.2023.3263172"},{"doi-asserted-by":"crossref","unstructured":"[34] K. Sano, O. Pell, W. Luk, and S. Yamamoto, \u201cFpga-based streaming computation for lattice boltzmann method,\u201d FPT, pp.233-236, 2007. 10.1109\/fpt.2007.4439254","key":"34","DOI":"10.1109\/FPT.2007.4439254"},{"doi-asserted-by":"crossref","unstructured":"[35] W. Altoyan and J.J. Alonso, \u201cAccelerating the lattice boltzmann method,\u201d 2023 IEEE Aerospace Conference, pp.1-20, 2023. 10.1109\/aero55745.2023.10115537","key":"35","DOI":"10.1109\/AERO55745.2023.10115537"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.jstage.jst.go.jp\/article\/transinf\/E109.D\/1\/E109.D_2025PAP0007\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,3]],"date-time":"2026-01-03T03:25:55Z","timestamp":1767410755000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.jstage.jst.go.jp\/article\/transinf\/E109.D\/1\/E109.D_2025PAP0007\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,1]]},"references-count":35,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026]]}},"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1587\/transinf.2025pap0007","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2026,1,1]]},"article-number":"2025PAP0007"}}