{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,2]],"date-time":"2025-11-02T13:57:31Z","timestamp":1762091851133,"version":"build-2065373602"},"reference-count":35,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,7,1]],"date-time":"2023-07-01T00:00:00Z","timestamp":1688169600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,7,1]],"date-time":"2023-07-01T00:00:00Z","timestamp":1688169600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,7]]},"DOI":"10.1109\/icme55011.2023.00330","type":"proceedings-article","created":{"date-parts":[[2023,8,25]],"date-time":"2023-08-25T17:18:08Z","timestamp":1692983888000},"page":"1925-1930","source":"Crossref","is-referenced-by-count":3,"title":["Improving Vision Transformers with Nested Multi-head Attentions"],"prefix":"10.1109","author":[{"given":"Jiquan","family":"Peng","sequence":"first","affiliation":[{"name":"Yanshan University,School of Information Science and Engineering,Qinhuangdao,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chaozhuo","family":"Li","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Zhao","sequence":"additional","affiliation":[{"name":"Yanshan University,School of Information Science and Engineering,Qinhuangdao,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuting","family":"Lin","sequence":"additional","affiliation":[{"name":"Yanshan University,School of Information Science and Engineering,Qinhuangdao,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaohan","family":"Fang","sequence":"additional","affiliation":[{"name":"Yanshan University,School of Information Science and Engineering,Qinhuangdao,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jibing","family":"Gong","sequence":"additional","affiliation":[{"name":"Yanshan University,School of Information Science and Engineering,Qinhuangdao,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"journal-title":"Cascaded head-colliding attention","year":"2021","author":"zheng","key":"ref13"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2013.77"},{"journal-title":"Repulsive attention Rethinking multi-head attention as bayesian inference","year":"2020","author":"an","key":"ref12"},{"journal-title":"Probabilistic Reasoning in Intelligent Systems Networks of Plausible Inference","year":"1988","author":"pearl","key":"ref34"},{"key":"ref15","first-page":"13209","article-title":"House: Knowledge graph embedding with householder parameterization","author":"li","year":"2022","journal-title":"International Conference on Machine Learning"},{"key":"ref14","first-page":"9377","article-title":"Going deeper into permutation-sensitive graph neural networks","author":"huang","year":"2022","journal-title":"International Conference on Machine Learning"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref30","first-page":"12270","article-title":"Auto-former: Searching transformers for visual recognition","author":"chen","year":"2021","journal-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision"},{"journal-title":"Multi-head attention with disagreement regularization","year":"2018","author":"li","key":"ref11"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531982"},{"key":"ref32","article-title":"Learning multiple layers of features from tiny images","author":"krizhevsky","year":"2009","journal-title":"Technical Report"},{"journal-title":"Learning on large-scale text-attributed graphs via variational inference","year":"2022","author":"zhao","key":"ref2"},{"key":"ref1","article-title":"Attention is all you need","volume":"30","author":"vaswani","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1989.1.4.541"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462926"},{"key":"ref19","first-page":"9355","article-title":"Twins: Revisiting the design of spatial attention in vision transformers","volume":"34","author":"chu","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.97"},{"journal-title":"DeepViT Towards Deeper Vision Transformer","year":"2021","author":"zhou","key":"ref24"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"ref26","first-page":"20014","article-title":"Xcit: Cross-covariance image transformers","volume":"34","author":"ali","year":"2021","journal-title":"Advances in neural information processing systems"},{"journal-title":"Kvt k-nn attention for boosting vision transformers","year":"2021","author":"wang","key":"ref25"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICME52920.2022.9859720"},{"journal-title":"Regionvit Regional-to-local attention for vision transformers","year":"2021","author":"chen","key":"ref22"},{"journal-title":"Shuffle transformer Rethinking spatial shuffle for vision transformer","year":"2021","author":"huang","key":"ref21"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00009"},{"key":"ref27","article-title":"Conditional positional encodings for vision transformers","author":"chu","year":"2021","journal-title":"arXiv preprint arXiv 2102 10496"},{"key":"ref29","first-page":"12259","article-title":"Levit: a vision transformer in convnet&#x2019;s clothing for faster inference","author":"graham","year":"2021","journal-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref7","first-page":"15908","article-title":"Transformer in transformer","volume":"34","author":"han","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01170"},{"journal-title":"An image is worth 16x16 words Transformers for image recognition at scale","year":"2020","author":"dosovitskiy","key":"ref4"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401057"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"ref5","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","author":"touvron","year":"2021","journal-title":"International Conference on Machine Learning"}],"event":{"name":"2023 IEEE International Conference on Multimedia and Expo (ICME)","start":{"date-parts":[[2023,7,10]]},"location":"Brisbane, Australia","end":{"date-parts":[[2023,7,14]]}},"container-title":["2023 IEEE International Conference on Multimedia and Expo (ICME)"],"original-title":[],"link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/http\/xplorestaging.ieee.org\/ielx7\/10219544\/10219545\/10219662.pdf?arnumber=10219662","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,9,18]],"date-time":"2023-09-18T17:42:14Z","timestamp":1695058934000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/ieeexplore.ieee.org\/document\/10219662\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,7]]},"references-count":35,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1109\/icme55011.2023.00330","relation":{},"subject":[],"published":{"date-parts":[[2023,7]]}}}