{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,10]],"date-time":"2026-05-10T21:38:58Z","timestamp":1778449138286,"version":"3.51.4"},"reference-count":173,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476065"],"award-info":[{"award-number":["62476065"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1016\/j.neucom.2025.130210","type":"journal-article","created":{"date-parts":[[2025,4,17]],"date-time":"2025-04-17T06:31:47Z","timestamp":1744871507000},"page":"130210","update-policy":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":7,"special_numbering":"C","title":["A review of transformer-based human pose estimation: Delving into the relation modeling"],"prefix":"10.1016","volume":"639","author":[{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0009-0000-2662-8996","authenticated-orcid":false,"given":"Jiangning","family":"Wei","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"TianJian","family":"Zou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yida","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinzhu","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minzhen","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dandan","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2025.130210_b1","doi-asserted-by":"crossref","unstructured":"Z. Cao, T. Simon, S.-E. Wei, Y. Sheikh, Realtime multi-person 2d pose estimation using part affinity fields, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017, pp. 7291\u20137299.","DOI":"10.1109\/CVPR.2017.143"},{"issue":"10","key":"10.1016\/j.neucom.2025.130210_b2","doi-asserted-by":"crossref","first-page":"3349","DOI":"10.1109\/TPAMI.2020.2983686","article-title":"Deep high-resolution representation learning for visual recognition","volume":"43","author":"Wang","year":"2020","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2025.130210_b3","series-title":"Ultralytics YOLO","author":"Jocher","year":"2023"},{"key":"10.1016\/j.neucom.2025.130210_b4","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b5","series-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"10.1016\/j.neucom.2025.130210_b6","series-title":"European Conference on Computer Vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.neucom.2025.130210_b7","doi-asserted-by":"crossref","unstructured":"G. Moon, J.Y. Chang, K.M. Lee, Posefix: Model-agnostic general human pose refinement network, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 7773\u20137781.","DOI":"10.1109\/CVPR.2019.00796"},{"key":"10.1016\/j.neucom.2025.130210_b8","doi-asserted-by":"crossref","unstructured":"F. Zhang, X. Zhu, H. Dai, M. Ye, C. Zhu, Distribution-aware coordinate representation for human pose estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 7093\u20137102.","DOI":"10.1109\/CVPR42600.2020.00712"},{"key":"10.1016\/j.neucom.2025.130210_b9","series-title":"Directpose: Direct end-to-end multi-person pose estimation","author":"Tian","year":"2019"},{"key":"10.1016\/j.neucom.2025.130210_b10","doi-asserted-by":"crossref","unstructured":"J. Li, W. Su, Z. Wang, Simple pose: Rethinking and improving a bottom-up approach for multi-person pose estimation, in: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, 2020, pp. 11354\u201311361, 07.","DOI":"10.1609\/aaai.v34i07.6797"},{"key":"10.1016\/j.neucom.2025.130210_b11","doi-asserted-by":"crossref","unstructured":"J. Huang, Z. Zhu, F. Guo, G. Huang, The devil is in the details: Delving into unbiased data processing for human pose estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 5700\u20135709.","DOI":"10.1109\/CVPR42600.2020.00574"},{"key":"10.1016\/j.neucom.2025.130210_b12","series-title":"Tfpose: Direct human pose estimation with transformers","author":"Mao","year":"2021"},{"key":"10.1016\/j.neucom.2025.130210_b13","doi-asserted-by":"crossref","unstructured":"Z. Geng, K. Sun, B. Xiao, Z. Zhang, J. Wang, Bottom-up human pose estimation via disentangled keypoint regression, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 14676\u201314686.","DOI":"10.1109\/CVPR46437.2021.01444"},{"key":"10.1016\/j.neucom.2025.130210_b14","doi-asserted-by":"crossref","unstructured":"Y. Li, S. Zhang, Z. Wang, S. Yang, W. Yang, S.-T. Xia, E. Zhou, Tokenpose: Learning keypoint tokens for human pose estimation, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 11313\u201311322.","DOI":"10.1109\/ICCV48922.2021.01112"},{"key":"10.1016\/j.neucom.2025.130210_b15","doi-asserted-by":"crossref","unstructured":"K. Li, S. Wang, X. Zhang, Y. Xu, W. Xu, Z. Tu, Pose recognition with cascade transformers, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 1944\u20131953.","DOI":"10.1109\/CVPR46437.2021.00198"},{"key":"10.1016\/j.neucom.2025.130210_b16","doi-asserted-by":"crossref","unstructured":"S. Yang, Z. Quan, M. Nie, W. Yang, Transpose: Keypoint localization via transformer, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 11802\u201311812.","DOI":"10.1109\/ICCV48922.2021.01159"},{"key":"10.1016\/j.neucom.2025.130210_b17","series-title":"Hrformer: High-resolution transformer for dense prediction","author":"Yuan","year":"2021"},{"key":"10.1016\/j.neucom.2025.130210_b18","doi-asserted-by":"crossref","unstructured":"D. Shi, X. Wei, L. Li, Y. Ren, W. Tan, End-to-end multi-person pose estimation with transformers, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 11069\u201311078.","DOI":"10.1109\/CVPR52688.2022.01079"},{"key":"10.1016\/j.neucom.2025.130210_b19","series-title":"2022 IEEE 8th International Conference on Computer and Communications","first-page":"1922","article-title":"VitPose: multi-view 3D human pose estimation with vision transformer","author":"Xia","year":"2022"},{"key":"10.1016\/j.neucom.2025.130210_b20","series-title":"European Conference on Computer Vision","first-page":"72","article-title":"Poseur: Direct human pose regression with transformers","author":"Mao","year":"2022"},{"key":"10.1016\/j.neucom.2025.130210_b21","series-title":"ViTPose++: Vision transformer foundation model for generic body pose estimation","author":"Xu","year":"2022"},{"key":"10.1016\/j.neucom.2025.130210_b22","doi-asserted-by":"crossref","unstructured":"Z. Geng, C. Wang, Y. Wei, Z. Liu, H. Li, H. Hu, Human pose as compositional tokens, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 660\u2013671.","DOI":"10.1109\/CVPR52729.2023.00071"},{"key":"10.1016\/j.neucom.2025.130210_b23","doi-asserted-by":"crossref","unstructured":"A. Kanazawa, M.J. Black, D.W. Jacobs, J. Malik, End-to-end recovery of human shape and pose, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 7122\u20137131.","DOI":"10.1109\/CVPR.2018.00744"},{"key":"10.1016\/j.neucom.2025.130210_b24","doi-asserted-by":"crossref","unstructured":"N. Kolotouros, G. Pavlakos, M.J. Black, K. Daniilidis, Learning to reconstruct 3D human pose and shape via model-fitting in the loop, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019, pp. 2252\u20132261.","DOI":"10.1109\/ICCV.2019.00234"},{"key":"10.1016\/j.neucom.2025.130210_b25","doi-asserted-by":"crossref","unstructured":"M. Kocabas, N. Athanasiou, M.J. Black, Vibe: Video inference for human body pose and shape estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 5253\u20135263.","DOI":"10.1109\/CVPR42600.2020.00530"},{"key":"10.1016\/j.neucom.2025.130210_b26","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part VII 16","first-page":"752","article-title":"I2l-meshnet: Image-to-lixel prediction network for accurate 3d human pose and mesh estimation from a single rgb image","author":"Moon","year":"2020"},{"key":"10.1016\/j.neucom.2025.130210_b27","doi-asserted-by":"crossref","unstructured":"H. Zhang, Y. Tian, X. Zhou, W. Ouyang, Y. Liu, L. Wang, Z. Sun, Pymaf: 3d human pose and shape regression with pyramidal mesh alignment feedback loop, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 11446\u201311456.","DOI":"10.1109\/ICCV48922.2021.01125"},{"key":"10.1016\/j.neucom.2025.130210_b28","doi-asserted-by":"crossref","unstructured":"J. Li, C. Xu, Z. Chen, S. Bian, L. Yang, C. Lu, Hybrik: A hybrid analytical-neural inverse kinematics solution for 3d human pose and shape estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 3383\u20133393.","DOI":"10.1109\/CVPR46437.2021.00339"},{"key":"10.1016\/j.neucom.2025.130210_b29","doi-asserted-by":"crossref","unstructured":"K. Lin, L. Wang, Z. Liu, Mesh graphormer, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 12939\u201312948.","DOI":"10.1109\/ICCV48922.2021.01270"},{"key":"10.1016\/j.neucom.2025.130210_b30","doi-asserted-by":"crossref","unstructured":"K. Lin, L. Wang, Z. Liu, End-to-end human pose and mesh reconstruction with transformers, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 1954\u20131963.","DOI":"10.1109\/CVPR46437.2021.00199"},{"key":"10.1016\/j.neucom.2025.130210_b31","doi-asserted-by":"crossref","unstructured":"C. Zheng, S. Zhu, M. Mendieta, T. Yang, C. Chen, Z. Ding, 3d human pose estimation with spatial and temporal transformers, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 11656\u201311665.","DOI":"10.1109\/ICCV48922.2021.01145"},{"key":"10.1016\/j.neucom.2025.130210_b32","doi-asserted-by":"crossref","unstructured":"W. Zhao, W. Wang, Y. Tian, Graformer: Graph-oriented transformer for 3d pose estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 20438\u201320447.","DOI":"10.1109\/CVPR52688.2022.01979"},{"key":"10.1016\/j.neucom.2025.130210_b33","doi-asserted-by":"crossref","unstructured":"S.K. Dwivedi, N. Athanasiou, M. Kocabas, M.J. Black, Learning to regress bodies from images using differentiable semantic rendering, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 11250\u201311259.","DOI":"10.1109\/ICCV48922.2021.01106"},{"key":"10.1016\/j.neucom.2025.130210_b34","doi-asserted-by":"crossref","first-page":"1282","DOI":"10.1109\/TMM.2022.3141231","article-title":"Exploiting temporal contexts with strided transformer for 3d human pose estimation","volume":"25","author":"Li","year":"2022","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.neucom.2025.130210_b35","doi-asserted-by":"crossref","unstructured":"W. Li, H. Liu, H. Tang, P. Wang, L. Van Gool, Mhformer: Multi-hypothesis transformer for 3d human pose estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 13147\u201313156.","DOI":"10.1109\/CVPR52688.2022.01280"},{"key":"10.1016\/j.neucom.2025.130210_b36","doi-asserted-by":"crossref","unstructured":"J. Zhang, Z. Tu, J. Yang, Y. Chen, J. Yuan, Mixste: Seq2seq mixed spatio-temporal encoder for 3d human pose estimation in video, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 13232\u201313242.","DOI":"10.1109\/CVPR52688.2022.01288"},{"key":"10.1016\/j.neucom.2025.130210_b37","series-title":"European Conference on Computer Vision","first-page":"342","article-title":"Cross-attention of disentangled modalities for 3d human mesh recovery with transformers","author":"Cho","year":"2022"},{"key":"10.1016\/j.neucom.2025.130210_b38","doi-asserted-by":"crossref","unstructured":"C. Zheng, M. Mendieta, P. Wang, A. Lu, C. Chen, A lightweight graph transformer network for human mesh reconstruction from 2d human pose, in: Proceedings of the 30th ACM International Conference on Multimedia, 2022, pp. 5496\u20135507.","DOI":"10.1145\/3503161.3547844"},{"key":"10.1016\/j.neucom.2025.130210_b39","doi-asserted-by":"crossref","unstructured":"M. Einfalt, K. Ludwig, R. Lienhart, Uplift and upsample: Efficient 3d human pose estimation with uplifting transformers, in: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, 2023, pp. 2903\u20132913.","DOI":"10.1109\/WACV56688.2023.00292"},{"key":"10.1016\/j.neucom.2025.130210_b40","doi-asserted-by":"crossref","unstructured":"H. Li, B. Shi, W. Dai, H. Zheng, B. Wang, Y. Sun, M. Guo, C. Li, J. Zou, H. Xiong, Pose-oriented transformer with uncertainty-guided refinement for 2d-to-3d human pose estimation, in: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, 2023, pp. 1296\u20131304, 1.","DOI":"10.1609\/aaai.v37i1.25213"},{"key":"10.1016\/j.neucom.2025.130210_b41","doi-asserted-by":"crossref","unstructured":"C. Zheng, X. Liu, G.-J. Qi, C. Chen, Potter: Pooling attention transformer for efficient human mesh recovery, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 1611\u20131620.","DOI":"10.1109\/CVPR52729.2023.00161"},{"key":"10.1016\/j.neucom.2025.130210_b42","doi-asserted-by":"crossref","unstructured":"Q. Zhao, C. Zheng, M. Liu, P. Wang, C. Chen, Poseformerv2: Exploring frequency domain for efficient and robust 3d human pose estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 8877\u20138886.","DOI":"10.1109\/CVPR52729.2023.00857"},{"key":"10.1016\/j.neucom.2025.130210_b43","doi-asserted-by":"crossref","unstructured":"C. Zheng, M. Mendieta, T. Yang, G.-J. Qi, C. Chen, Feater: An efficient network for human reconstruction via feature map-based transformer, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 13945\u201313954.","DOI":"10.1109\/CVPR52729.2023.01340"},{"key":"10.1016\/j.neucom.2025.130210_b44","doi-asserted-by":"crossref","unstructured":"W. Zhu, X. Ma, Z. Liu, L. Liu, W. Wu, Y. Wang, Motionbert: A unified perspective on learning human motion representations, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 15085\u201315099.","DOI":"10.1109\/ICCV51070.2023.01385"},{"key":"10.1016\/j.neucom.2025.130210_b45","doi-asserted-by":"crossref","unstructured":"Z. Dou, Q. Wu, C. Lin, Z. Cao, Q. Wu, W. Wan, T. Komura, W. Wang, Tore: Token reduction for efficient human mesh recovery with transformer, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 15143\u201315155.","DOI":"10.1109\/ICCV51070.2023.01390"},{"key":"10.1016\/j.neucom.2025.130210_b46","series-title":"Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"issue":"7","key":"10.1016\/j.neucom.2025.130210_b47","doi-asserted-by":"crossref","first-page":"1325","DOI":"10.1109\/TPAMI.2013.248","article-title":"Human3. 6m: Large scale datasets and predictive methods for 3d human sensing in natural environments","volume":"36","author":"Ionescu","year":"2013","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2025.130210_b48","doi-asserted-by":"crossref","first-page":"10","DOI":"10.1016\/j.jvcir.2015.06.013","article-title":"A survey of human pose estimation: the body parts parsing based methods","volume":"32","author":"Liu","year":"2015","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.neucom.2025.130210_b49","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2019.102897","article-title":"Monocular human pose estimation: A survey of deep learning-based methods","volume":"192","author":"Chen","year":"2020","journal-title":"Comput. Vis. Image Underst."},{"issue":"5","key":"10.1016\/j.neucom.2025.130210_b50","doi-asserted-by":"crossref","first-page":"538","DOI":"10.1109\/JSTSP.2012.2196975","article-title":"Human pose estimation and activity recognition from multi-view videos: Comparative explorations of recent developments","volume":"6","author":"Holte","year":"2012","journal-title":"IEEE J. Sel. Top. Signal Process."},{"issue":"1","key":"10.1016\/j.neucom.2025.130210_b51","doi-asserted-by":"crossref","first-page":"167","DOI":"10.1007\/s00530-022-00980-0","article-title":"A comprehensive survey on human pose estimation approaches","volume":"29","author":"Dubey","year":"2023","journal-title":"Multimedia Syst."},{"issue":"1","key":"10.1016\/j.neucom.2025.130210_b52","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3603618","article-title":"Deep learning-based human pose estimation: A survey","volume":"56","author":"Zheng","year":"2023","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.neucom.2025.130210_b53","series-title":"A survey on visual transformer","author":"Han","year":"2020"},{"issue":"10s","key":"10.1016\/j.neucom.2025.130210_b54","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3505244","article-title":"Transformers in vision: A survey","volume":"54","author":"Khan","year":"2022","journal-title":"ACM Comput. Surv."},{"issue":"18","key":"10.1016\/j.neucom.2025.130210_b55","doi-asserted-by":"crossref","first-page":"5996","DOI":"10.3390\/s21185996","article-title":"A systematic review of the application of camera-based human pose estimation in the field of sport and physical exercise","volume":"21","author":"Badiola-Bengoa","year":"2021","journal-title":"Sensors"},{"key":"10.1016\/j.neucom.2025.130210_b56","doi-asserted-by":"crossref","first-page":"39","DOI":"10.1016\/j.neunet.2017.02.005","article-title":"Application of structured support vector machine backpropagation to a convolutional neural network for human pose estimation","volume":"92","author":"Witoonchart","year":"2017","journal-title":"Neural Netw."},{"key":"10.1016\/j.neucom.2025.130210_b57","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2021.103055","article-title":"Human pose estimation and its application to action recognition: A survey","volume":"76","author":"Song","year":"2021","journal-title":"J. Vis. Commun. Image Represent."},{"issue":"12","key":"10.1016\/j.neucom.2025.130210_b58","doi-asserted-by":"crossref","first-page":"1966","DOI":"10.3390\/s16121966","article-title":"Human pose estimation from monocular images: A comprehensive survey","volume":"16","author":"Gong","year":"2016","journal-title":"Sensors"},{"key":"10.1016\/j.neucom.2025.130210_b59","doi-asserted-by":"crossref","DOI":"10.1109\/TPAMI.2023.3298850","article-title":"Recovering 3d human mesh from monocular images: A survey","author":"Tian","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2025.130210_b60","article-title":"A review of 3D human body pose estimation and mesh recovery","volume":"128","author":"Huang","year":"2022","journal-title":"Digit. Signal Process."},{"key":"10.1016\/j.neucom.2025.130210_b61","doi-asserted-by":"crossref","DOI":"10.1016\/j.cag.2017.11.008","article-title":"Parametric modeling of 3D human body shape\u2014A survey","author":"Cheng","year":"2018","journal-title":"Comput. Graph."},{"key":"10.1016\/j.neucom.2025.130210_b62","series-title":"Deep learning for 3D human pose estimation and mesh recovery: A survey","author":"Liu","year":"2024"},{"issue":"6","key":"10.1016\/j.neucom.2025.130210_b63","doi-asserted-by":"crossref","first-page":"7616","DOI":"10.1007\/s11227-021-04184-7","article-title":"Human pose, hand and mesh estimation using deep learning: a survey","volume":"78","author":"Toshpulatov","year":"2022","journal-title":"J. Supercomput."},{"key":"10.1016\/j.neucom.2025.130210_b64","article-title":"White-box transformers via sparse rate reduction","volume":"36","author":"Yu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b65","first-page":"9422","article-title":"Learning diverse and discriminative representations via the principle of maximal coding rate reduction","volume":"33","author":"Yu","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b66","series-title":"The information bottleneck method","author":"Tishby","year":"2000"},{"key":"10.1016\/j.neucom.2025.130210_b67","series-title":"2015 IEEE Information Theory Workshop","first-page":"1","article-title":"Deep learning and the information bottleneck principle","author":"Tishby","year":"2015"},{"key":"10.1016\/j.neucom.2025.130210_b68","series-title":"Opening the black box of deep neural networks via information","author":"Shwartz-Ziv","year":"2017"},{"key":"10.1016\/j.neucom.2025.130210_b69","series-title":"Deep variational information bottleneck","author":"Alemi","year":"2016"},{"key":"10.1016\/j.neucom.2025.130210_b70","series-title":"On the relationship between self-attention and convolutional layers","author":"Cordonnier","year":"2019"},{"key":"10.1016\/j.neucom.2025.130210_b71","first-page":"30392","article-title":"Early convolutions help transformers see better","volume":"34","author":"Xiao","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b72","first-page":"12116","article-title":"Do vision transformers see like convolutional neural networks?","volume":"34","author":"Raghu","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b73","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Deep residual learning for image recognition, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"issue":"10","key":"10.1016\/j.neucom.2025.130210_b74","doi-asserted-by":"crossref","DOI":"10.1016\/j.jksuci.2023.101819","article-title":"Joint graph convolution networks and transformer for human pose estimation in sports technique analysis","volume":"35","author":"Cheng","year":"2023","journal-title":"J. King Saud Univ.-Comput. Inf. Sci."},{"key":"10.1016\/j.neucom.2025.130210_b75","first-page":"38571","article-title":"Vitpose: Simple vision transformer baselines for human pose estimation","volume":"35","author":"Xu","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b76","series-title":"End-to-end trainable multi-instance pose estimation with transformers","author":"Stoffl","year":"2021"},{"key":"10.1016\/j.neucom.2025.130210_b77","doi-asserted-by":"crossref","unstructured":"Z. Qiu, Q. Yang, J. Wang, H. Feng, J. Han, E. Ding, C. Xu, D. Fu, J. Wang, PSVT: End-to-end multi-person 3D pose and shape estimation with progressive video transformers, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 21254\u201321263.","DOI":"10.1109\/CVPR52729.2023.02036"},{"key":"10.1016\/j.neucom.2025.130210_b78","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b79","series-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"10.1016\/j.neucom.2025.130210_b80","series-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"10.1016\/j.neucom.2025.130210_b81","series-title":"Gemini: a family of highly capable multimodal models","author":"Team","year":"2023"},{"key":"10.1016\/j.neucom.2025.130210_b82","series-title":"European Conference on Computer Vision","first-page":"424","article-title":"Ppt: token-pruned pose transformer for monocular and multi-view human pose estimation","author":"Ma","year":"2022"},{"key":"10.1016\/j.neucom.2025.130210_b83","doi-asserted-by":"crossref","unstructured":"R. Feng, Y. Gao, T.H.E. Tse, X. Ma, H.J. Chang, Diffpose: Spatiotemporal diffusion model for video-based human pose estimation, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 14861\u201314872.","DOI":"10.1109\/ICCV51070.2023.01365"},{"key":"10.1016\/j.neucom.2025.130210_b84","article-title":"Smpler-x: Scaling up expressive human pose and shape estimation","volume":"36","author":"Cai","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b85","doi-asserted-by":"crossref","unstructured":"J. Li, Z. Yang, X. Wang, J. Ma, C. Zhou, Y. Yang, JOTR: 3D joint contrastive learning with transformers for occluded human mesh recovery, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 9110\u20139121.","DOI":"10.1109\/ICCV51070.2023.00836"},{"key":"10.1016\/j.neucom.2025.130210_b86","doi-asserted-by":"crossref","unstructured":"J. Lin, A. Zeng, H. Wang, L. Zhang, Y. Li, One-stage 3d whole-body mesh recovery with component aware transformer, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 21159\u201321168.","DOI":"10.1109\/CVPR52729.2023.02027"},{"key":"10.1016\/j.neucom.2025.130210_b87","doi-asserted-by":"crossref","unstructured":"X. Shen, Z. Yang, X. Wang, J. Ma, C. Zhou, Y. Yang, Global-to-local modeling for video-based 3d human pose and shape estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 8887\u20138896.","DOI":"10.1109\/CVPR52729.2023.00858"},{"key":"10.1016\/j.neucom.2025.130210_b88","doi-asserted-by":"crossref","unstructured":"J. Tompson, R. Goroshin, A. Jain, Y. LeCun, C. Bregler, Efficient object localization using convolutional networks, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2015, pp. 648\u2013656.","DOI":"10.1109\/CVPR.2015.7298664"},{"key":"10.1016\/j.neucom.2025.130210_b89","doi-asserted-by":"crossref","unstructured":"S.-E. Wei, V. Ramakrishna, T. Kanade, Y. Sheikh, Convolutional pose machines, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 4724\u20134732.","DOI":"10.1109\/CVPR.2016.511"},{"key":"10.1016\/j.neucom.2025.130210_b90","series-title":"Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, the Netherlands, October 11-14, 2016, Proceedings, Part VIII 14","first-page":"483","article-title":"Stacked hourglass networks for human pose estimation","author":"Newell","year":"2016"},{"key":"10.1016\/j.neucom.2025.130210_b91","doi-asserted-by":"crossref","unstructured":"W. Yang, S. Li, W. Ouyang, H. Li, X. Wang, Learning feature pyramids for human pose estimation, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 1281\u20131290.","DOI":"10.1109\/ICCV.2017.144"},{"key":"10.1016\/j.neucom.2025.130210_b92","doi-asserted-by":"crossref","unstructured":"Y. Chen, Z. Wang, Y. Peng, Z. Zhang, G. Yu, J. Sun, Cascaded pyramid network for multi-person pose estimation, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 7103\u20137112.","DOI":"10.1109\/CVPR.2018.00742"},{"key":"10.1016\/j.neucom.2025.130210_b93","doi-asserted-by":"crossref","unstructured":"B. Cheng, B. Xiao, J. Wang, H. Shi, T.S. Huang, L. Zhang, Higherhrnet: Scale-aware representation learning for bottom-up human pose estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 5386\u20135395.","DOI":"10.1109\/CVPR42600.2020.00543"},{"key":"10.1016\/j.neucom.2025.130210_b94","series-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"10.1016\/j.neucom.2025.130210_b95","doi-asserted-by":"crossref","unstructured":"Z. Liu, H. Chen, R. Feng, S. Wu, S. Ji, B. Yang, X. Wang, Deep dual consecutive network for human pose estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 525\u2013534.","DOI":"10.1109\/CVPR46437.2021.00059"},{"key":"10.1016\/j.neucom.2025.130210_b96","doi-asserted-by":"crossref","unstructured":"Z. Liu, R. Feng, H. Chen, S. Wu, Y. Gao, Y. Gao, X. Wang, Temporal feature alignment and mutual information maximization for video-based human pose estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 11006\u201311016.","DOI":"10.1109\/CVPR52688.2022.01073"},{"key":"10.1016\/j.neucom.2025.130210_b97","doi-asserted-by":"crossref","unstructured":"R. Feng, Y. Gao, X. Ma, T.H.E. Tse, H.J. Chang, Mutual information-based temporal difference learning for human pose estimation in video, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 17131\u201317141.","DOI":"10.1109\/CVPR52729.2023.01643"},{"key":"10.1016\/j.neucom.2025.130210_b98","doi-asserted-by":"crossref","unstructured":"J. Dai, H. Qi, Y. Xiong, Y. Li, G. Zhang, H. Hu, Y. Wei, Deformable convolutional networks, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 764\u2013773.","DOI":"10.1109\/ICCV.2017.89"},{"key":"10.1016\/j.neucom.2025.130210_b99","doi-asserted-by":"crossref","unstructured":"J. Li, S. Bian, A. Zeng, C. Wang, B. Pang, W. Liu, C. Lu, Human pose regression with residual log-likelihood estimation, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 11025\u201311034.","DOI":"10.1109\/ICCV48922.2021.01084"},{"key":"10.1016\/j.neucom.2025.130210_b100","doi-asserted-by":"crossref","unstructured":"J. Hu, L. Shen, G. Sun, Squeeze-and-excitation networks, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 7132\u20137141.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"10.1016\/j.neucom.2025.130210_b101","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part VII 16","first-page":"769","article-title":"Pose2mesh: Graph convolutional network for 3d human pose and mesh recovery from a 2d human pose","author":"Choi","year":"2020"},{"key":"10.1016\/j.neucom.2025.130210_b102","series-title":"Seminal Graphics Papers: Pushing the Boundaries, Volume 2","first-page":"851","article-title":"SMPL: A skinned multi-person linear model","author":"Loper","year":"2023"},{"key":"10.1016\/j.neucom.2025.130210_b103","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part VI 16","first-page":"598","article-title":"Star: Sparse trained articulated human body regressor","author":"Osman","year":"2020"},{"key":"10.1016\/j.neucom.2025.130210_b104","series-title":"Embodied hands: Modeling and capturing hands and bodies together","author":"Romero","year":"2022"},{"key":"10.1016\/j.neucom.2025.130210_b105","doi-asserted-by":"crossref","unstructured":"M. Zanfir, A. Zanfir, E.G. Bazavan, W.T. Freeman, R. Sukthankar, C. Sminchisescu, Thundr: Transformer-based 3d human reconstruction with markers, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 12971\u201312980.","DOI":"10.1109\/ICCV48922.2021.01273"},{"key":"10.1016\/j.neucom.2025.130210_b106","doi-asserted-by":"crossref","unstructured":"H. Xu, E.G. Bazavan, A. Zanfir, W.T. Freeman, R. Sukthankar, C. Sminchisescu, Ghum & ghuml: Generative 3d human shape and articulated pose models, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 6184\u20136193.","DOI":"10.1109\/CVPR42600.2020.00622"},{"key":"10.1016\/j.neucom.2025.130210_b107","doi-asserted-by":"crossref","unstructured":"A. Kamath, M. Singh, Y. LeCun, G. Synnaeve, I. Misra, N. Carion, Mdetr-modulated detection for end-to-end multi-modal understanding, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 1780\u20131790.","DOI":"10.1109\/ICCV48922.2021.00180"},{"key":"10.1016\/j.neucom.2025.130210_b108","first-page":"11846","article-title":"Detecting moments and highlights in videos via natural language queries","volume":"34","author":"Lei","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b109","doi-asserted-by":"crossref","unstructured":"G. Moon, H. Choi, K.M. Lee, Accurate 3D hand pose estimation for whole-body 3D human mesh estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 2308\u20132317.","DOI":"10.1109\/CVPRW56347.2022.00257"},{"key":"10.1016\/j.neucom.2025.130210_b110","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part X 16","first-page":"20","article-title":"Monocular expressive body regression through body-driven attention","author":"Choutas","year":"2020"},{"key":"10.1016\/j.neucom.2025.130210_b111","doi-asserted-by":"crossref","unstructured":"Y. Rong, T. Shiratori, H. Joo, Frankmocap: A monocular 3d whole-body pose estimation system via regression and integration, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 1749\u20131759.","DOI":"10.1109\/ICCVW54120.2021.00201"},{"key":"10.1016\/j.neucom.2025.130210_b112","doi-asserted-by":"crossref","unstructured":"R. Girshick, Fast r-cnn, in: Proceedings of the IEEE International Conference on Computer Vision, 2015, pp. 1440\u20131448.","DOI":"10.1109\/ICCV.2015.169"},{"key":"10.1016\/j.neucom.2025.130210_b113","doi-asserted-by":"crossref","unstructured":"B. Xiao, H. Wu, Y. Wei, Simple baselines for human pose estimation and tracking, in: Proceedings of the European Conference on Computer Vision, ECCV, 2018, pp. 466\u2013481.","DOI":"10.1007\/978-3-030-01231-1_29"},{"key":"10.1016\/j.neucom.2025.130210_b114","doi-asserted-by":"crossref","unstructured":"L. Pishchulin, E. Insafutdinov, S. Tang, B. Andres, M. Andriluka, P.V. Gehler, B. Schiele, Deepcut: Joint subset partition and labeling for multi person pose estimation, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 4929\u20134937.","DOI":"10.1109\/CVPR.2016.533"},{"key":"10.1016\/j.neucom.2025.130210_b115","series-title":"Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, the Netherlands, October 11-14, 2016, Proceedings, Part VI 14","first-page":"34","article-title":"Deepercut: A deeper, stronger, and faster multi-person pose estimation model","author":"Insafutdinov","year":"2016"},{"key":"10.1016\/j.neucom.2025.130210_b116","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part VII 16","first-page":"718","article-title":"Differentiable hierarchical graph grouping for multi-person pose estimation","author":"Jin","year":"2020"},{"key":"10.1016\/j.neucom.2025.130210_b117","doi-asserted-by":"crossref","unstructured":"X. Nie, J. Feng, J. Zhang, S. Yan, Single-stage multi-person pose machines, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019, pp. 6951\u20136960.","DOI":"10.1109\/ICCV.2019.00705"},{"key":"10.1016\/j.neucom.2025.130210_b118","series-title":"Objects as points","author":"Zhou","year":"2019"},{"key":"10.1016\/j.neucom.2025.130210_b119","doi-asserted-by":"crossref","unstructured":"W. Mao, Z. Tian, X. Wang, C. Shen, Fcpose: Fully convolutional multi-person pose estimation with dynamic instance-aware convolutions, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 9034\u20139043.","DOI":"10.1109\/CVPR46437.2021.00892"},{"key":"10.1016\/j.neucom.2025.130210_b120","doi-asserted-by":"crossref","unstructured":"S.-H. Zhang, R. Li, X. Dong, P. Rosin, Z. Cai, X. Han, D. Yang, H. Huang, S.-M. Hu, Pose2seg: Detection free human instance segmentation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 889\u2013898.","DOI":"10.1109\/CVPR.2019.00098"},{"key":"10.1016\/j.neucom.2025.130210_b121","doi-asserted-by":"crossref","unstructured":"M. Andriluka, L. Pishchulin, P. Gehler, B. Schiele, 2d human pose estimation: New benchmark and state of the art analysis, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2014, pp. 3686\u20133693.","DOI":"10.1109\/CVPR.2014.471"},{"key":"10.1016\/j.neucom.2025.130210_b122","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IX 16","first-page":"196","article-title":"Whole-body human pose estimation in the wild","author":"Jin","year":"2020"},{"key":"10.1016\/j.neucom.2025.130210_b123","series-title":"Ap-10k: A benchmark for animal pose estimation in the wild","author":"Yu","year":"2021"},{"key":"10.1016\/j.neucom.2025.130210_b124","first-page":"17301","article-title":"Apt-36k: A large-scale benchmark for animal pose estimation and tracking","volume":"35","author":"Yang","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b125","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXV 16","first-page":"17","article-title":"Hand-transformer: Non-autoregressive structured modeling for 3d hand pose estimation","author":"Huang","year":"2020"},{"key":"10.1016\/j.neucom.2025.130210_b126","doi-asserted-by":"crossref","unstructured":"H. Qiu, C. Wang, J. Wang, N. Wang, W. Zeng, Cross view fusion for 3d human pose estimation, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019, pp. 4342\u20134351.","DOI":"10.1109\/ICCV.2019.00444"},{"key":"10.1016\/j.neucom.2025.130210_b127","series-title":"Transfusion: Cross-view fusion with transformer for 3d human pose estimation","author":"Ma","year":"2021"},{"issue":"10.5555","key":"10.1016\/j.neucom.2025.130210_b128","first-page":"2969239","article-title":"Towards real-time object detection with region proposal networks","volume":"9199","author":"Faster","year":"2015","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"5","key":"10.1016\/j.neucom.2025.130210_b129","doi-asserted-by":"crossref","first-page":"1483","DOI":"10.1109\/TPAMI.2019.2956516","article-title":"Cascade R-CNN: High quality object detection and instance segmentation","volume":"43","author":"Cai","year":"2019","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2025.130210_b130","doi-asserted-by":"crossref","unstructured":"T.-Y. Lin, P. Goyal, R. Girshick, K. He, P. Doll\u00e1r, Focal loss for dense object detection, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 2980\u20132988.","DOI":"10.1109\/ICCV.2017.324"},{"key":"10.1016\/j.neucom.2025.130210_b131","doi-asserted-by":"crossref","unstructured":"R. Stewart, M. Andriluka, A.Y. Ng, End-to-end people detection in crowded scenes, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 2325\u20132333.","DOI":"10.1109\/CVPR.2016.255"},{"key":"10.1016\/j.neucom.2025.130210_b132","article-title":"Pointnet++: Deep hierarchical feature learning on point sets in a metric space","volume":"30","author":"Qi","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b133","series-title":"Computer Graphics Forum","first-page":"35","article-title":"Inverse kinematics techniques in computer graphics: A survey","volume":"vol. 37","author":"Aristidou","year":"2018"},{"key":"10.1016\/j.neucom.2025.130210_b134","doi-asserted-by":"crossref","first-page":"3973","DOI":"10.1109\/TIP.2022.3177959","article-title":"Relation-based associative joint location for human pose estimation in videos","volume":"31","author":"Dang","year":"2022","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.neucom.2025.130210_b135","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110287","article-title":"Kinematics modeling network for video-based human pose estimation","volume":"150","author":"Dang","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.neucom.2025.130210_b136","doi-asserted-by":"crossref","unstructured":"J. Li, C. Wang, H. Zhu, Y. Mao, H.-S. Fang, C. Lu, Crowdpose: Efficient crowded scenes pose estimation and a new benchmark, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 10863\u201310872.","DOI":"10.1109\/CVPR.2019.01112"},{"key":"10.1016\/j.neucom.2025.130210_b137","doi-asserted-by":"crossref","unstructured":"T. Von Marcard, R. Henschel, M.J. Black, B. Rosenhahn, G. Pons-Moll, Recovering accurate 3d human pose in the wild using imus and a moving camera, in: Proceedings of the European Conference on Computer Vision, ECCV, 2018, pp. 601\u2013617.","DOI":"10.1007\/978-3-030-01249-6_37"},{"key":"10.1016\/j.neucom.2025.130210_b138","series-title":"2017 International Conference on 3D Vision","first-page":"506","article-title":"Monocular 3d human pose estimation in the wild using improved cnn supervision","author":"Mehta","year":"2017"},{"issue":"1","key":"10.1016\/j.neucom.2025.130210_b139","doi-asserted-by":"crossref","first-page":"4","DOI":"10.1007\/s11263-009-0273-6","article-title":"Humaneva: Synchronized video and motion capture dataset and baseline algorithm for evaluation of articulated human motion","volume":"87","author":"Sigal","year":"2010","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2025.130210_b140","doi-asserted-by":"crossref","unstructured":"P. Patel, C.-H.P. Huang, J. Tesch, D.T. Hoffmann, S. Tripathi, M.J. Black, AGORA: Avatars in geography optimized for regression analysis, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 13468\u201313478.","DOI":"10.1109\/CVPR46437.2021.01326"},{"key":"10.1016\/j.neucom.2025.130210_b141","series-title":"Panoptic studio: A massively multiview system for social interaction capture","author":"Joo","year":"2017"},{"key":"10.1016\/j.neucom.2025.130210_b142","doi-asserted-by":"crossref","first-page":"703","DOI":"10.1007\/s11263-020-01398-9","article-title":"Adafuse: Adaptive multiview fusion for accurate human pose estimation in the wild","volume":"129","author":"Zhang","year":"2021","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2025.130210_b143","doi-asserted-by":"crossref","unstructured":"N. Kolotouros, G. Pavlakos, K. Daniilidis, Convolutional mesh regression for single-image human shape reconstruction, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 4501\u20134510.","DOI":"10.1109\/CVPR.2019.00463"},{"key":"10.1016\/j.neucom.2025.130210_b144","doi-asserted-by":"crossref","unstructured":"K. Lee, I. Lee, S. Lee, Propagating lstm: 3d pose estimation based on joint interdependency, in: Proceedings of the European Conference on Computer Vision, ECCV, 2018, pp. 119\u2013135.","DOI":"10.1007\/978-3-030-01234-2_8"},{"key":"10.1016\/j.neucom.2025.130210_b145","doi-asserted-by":"crossref","first-page":"335","DOI":"10.1016\/j.neucom.2018.10.009","article-title":"Multiview 3D human pose estimation using improved least-squares and LSTM networks","volume":"323","author":"N\u00fa\u00f1ez","year":"2019","journal-title":"Neurocomputing"},{"issue":"13","key":"10.1016\/j.neucom.2025.130210_b146","doi-asserted-by":"crossref","first-page":"15690","DOI":"10.1007\/s10489-022-03312-x","article-title":"High-order local connection network for 3D human pose estimation based on GCN","volume":"52","author":"Wu","year":"2022","journal-title":"Appl. Intell."},{"key":"10.1016\/j.neucom.2025.130210_b147","unstructured":"B.X. Yu, Z. Zhang, Y. Liu, S.-h. Zhong, Y. Liu, C.W. Chen, Gla-gcn: Global-local adaptive graph convolutional network for 3d human pose estimation from monocular video, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 8818\u20138829."},{"key":"10.1016\/j.neucom.2025.130210_b148","doi-asserted-by":"crossref","unstructured":"J. Xu, Y. Guo, Y. Peng, FinePOSE: Fine-Grained Prompt-Driven 3D Human Pose Estimation via Diffusion Models, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 561\u2013570.","DOI":"10.1109\/CVPR52733.2024.00060"},{"key":"10.1016\/j.neucom.2025.130210_b149","series-title":"VLPose: Bridging the domain gap in pose estimation with language-vision tuning","author":"Li","year":"2024"},{"key":"10.1016\/j.neucom.2025.130210_b150","series-title":"Wild2Avatar: Rendering humans behind occlusions","author":"Xiang","year":"2023"},{"key":"10.1016\/j.neucom.2025.130210_b151","series-title":"Occnerf: Self-supervised multi-camera occupancy prediction with neural radiance fields","author":"Zhang","year":"2023"},{"key":"10.1016\/j.neucom.2025.130210_b152","doi-asserted-by":"crossref","unstructured":"C. Guo, T. Jiang, X. Chen, J. Song, O. Hilliges, Vid2avatar: 3d avatar reconstruction from videos in the wild via self-supervised scene decomposition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 12858\u201312868.","DOI":"10.1109\/CVPR52729.2023.01236"},{"key":"10.1016\/j.neucom.2025.130210_b153","doi-asserted-by":"crossref","unstructured":"Y. Chen, S. Huang, T. Yuan, S. Qi, Y. Zhu, S.-C. Zhu, Holistic++ scene understanding: Single-view 3d holistic scene parsing and human pose estimation with human-object interaction and physical commonsense, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019, pp. 8648\u20138657.","DOI":"10.1109\/ICCV.2019.00874"},{"key":"10.1016\/j.neucom.2025.130210_b154","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b155","doi-asserted-by":"crossref","unstructured":"R. Rombach, A. Blattmann, D. Lorenz, P. Esser, B. Ommer, High-resolution image synthesis with latent diffusion models, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 10684\u201310695.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"10.1016\/j.neucom.2025.130210_b156","doi-asserted-by":"crossref","unstructured":"S. Yan, Y. Xiong, D. Lin, Spatial temporal graph convolutional networks for skeleton-based action recognition, in: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32, 2018, 1.","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"10.1016\/j.neucom.2025.130210_b157","doi-asserted-by":"crossref","unstructured":"L. Shi, Y. Zhang, J. Cheng, H. Lu, Two-stream adaptive graph convolutional networks for skeleton-based action recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 12026\u201312035.","DOI":"10.1109\/CVPR.2019.01230"},{"key":"10.1016\/j.neucom.2025.130210_b158","doi-asserted-by":"crossref","unstructured":"Y. Chen, Z. Zhang, C. Yuan, B. Li, Y. Deng, W. Hu, Channel-wise topology refinement graph convolution for skeleton-based action recognition, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 13359\u201313368.","DOI":"10.1109\/ICCV48922.2021.01311"},{"key":"10.1016\/j.neucom.2025.130210_b159","article-title":"Fusing higher-order features in graph neural networks for skeleton-based action recognition","author":"Qin","year":"2022","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.neucom.2025.130210_b160","doi-asserted-by":"crossref","unstructured":"Z. Liu, H. Zhang, Z. Chen, Z. Wang, W. Ouyang, Disentangling and unifying graph convolutions for skeleton-based action recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 143\u2013152.","DOI":"10.1109\/CVPR42600.2020.00022"},{"issue":"4","key":"10.1016\/j.neucom.2025.130210_b161","doi-asserted-by":"crossref","first-page":"4122","DOI":"10.1109\/TPAMI.2022.3188716","article-title":"Adaptive multi-view and temporal fusing transformer for 3d human pose estimation","volume":"45","author":"Shuai","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2025.130210_b162","doi-asserted-by":"crossref","unstructured":"Y. He, R. Yan, K. Fragkiadaki, S.-I. Yu, Epipolar transformer for multi-view human pose estimation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, 2020, pp. 1036\u20131037.","DOI":"10.1109\/CVPRW50498.2020.00526"},{"key":"10.1016\/j.neucom.2025.130210_b163","doi-asserted-by":"crossref","unstructured":"K. Zhou, L. Zhang, F. Lu, X.-D. Zhou, Y. Shi, Efficient Hierarchical Multi-view Fusion Transformer for 3D Human Pose Estimation, in: Proceedings of the 31st ACM International Conference on Multimedia, 2023, pp. 7512\u20137520.","DOI":"10.1145\/3581783.3612098"},{"key":"10.1016\/j.neucom.2025.130210_b164","series-title":"RSB-pose: Robust short-baseline binocular 3D human pose estimation with occlusion handling","author":"Wan","year":"2023"},{"key":"10.1016\/j.neucom.2025.130210_b165","doi-asserted-by":"crossref","DOI":"10.1016\/j.optlaseng.2019.105817","article-title":"A calibration method for binocular stereo vision sensor with short-baseline based on 3D flexible control field","volume":"124","author":"Yang","year":"2020","journal-title":"Opt. Lasers Eng."},{"key":"10.1016\/j.neucom.2025.130210_b166","series-title":"Google mediapipe","author":"Google","year":"2023"},{"key":"10.1016\/j.neucom.2025.130210_b167","doi-asserted-by":"crossref","unstructured":"Z. Liu, Y. Lin, Y. Cao, H. Hu, Y. Wei, Z. Zhang, S. Lin, B. Guo, Swin transformer: Hierarchical vision transformer using shifted windows, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.neucom.2025.130210_b168","series-title":"International Conference on Machine Learning","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","author":"Touvron","year":"2021"},{"key":"10.1016\/j.neucom.2025.130210_b169","doi-asserted-by":"crossref","unstructured":"B. Graham, A. El-Nouby, H. Touvron, P. Stock, A. Joulin, H. J\u00e9gou, M. Douze, Levit: a vision transformer in convnet\u2019s clothing for faster inference, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 12259\u201312269.","DOI":"10.1109\/ICCV48922.2021.01204"},{"key":"10.1016\/j.neucom.2025.130210_b170","doi-asserted-by":"crossref","unstructured":"J. Wei, L. Qin, B. Yu, T. Zou, C. Yan, D. Xiao, Y. Yu, L. Yang, K. Li, J. Liu, VA-AR: Learning Velocity-Aware Action Representations with Mixture of Window Attention, in: Proceedings of the AAAI Conference on Artificial Intelligence, 39, (8) 2025, pp. 8286\u20138294.","DOI":"10.1609\/aaai.v39i8.32894"},{"key":"10.1016\/j.neucom.2025.130210_b171","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.neucom.2025.130210_b172","doi-asserted-by":"crossref","DOI":"10.1109\/TCSVT.2024.3397997","article-title":"Clipose: Category-level object pose estimation with pre-trained vision-language knowledge","author":"Lin","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.neucom.2025.130210_b173","series-title":"Language knowledge-assisted representation learning for skeleton-based action recognition","author":"Xu","year":"2023"}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S0925231225008823?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S0925231225008823?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2025,7,22]],"date-time":"2025-07-22T14:47:02Z","timestamp":1753195622000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231225008823"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7]]},"references-count":173,"alternative-id":["S0925231225008823"],"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.neucom.2025.130210","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2025,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A review of transformer-based human pose estimation: Delving into the relation modeling","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.neucom.2025.130210","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2025 Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"130210"}}