{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T21:07:04Z","timestamp":1781816824447,"version":"3.54.5"},"reference-count":63,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Engineering Applications of Artificial Intelligence"],"published-print":{"date-parts":[[2024,8]]},"DOI":"10.1016\/j.engappai.2024.108682","type":"journal-article","created":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T08:53:54Z","timestamp":1717232034000},"page":"108682","update-policy":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":6,"special_numbering":"C","title":["Unified spatio-temporal attention mixformer for visual object tracking"],"prefix":"10.1016","volume":"134","author":[{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0009-0003-3001-7452","authenticated-orcid":false,"given":"Minho","family":"Park","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0000-0002-0654-491X","authenticated-orcid":false,"given":"Gang-Joon","family":"Yoon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinjoo","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0000-0003-0001-1845","authenticated-orcid":false,"given":"Sang Min","family":"Yoon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"4","key":"10.1016\/j.engappai.2024.108682_b1","doi-asserted-by":"crossref","first-page":"573","DOI":"10.3390\/rs13040573","article-title":"Multeye: Monitoring system for real-time vehicle detection, tracking and speed estimation from UAV imagery on edge-computing platforms","volume":"13","author":"Balamuralidhar","year":"2021","journal-title":"Remote Sens."},{"key":"10.1016\/j.engappai.2024.108682_b2","doi-asserted-by":"crossref","unstructured":"Bertinetto, L., Valmadre, J., Henriques, J.F., Vedaldi, A., Torr, P.H., 2016. Fully-convolutional siamese networks for object tracking. In: Proc. ECCV. pp. 850\u2013865.","DOI":"10.1007\/978-3-319-48881-3_56"},{"key":"10.1016\/j.engappai.2024.108682_b3","doi-asserted-by":"crossref","unstructured":"Bhat, G., Johnander, J., Danelljan, M., Khan, F.S., Felsberg, M., 2018. Unveiling the power of deep tracking. In: Proc. ECCV. pp. 483\u2013498.","DOI":"10.1007\/978-3-030-01216-8_30"},{"key":"10.1016\/j.engappai.2024.108682_b4","article-title":"Signature verification using a siamese time delay neural network","volume":"6","author":"Bromley","year":"1993","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2024.108682_b5","doi-asserted-by":"crossref","unstructured":"Chen, H., Wang, Y., Guo, T., Xu, C., Deng, Y., Liu, Z., Ma, S., Xu, C., Xu, C., Gao, W., 2021a. Pre-trained image processing transformer. In: Proc. CVPR. pp. 12299\u201312310.","DOI":"10.1109\/CVPR46437.2021.01212"},{"key":"10.1016\/j.engappai.2024.108682_b6","doi-asserted-by":"crossref","unstructured":"Chen, X., Yan, B., Zhu, J., Wang, D., Yang, X., Lu, H., 2021b. Transformer tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 8126\u20138135.","DOI":"10.1109\/CVPR46437.2021.00803"},{"key":"10.1016\/j.engappai.2024.108682_b7","doi-asserted-by":"crossref","unstructured":"Chen, Z., Zhong, B., Li, G., Zhang, S., Ji, R., 2020. Siamese box adaptive network for visual tracking. In: Proc. CVPR. pp. 6668\u20136677.","DOI":"10.1109\/CVPR42600.2020.00670"},{"key":"10.1016\/j.engappai.2024.108682_b8","doi-asserted-by":"crossref","unstructured":"Cheng, S., Zhong, B., Li, G., Liu, X., Tang, Z., Li, X., Wang, J., 2021. Learning to filter: Siamese relation network for robust tracking. In: Proc. CVPR. pp. 4421\u20134431.","DOI":"10.1109\/CVPR46437.2021.00440"},{"key":"10.1016\/j.engappai.2024.108682_b9","doi-asserted-by":"crossref","unstructured":"Cui, Y., Jiang, C., Wang, L., Wu, G., 2022. Mixformer: End-to-end tracking with iterative mixed attention. In: Proc. CVPR. pp. 13608\u201313618.","DOI":"10.1109\/CVPR52688.2022.01324"},{"key":"10.1016\/j.engappai.2024.108682_b10","series-title":"Proc. ICLR","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2021"},{"key":"10.1016\/j.engappai.2024.108682_b11","doi-asserted-by":"crossref","unstructured":"Fan, H., Lin, L., Yang, F., Chu, P., Deng, G., Yu, S., Bai, H., Xu, Y., Liao, C., Ling, H., 2019. Lasot: A high-quality benchmark for large-scale single object tracking. In: Proc. CVPR. pp. 5374\u20135383.","DOI":"10.1109\/CVPR.2019.00552"},{"key":"10.1016\/j.engappai.2024.108682_b12","series-title":"IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021, Virtual, June 19-25, 2021","first-page":"13774","article-title":"Stmtrack: Template-free visual tracking with space\u2013time memory networks","author":"Fu","year":"2021"},{"key":"10.1016\/j.engappai.2024.108682_b13","doi-asserted-by":"crossref","unstructured":"Gao, S., Zhou, C., Ma, C., Wang, X., Yuan, J., 2022. Aiatrack: Attention in attention for transformer visual tracking. In: Proc. ECCV. pp. 146\u2013164.","DOI":"10.1007\/978-3-031-20047-2_9"},{"key":"10.1016\/j.engappai.2024.108682_b14","doi-asserted-by":"crossref","unstructured":"Gao, S., Zhou, C., Zhang, J., 2023. Generalized relation modeling for transformer tracking. In: Proc. CVPR. pp. 18686\u201318695.","DOI":"10.1109\/CVPR52729.2023.01792"},{"issue":"5","key":"10.1016\/j.engappai.2024.108682_b15","doi-asserted-by":"crossref","first-page":"2526","DOI":"10.1109\/TIP.2018.2806280","article-title":"Good features to correlate for visual tracking","volume":"27","author":"Gundogdu","year":"2018","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.engappai.2024.108682_b16","doi-asserted-by":"crossref","unstructured":"Guo, D., Shao, Y., Cui, Y., Wang, Z., Zhang, L., Shen, C., 2021. Graph attention tracking. In: Proc. CVPR. pp. 9543\u20139552.","DOI":"10.1109\/CVPR46437.2021.00942"},{"issue":"1","key":"10.1016\/j.engappai.2024.108682_b17","doi-asserted-by":"crossref","first-page":"155","DOI":"10.1109\/TCSVT.2018.2888492","article-title":"Adaptive discriminative deep correlation filter for visual object tracking","volume":"30","author":"Han","year":"2018","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.engappai.2024.108682_b18","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R., 2022. Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 16000\u201316009.","DOI":"10.1109\/CVPR52688.2022.01553"},{"issue":"5","key":"10.1016\/j.engappai.2024.108682_b19","doi-asserted-by":"crossref","first-page":"1562","DOI":"10.1109\/TPAMI.2019.2957464","article-title":"Got-10k: A large high-diversity benchmark for generic object tracking in the wild","volume":"43","author":"Huang","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2024.108682_b20","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.111604","article-title":"Autonomous obstacle avoidance and target tracking of uav: Transformer for observation sequence in reinforcement learning","volume":"290","author":"Jiang","year":"2024","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.engappai.2024.108682_b21","doi-asserted-by":"crossref","unstructured":"Li, Y., Fu, C., Huang, Z., Zhang, Y., Pan, J., 2020. Keyfilter-aware real-time uav object tracking. In: Proc. Int. Conf. Robotics and Automation. ICRA, pp. 193\u2013199.","DOI":"10.1109\/ICRA40945.2020.9196943"},{"key":"10.1016\/j.engappai.2024.108682_b22","doi-asserted-by":"crossref","unstructured":"Li, F., Tian, C., Zuo, W., Zhang, L., Yang, M.-H., 2018a. Learning spatial\u2013temporal regularized correlation filters for visual tracking. In: Proc. CVPR. pp. 4904\u20134913.","DOI":"10.1109\/CVPR.2018.00515"},{"key":"10.1016\/j.engappai.2024.108682_b23","doi-asserted-by":"crossref","unstructured":"Li, B., Wu, W., Wang, Q., Zhang, F., Xing, J., Yan, J., 2019. Siamrpn++: Evolution of siamese visual tracking with very deep networks. In: Proc. CVPR. pp. 4282\u20134291.","DOI":"10.1109\/CVPR.2019.00441"},{"key":"10.1016\/j.engappai.2024.108682_b24","doi-asserted-by":"crossref","unstructured":"Li, B., Yan, J., Wu, W., Zhu, Z., Hu, X., 2018b. High performance visual tracking with siamese region proposal network. In: Proc. CVPR. pp. 8971\u20138980.","DOI":"10.1109\/CVPR.2018.00935"},{"issue":"1","key":"10.1016\/j.engappai.2024.108682_b25","doi-asserted-by":"crossref","first-page":"179","DOI":"10.1109\/TCSVT.2018.2889457","article-title":"Robust visual tracking via hierarchical particle filter and ensemble deep features","volume":"30","author":"Li","year":"2018","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.engappai.2024.108682_b26","first-page":"16743","article-title":"Swintrack: A simple and strong baseline for transformer tracking","volume":"35","author":"Lin","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2024.108682_b27","series-title":"Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.engappai.2024.108682_b28","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B., 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In: Proc. ICCV. pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.engappai.2024.108682_b29","doi-asserted-by":"crossref","unstructured":"Ma, F., Shou, M.Z., Zhu, L., Fan, H., Xu, Y., Yang, Y., Yan, Z., 2022. Unified transformer tracker for object tracking. In: Proc. CVPR. pp. 8781\u20138790.","DOI":"10.1109\/CVPR52688.2022.00858"},{"key":"10.1016\/j.engappai.2024.108682_b30","doi-asserted-by":"crossref","unstructured":"Ma, Y., Yuan, C., Gao, P., Wang, F., 2019. Efficient multi-level correlating for visual tracking. In: Proc. ACCV. pp. 452\u2013465.","DOI":"10.1007\/978-3-030-20873-8_29"},{"key":"10.1016\/j.engappai.2024.108682_b31","doi-asserted-by":"crossref","unstructured":"Mayer, C., Danelljan, M., Bhat, G., Paul, M., Paudel, D.P., Yu, F., Van Gool, L., 2022. Transforming model prediction for tracking. In: Proc. CVPR. pp. 8731\u20138740.","DOI":"10.1109\/CVPR52688.2022.00853"},{"key":"10.1016\/j.engappai.2024.108682_b32","doi-asserted-by":"crossref","unstructured":"Mayer, C., Danelljan, M., Paudel, D.P., Van Gool, L., 2021. Learning target candidate association to keep track of what not to track. In: Proc. ICCV. pp. 13444\u201313454.","DOI":"10.1109\/ICCV48922.2021.01319"},{"key":"10.1016\/j.engappai.2024.108682_b33","doi-asserted-by":"crossref","unstructured":"Meinhardt, T., Kirillov, A., Leal-Taix\u00e9, L., Feichtenhofer, C., 2022. Trackformer: Multi-object tracking with transformers. In: Proc. CVPR. pp. 8834\u20138844.","DOI":"10.1109\/CVPR52688.2022.00864"},{"key":"10.1016\/j.engappai.2024.108682_b34","series-title":"Lost vibration test data recovery using convolutional neural network: a case study","author":"Moeinifard","year":"2022"},{"key":"10.1016\/j.engappai.2024.108682_b35","doi-asserted-by":"crossref","unstructured":"M\u00fcller, M., Bibi, A., Giancola, S., Al-Subaihi, S., Ghanem, B., 2018. Trackingnet: A large-scale dataset and benchmark for object tracking in the wild. In: Proc. ECCV. Vol. 11205, pp. 310\u2013327.","DOI":"10.1007\/978-3-030-01246-5_19"},{"key":"10.1016\/j.engappai.2024.108682_b36","unstructured":"Parmar, N., Vaswani, A., Uszkoreit, J., Kaiser, L., Shazeer, N., Ku, A., Tran, D., 2018. Image transformer. In: Proc. ICML. pp. 4055\u20134064."},{"key":"10.1016\/j.engappai.2024.108682_b37","article-title":"Deep attentive tracking via reciprocative learning","volume":"31","author":"Pu","year":"2018","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2024.108682_b38","unstructured":"Ramachandran, P., Parmar, N., Vaswani, A., Bello, I., Levskaya, A., Shlens, J., 2019. Stand-alone self-attention in vision models. In: Proc. NeurIPS."},{"key":"10.1016\/j.engappai.2024.108682_b39","doi-asserted-by":"crossref","unstructured":"Rezatofighi, H., Tsoi, N., Gwak, J., Sadeghian, A., Reid, I., Savarese, S., 2019. Generalized intersection over union: A metric and a loss for bounding box regression. In: Proc. CVPR. pp. 658\u2013666.","DOI":"10.1109\/CVPR.2019.00075"},{"issue":"1","key":"10.1016\/j.engappai.2024.108682_b40","first-page":"31","article-title":"Neural network controller application on a visual based object tracking and following robot","volume":"8","author":"Risma","year":"2019","journal-title":"Comput. Eng. Appl. J."},{"issue":"1","key":"10.1016\/j.engappai.2024.108682_b41","doi-asserted-by":"crossref","first-page":"295","DOI":"10.1109\/JETCAS.2023.3243604","article-title":"Stochastic computing design and implementation of a sound source localization system","volume":"13","author":"Schober","year":"2023","journal-title":"IEEE J. Emerg. Sel. Top. Circuits Syst."},{"key":"10.1016\/j.engappai.2024.108682_b42","series-title":"Proc. NeurIPS","first-page":"5998","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.engappai.2024.108682_b43","doi-asserted-by":"crossref","unstructured":"Voigtlaender, P., Luiten, J., Torr, P.H., Leibe, B., 2020. Siam r-cnn: Visual tracking by re-detection. In: Proc. CVPR. pp. 6578\u20136588.","DOI":"10.1109\/CVPR42600.2020.00661"},{"key":"10.1016\/j.engappai.2024.108682_b44","doi-asserted-by":"crossref","unstructured":"Wang, N., Zhou, W., Tian, Q., Hong, R., Wang, M., Li, H., 2018. Multi-cue correlation filters for robust visual tracking. In: Proc. CVPR. pp. 4844\u20134853.","DOI":"10.1109\/CVPR.2018.00509"},{"key":"10.1016\/j.engappai.2024.108682_b45","doi-asserted-by":"crossref","unstructured":"Wang, N., Zhou, W., Wang, J., Li, H., 2021a. Transformer meets tracker: Exploiting temporal context for robust visual tracking. In: Proc. CVPR. pp. 1571\u20131580.","DOI":"10.1109\/CVPR46437.2021.00162"},{"key":"10.1016\/j.engappai.2024.108682_b46","series-title":"IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021, Virtual, June 19-25, 2021","first-page":"1571","article-title":"Transformer meets tracker: Exploiting temporal context for robust visual tracking","author":"Wang","year":"2021"},{"key":"10.1016\/j.engappai.2024.108682_b47","doi-asserted-by":"crossref","unstructured":"Wei, X., Bai, Y., Zheng, Y., Shi, D., Gong, Y., 2023. Autoregressive visual tracking. In: Proc. CVPR. pp. 9697\u20139706.","DOI":"10.1109\/CVPR52729.2023.00935"},{"key":"10.1016\/j.engappai.2024.108682_b48","doi-asserted-by":"crossref","unstructured":"Wu, Q., Yan, Y., Liang, Y., Liu, Y., Wang, H., 2019. Dsnet: Deep and shallow feature learning for efficient visual tracking. In: Proc. ACCV. pp. 119\u2013134.","DOI":"10.1007\/978-3-030-20873-8_8"},{"key":"10.1016\/j.engappai.2024.108682_b49","doi-asserted-by":"crossref","unstructured":"Wu, Q., Yang, T., Liu, Z., Wu, B., Shan, Y., Chan, A.B., 2023. Dropmae: Masked autoencoders with spatial-attention dropout for tracking tasks. In: Proc. CVPR. pp. 14561\u201314571.","DOI":"10.1109\/CVPR52729.2023.01399"},{"key":"10.1016\/j.engappai.2024.108682_b50","doi-asserted-by":"crossref","unstructured":"Xie, F., Chu, L., Li, J., Lu, Y., Ma, C., 2023. Videotrack: Learning to track objects via video transformer. In: Proc. CVPR. pp. 22826\u201322835.","DOI":"10.1109\/CVPR52729.2023.02186"},{"key":"10.1016\/j.engappai.2024.108682_b51","doi-asserted-by":"crossref","unstructured":"Xie, F., Wang, C., Wang, G., Cao, Y., Yang, W., Zeng, W., 2022. Correlation-aware deep tracking. In: Proc. CVPR. pp. 8751\u20138760.","DOI":"10.1109\/CVPR52688.2022.00855"},{"key":"10.1016\/j.engappai.2024.108682_b52","doi-asserted-by":"crossref","unstructured":"Xie, F., Wang, C., Wang, G., Yang, W., Zeng, W., 2021. Learning tracking representations via dual-branch fully transformer networks. In: Proc. ICCV. pp. 2688\u20132697.","DOI":"10.1109\/ICCVW54120.2021.00303"},{"issue":"6","key":"10.1016\/j.engappai.2024.108682_b53","doi-asserted-by":"crossref","first-page":"7820","DOI":"10.1109\/TPAMI.2022.3225078","article-title":"Transcenter: Transformers with dense representations for multiple-object tracking","volume":"45","author":"Xu","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"4","key":"10.1016\/j.engappai.2024.108682_b54","doi-asserted-by":"crossref","first-page":"2280","DOI":"10.1109\/TASE.2022.3213730","article-title":"A learning-based object tracking strategy using visual sensors and intelligent robot arm","volume":"20","author":"Xu","year":"2023","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"key":"10.1016\/j.engappai.2024.108682_b55","doi-asserted-by":"crossref","unstructured":"Yan, B., Peng, H., Fu, J., Wang, D., Lu, H., 2021. Learning spatio-temporal transformer for visual tracking. In: Proc. ICCV. pp. 10448\u201310457.","DOI":"10.1109\/ICCV48922.2021.01028"},{"key":"10.1016\/j.engappai.2024.108682_b56","article-title":"Bandt: A border-aware network with deformable transformers for visual tracking","author":"Yang","year":"2023","journal-title":"IEEE Trans. Consum. Electron."},{"key":"10.1016\/j.engappai.2024.108682_b57","doi-asserted-by":"crossref","unstructured":"Ye, B., Chang, H., Ma, B., Shan, S., Chen, X., 2022. Joint feature learning and relation modeling for tracking: A one-stream framework. In: Proc. ECCV. pp. 341\u2013357.","DOI":"10.1007\/978-3-031-20047-2_20"},{"key":"10.1016\/j.engappai.2024.108682_b58","doi-asserted-by":"crossref","unstructured":"Yu, B., Tang, M., Zheng, L., Zhu, G., Wang, J., Feng, H., Feng, X., Lu, H., 2021a. High-performance discriminative tracking with transformers. In: Proc. ICCV. pp. 9856\u20139865.","DOI":"10.1109\/ICCV48922.2021.00971"},{"key":"10.1016\/j.engappai.2024.108682_b59","doi-asserted-by":"crossref","unstructured":"Yu, B., Tang, M., Zheng, L., Zhu, G., Wang, J., Feng, H., Feng, X., Lu, H., 2021b. High-performance discriminative tracking with transformers. In: Proc. ICCV. pp. 9836\u20139845.","DOI":"10.1109\/ICCV48922.2021.00971"},{"key":"10.1016\/j.engappai.2024.108682_b60","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Peng, H., Fu, J., Li, B., Hu, W., 2020. Ocean: Object-aware anchor-free tracking. In: Proc. ECCV. pp. 771\u2013787.","DOI":"10.1007\/978-3-030-58589-1_46"},{"key":"10.1016\/j.engappai.2024.108682_b61","series-title":"Trtr: Visual tracking with transformer","author":"Zhao","year":"2021"},{"key":"10.1016\/j.engappai.2024.108682_b62","doi-asserted-by":"crossref","unstructured":"Zhong, M., Chen, F., Xu, J., Lu, G., 2022. Correlation-based transformer tracking. In: Int. Conf. Artificial Neural Networks. pp. 85\u201396.","DOI":"10.1007\/978-3-031-15919-0_8"},{"key":"10.1016\/j.engappai.2024.108682_b63","doi-asserted-by":"crossref","unstructured":"Zhou, X., Yin, T., Koltun, V., Kr\u00e4henb\u00fchl, P., 2022. Global tracking transformers. In: Proc. CVPR. pp. 8761\u20138770.","DOI":"10.1109\/CVPR52688.2022.00857"}],"container-title":["Engineering Applications of Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S0952197624008406?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S0952197624008406?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2024,6,13]],"date-time":"2024-06-13T03:04:19Z","timestamp":1718247859000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/linkinghub.elsevier.com\/retrieve\/pii\/S0952197624008406"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8]]},"references-count":63,"alternative-id":["S0952197624008406"],"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.engappai.2024.108682","relation":{},"ISSN":["0952-1976"],"issn-type":[{"value":"0952-1976","type":"print"}],"subject":[],"published":{"date-parts":[[2024,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Unified spatio-temporal attention mixformer for visual object tracking","name":"articletitle","label":"Article Title"},{"value":"Engineering Applications of Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.engappai.2024.108682","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2024 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"108682"}}