{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T03:39:41Z","timestamp":1784691581963,"version":"3.55.0"},"reference-count":81,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Engineering Applications of Artificial Intelligence"],"published-print":{"date-parts":[[2023,11]]},"DOI":"10.1016\/j.engappai.2023.106669","type":"journal-article","created":{"date-parts":[[2023,7,29]],"date-time":"2023-07-29T13:33:27Z","timestamp":1690637607000},"page":"106669","update-policy":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":259,"special_numbering":"PA","title":["Semantic segmentation using Vision Transformers: A survey"],"prefix":"10.1016","volume":"126","author":[{"given":"Hans","family":"Thisanke","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chamli","family":"Deshan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kavindu","family":"Chamith","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0000-0001-9094-2736","authenticated-orcid":false,"given":"Sachith","family":"Seneviratne","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0000-0003-1645-0084","authenticated-orcid":false,"given":"Rajith","family":"Vidanaarachchi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Damayanthi","family":"Herath","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"4","key":"10.1016\/j.engappai.2023.106669_b1","doi-asserted-by":"crossref","first-page":"5526","DOI":"10.1109\/LRA.2020.3009075","article-title":"IDDA: A large-scale multi-domain dataset for autonomous driving","volume":"5","author":"Alberti","year":"2020","journal-title":"IEEE Robot. Autom. Lett."},{"issue":"4","key":"10.1016\/j.engappai.2023.106669_b2","doi-asserted-by":"crossref","first-page":"355","DOI":"10.1162\/pres.1997.6.4.355","article-title":"A survey of augmented reality","volume":"6","author":"Azuma","year":"1997","journal-title":"Presence: Teleoperators Virtual Environ."},{"key":"10.1016\/j.engappai.2023.106669_b3","series-title":"The liver tumor segmentation benchmark (lits)","author":"Bilic","year":"2019"},{"key":"10.1016\/j.engappai.2023.106669_b4","doi-asserted-by":"crossref","unstructured":"Boguszewski,\u00a0A., Batorski,\u00a0D., Ziemba-Jankowska,\u00a0N., Dziedzic,\u00a0T., Zambrzycka,\u00a0A., 2021. LandCover. ai: Dataset for automatic mapping of buildings, woodlands, water and roads from aerial imagery. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 1102\u20131110.","DOI":"10.1109\/CVPRW53098.2021.00121"},{"issue":"2","key":"10.1016\/j.engappai.2023.106669_b5","doi-asserted-by":"crossref","first-page":"88","DOI":"10.1016\/j.patrec.2008.04.005","article-title":"Semantic object classes in video: A high-definition ground truth database","volume":"30","author":"Brostow","year":"2009","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.engappai.2023.106669_b6","series-title":"Computer Vision\u2013ECCV 2022 Workshops: Tel Aviv, Israel","first-page":"205","article-title":"Swin-unet: Unet-like pure transformer for medical image segmentation","author":"Cao","year":"2023"},{"key":"10.1016\/j.engappai.2023.106669_b7","doi-asserted-by":"crossref","unstructured":"Caron,\u00a0M., Touvron,\u00a0H., Misra,\u00a0I., J\u00e9gou,\u00a0H., Mairal,\u00a0J., Bojanowski,\u00a0P., Joulin,\u00a0A., 2021. Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 9650\u20139660.","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"10.1016\/j.engappai.2023.106669_b8","doi-asserted-by":"crossref","unstructured":"Chen,\u00a0C.-F.R., Fan,\u00a0Q., Panda,\u00a0R., 2021a. Crossvit: Cross-attention multi-scale vision transformer for image classification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 357\u2013366.","DOI":"10.1109\/ICCV48922.2021.00041"},{"key":"10.1016\/j.engappai.2023.106669_b9","series-title":"Transunet: Transformers make strong encoders for medical image segmentation","author":"Chen","year":"2021"},{"key":"10.1016\/j.engappai.2023.106669_b10","doi-asserted-by":"crossref","unstructured":"Cheng,\u00a0B., Misra,\u00a0I., Schwing,\u00a0A.G., Kirillov,\u00a0A., Girdhar,\u00a0R., 2022. Masked-attention mask transformer for universal image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 1290\u20131299.","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"10.1016\/j.engappai.2023.106669_b11","first-page":"17864","article-title":"Per-pixel classification is not all you need for semantic segmentation","volume":"34","author":"Cheng","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2023.106669_b12","first-page":"9355","article-title":"Twins: Revisiting the design of spatial attention in vision transformers","volume":"34","author":"Chu","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2023.106669_b13","series-title":"Conditional positional encodings for vision transformers","author":"Chu","year":"2021"},{"key":"10.1016\/j.engappai.2023.106669_b14","series-title":"2018 IEEE 15th International Symposium on Biomedical Imaging","first-page":"168","article-title":"Skin lesion analysis toward melanoma detection: A challenge at the 2017 international symposium on biomedical imaging (isbi), hosted by the international skin imaging collaboration (isic)","author":"Codella","year":"2018"},{"key":"10.1016\/j.engappai.2023.106669_b15","doi-asserted-by":"crossref","unstructured":"Cordts,\u00a0M., Omran,\u00a0M., Ramos,\u00a0S., Rehfeld,\u00a0T., Enzweiler,\u00a0M., Benenson,\u00a0R., Franke,\u00a0U., Roth,\u00a0S., Schiele,\u00a0B., 2016. The cityscapes dataset for semantic urban scene understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 3213\u20133223.","DOI":"10.1109\/CVPR.2016.350"},{"key":"10.1016\/j.engappai.2023.106669_b16","doi-asserted-by":"crossref","unstructured":"Dai,\u00a0X., Chen,\u00a0Y., Yang,\u00a0J., Zhang,\u00a0P., Yuan,\u00a0L., Zhang,\u00a0L., 2021. Dynamic detr: End-to-end object detection with dynamic attention. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 2988\u20132997.","DOI":"10.1109\/ICCV48922.2021.00298"},{"key":"10.1016\/j.engappai.2023.106669_b17","doi-asserted-by":"crossref","unstructured":"Demir,\u00a0I., Koperski,\u00a0K., Lindenbaum,\u00a0D., Pang,\u00a0G., Huang,\u00a0J., Basu,\u00a0S., Hughes,\u00a0F., Tuia,\u00a0D., Raskar,\u00a0R., 2018. Deepglobe 2018: A challenge to parse the earth through satellite images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops. pp. 172\u2013181.","DOI":"10.1109\/CVPRW.2018.00031"},{"key":"10.1016\/j.engappai.2023.106669_b18","doi-asserted-by":"crossref","first-page":"94","DOI":"10.1016\/j.isprsjprs.2020.01.013","article-title":"ResUNet-a: A deep learning framework for semantic segmentation of remotely sensed data","volume":"162","author":"Diakogiannis","year":"2020","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"key":"10.1016\/j.engappai.2023.106669_b19","first-page":"1","article-title":"Looking outside the window: Wide-context transformer for the semantic segmentation of high-resolution remote sensing images","volume":"60","author":"Ding","year":"2022","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.engappai.2023.106669_b20","doi-asserted-by":"crossref","unstructured":"Ding,\u00a0M., Wang,\u00a0Z., Zhou,\u00a0B., Shi,\u00a0J., Lu,\u00a0Z., Luo,\u00a0P., 2020. Every frame counts: Joint learning of video segmentation and optical flow. In: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 34. pp. 10713\u201310720.","DOI":"10.1609\/aaai.v34i07.6699"},{"key":"10.1016\/j.engappai.2023.106669_b21","series-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"10.1016\/j.engappai.2023.106669_b22","first-page":"1","article-title":"The PASCAL visual object classes challenge 2012 (VOC2012) development kit","volume":"2007","author":"Everingham","year":"2012","journal-title":"Pattern Anal. Stat. Model. Comput. Learn., Tech. Rep"},{"key":"10.1016\/j.engappai.2023.106669_b23","doi-asserted-by":"crossref","unstructured":"Gaidon,\u00a0A., Wang,\u00a0Q., Cabon,\u00a0Y., Vig,\u00a0E., 2016. Virtual worlds as proxy for multi-object tracking analysis. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 4340\u20134349.","DOI":"10.1109\/CVPR.2016.470"},{"key":"10.1016\/j.engappai.2023.106669_b24","doi-asserted-by":"crossref","first-page":"41","DOI":"10.1016\/j.asoc.2018.05.018","article-title":"A survey on deep learning techniques for image and video semantic segmentation","volume":"70","author":"Garcia-Garcia","year":"2018","journal-title":"Appl. Soft Comput."},{"key":"10.1016\/j.engappai.2023.106669_b25","first-page":"1","article-title":"Image search engines: An overview","author":"Gevers","year":"2004","journal-title":"Emerg. Top. Comput. Vis."},{"key":"10.1016\/j.engappai.2023.106669_b26","series-title":"2014 12th IEEE International Conference on Industrial Informatics","first-page":"289","article-title":"Human-machine-interaction in the industry 4.0 era","author":"Gorecky","year":"2014"},{"issue":"10","key":"10.1016\/j.engappai.2023.106669_b27","doi-asserted-by":"crossref","first-page":"2281","DOI":"10.1109\/TMI.2019.2903562","article-title":"Ce-net: Context encoder network for 2d medical image segmentation","volume":"38","author":"Gu","year":"2019","journal-title":"IEEE Trans. Med. Imaging"},{"key":"10.1016\/j.engappai.2023.106669_b28","series-title":"Object detection and semantic segmentation using self-supervised learning","author":"Gustavsson","year":"2021"},{"key":"10.1016\/j.engappai.2023.106669_b29","article-title":"A survey on vision transformer","author":"Han","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2023.106669_b30","doi-asserted-by":"crossref","unstructured":"He,\u00a0K., Zhang,\u00a0X., Ren,\u00a0S., Sun,\u00a0J., 2016. Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"issue":"02","key":"10.1016\/j.engappai.2023.106669_b31","doi-asserted-by":"crossref","first-page":"107","DOI":"10.1142\/S0218488598000094","article-title":"The vanishing gradient problem during learning recurrent neural nets and problem solutions","volume":"6","author":"Hochreiter","year":"1998","journal-title":"Int. J. Uncertain. Fuzziness Knowl.-Based Syst."},{"key":"10.1016\/j.engappai.2023.106669_b32","doi-asserted-by":"crossref","unstructured":"Hu,\u00a0H., Zhang,\u00a0Z., Xie,\u00a0Z., Lin,\u00a0S., 2019. Local relation networks for image recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 3464\u20133473.","DOI":"10.1109\/ICCV.2019.00356"},{"key":"10.1016\/j.engappai.2023.106669_b33","series-title":"ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1055","article-title":"Unet 3+: A full-scale connected unet for medical image segmentation","author":"Huang","year":"2020"},{"key":"10.1016\/j.engappai.2023.106669_b34","doi-asserted-by":"crossref","first-page":"317","DOI":"10.1016\/j.procs.2016.09.407","article-title":"Review of MRI-based brain tumor image segmentation using deep learning methods","volume":"102","author":"I\u015f\u0131n","year":"2016","journal-title":"Procedia Comput. Sci."},{"key":"10.1016\/j.engappai.2023.106669_b35","series-title":"2020 IEEE Conference on Computational Intelligence in Bioinformatics and Computational Biology","first-page":"1","article-title":"A survey of loss functions for semantic segmentation","author":"Jadon","year":"2020"},{"key":"10.1016\/j.engappai.2023.106669_b36","doi-asserted-by":"crossref","unstructured":"Jain,\u00a0S., Wang,\u00a0X., Gonzalez,\u00a0J.E., 2019. Accel: A corrective fusion network for efficient semantic segmentation on video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 8866\u20138875.","DOI":"10.1109\/CVPR.2019.00907"},{"issue":"1\u20133","key":"10.1016\/j.engappai.2023.106669_b37","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1561\/0600000079","article-title":"Computer vision for autonomous vehicles: Problems, datasets and state of the art","volume":"12","author":"Janai","year":"2020","journal-title":"Found. Trends\u00ae Comput. Graph. Vis."},{"issue":"10s","key":"10.1016\/j.engappai.2023.106669_b38","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3505244","article-title":"Transformers in vision: A survey","volume":"54","author":"Khan","year":"2022","journal-title":"ACM Comput. Surv."},{"issue":"2","key":"10.1016\/j.engappai.2023.106669_b39","doi-asserted-by":"crossref","first-page":"712","DOI":"10.1109\/TITS.2019.2962338","article-title":"A survey of deep learning applications to autonomous vehicle control","volume":"22","author":"Kuutti","year":"2020","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.engappai.2023.106669_b40","doi-asserted-by":"crossref","unstructured":"Lin,\u00a0T.-Y., Doll\u00e1r,\u00a0P., Girshick,\u00a0R., He,\u00a0K., Hariharan,\u00a0B., Belongie,\u00a0S., 2017a. Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 2117\u20132125.","DOI":"10.1109\/CVPR.2017.106"},{"key":"10.1016\/j.engappai.2023.106669_b41","doi-asserted-by":"crossref","unstructured":"Lin,\u00a0T.-Y., Goyal,\u00a0P., Girshick,\u00a0R., He,\u00a0K., Doll\u00e1r,\u00a0P., 2017b. Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision. pp. 2980\u20132988.","DOI":"10.1109\/ICCV.2017.324"},{"key":"10.1016\/j.engappai.2023.106669_b42","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0Z., Lin,\u00a0Y., Cao,\u00a0Y., Hu,\u00a0H., Wei,\u00a0Y., Zhang,\u00a0Z., Lin,\u00a0S., Guo,\u00a0B., 2021a. Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.engappai.2023.106669_b43","doi-asserted-by":"crossref","DOI":"10.1109\/TKDE.2021.3090866","article-title":"Self-supervised learning: Generative or contrastive","author":"Liu","year":"2021","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"10.1016\/j.engappai.2023.106669_b44","article-title":"A survey of visual transformers","author":"Liu","year":"2023","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.engappai.2023.106669_b45","doi-asserted-by":"crossref","unstructured":"Long,\u00a0J., Shelhamer,\u00a0E., Darrell,\u00a0T., 2015. Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 3431\u20133440.","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"10.1016\/j.engappai.2023.106669_b46","doi-asserted-by":"crossref","unstructured":"Mahasseni,\u00a0B., Todorovic,\u00a0S., Fern,\u00a0A., 2017. Budget-aware deep semantic video segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 1029\u20131038.","DOI":"10.1109\/CVPR.2017.224"},{"issue":"10","key":"10.1016\/j.engappai.2023.106669_b47","doi-asserted-by":"crossref","first-page":"1993","DOI":"10.1109\/TMI.2014.2377694","article-title":"The multimodal brain tumor image segmentation benchmark (BRATS)","volume":"34","author":"Menze","year":"2014","journal-title":"IEEE Trans. Med. Imaging"},{"key":"10.1016\/j.engappai.2023.106669_b48","series-title":"2016 Fourth International Conference on 3D Vision (3DV)","first-page":"565","article-title":"V-net: Fully convolutional neural networks for volumetric medical image segmentation","author":"Milletari","year":"2016"},{"key":"10.1016\/j.engappai.2023.106669_b49","doi-asserted-by":"crossref","unstructured":"Mottaghi,\u00a0R., Chen,\u00a0X., Liu,\u00a0X., Cho,\u00a0N.-G., Lee,\u00a0S.-W., Fidler,\u00a0S., Urtasun,\u00a0R., Yuille,\u00a0A., 2014. The role of context for object detection and semantic segmentation in the wild. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 891\u2013898.","DOI":"10.1109\/CVPR.2014.119"},{"key":"10.1016\/j.engappai.2023.106669_b50","series-title":"Attention u-net: Learning where to look for the pancreas","author":"Oktay","year":"2018"},{"issue":"2","key":"10.1016\/j.engappai.2023.106669_b51","doi-asserted-by":"crossref","first-page":"127","DOI":"10.1016\/S1361-8415(00)00041-4","article-title":"Interaction in the segmentation of medical images: A survey","volume":"5","author":"Olabarriaga","year":"2001","journal-title":"Med. Image Anal."},{"issue":"2","key":"10.1016\/j.engappai.2023.106669_b52","doi-asserted-by":"crossref","DOI":"10.1088\/1748-9326\/3\/2\/025011","article-title":"Reference scenarios for deforestation and forest degradation in support of REDD: A review of data and methods","volume":"3","author":"Olander","year":"2008","journal-title":"Environ. Res. Lett."},{"key":"10.1016\/j.engappai.2023.106669_b53","article-title":"A review on deep learning in UAV remote sensing","volume":"102","author":"Osco","year":"2021","journal-title":"Int. J. Appl. Earth Obs. Geoinf."},{"issue":"9","key":"10.1016\/j.engappai.2023.106669_b54","doi-asserted-by":"crossref","first-page":"2940","DOI":"10.1109\/TGRS.2007.902824","article-title":"An innovative neural-net method to detect temporal changes in high-resolution optical satellite imagery","volume":"45","author":"Pacifici","year":"2007","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.engappai.2023.106669_b55","series-title":"How do vision transformers work?","author":"Park","year":"2022"},{"key":"10.1016\/j.engappai.2023.106669_b56","article-title":"Stand-alone self-attention in vision models","volume":"32","author":"Ramachandran","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2023.106669_b57","doi-asserted-by":"crossref","unstructured":"Ranftl,\u00a0R., Bochkovskiy,\u00a0A., Koltun,\u00a0V., 2021. Vision transformers for dense prediction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 12179\u201312188.","DOI":"10.1109\/ICCV48922.2021.01196"},{"key":"10.1016\/j.engappai.2023.106669_b58","series-title":"European Conference on Computer Vision","first-page":"102","article-title":"Playing for data: Ground truth from computer games","author":"Richter","year":"2016"},{"key":"10.1016\/j.engappai.2023.106669_b59","series-title":"International Conference on Medical Image Computing and Computer-Assisted Intervention","first-page":"234","article-title":"U-net: Convolutional networks for biomedical image segmentation","author":"Ronneberger","year":"2015"},{"key":"10.1016\/j.engappai.2023.106669_b60","series-title":"Weakly supervised semantic segmentation of satellite images for land cover mapping\u2013challenges and opportunities","author":"Schmitt","year":"2020"},{"key":"10.1016\/j.engappai.2023.106669_b61","series-title":"European Conference on Computer Vision","first-page":"852","article-title":"Clockwork convnets for video semantic segmentation","author":"Shelhamer","year":"2016"},{"key":"10.1016\/j.engappai.2023.106669_b62","doi-asserted-by":"crossref","unstructured":"Strudel,\u00a0R., Garcia,\u00a0R., Laptev,\u00a0I., Schmid,\u00a0C., 2021. Segmenter: Transformer for semantic segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 7262\u20137272.","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"10.1016\/j.engappai.2023.106669_b63","series-title":"International Conference on Machine Learning","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","author":"Touvron","year":"2021"},{"key":"10.1016\/j.engappai.2023.106669_b64","series-title":"2019 IEEE Winter Conference on Applications of Computer Vision","first-page":"1743","article-title":"IDD: A dataset for exploring problems of autonomous navigation in unconstrained environments","author":"Varma","year":"2019"},{"key":"10.1016\/j.engappai.2023.106669_b65","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"10","key":"10.1016\/j.engappai.2023.106669_b66","doi-asserted-by":"crossref","first-page":"3349","DOI":"10.1109\/TPAMI.2020.2983686","article-title":"Deep high-resolution representation learning for visual recognition","volume":"43","author":"Wang","year":"2020","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2023.106669_b67","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0W., Xie,\u00a0E., Li,\u00a0X., Fan,\u00a0D.-P., Song,\u00a0K., Liang,\u00a0D., Lu,\u00a0T., Luo,\u00a0P., Shao,\u00a0L., 2021a. Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 568\u2013578.","DOI":"10.1109\/ICCV48922.2021.00061"},{"issue":"3","key":"10.1016\/j.engappai.2023.106669_b68","doi-asserted-by":"crossref","first-page":"415","DOI":"10.1007\/s41095-022-0274-8","article-title":"Pvt v2: Improved baselines with pyramid vision transformer","volume":"8","author":"Wang","year":"2022","journal-title":"Comput. Vis. Media"},{"key":"10.1016\/j.engappai.2023.106669_b69","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0Y., Xu,\u00a0Z., Wang,\u00a0X., Shen,\u00a0C., Cheng,\u00a0B., Shen,\u00a0H., Xia,\u00a0H., 2021b. End-to-end video instance segmentation with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 8741\u20138750.","DOI":"10.1109\/CVPR46437.2021.00863"},{"key":"10.1016\/j.engappai.2023.106669_b70","series-title":"LoveDA: A remote sensing land-cover dataset for domain adaptive semantic segmentation","author":"Wang","year":"2021"},{"key":"10.1016\/j.engappai.2023.106669_b71","series-title":"Computer Vision\u2013ECCV 2022: 17th European Conference","first-page":"553","article-title":"Seqformer: Sequential transformer for video instance segmentation","author":"Wu","year":"2022"},{"key":"10.1016\/j.engappai.2023.106669_b72","first-page":"12077","article-title":"SegFormer: Simple and efficient design for semantic segmentation with transformers","volume":"34","author":"Xie","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"18","key":"10.1016\/j.engappai.2023.106669_b73","doi-asserted-by":"crossref","first-page":"3585","DOI":"10.3390\/rs13183585","article-title":"Efficient transformer for remote sensing image segmentation","volume":"13","author":"Xu","year":"2021","journal-title":"Remote Sens."},{"key":"10.1016\/j.engappai.2023.106669_b74","doi-asserted-by":"crossref","unstructured":"Yang,\u00a0S., Wang,\u00a0X., Li,\u00a0Y., Fang,\u00a0Y., Fang,\u00a0J., Liu,\u00a0W., Zhao,\u00a0X., Shan,\u00a0Y., 2022. Temporally efficient vision transformer for video instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2885\u20132895.","DOI":"10.1109\/CVPR52688.2022.00290"},{"key":"10.1016\/j.engappai.2023.106669_b75","first-page":"7281","article-title":"Hrformer: High-resolution vision transformer for dense predict","volume":"34","author":"Yuan","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2023.106669_b76","doi-asserted-by":"crossref","unstructured":"Zheng,\u00a0S., Lu,\u00a0J., Zhao,\u00a0H., Zhu,\u00a0X., Luo,\u00a0Z., Wang,\u00a0Y., Fu,\u00a0Y., Feng,\u00a0J., Xiang,\u00a0T., Torr,\u00a0P.H., et al., 2021. Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 6881\u20136890.","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"10.1016\/j.engappai.2023.106669_b77","series-title":"Deep Learning in Medical Image Analysis and Multimodal Learning for Clinical Decision Support","first-page":"3","article-title":"Unet++: A nested u-net architecture for medical image segmentation","author":"Zhou","year":"2018"},{"key":"10.1016\/j.engappai.2023.106669_b78","doi-asserted-by":"crossref","unstructured":"Zhou,\u00a0B., Zhao,\u00a0H., Puig,\u00a0X., Fidler,\u00a0S., Barriuso,\u00a0A., Torralba,\u00a0A., 2017. Scene parsing through ade20k dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 633\u2013641.","DOI":"10.1109\/CVPR.2017.544"},{"key":"10.1016\/j.engappai.2023.106669_b79","series-title":"Deformable detr: Deformable transformers for end-to-end object detection","author":"Zhu","year":"2020"},{"key":"10.1016\/j.engappai.2023.106669_b80","series-title":"Multi-Purposeful Application of Geospatial Data","first-page":"19","article-title":"A review: Remote sensing sensors","author":"Zhu","year":"2018"},{"key":"10.1016\/j.engappai.2023.106669_b81","doi-asserted-by":"crossref","unstructured":"Zhu,\u00a0X., Xiong,\u00a0Y., Dai,\u00a0J., Yuan,\u00a0L., Wei,\u00a0Y., 2017. Deep feature flow for video recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 2349\u20132358.","DOI":"10.1109\/CVPR.2017.441"}],"container-title":["Engineering Applications of Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S0952197623008539?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S0952197623008539?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2025,10,16]],"date-time":"2025-10-16T07:53:26Z","timestamp":1760601206000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/linkinghub.elsevier.com\/retrieve\/pii\/S0952197623008539"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11]]},"references-count":81,"alternative-id":["S0952197623008539"],"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.engappai.2023.106669","relation":{},"ISSN":["0952-1976"],"issn-type":[{"value":"0952-1976","type":"print"}],"subject":[],"published":{"date-parts":[[2023,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Semantic segmentation using Vision Transformers: A survey","name":"articletitle","label":"Article Title"},{"value":"Engineering Applications of Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.engappai.2023.106669","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2023 Elsevier Ltd. All rights reserved.","name":"copyright","label":"Copyright"}],"article-number":"106669"}}