{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T08:04:44Z","timestamp":1784534684571,"version":"3.55.0"},"reference-count":45,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62176144"],"award-info":[{"award-number":["62176144"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Journal of Visual Communication and Image Representation"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.jvcir.2026.104852","type":"journal-article","created":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T15:42:59Z","timestamp":1780501379000},"page":"104852","update-policy":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Region-guided representation fusion and background consistency in mask-free image editing"],"prefix":"10.1016","volume":"119","author":[{"given":"YiXin","family":"Fang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0000-0001-6259-7533","authenticated-orcid":false,"given":"Huaxiang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hua","family":"Ji","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaolong","family":"Dai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chunqiang","family":"Yao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.jvcir.2026.104852_b1","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1316","article-title":"AttnGAN: Fine-grained text to image generation with attentional generative adversarial networks","author":"Xu","year":"2018"},{"key":"10.1016\/j.jvcir.2026.104852_b2","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2023.104031","article-title":"Facial attribute editing method combined with parallel GAN for attribute separation","volume":"98","author":"Wei","year":"2024","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104852_b3","series-title":"2025 IEEE International Conference on Consumer Electronics","first-page":"1","article-title":"A study of style transfer based on text-to-image diffusion models","author":"Kim","year":"2025"},{"key":"10.1016\/j.jvcir.2026.104852_b4","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6038","article-title":"Null-text inversion for editing real images using guided diffusion models","author":"Mokady","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104852_b5","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1921","article-title":"Plug-and-play diffusion features for text-driven image-to-image translation","author":"Tumanyan","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104852_b6","doi-asserted-by":"crossref","first-page":"112948","DOI":"10.1109\/ACCESS.2024.3442296","article-title":"Diff-KT: Text-driven image editing by knowledge enhancement and mask transformer","volume":"12","author":"Zhao","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.jvcir.2026.104852_b7","series-title":"ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"CiGA: A cross-layer fine-grained attention correction method for large language model","author":"Li","year":"2025"},{"key":"10.1016\/j.jvcir.2026.104852_b8","series-title":"2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"26764","article-title":"Rethinking multi-view representation learning via distilled disentangling","author":"Ke","year":"2024"},{"issue":"12","key":"10.1016\/j.jvcir.2026.104852_b9","doi-asserted-by":"crossref","first-page":"9904","DOI":"10.1109\/TPAMI.2021.3132068","article-title":"CTNet: Context-based tandem network for semantic segmentation","volume":"44","author":"Li","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.jvcir.2026.104852_b10","series-title":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1740","article-title":"Optimizing latent space directions for gan-based local image editing","author":"Pajouheshgar","year":"2022"},{"issue":"1","key":"10.1016\/j.jvcir.2026.104852_b11","doi-asserted-by":"crossref","first-page":"179","DOI":"10.26599\/CVM.2025.9450310","article-title":"LucIE: Language-guided local image editing for fashion images","volume":"11","author":"Wen","year":"2025","journal-title":"Comput. Vis. Media"},{"key":"10.1016\/j.jvcir.2026.104852_b12","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"7059","article-title":"Text-driven image editing via learnable regions","author":"Lin","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104852_b13","series-title":"2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"8362","article-title":"SmartEdit: Exploring complex instruction-based image editing with multimodal large language models","author":"Huang","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104852_b14","series-title":"2025 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"7877","article-title":"Stable flow: Vital layers for training-free image editing","author":"Avrahami","year":"2025"},{"key":"10.1016\/j.jvcir.2026.104852_b15","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10686","article-title":"Vector quantized diffusion model for text-to-image synthesis","author":"Gu","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104852_b16","series-title":"2025 IEEE International Conference on Consumer Electronics","first-page":"1","article-title":"Privacy-diffusion: Privacy-preserving stable diffusion without homomorphic encryption","author":"Hsu","year":"2025"},{"key":"10.1016\/j.jvcir.2026.104852_b17","series-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18392","article-title":"InstructPix2Pix: Learning to follow image editing instructions","author":"Brooks","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104852_b18","series-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6007","article-title":"Imagic: Text-based real image editing with diffusion models","author":"Kawar","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104852_b19","doi-asserted-by":"crossref","unstructured":"B. Gao, X. Gao, X. Wu, Y. Zhou, Y. Qiao, L. Niu, X. Chen, Y. Wang, The Devil is in the Prompts: Retrieval-Augmented Prompt Optimization for Text-to-Video Generation, in: Proceedings of the Computer Vision and Pattern Recognition Conference, CVPR, 2025, pp. 3173\u20133183.","DOI":"10.1109\/CVPR52734.2025.00302"},{"key":"10.1016\/j.jvcir.2026.104852_b20","doi-asserted-by":"crossref","unstructured":"N. Wasserman, N. Rotstein, R. Ganz, R. Kimmel, Paint by Inpaint: Learning to Add Image Objects by Removing Them First, in: Proceedings of the Computer Vision and Pattern Recognition Conference, CVPR, 2025, pp. 18313\u201318324.","DOI":"10.1109\/CVPR52734.2025.01707"},{"key":"10.1016\/j.jvcir.2026.104852_b21","series-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18187","article-title":"Blended diffusion for text-driven editing of natural images","author":"Avrahami","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104852_b22","doi-asserted-by":"crossref","unstructured":"O. Madar, O. Fried, Tiled Diffusion, in: Proceedings of the Computer Vision and Pattern Recognition Conference, CVPR, 2025, pp. 7795\u20137804.","DOI":"10.1109\/CVPR52734.2025.00730"},{"key":"10.1016\/j.jvcir.2026.104852_b23","series-title":"Proceedings of the 32nd ACM International Conference on Multimedia","first-page":"10910","article-title":"GG-editor: Locally editing 3D avatars with multimodal large language model guidance","author":"Xu","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104852_b24","series-title":"BideDPO: Conditional image generation with simultaneous text and condition alignment","author":"Zhou","year":"2025"},{"issue":"6","key":"10.1016\/j.jvcir.2026.104852_b25","doi-asserted-by":"crossref","first-page":"4409","DOI":"10.1109\/TPAMI.2025.3541625","article-title":"Diffusion model-based image editing: A survey","volume":"47","author":"Huang","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.jvcir.2026.104852_b26","series-title":"2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"12868","article-title":"Taming transformers for high-resolution image synthesis","author":"Esser","year":"2021"},{"issue":"11","key":"10.1016\/j.jvcir.2026.104852_b27","doi-asserted-by":"crossref","first-page":"12960","DOI":"10.1109\/TPAMI.2022.3207091","article-title":"Transformer for image harmonization and beyond","volume":"45","author":"Guo","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"9","key":"10.1016\/j.jvcir.2026.104852_b28","doi-asserted-by":"crossref","first-page":"2070","DOI":"10.1109\/TPAMI.2018.2852750","article-title":"Deep collaborative embedding for social image understanding","volume":"41","author":"Li","year":"2019","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.jvcir.2026.104852_b29","series-title":"2021 IEEE\/CVF International Conference on Computer Vision","first-page":"9630","article-title":"Emerging properties in self-supervised vision transformers","author":"Caron","year":"2021"},{"key":"10.1016\/j.jvcir.2026.104852_b30","series-title":"2021 IEEE\/CVF International Conference on Computer Vision","first-page":"8219","article-title":"Self-supervised video representation learning with meta-contrastive network","author":"Lin","year":"2021"},{"issue":"3","key":"10.1016\/j.jvcir.2026.104852_b31","doi-asserted-by":"crossref","first-page":"10","DOI":"10.1109\/MCI.2025.3563859","article-title":"Don\u2019t forget your inverse DDIM for image editing","volume":"20","author":"Gomez-Trenado","year":"2025","journal-title":"IEEE Comput. Intell. Mag."},{"key":"10.1016\/j.jvcir.2026.104852_b32","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"2920","article-title":"Entwined inversion: Tune-free inversion for real image faithful reconstruction and editing","author":"Huang","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104852_b33","series-title":"MultiMedia Modeling","first-page":"648","article-title":"GAS: Geometry-appearance synergy for consistent video customization","author":"Jia","year":"2026"},{"key":"10.1016\/j.jvcir.2026.104852_b34","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"11305","article-title":"MaskGIT: Masked generative image transformer","author":"Chang","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104852_b35","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2025.104519","article-title":"Self-supervised panoramic stitched image quality assessment based on contrastive learning","volume":"111","author":"Li","year":"2025","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104852_b36","series-title":"2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10043","article-title":"Style transfer by relaxed optimal transport and self-similarity","author":"Kolkin","year":"2019"},{"key":"10.1016\/j.jvcir.2026.104852_b37","series-title":"2007 IEEE Computer Society Conference on Computer Vision and Pattern Recognition","article-title":"Matching local self-similarities across images and videos","author":"Shechtman","year":"2007"},{"key":"10.1016\/j.jvcir.2026.104852_b38","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2024.104347","article-title":"Global\u2013local prompts guided image-text embedding, alignment and aggregation for multi-label zero-shot learning","volume":"106","author":"Song","year":"2025","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104852_b39","series-title":"2024 IEEE International Workshop on Metrology for Living Environment","first-page":"415","article-title":"Denoising probabilistic diffusion models for synthetic healthcare image generation","author":"Iuliano","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104852_b40","series-title":"2021 IEEE\/CVF International Conference on Computer Vision","first-page":"2065","article-title":"StyleCLIP: Text-driven manipulation of stylegan imagery","author":"Patashnik","year":"2021"},{"key":"10.1016\/j.jvcir.2026.104852_b41","series-title":"An image is worth 16 \u00d7 16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"10.1016\/j.jvcir.2026.104852_b42","series-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2019"},{"key":"10.1016\/j.jvcir.2026.104852_b43","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2024.106777","article-title":"PFB-diff: Progressive feature blending diffusion for text-driven image editing","volume":"181","author":"Huang","year":"2025","journal-title":"Neural Netw."},{"key":"10.1016\/j.jvcir.2026.104852_b44","series-title":"KV-edit: Training-free image editing for precise background preservation","author":"Zhu","year":"2025"},{"key":"10.1016\/j.jvcir.2026.104852_b45","series-title":"Taming rectified flow for inversion and editing","author":"Wang","year":"2024"}],"container-title":["Journal of Visual Communication and Image Representation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S1047320326001471?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S1047320326001471?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T07:26:26Z","timestamp":1784532386000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/linkinghub.elsevier.com\/retrieve\/pii\/S1047320326001471"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":45,"alternative-id":["S1047320326001471"],"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.jvcir.2026.104852","relation":{},"ISSN":["1047-3203"],"issn-type":[{"value":"1047-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Region-guided representation fusion and background consistency in mask-free image editing","name":"articletitle","label":"Article Title"},{"value":"Journal of Visual Communication and Image Representation","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.jvcir.2026.104852","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104852"}}