{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T06:13:46Z","timestamp":1777875226151,"version":"3.51.4"},"reference-count":55,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100005290","name":"South-Central Minzu University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100005290","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1016\/j.neucom.2026.133150","type":"journal-article","created":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T08:51:37Z","timestamp":1772095897000},"page":"133150","update-policy":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["TINCLIP: Improving compositional reasoning of CLIP via textual inversion with no"],"prefix":"10.1016","volume":"679","author":[{"given":"Jiahe","family":"Wan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Ge","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"ZhongHao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/2.zoppoz.workers.dev:443\/https\/orcid.org\/0009-0006-2874-2251","authenticated-orcid":false,"given":"Zheng","family":"Ye","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.133150_bib0005","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"15338","article-title":"Zero-shot composed image retrieval with textual inversion","author":"Baldrati","year":"2023"},{"key":"10.1016\/j.neucom.2026.133150_bib0010","series-title":"The Thirteenth International Conference on Learning Representations (ICLR)","article-title":"Natural language inference improves compositionality in vision-language models","author":"Cascante-Bonilla","year":"2024"},{"key":"10.1016\/j.neucom.2026.133150_bib0015","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"3558","article-title":"Conceptual 12m: pushing web-scale image-text pre-training to recognize long-tail visual concepts","author":"Changpinyo","year":"2021"},{"key":"10.1016\/j.neucom.2026.133150_bib0020","series-title":"Proceedings of the 37th International Conference on Machine Learning (ICML)","first-page":"1597","article-title":"A simple framework for contrastive learning of visual representations","volume":"vol. 119","author":"Chen","year":"2020"},{"key":"10.1016\/j.neucom.2026.133150_bib0025","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"2818","article-title":"Reproducible scaling laws for contrastive language-image learning","author":"Cherti","year":"2023"},{"key":"10.1016\/j.neucom.2026.133150_bib0030","series-title":"Proceedings of the 17th European Conference on Computer Vision (ECCV)","first-page":"558","article-title":"\u201cThis is my unicorn, fluffy\u201d: personalizing frozen vision-language representations","author":"Cohen","year":"2022"},{"key":"10.1016\/j.neucom.2026.133150_bib0035","series-title":"Logics and Languages","author":"Cresswell","year":"2016"},{"key":"10.1016\/j.neucom.2026.133150_bib0040","author":"Dubey"},{"key":"10.1016\/j.neucom.2026.133150_bib0045","doi-asserted-by":"crossref","first-page":"17972","DOI":"10.52202\/079017-0571","article-title":"Sugarcrepe++ dataset: vision-language model sensitivity to semantic and lexical alterations","author":"Dumpala","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133150_bib0050","first-page":"27092","article-title":"Datacomp: in search of the next generation of multimodal datasets","volume":"36","author":"Gadre","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133150_bib0055","series-title":"Proceedings of the 11th International Conference on Learning Representations (ICLR)","first-page":"1","article-title":"An image is worth one word: personalizing text-to-image generation using textual inversion","author":"Gal","year":"2023"},{"key":"10.1016\/j.neucom.2026.133150_bib0060","author":"Gao"},{"key":"10.1016\/j.neucom.2026.133150_bib0065","author":"Hendrycks"},{"key":"10.1016\/j.neucom.2026.133150_bib0070","doi-asserted-by":"crossref","first-page":"31096","DOI":"10.52202\/075280-1355","article-title":"Sugarcrepe: fixing hackable benchmarks for vision-language compositionality","volume":"36","author":"Hsieh","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133150_bib0075","series-title":"Proceedings of the 33rd ACM International Conference on Multimedia (ACM MM)","first-page":"3251","article-title":"Decoupled global-local alignment for improving compositional understanding","author":"Hu","year":"2025"},{"key":"10.1016\/j.neucom.2026.133150_bib0080","series-title":"Proceedings of the 38th AAAI Conference on Artificial Intelligence (AAAI)","first-page":"2417","article-title":"Structure-clip: towards scene graph knowledge to enhance multi-modal structured representations","author":"Huang","year":"2024"},{"issue":"1","key":"10.1016\/j.neucom.2026.133150_bib0085","doi-asserted-by":"crossref","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","article-title":"Visual genome: connecting language and vision using crowdsourced dense image annotations","volume":"123","author":"Krishna","year":"2017","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.133150_bib0090","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"1931","article-title":"Multi-concept customization of text-to-image diffusion","author":"Kumari","year":"2023"},{"issue":"7","key":"10.1016\/j.neucom.2026.133150_bib0095","doi-asserted-by":"crossref","first-page":"1956","DOI":"10.1007\/s11263-020-01316-z","article-title":"The open images dataset v4: unified image classification, object detection, and visual relationship detection at scale","volume":"128","author":"Kuznetsova","year":"2020","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.133150_bib0100","series-title":"Proceedings of the 40th International Conference on Machine Learning (ICML)","first-page":"19730","article-title":"Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models","volume":"vol. 202","author":"Li","year":"2023"},{"key":"10.1016\/j.neucom.2026.133150_bib0105","series-title":"Proceedings of the 39th International Conference on Machine Learning (ICML)","first-page":"12888","article-title":"BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation","volume":"vol. 162","author":"Li","year":"2022"},{"key":"10.1016\/j.neucom.2026.133150_bib0110","first-page":"9694","article-title":"Align before fuse: vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133150_bib0115","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing (EMNLP)","first-page":"14616","article-title":"Interpretable composition attribution enhancement for visio-linguistic compositional understanding","author":"Li","year":"2024"},{"key":"10.1016\/j.neucom.2026.133150_bib0120","series-title":"Proceedings of the 13th European Conference on Computer Vision (ECCV)","first-page":"740","article-title":"Microsoft COCO: common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.neucom.2026.133150_bib0125","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"26296","article-title":"Improved baselines with visual instruction tuning","author":"Liu","year":"2024"},{"key":"10.1016\/j.neucom.2026.133150_bib0130","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"10910","article-title":"Crepe: can vision-language foundation models reason compositionally?","author":"Ma","year":"2023"},{"key":"10.1016\/j.neucom.2026.133150_bib0135","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing (EMNLP)","first-page":"19060","article-title":"Preserving multi-modal capabilities of pre-trained vlms for improving vision-linguistic compositionality","author":"Oh","year":"2024"},{"key":"10.1016\/j.neucom.2026.133150_bib0140","first-page":"1","article-title":"Dinov2: learning robust visual features without supervision","volume":"2024","author":"Oquab","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.133150_bib0145","author":"Park"},{"key":"10.1016\/j.neucom.2026.133150_bib0150","first-page":"32731","article-title":"Tripletclip: improving compositional reasoning of CLIP via synthetic vision-language negatives","volume":"37","author":"Patel","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133150_bib0155","series-title":"Proceedings of the IEEE International Conference on Computer Vision (ICCV)","first-page":"2641","article-title":"Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models","author":"Plummer","year":"2015"},{"key":"10.1016\/j.neucom.2026.133150_bib0160","series-title":"Proceedings of the 38th International Conference on Machine Learning (ICML)","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.neucom.2026.133150_bib0165","author":"Ramesh"},{"issue":"3","key":"10.1016\/j.neucom.2026.133150_bib0170","doi-asserted-by":"crossref","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","article-title":"Imagenet large scale visual recognition challenge","volume":"115","author":"Russakovsky","year":"2015","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.133150_bib0175","first-page":"36479","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume":"35","author":"Saharia","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133150_bib0180","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"19305","article-title":"Pic2word: mapping pictures to words for zero-shot composed image retrieval","author":"Saito","year":"2023"},{"key":"10.1016\/j.neucom.2026.133150_bib0185","first-page":"25278","article-title":"Laion-5b: an open large-scale dataset for training next generation image-text models","volume":"35","author":"Schuhmann","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.133150_bib0190","series-title":"Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics","first-page":"2556","article-title":"Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning","author":"Sharma","year":"2018"},{"key":"10.1016\/j.neucom.2026.133150_bib0195","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","first-page":"7991","article-title":"Learning the power of \u201cno\u201d: foundation models with negations","author":"Singh","year":"2025"},{"key":"10.1016\/j.neucom.2026.133150_bib0200","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"5238","article-title":"Winoground: probing vision and language models for visio-linguistic compositionality","author":"Thrush","year":"2022"},{"key":"10.1016\/j.neucom.2026.133150_bib0205","series-title":"Proceedings of the 19th IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"1802","article-title":"Clipn for zero-shot OOD detection: teaching CLIP to say no","author":"Wang","year":"2023"},{"key":"10.1016\/j.neucom.2026.133150_bib0210","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"19175","article-title":"Image as a foreign language: beit pretraining for vision and vision-language tasks","author":"Wang","year":"2023"},{"key":"10.1016\/j.neucom.2026.133150_bib0215","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence (AAAI)","first-page":"8060","article-title":"Enhancing fine-grained vision-language pretraining with negative augmented samples","author":"Wang","year":"2025"},{"key":"10.1016\/j.neucom.2026.133150_bib0220","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"11307","article-title":"Fashion iq: a new dataset towards retrieving images by natural language feedback","author":"Wu","year":"2021"},{"key":"10.1016\/j.neucom.2026.133150_bib0225","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"15952","article-title":"CLIP-kd: an empirical study of CLIP model distillation","author":"Yang","year":"2024"},{"key":"10.1016\/j.neucom.2026.133150_bib0230","author":"Yang"},{"key":"10.1016\/j.neucom.2026.133150_bib0235","series-title":"Proceedings of the 32nd ACM International Conference on Multimedia (ACM MM)","first-page":"1245","article-title":"Semantic editing increment benefits zero-shot composed image retrieval","author":"Yang","year":"2024"},{"key":"10.1016\/j.neucom.2026.133150_bib0240","series-title":"Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR)","first-page":"80","article-title":"Ldre: llm-based divergent reasoning and ensemble for zero-shot composed image retrieval","author":"Yang","year":"2024"},{"key":"10.1016\/j.neucom.2026.133150_bib0245","first-page":"1","article-title":"Coca: contrastive captioners are image-text foundation models","author":"Yu","year":"2022","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.133150_bib0250","series-title":"Proceedings of the 11th International Conference on Learning Representations (ICLR)","first-page":"1","article-title":"When and why vision-language models behave like bags-of-words, and what to do about IT?","author":"Yuksekgonul","year":"2023"},{"key":"10.1016\/j.neucom.2026.133150_bib0255","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"18123","article-title":"Lit: zero-shot transfer with locked-image text tuning","author":"Zhai","year":"2022"},{"key":"10.1016\/j.neucom.2026.133150_bib0260","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","first-page":"5905","article-title":"Enhancing vision-language few-shot adaptation with negative learning","author":"Zhang","year":"2025"},{"key":"10.1016\/j.neucom.2026.133150_bib0265","author":"Zhang"},{"key":"10.1016\/j.neucom.2026.133150_bib0270","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"13774","article-title":"Contrasting intra-modal and ranking cross-modal hard negatives to enhance visio-linguistic compositional understanding","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neucom.2026.133150_bib0275","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"13774","article-title":"Contrasting intra-modal and ranking cross-modal hard negatives to enhance visio-linguistic compositional understanding","author":"Zhang","year":"2024"}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S0925231226005473?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/api.elsevier.com\/content\/article\/PII:S0925231226005473?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T21:21:09Z","timestamp":1777584069000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226005473"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":55,"alternative-id":["S0925231226005473"],"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.neucom.2026.133150","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,5]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"TINCLIP: Improving compositional reasoning of CLIP via textual inversion with no","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1016\/j.neucom.2026.133150","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133150"}}