{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T19:40:34Z","timestamp":1785699634653,"version":"3.56.0"},"reference-count":81,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,6,16]],"date-time":"2024-06-16T00:00:00Z","timestamp":1718496000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,6,16]],"date-time":"2024-06-16T00:00:00Z","timestamp":1718496000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key R&D Program of China","doi-asserted-by":"publisher","award":["2022ZD0162000"],"award-info":[{"award-number":["2022ZD0162000"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62106219,62206046"],"award-info":[{"award-number":["62106219,62206046"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,6,16]]},"DOI":"10.1109\/cvpr52733.2024.01725","type":"proceedings-article","created":{"date-parts":[[2024,9,16]],"date-time":"2024-09-16T17:34:53Z","timestamp":1726508093000},"page":"18221-18232","source":"Crossref","is-referenced-by-count":178,"title":["MovieChat: From Dense Token to Sparse Memory for Long Video Understanding"],"prefix":"10.1109","author":[{"given":"Enxin","family":"Song","sequence":"first","affiliation":[{"name":"Zhejiang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenhao","family":"Chai","sequence":"additional","affiliation":[{"name":"University of Washington"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guanhong","family":"Wang","sequence":"additional","affiliation":[{"name":"Zhejiang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yucheng","family":"Zhang","sequence":"additional","affiliation":[{"name":"Zhejiang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haoyang","family":"Zhou","sequence":"additional","affiliation":[{"name":"Zhejiang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Feiyang","family":"Wu","sequence":"additional","affiliation":[{"name":"Zhejiang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haozhe","family":"Chi","sequence":"additional","affiliation":[{"name":"Zhejiang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xun","family":"Guo","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tian","family":"Ye","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology (GZ)"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanting","family":"Zhang","sequence":"additional","affiliation":[{"name":"Donghua University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yan","family":"Lu","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jenq-Neng","family":"Hwang","sequence":"additional","affiliation":[{"name":"University of Washington"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gaoang","family":"Wang","sequence":"additional","affiliation":[{"name":"Zhejiang University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"23716","article-title":"Flamingo: a visual language model for few-shot learning","volume":"35","author":"Alayrac","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1080\/02724980543000097"},{"key":"ref3","year":"2023","journal-title":"Meet claude"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/S0079-7421(08)60422-3"},{"key":"ref6","author":"Awad","year":"2021","journal-title":"Trecvid 2020: A comprehensive campaign for evaluating video retrieval tasks across multiple application domains."},{"key":"ref7","author":"Awad","year":"2020","journal-title":"Trecvid 2019: An evaluation campaign to benchmark video activity detection, video captioning and matching, and video search & retrieval."},{"key":"ref8","article-title":"Trecvid 2018: Benchmarking video activity detection, video captioning and matching, video storytelling linking and video search","volume-title":"Proceedings of TRECVID 2018","author":"Awad","year":"2018"},{"key":"ref9","article-title":"Trecvid 2017: evaluating ad-hoc and instance video search, events detection, video captioning, and hyperlinking","author":"Awad","year":"2017","journal-title":"TREC Video Retrieval Evaluation (TRECVID)"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"ref11","article-title":"Token merging: Your vit but faster","volume-title":"The Eleventh International Conference on Learning Representations","author":"Bolya","year":"2022"},{"key":"ref12","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00792"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.3390\/app12136588"},{"key":"ref16","first-page":"190","article-title":"Collecting highly parallel data for paraphrase evaluation","volume-title":"Proceedings of the 49th annual meeting of the association for computational linguistics: human language technologies","author":"Chen","year":"2011"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19815-1_37"},{"key":"ref18","volume-title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality.","author":"Chiang","year":"2023"},{"key":"ref19","author":"Dai","year":"2023","journal-title":"Instructblip: Towards general-purpose vision-language models with instruction tuning."},{"key":"ref20","author":"Driess","year":"2023","journal-title":"Palme: An embodied multimodal language model."},{"key":"ref21","author":"Fang","year":"2022","journal-title":"Eva: Exploring the limits of masked visual representation learning at scale."},{"key":"ref22","author":"Fu","year":"2023","journal-title":"Mme: A comprehensive evaluation benchmark for multimodal large language models."},{"key":"ref23","author":"Gao","year":"2023","journal-title":"Llama-adapter v2: Parameter-efficient visual instruction model."},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"ref25","author":"Gong","year":"2023","journal-title":"Multimodal-gpt: A vision and language model for dialogue with humans."},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.3389\/fmars.2022.1071618"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00413"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58548-8_41"},{"key":"ref29","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","volume-title":"Proceedings of NAACL-HLT","author":"Kenton","year":"2019"},{"key":"ref30","author":"Li","year":"2023","journal-title":"Otter: A multi-modal model with in-context instruction tuning."},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/springerreference_9081"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/springerreference_9081"},{"key":"ref33","first-page":"1288812900","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","volume-title":"International Conference on Machine Learning","author":"Li"},{"key":"ref34","author":"Li","year":"2023","journal-title":"Videochat: Chat-centric video understanding."},{"key":"ref35","author":"Liu","year":"2017","journal-title":"Mavot: Memory-augmented video object tracking."},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.33540\/2168"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"ref38","author":"Lyu","year":"2023","journal-title":"Macaw-llm: Multi-modal language modeling with image, audio, video, and text integration."},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-018-1076-4"},{"key":"ref40","author":"Maaz","year":"2023","journal-title":"Video-chatgpt: Towards detailed video understanding via large vision and language models."},{"key":"ref41","author":"Mangalam","year":"2023","journal-title":"Egoschema: A diagnostic benchmark for very long-form video language understanding."},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00272"},{"key":"ref43","year":"2021","journal-title":"Gpt3.5, 2021."},{"key":"ref44","year":"2023","journal-title":"Gpt-4 technical report."},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00207"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-11752-2_15"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.447"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2012.6247801"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33718-5_11"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0851-8"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58517-4_10"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58542-6_38"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01265"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00797"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00497"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1101\/cshperspect.a021766"},{"key":"ref57","volume-title":"Bert position encoding.","author":"Su","year":"2023"},{"key":"ref58","author":"Su","year":"2023","journal-title":"Pandagpt: One model to instruction-follow them all."},{"key":"ref59","author":"Taori","year":"2023","journal-title":"Stanford alpaca: An instruction-following llama model"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.501"},{"key":"ref61","author":"Touvron","year":"2023","journal-title":"Llama: Open and efficient foundation language models."},{"key":"ref62","author":"Touvron","year":"2023","journal-title":"Llama 2: Open foundation and fine-tuned chat models."},{"key":"ref63","author":"Wang","year":"2023","journal-title":"Chatvideo: A tracklet-centric multimodal and versatile video understanding system."},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01271"},{"key":"ref65","author":"Wang","year":"2023","journal-title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks."},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00037"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00192"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01322"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVIDLICCEA56201.2022.9825193"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"ref72","first-page":"124","article-title":"Zero-shot video question answering via frozen bidirectional language models","volume":"35","author":"Yang","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01240-3_10"},{"key":"ref74","author":"Ye","year":"2023","journal-title":"mplug-owl: Modularization empowers large language models with multimodality."},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019127"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02208"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_38"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"ref79","author":"Zhang","year":"2023","journal-title":"Llama-adapter: Efficient fine-tuning of language models with zero-init attention."},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3272319"},{"key":"ref81","author":"Zhu","year":"2023","journal-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models."}],"event":{"name":"2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","location":"Seattle, WA, USA","start":{"date-parts":[[2024,6,16]]},"end":{"date-parts":[[2024,6,22]]}},"container-title":["2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"],"original-title":[],"link":[{"URL":"https:\/\/2.zoppoz.workers.dev:443\/http\/xplorestaging.ieee.org\/ielx8\/10654794\/10654797\/10657734.pdf?arnumber=10657734","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,19]],"date-time":"2024-09-19T06:38:15Z","timestamp":1726727895000},"score":1,"resource":{"primary":{"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/ieeexplore.ieee.org\/document\/10657734\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,16]]},"references-count":81,"URL":"https:\/\/2.zoppoz.workers.dev:443\/https\/doi.org\/10.1109\/cvpr52733.2024.01725","relation":{},"subject":[],"published":{"date-parts":[[2024,6,16]]}}}