{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,10]],"date-time":"2025-04-10T04:09:38Z","timestamp":1744258178581,"version":"3.40.4"},"reference-count":66,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,2,26]],"date-time":"2025-02-26T00:00:00Z","timestamp":1740528000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,2,26]],"date-time":"2025-02-26T00:00:00Z","timestamp":1740528000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100022984","name":"Amazon","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100022984","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,2,26]]},"DOI":"10.1109\/wacv61041.2025.00463","type":"proceedings-article","created":{"date-parts":[[2025,4,8]],"date-time":"2025-04-08T17:08:13Z","timestamp":1744132093000},"page":"4725-4735","source":"Crossref","is-referenced-by-count":0,"title":["GEXIA: Granularity Expansion and Iterative Approximation for Scalable Multi-Grained Video-Language Learning"],"prefix":"10.1109","author":[{"given":"Yicheng","family":"Wang","sequence":"first","affiliation":[{"name":"Texas A&#x0026;M University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhikang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Amazon"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jue","family":"Wang","sequence":"additional","affiliation":[{"name":"Amazon"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David","family":"Fan","sequence":"additional","affiliation":[{"name":"Amazon"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhenlin","family":"Xu","sequence":"additional","affiliation":[{"name":"Amazon"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Linda","family":"Liu","sequence":"additional","affiliation":[{"name":"Amazon"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiang","family":"Hao","sequence":"additional","affiliation":[{"name":"Amazon"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vimal","family":"Bhat","sequence":"additional","affiliation":[{"name":"Amazon"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinyu","family":"Li","sequence":"additional","affiliation":[{"name":"Amazon"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Flamingo: a visual language model for few-shot learning","volume":"abs\/2204.14198","author":"Alayrac","year":"2022","journal-title":"ArXiv"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02209"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2102.05095"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00293"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19809-0_3"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01449"},{"journal-title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality","year":"2023","author":"Chiang","key":"ref9"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093511"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01434"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72652-1_17"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"key":"ref14","article-title":"Violet: End-to-end video-language transformers with masked visual-token modeling","author":"Fu","year":"2021","journal-title":"arXiv preprint"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01282"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_6"},{"key":"ref19","first-page":"18198","article-title":"Video re-cap: Recursive captioning of hour-long videos","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Islam","year":"2024"},{"key":"ref20","first-page":"4651","article-title":"Perceiver: General perception with iterative attention","volume-title":"International conference on machine learning","author":"Jaegle","year":"2021"},{"key":"ref21","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","volume-title":"International Conference on Machine Learning","author":"Jia","year":"2021"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.83"},{"key":"ref23","article-title":"Video token merging for long-form video understanding","author":"Lee","year":"2024","journal-title":"arXiv preprint"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.161"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02214"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00687"},{"key":"ref28","first-page":"7575","article-title":"Egocentric video-language pretraining","volume":"35","author":"Lin","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01042"},{"key":"ref30","article-title":"Sgdr: Stochastic gradient descent with warm restarts","volume-title":"International Conference on Learning Representations","author":"Loshchilov","year":"2016"},{"key":"ref31","article-title":"Decoupled weight decay regularization","volume-title":"International Conference on Learning Representations","author":"Loshchilov","year":"2018"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.07.028"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547910"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00272"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2020.3016905"},{"key":"ref37","first-page":"12493","article-title":"Keeping your eye on the ball: Trajectory attention in video transformers","volume":"34","author":"Patrick","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref38","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford","year":"2021"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0987-1"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00772"},{"key":"ref41","article-title":"Charades-ego: A large-scale dataset of paired third and first person videos","author":"Sigurdsson","year":"2018","journal-title":"arXiv preprint"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00756"},{"key":"ref43","first-page":"38032","article-title":"Long-form video-language pretraining with multimodal temporal contrastive learning","volume":"35","author":"Sun","year":"2022","journal-title":"Advances in neural information processing systems"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00130"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00439"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00618"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00618"},{"key":"ref49","article-title":"Internvid: A large-scale video-text dataset for multimodal understanding and generation","author":"Wang","year":"2023","journal-title":"arXiv preprint"},{"key":"ref50","article-title":"Dhp benchmark: Are llms good nlg evaluators?","author":"Wang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00264"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00192"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00498"},{"key":"ref56","article-title":"Clip-vip: Adapting pretrained image-text model to video-language alignment","volume-title":"The Eleventh International Conference on Learning Representations","author":"Xue","year":"2023"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00171"},{"key":"ref58","first-page":"124","article-title":"Zero-shot video question answering via frozen bidirectional language models","volume":"35","author":"Yang","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01136"},{"key":"ref60","article-title":"Scaling white-box transformers for vision","author":"Yang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01413"},{"key":"ref62","article-title":"Self-chained image-language model for video localization and question answering","author":"Yu","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref63","first-page":"26462","article-title":"Learning from inside: Self-driven siamese sampling and reasoning for video question answering","volume":"34","author":"Yu","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref64","first-page":"9422","article-title":"White-box transformers via sparse rate reduction","volume":"36","author":"Yu","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Videoprism: A foundational visual encoder for video understanding","volume-title":"Forty-first International Conference on Machine Learning","author":"Zhao","key":"ref65"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531950"}],"event":{"name":"2025 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","start":{"date-parts":[[2025,2,26]]},"location":"Tucson, AZ, USA","end":{"date-parts":[[2025,3,6]]}},"container-title":["2025 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10943266\/10943193\/10944078.pdf?arnumber=10944078","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,9]],"date-time":"2025-04-09T05:52:17Z","timestamp":1744177937000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10944078\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2,26]]},"references-count":66,"URL":"https:\/\/doi.org\/10.1109\/wacv61041.2025.00463","relation":{},"subject":[],"published":{"date-parts":[[2025,2,26]]}}}