@inproceedings{bb247200,
        AUTHOR = "Kaduri, O. and Bagon, S. and Dekel, T.",
        TITLE = "What's in the Image? A Deep-Dive into the Vision of Vision Language
Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "14549-14558",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242093"}

@inproceedings{bb247201,
        AUTHOR = "Xing, L. and Huang, Q.D. and Dong, X.Y. and Lu, J.J. and Zhang, P. and Zang, Y.H. and Cao, Y.H. and He, C.H. and Wang, J.Q. and Wu, F. and Lin, D.",
        TITLE = "Conical Visual Concentration for Efficient Large Vision-Language
Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "14593-14603",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242094"}

@inproceedings{bb247202,
        AUTHOR = "Zhang, L. and Yang, Q. and Agrawal, A.",
        TITLE = "Assessing and Learning Alignment of Unimodal Vision and Language
Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "14604-14614",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242095"}

@inproceedings{bb247203,
        AUTHOR = "Sehgal, A. and Yuan, P. and Hu, Z. and Yue, Y.S. and Sun, J.J. and Chaudhuri, S.",
        TITLE = "Self-Evolving Visual Concept Library using Vision-Language Critics",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "13124-13134",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242096"}

@inproceedings{bb247204,
        AUTHOR = "Wang, W.H. and Wang, L. and Gu, X.T. and Huang, S.Y. and Dong, Y.X. and Tang, J.",
        TITLE = "MotionBench: Benchmarking and Improving Fine-Grained Video Motion
Understanding for Vision Language Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "8450-8460",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242097"}

@inproceedings{bb247205,
        AUTHOR = "Nacson, M.S. and Aberdam, A. and Ganz, R. and Avraham, E.B. and Golts, A. and Kittenplon, Y. and Mazor, S. and Litman, R.",
        TITLE = "DocVLM: Make Your VLM an Efficient Reader",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "29005-29015",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242098"}

@inproceedings{bb247206,
        AUTHOR = "Alhamoud, K. and Alshammari, S. and Tian, Y.L. and Li, G.H. and Torr, P.H.S. and Kim, Y. and Ghassemi, M.",
        TITLE = "Vision-Language Models Do Not Understand Negation",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "29612-29622",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242099"}

@inproceedings{bb247207,
        AUTHOR = "Schmalfuss, J. and Chang, N. and VS, V. and Shen, M. and Bruhn, A. and Alvarez, J.M.",
        TITLE = "PARC: A Quantitative Framework Uncovering the Symmetries within
Vision Language Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "25081-25091",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242100"}

@inproceedings{bb247208,
        AUTHOR = "Xiao, J.Q. and Sang, S. and Zhi, T.C. and Liu, J. and Yan, Q. and Luo, L.J. and Yuan, B.",
        TITLE = "COAP: Memory-Efficient Training with Correlation-Aware Gradient
Projection",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "30116-30126",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242101"}

@inproceedings{bb247209,
        AUTHOR = "Zhu, Y.Q. and Wang, Z.Y. and Zhang, C. and Li, P. and Liu, Y.",
        TITLE = "CoSpace: Benchmarking Continuous Space Perception Ability for
Vision-Language Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "29569-29579",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242102"}

@inproceedings{bb247210,
        AUTHOR = "Kang, H.Q. and Sachdeva, E. and Gupta, P. and Bae, S.J. and Lee, K.",
        TITLE = "GFlowVLM: Enhancing Multi-step Reasoning in Vision-Language Models
with Generative Flow Networks",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "3815-3825",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242103"}

@inproceedings{bb247211,
        AUTHOR = "Zhang, K. and Li, J.Y. and Li, Z. and Zhou, S.K.",
        TITLE = "DH-Set: Improving Vision-Language Alignment with Diverse and Hybrid
Set-Embeddings Learning",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "24993-25003",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242104"}

@inproceedings{bb247212,
        AUTHOR = "Saravanan, D. and Gupta, V. and Singh, D. and Khan, Z. and Gandhi, V. and Tapaswi, M.",
        TITLE = "VELOCITI: Benchmarking Video-Language Compositional Reasoning with
Strict Entailment",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "18914-18924",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242105"}

@inproceedings{bb247213,
        AUTHOR = "Pan, B. and Li, Q. and Tang, X.Y. and Huang, W. and Fang, Z. and Liu, F. and Wang, J.Y. and Yu, J.Y. and Shi, Y.",
        TITLE = "NLPrompt: Noise-Label Prompt Learning for Vision-Language Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "19963-19973",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242106"}

@inproceedings{bb247214,
        AUTHOR = "Zhang, Y.T. and Chen, L. and Zheng, G.D. and Gao, Y.F. and Zheng, R. and Fu, J. and Yin, Z.F. and Jin, S. and Qiao, Y. and Huang, X.J. and Zhao, F. and Gui, T. and Shao, J.",
        TITLE = "SPA-VL: A Comprehensive Safety Preference Alignment Dataset for
Vision Language Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "19867-19878",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242107"}

@inproceedings{bb247215,
        AUTHOR = "Zhou, E. and Su, Q. and Chi, C. and Zhang, Z.Z. and Wang, Z.Y. and Huang, T.J. and Sheng, L. and Wang, H.",
        TITLE = "Code-as-Monitor: Constraint-aware Visual Programming for Reactive and
Proactive Robotic Failure Detection",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "6919-6929",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242108"}

@inproceedings{bb247216,
        AUTHOR = "Song, C.H. and Blukis, V. and Tremblay, J. and Tyree, S. and Su, Y. and Birchfield, S.",
        TITLE = "RoboSpatial: Teaching Spatial Understanding to 2D and 3D
Vision-Language Models for Robotics",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "15768-15780",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242109"}

@inproceedings{bb247217,
        AUTHOR = "Lozano, A. and Sun, M.W. and Burgess, J. and Chen, L. and Nirschl, J.J. and Gu, J. and Lopez, I. and Aklilu, J. and Rau, A. and Katzer, A.W. and Zhang, Y.H. and Chiu, C. and Wang, X.H. and Song, A.S. and Tibshirani, R. and Yeung Levy, S.",
        TITLE = "BIOMEDICA: An Open Biomedical Image-Caption Archive, Dataset, and
Vision-Language Models Derived from Scientific Literature",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "19724-19735",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242110"}

@inproceedings{bb247218,
        AUTHOR = "Xiao, R. and Kim, S. and Georgescu, M.I. and Akata, Z. and Alaniz, S.",
        TITLE = "FLAIR: VLM with Fine-grained Language-informed Image Representations",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "24884-24894",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242111"}

@inproceedings{bb247219,
        AUTHOR = "Vasu, P.K.A. and Faghri, F. and Li, C.L. and Koc, C. and True, N. and Antony, A. and Santhanam, G. and Gabriel, J. and Grasch, P. and Tuzel, O. and Pouransari, H.",
        TITLE = "FastVLM: Efficient Vision Encoding for Vision Language Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "19769-19780",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242112"}

@inproceedings{bb247220,
        AUTHOR = "Chen, Q.Z. and Wang, C. and Wang, D. and Zhang, T. and Li, W. and He, X.F.",
        TITLE = "Lifelong Knowledge Editing for Vision Language Models with Low-Rank
Mixture-of-Experts",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "9455-9466",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242113"}

@inproceedings{bb247221,
        AUTHOR = "Liu, Z.J. and Zhu, L. and Shi, B. and Zhang, Z.Y. and Lou, Y.M. and Yang, S. and Xi, H.C. and Cao, S.Y. and Gu, Y.X. and Li, D.C. and Li, X. and Tang, H.T. and Fang, Y.H. and Chen, Y. and Hsieh, C.Y. and Huang, D.A. and Cheng, A.C. and Hu, J.Y. and Liu, S. and Krishna, R. and Molchanov, P. and Kautz, J. and Yin, H.X. and Han, S. and Lu, Y.",
        TITLE = "NVILA: Efficient Frontier Visual Language Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "4122-4134",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242114"}

@inproceedings{bb247222,
        AUTHOR = "Zhang, H.Y. and Guo, Y.Y. and Kankanhalli, M.",
        TITLE = "Joint Vision-Language Social Bias Removal for CLIP",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "4246-4255",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242115"}

@inproceedings{bb247223,
        AUTHOR = "Deng, A. and Cao, T. and Chen, Z. and Hooi, B.",
        TITLE = "Words or Vision: Do Vision-Language Models Have Blind Faith in Text?",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "3867-3876",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242116"}

@inproceedings{bb247224,
        AUTHOR = "Huang, R. and Ding, X.P. and Wang, C.W. and Han, J.H. and Liu, Y.L. and Zhao, H.S. and Xu, H. and Hou, L. and Zhang, W. and Liang, X.D.",
        TITLE = "HiRes-LLaVA: Restoring Fragmentation Input in High-Resolution Large
Vision-Language Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "29814-29824",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242117"}

@inproceedings{bb247225,
        AUTHOR = "Wang, S. and Zhang, Y.J. and Zhu, Y. and Li, J.N. and Wang, Z.Z. and Liu, Y.W. and Ji, X.Y.",
        TITLE = "Towards Understanding How Knowledge Evolves in Large Vision-Language
Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "29858-29868",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242118"}

@inproceedings{bb247226,
        AUTHOR = "Zhao, W. and Han, Y.Z. and Tang, J.S. and Li, Z. and Song, Y.B. and Wang, K. and Wang, Z.Y. and You, Y.",
        TITLE = "A Stitch in Time Saves Nine: Small VLM is a Precise Guidance for
Accelerating Large VLMs",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "19814-19824",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242119"}

@inproceedings{bb247227,
        AUTHOR = "Safaei, B. and Patel, V.M.",
        TITLE = "Active Learning for Vision-Language Models",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "4902-4912",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242120"}

@inproceedings{bb247228,
        AUTHOR = "Wang, Y.C. and Zhang, Z.K. and Wang, J. and Fan, D. and Xu, Z.L. and Liu, L. and Hao, X. and Bhat, V. and Li, X.Y.",
        TITLE = "GEXIA: Granularity Expansion and Iterative Approximation for Scalable
Multi-Grained Video-Language Learning",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "4725-4735",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242121"}

@inproceedings{bb247229,
        AUTHOR = "Colman, R. and Vu, M. and Bhattarai, M. and Ma, M. and Viswanathan, H. and O'Malley, D. and Santos, J.E.",
        TITLE = "PatchFinder: Leveraging Visual Language Models for Accurate
Information Retrieval Using Model Uncertainty",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "9146-9155",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242122"}

@inproceedings{bb247230,
        AUTHOR = "Jawade, B. and Soares, J.V.B. and Thadani, K. and Mohan, D.D. and Eshratifar, A.E. and Culpepper, B. and de Juan, P. and Setlur, S. and Govindaraju, V.",
        TITLE = "SCOT: Self-Supervised Contrastive Pretraining for Zero-Shot
Compositional Retrieval",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "5509-5519",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242123"}

@inproceedings{bb247231,
        AUTHOR = "Chang, H.S. and Wang, C.Y. and Wang, R.R. and Chou, G. and Liao, H.Y.M.",
        TITLE = "Generalist YOLO: Towards Real-Time End-to-End Multi-Task Visual
Language Models",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "6217-6227",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242124"}

@inproceedings{bb247232,
        AUTHOR = "Chen, H.N. and Ni, Y. and Huang, W.J. and Liu, Y. and Jeong, S. and Wen, F. and Bastian, N.D. and Latapie, H. and Imani, M.",
        TITLE = "VLTP: Vision-Language Guided Token Pruning for Task-Oriented
Segmentation",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "9353-9363",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242125"}

@inproceedings{bb247233,
        AUTHOR = "Yamada, M. and Dharamshi, N. and Kohli, A. and Kasu, P. and Khan, A. and Ghulyani, M.",
        TITLE = "Unleashing Potentials of Vision-Language Models for Zero-Shot HOI
Detection",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "5751-5760",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242126"}

@inproceedings{bb247234,
        AUTHOR = "Ghoddoosian, R. and Agarwal, N. and Dwivedi, I. and Darisuh, B.",
        TITLE = "ACE: Action Concept Enhancement of Video-Language Models in
Procedural Videos",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "9521-9531",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242127"}

@inproceedings{bb247235,
        AUTHOR = "Onoe, Y. and Rane, S. and Berger, Z. and Bitton, Y. and Cho, J. and Garg, R. and Ku, A. and Parekh, Z. and Pont Tuset, J. and Tanzer, G. and Wang, S. and Baldridge, J.",
        TITLE = "DOCCI: Descriptions of Connected and Contrasting Images",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "LX: 291-309",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242128"}

@inproceedings{bb247236,
        AUTHOR = "Li, T. and Ma, M.M. and Peng, X.",
        TITLE = "DEAL: Disentangle and Localize Concept-level Explanations for VLMs",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XXXIX: 383-401",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242129"}

@inproceedings{bb247237,
        AUTHOR = "Li, S.C. and Li, L. and Liu, Y. and Ren, S.H. and Liu, Y.X. and Gao, R.D. and Sun, X. and Hou, L.",
        TITLE = "Vitatecs: A Diagnostic Dataset for Temporal Concept Understanding of
Video-language Models",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "LXX: 331-348",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242130"}

@inproceedings{bb247238,
        AUTHOR = "Rahmanzadehgervi, P. and Bolton, L. and Taesiri, M.R. and Nguyen, A.T.",
        TITLE = "Vision Language Models are blind",
        BOOKTITLE = ACCV24,
        YEAR = "2024",
        PAGES = "V: 293-309",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242131"}

@inproceedings{bb247239,
        AUTHOR = "Chytas, S.P. and Kim, H.W.J. and Singh, V.",
        TITLE = "Understanding Multi-compositional Learning in Vision and Language
Models via Category Theory",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XLVIII: 324-341",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242132"}

@inproceedings{bb247240,
        AUTHOR = "Song, Y.Z. and Chen, Y.S. and Lin, T.L. and Liu, B. and Fu, J.L. and Shuai, H.H.",
        TITLE = "Capture Concept Through Comparison: Vision-and-language Representation
Learning with Intrinsic Information Mining",
        BOOKTITLE = ACCV24,
        YEAR = "2024",
        PAGES = "III: 220-238",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242133"}

@inproceedings{bb247241,
        AUTHOR = "He, H.C. and Liu, W.B. and Xing, W.W.",
        TITLE = "Biefficient: Bidirectionally Prompting Vision-language Models for
Parameter-efficient Video Recognition",
        BOOKTITLE = ACCV24,
        YEAR = "2024",
        PAGES = "III: 257-274",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242134"}

@inproceedings{bb247242,
        AUTHOR = "Yang, J.K. and Dong, Y.H. and Liu, S. and Li, B. and Wang, Z.Y. and Tan, H.R. and Jiang, C.C. and Kang, J. and Zhang, Y.H. and Zhou, K.Y. and Liu, Z.W.",
        TITLE = "Octopus: Embodied Vision-language Programmer from Environmental
Feedback",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "I: 20-38",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242135"}

@inproceedings{bb247243,
        AUTHOR = "Kar, O.F. and Tonioni, A. and Poklukar, P. and Kulshrestha, A. and Zamir, A. and Tombari, F.",
        TITLE = "Brave: Broadening the Visual Encoding of Vision-language Models",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XVI: 113-132",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242136"}

@inproceedings{bb247244,
        AUTHOR = "Kamath, A. and Hsieh, C.Y. and Chang, K.W. and Krishna, R.",
        TITLE = "The Hard Positive Truth About Vision-language Compositionality",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XIV: 37-54",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242137"}

@inproceedings{bb247245,
        AUTHOR = "Jia, B.X. and Chen, Y.X. and Yu, H.Y. and Wang, Y. and Niu, X.S. and Liu, T.Y. and Li, Q. and Huang, S.Y.",
        TITLE = "Sceneverse: Scaling 3d Vision-language Learning for Grounded Scene
Understanding",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "IX: 289-310",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242138"}

@inproceedings{bb247246,
        AUTHOR = "Zhang, Y.F. and Jiang, M. and Zhao, Q.",
        TITLE = "Learning Chain of Counterfactual Thought for Bias-robust
Vision-language Reasoning",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "VIII: 334-351",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242139"}

@inproceedings{bb247247,
        AUTHOR = "Li, J. and Chen, D. and Cai, T. and Chen, P.H. and Hong, Y. and Chen, Z.F. and Shen, Y.K. and Gan, C.",
        TITLE = "Flexattention for Efficient High-resolution Vision-language Models",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XXV: 286-302",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242140"}

@inproceedings{bb247248,
        AUTHOR = "Li, X. and Ding, J. and Chen, Z.Y. and Elhoseiny, M.",
        TITLE = "UNI3DL: A Unified Model for 3d Vision-language Understanding",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XXIII: 74-92",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242141"}

@inproceedings{bb247249,
        AUTHOR = "Hao, T.X. and Ding, X.H. and Feng, J.X. and Yang, Y.H. and Chen, H. and Ding, G.",
        TITLE = "Quantized Prompt for Efficient Generalization of Vision-language Models",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XIX: 54-73",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242142"}

@inproceedings{bb247250,
        AUTHOR = "Xu, H.B. and Ke, X. and Li, Y.Z. and Xu, R. and Wu, H.Q. and Lin, X.F. and Guo, W.Z.",
        TITLE = "Vision-language Action Knowledge Learning for Semantic-aware Action
Quality Assessment",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XLII: 423-440",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242143"}

@inproceedings{bb247251,
        AUTHOR = "Zhu, Z.Y. and Zhang, Z. and Ma, X.J. and Niu, X.S. and Chen, Y.X. and Jia, B.X. and Deng, Z.D. and Huang, S.Y. and Li, Q.",
        TITLE = "Unifying 3d Vision-language Understanding via Promptable Queries",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XLIV: 188-206",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242144"}

@inproceedings{bb247252,
        AUTHOR = "Jiang, H.B. and Yue, J.P. and Luo, H. and Ding, Z. and Lu, Z.Q.",
        TITLE = "Reinforcement Learning Friendly Vision-language Model for Minecraft",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "LXVIII: 1-17",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242145"}

@inproceedings{bb247253,
        AUTHOR = "Nguyen, A.T. and Tai, K.S. and Chen, B.C. and Shukla, S.N. and Yu, H.C. and Torr, P.H.S. and Tian, T.P. and Lim, S.N.",
        TITLE = "ucap: An Unsupervised Prompting Method for Vision-language Models",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "LXXIV: 425-439",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242146"}

@inproceedings{bb247254,
        AUTHOR = "Zhang, Y. and Yu, K. and Wu, S.Q. and He, Z.H.",
        TITLE = "Conceptual Codebook Learning for Vision-language Models",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "LXXVII: 235-251",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242147"}

@inproceedings{bb247255,
        AUTHOR = "Chatterjee, A. and Luo, Y.R. and Gokhale, T. and Yang, Y.Z. and Baral, C.",
        TITLE = "Revision: Rendering Tools Enable Spatial Fidelity in Vision-language
Models",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XXX: 339-357",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242148"}

@inproceedings{bb247256,
        AUTHOR = "Sharma, P. and Shaham, T.R. and Baradad, M. and Rodriiuez Munoz, A. and Duggal, S. and Isola, P. and Torralba, A. and Fu, S.",
        TITLE = "A Vision Check-up for Language Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14410-14419",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242149"}

@inproceedings{bb247257,
        AUTHOR = "Parodi, F. and Matelsky, J.K. and Regla Vargas, A. and Foglia, E.E. and Lim, C. and Weinberg, D. and Kording, K.P. and Herrick, H.M. and Platt, M.L.",
        TITLE = "Vision-language models for decoding provider attention during
neonatal resuscitation",
        BOOKTITLE = CVPM24,
        YEAR = "2024",
        PAGES = "343-353",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242150"}

@inproceedings{bb247258,
        AUTHOR = "Li, L. and Guan, H.Y. and Qiu, J.N. and Spratling, M.",
        TITLE = "One Prompt Word is Enough to Boost Adversarial Robustness for
Pre-Trained Vision-Language Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "24408-24419",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242151"}

@inproceedings{bb247259,
        AUTHOR = "Cui, J.Q. and Zhu, B. and Wen, X. and Qi, X.J. and Yu, B. and Zhang, H.W.",
        TITLE = "Classes Are Not Equal: An Empirical Study on Image Recognition
Fairness",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "23283-23292",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242152"}

@inproceedings{bb247260,
        AUTHOR = "Stojnic, V. and Kalantidis, Y. and Tolias, G.",
        TITLE = "Label Propagation for Zero-shot Classification with Vision-Language
Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "23209-23218",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242153"}

@inproceedings{bb247261,
        AUTHOR = "Yuan, T.T. and Zhang, X. and Liu, K. and Liu, B. and Chen, C. and Jin, J. and Jiao, Z.Z.",
        TITLE = "Towards Surveillance Video-and-Language Understanding: New Dataset,
Baselines, and Challenges",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "22052-22061",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242154"}

@inproceedings{bb247262,
        AUTHOR = "Mittal, H. and Agarwal, N. and Lo, S.Y. and Lee, K.",
        TITLE = "Can't make an Omelette without Breaking some Eggs: Plausible Action
Anticipation using Large Video-Language Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "18580-18590",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242155"}

@inproceedings{bb247263,
        AUTHOR = "Zhao, G.L. and Li, G.B. and Chen, W.K. and Yu, Y.Z.",
        TITLE = "OVER-NAV: Elevating Iterative Vision-and-Language Navigation with
Open-Vocabulary Detection and StructurEd Representation",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "16296-16306",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242156"}

@inproceedings{bb247264,
        AUTHOR = "Li, X. and Wu, Y.F. and Jiang, X.H. and Guo, Z.H. and Gong, M.M. and Cao, H.Y. and Liu, Y.S. and Jiang, D.Q. and Sun, X.",
        TITLE = "Enhancing Visual Document Understanding with Contrastive Learning in
Large Visual-Language Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "15546-15555",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242157"}

@inproceedings{bb247265,
        AUTHOR = "Pham, K. and Huynh, C. and Lim, S.N. and Shrivastava, A.",
        TITLE = "Composing Object Relations and Attributes for Image-Text Matching",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14354-14363",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242158"}

@inproceedings{bb247266,
        AUTHOR = "Xu, Z.L. and Zhu, Y. and Deng, S.Q. and Mittal, A. and Chen, Y.B. and Wang, M. and Favaro, P. and Tighe, J. and Modolo, D.",
        TITLE = "Benchmarking Zero-Shot Recognition with Vision-Language Models:
Challenges on Granularity and Specificity",
        BOOKTITLE = WhatNext24,
        YEAR = "2024",
        PAGES = "1827-1836",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242159"}

@inproceedings{bb247267,
        AUTHOR = "Luo, Z.W. and Gustafsson, F.K. and Zhao, Z. and Sjolund, J. and Schon, T.B.",
        TITLE = "Photo-Realistic Image Restoration in the Wild with Controlled
Vision-Language Models",
        BOOKTITLE = NTIRE24,
        YEAR = "2024",
        PAGES = "6641-6651",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242160"}

@inproceedings{bb247268,
        AUTHOR = "Pan, C. and Yaman, B. and Nesti, T. and Mallik, A. and Allievi, A.G. and Velipasalar, S. and Ren, L.",
        TITLE = "VLP: Vision Language Planning for Autonomous Driving",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14760-14769",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242161"}

@inproceedings{bb247269,
        AUTHOR = "Liang, M. and Su, J.C. and Schulter, S. and Garg, S. and Zhao, S.Y. and Wu, Y. and Chandraker, M.",
        TITLE = "AIDE: An Automatic Data Engine for Object Detection in Autonomous
Driving",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14695-14706",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242162"}

@inproceedings{bb247270,
        AUTHOR = "Li, Z. and Li, X. and Fu, X. and Zhang, X. and Wang, W.Q. and Chen, S. and Yang, J.",
        TITLE = "PromptKD: Unsupervised Prompt Distillation for Vision-Language Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "26607-26616",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242163"}

@inproceedings{bb247271,
        AUTHOR = "Zhang, L. and Awal, R. and Agrawal, A.",
        TITLE = "Contrasting Intra-Modal and Ranking Cross-Modal Hard Negatives to
Enhance Visio-Linguistic Compositional Understanding",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13774-13784",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242164"}

@inproceedings{bb247272,
        AUTHOR = "Rosasco, A. and Berti, S. and Pasquale, G. and Malafronte, D. and Sato, S. and Segawa, H. and Inada, T. and Natale, L.",
        TITLE = "ConCon-Chi: Concept-Context Chimera Benchmark for Personalized
Vision-Language Tasks",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "22239-22248",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242165"}

@inproceedings{bb247273,
        AUTHOR = "Cheng, S. and Guo, Z.C. and Wu, J. and Fang, K. and Li, P. and Liu, H.P. and Liu, Y.",
        TITLE = "EgoThink: Evaluating First-Person Perspective Thinking Capability of
Vision-Language Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14291-14302",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242166"}

@inproceedings{bb247274,
        AUTHOR = "Kil, J. and Song, C.H. and Zheng, B. and Deng, X. and Su, Y. and Chao, W.L.",
        TITLE = "Dual-View Visual Contextualization for Web Navigation",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14445-14454",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242167"}

@inproceedings{bb247275,
        AUTHOR = "Guo, Y.Y. and Wang, G.Z. and Kankanhalli, M.",
        TITLE = "PELA: Learning Parameter-Efficient Models with Low-Rank Approximation",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "15699-15709",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242168"}

@inproceedings{bb247276,
        AUTHOR = "Farina, M. and Mancini, M. and Cunegatti, E. and Cunegatti, E. and Iacca, G. and Ricci, E.",
        TITLE = "MULTIFLOW: Shifting Towards Task-Agnostic Vision-Language Pruning",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "16185-16195",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242169"}

@inproceedings{bb247277,
        AUTHOR = "Mu, F.Z. and Mo, S.C. and Li, Y.",
        TITLE = "SnAG: Scalable and Accurate Video Grounding",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "18930-18940",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242170"}

@inproceedings{bb247278,
        AUTHOR = "Cao, Y.H. and Ji, K.X. and Huang, Z.Y. and Zheng, C.Y. and Liu, J.J. and Wang, J. and Chen, J.D. and Yang, M.",
        TITLE = "Towards Better Vision-Inspired Vision-Language Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13537-13547",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242171"}

@inproceedings{bb247279,
        AUTHOR = "Shi, K.Y. and Dong, Q. and Goncalves, L. and Tu, Z.W. and Soatto, S.",
        TITLE = "Non-autoregressive Sequence-to-Sequence Vision-Language Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13603-13612",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242172"}

@inproceedings{bb247280,
        AUTHOR = "Man, Y.Z. and Gui, L.Y. and Wang, Y.X.",
        TITLE = "Situational Awareness Matters in 3D Vision Language Reasoning",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13678-13688",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242173"}

@inproceedings{bb247281,
        AUTHOR = "Zheng, C.H. and Zhang, J. and Kembhavi, A. and Krishna, R.",
        TITLE = "Iterated Learning Improves Compositionality in Large Vision-Language
Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13785-13795",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242174"}

@inproceedings{bb247282,
        AUTHOR = "Song, C.H. and Hwang, T. and Yoon, J.Y. and Choi, S. and Gu, Y.H.",
        TITLE = "SyncMask: Synchronized Attentional Masking for Fashion-centric
Vision-Language Pretraining",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13948-13957",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242175"}

@inproceedings{bb247283,
        AUTHOR = "Pramanick, S. and Han, G.X. and Hou, R. and Nag, S. and Lim, S.N. and Ballas, N. and Wang, Q.F. and Chellappa, R. and Almahairi, A.",
        TITLE = "Jack of All Tasks, Master of Many: Designing General-purpose
Coarse-to-Fine Vision-Language Model",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14076-14088",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242176"}

@inproceedings{bb247284,
        AUTHOR = "Zeng, Y. and Huang, Y. and Zhang, J.J. and Jie, Z.Q. and Chai, Z.H. and Wang, L.",
        TITLE = "Investigating Compositional Challenges in Vision-Language Models for
Visual Grounding",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14141-14151",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242177"}

@inproceedings{bb247285,
        AUTHOR = "Sameni, S. and Kafle, K. and Tan, H. and Jenni, S.",
        TITLE = "Building Vision-Language Models on Solid Foundations with Masked
Distillation",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14216-14226",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242178"}

@inproceedings{bb247286,
        AUTHOR = "Peng, W. and Xie, S.C. and You, Z. and Lan, S.Y. and Wu, Z.X.",
        TITLE = "Synthesize, Diagnose, and Optimize: Towards Fine-Grained
Vision-Language Understanding",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13279-13288",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242179"}

@inproceedings{bb247287,
        AUTHOR = "Chen, J.N. and Yu, Q.H. and Shen, X.H. and Yuille, A.L. and Chen, L.C.",
        TITLE = "ViTamin: Designing Scalable Vision Models in the Vision-Language Era",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "12954-12966",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242180"}

@inproceedings{bb247288,
        AUTHOR = "Liu, S.H. and Yu, S. and Lin, Z.Q. and Pathak, D. and Ramanan, D.",
        TITLE = "Language Models as Black-Box Optimizers for Vision-Language Models",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "12687-12697",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242181"}

@inproceedings{bb247289,
        AUTHOR = "Howard, P. and Madasu, A. and Le, T. and Moreno, G.L. and Bhiwandiwalla, A. and Lal, V.",
        TITLE = "SocialCounterfactuals: Probing and Mitigating Intersectional Social
Biases in Vision-Language Models with Counterfactual Examples",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "11975-11985",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242182"}

@inproceedings{bb247290,
        AUTHOR = "Jiang, Y.K. and Huang, Z.Z. and Zhang, R.Z. and Zhang, X.F. and Zhang, S.T.",
        TITLE = "ZePT: Zero-Shot Pan-Tumor Segmentation via Query-Disentangling and
Self-Prompting",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "11386-11397",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242183"}

@inproceedings{bb247291,
        AUTHOR = "Kim, Y. and Mo, S. and Kim, M. and Lee, K. and Lee, J. and Shin, J.",
        TITLE = "Discovering and Mitigating Visual Biases Through Keyword Explanation",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "11082-11092",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242184"}

@inproceedings{bb247292,
        AUTHOR = "Li, R. and Fischer, T. and Segu, M. and Pollefeys, M. and Van Gool, L.J. and Tombari, F.",
        TITLE = "Know Your Neighbors: Improving Single-View Reconstruction via Spatial
Vision-Language Reasoning",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "9848-9858",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242185"}

@inproceedings{bb247293,
        AUTHOR = "Zeng, Z. and Wang, D. and Yang, F.Y. and Park, H. and Soatto, S. and Lao, D. and Wong, A.",
        TITLE = "WorDepth: Variational Language Prior for Monocular Depth Estimation",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "9708-9719",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242186"}

@inproceedings{bb247294,
        AUTHOR = "Yang, C. and Xu, R. and Guo, Y. and Huang, P.X. and Chen, Y. and Ding, W. and Wang, Z.Y. and Zhou, H.",
        TITLE = "Improving Vision-and-Language Reasoning via Spatial Relations
Modeling",
        BOOKTITLE = WACV24,
        YEAR = "2024",
        PAGES = "758-767",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242187"}

@inproceedings{bb247295,
        AUTHOR = "Zhang, G.Y. and Zhang, Y.R. and Zhang, K. and Tresp, V.",
        TITLE = "Can Vision-Language Models be a Good Guesser? Exploring VLMs for
Times and Location Reasoning",
        BOOKTITLE = WACV24,
        YEAR = "2024",
        PAGES = "625-634",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242188"}

@inproceedings{bb247296,
        AUTHOR = "Ganz, R. and Nuriel, O. and Aberdam, A. and Kittenplon, Y. and Mazor, S. and Litman, R.",
        TITLE = "Towards Models that Can See and Read",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "21661-21671",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242189"}

@inproceedings{bb247297,
        AUTHOR = "Zhang, H. and Liu, D. and Lv, Z. and Su, B. and Tao, D.C.",
        TITLE = "Exploring Temporal Concurrency for Video-Language Representation
Learning",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15522-15532",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242190"}

@inproceedings{bb247298,
        AUTHOR = "Shukor, M. and Dancette, C. and Cord, M.",
        TITLE = "eP-ALM: Efficient Perceptual Augmentation of Language Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "21999-22012",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242191"}

@inproceedings{bb247299,
        AUTHOR = "Schulter, S. and Kumar, B.G.V. and Suh, Y.M. and Dafnis, K.M. and Zhang, Z.X. and Zhao, S.Y. and Metaxas, D.N.",
        TITLE = "OmniLabel: A Challenging Benchmark for Language-Based Object
Detection",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "11919-11928",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT242192"}

Last update:Sep 30, 2026 at 11:45:00